officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -1,5 +1,73 @@
1
1
  // Generated by dts-bundle-generator v9.5.1
2
2
 
3
+ /**
4
+ * Standard error types for OfficeParser.
5
+ * Use these to identify the kind of error being reported.
6
+ */
7
+ export declare enum OfficeErrorType {
8
+ /** Unsupported file extension */
9
+ EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED",
10
+ /** File appears to be corrupted or malformed */
11
+ FILE_CORRUPTED = "FILE_CORRUPTED",
12
+ /** File could not be found at the specified path */
13
+ FILE_DOES_NOT_EXIST = "FILE_DOES_NOT_EXIST",
14
+ /** Specified location/directory is not reachable or is a directory */
15
+ LOCATION_NOT_FOUND = "LOCATION_NOT_FOUND",
16
+ /** Arguments passed to the function are missing or invalid */
17
+ IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS",
18
+ /** Error occurred while reading or processing file buffers */
19
+ IMPROPER_BUFFERS = "IMPROPER_BUFFERS",
20
+ /** Input type is not a supported type (string, Buffer, ArrayBuffer) */
21
+ INVALID_INPUT = "INVALID_INPUT",
22
+ /** PDF worker source is missing (required in browser) */
23
+ PDF_WORKER_MISSING = "PDF_WORKER_MISSING",
24
+ /** Attempted to use Node.js-only features in a browser environment */
25
+ FEATURE_NOT_SUPPORTED_IN_BROWSER = "FEATURE_NOT_SUPPORTED_IN_BROWSER",
26
+ /** Style mapping string is malformed */
27
+ INVALID_STYLE_MAPPING = "INVALID_STYLE_MAPPING",
28
+ /** Selector in style mapping is invalid */
29
+ INVALID_SELECTOR = "INVALID_SELECTOR",
30
+ /** Output mapping in style mapping is invalid */
31
+ INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
32
+ /** Semantic chunking strategy is selected but no embedding function is provided */
33
+ MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
34
+ }
35
+ /**
36
+ * Standard warning types for OfficeParser.
37
+ * Use these for reporting non-fatal issues or performance tips.
38
+ */
39
+ export declare enum OfficeWarningType {
40
+ /** Performance advice (e.g., Rosetta translation on Mac) */
41
+ PERFORMANCE_TIP = "PERFORMANCE_TIP",
42
+ /** OCR processing failed for an attachment */
43
+ OCR_FAILED = "OCR_FAILED",
44
+ /** Extraction of structured chart data failed */
45
+ CHART_DATA_EXTRACTION_FAILED = "CHART_DATA_EXTRACTION_FAILED",
46
+ /** Automatic worker path failed, falling back to CDN */
47
+ PDF_WORKER_FALLBACK = "PDF_WORKER_FALLBACK",
48
+ /** General attachment extraction failure */
49
+ ATTACHMENT_EXTRACTION_FAILED = "ATTACHMENT_EXTRACTION_FAILED",
50
+ /** Failed to load a specific page in a multi-page document */
51
+ PAGE_LOAD_FAILED = "PAGE_LOAD_FAILED",
52
+ /** Failed to load a required dynamic dependency */
53
+ DEPENDENCY_LOAD_FAILED = "DEPENDENCY_LOAD_FAILED",
54
+ /** Failed to extract images from a source */
55
+ IMAGE_EXTRACTION_FAILED = "IMAGE_EXTRACTION_FAILED",
56
+ /** Failed to extract annotations from a document */
57
+ ANNOTATION_EXTRACTION_FAILED = "ANNOTATION_EXTRACTION_FAILED",
58
+ /** Failed to process an extracted image bitmap */
59
+ IMAGE_PROCESSING_FAILED = "IMAGE_PROCESSING_FAILED",
60
+ /** Warning about limitations of browser-based generation */
61
+ BROWSER_GENERATION_LIMITATION = "BROWSER_GENERATION_LIMITATION",
62
+ /** Specified sheet range in Excel/ODS export was not found */
63
+ SHEET_RANGE_NOT_FOUND = "SHEET_RANGE_NOT_FOUND",
64
+ /** Buffer content type does not match the provided or expected file extension */
65
+ BUFFER_TYPE_MISMATCH = "BUFFER_TYPE_MISMATCH",
66
+ /** No chunks were generated for the document given the current strategy */
67
+ EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
68
+ /** A node was skipped because it only contained whitespace */
69
+ WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
70
+ }
3
71
  /**
4
72
  * Configuration options for OCR.
5
73
  */
@@ -18,16 +86,19 @@ export interface OcrConfig {
18
86
  /**
19
87
  * Path to the Tesseract worker script.
20
88
  * Primarily used for offline/air-gapped environments.
89
+ * Default is ''.
21
90
  */
22
91
  workerPath?: string;
23
92
  /**
24
93
  * Path to the Tesseract core script.
25
94
  * Primarily used for offline/air-gapped environments.
95
+ * Default is ''.
26
96
  */
27
97
  corePath?: string;
28
98
  /**
29
99
  * Path for Tesseract language files (traineddata).
30
100
  * Primarily used for offline/air-gapped environments.
101
+ * Default is ''.
31
102
  */
32
103
  langPath?: string;
33
104
  /**
@@ -42,10 +113,17 @@ export interface OcrConfig {
42
113
  */
43
114
  export interface OfficeParserConfig {
44
115
  /**
116
+ * @deprecated Use `onWarning` instead.
45
117
  * Flag to show all the logs to console in case of an error irrespective of your own handling.
46
118
  * Default is false.
47
119
  */
48
120
  outputErrorToConsole?: boolean;
121
+ /**
122
+ * Callback for warnings or non-fatal errors encountered during parsing.
123
+ * Allows you to capture issues like OCR failures or attachment extraction errors
124
+ * without stopping the parsing process.
125
+ */
126
+ onWarning?: (issue: OfficeIssue) => void;
49
127
  /**
50
128
  * The delimiter used for every new line in places that allow multiline text like word.
51
129
  * Default is \n.
@@ -114,23 +192,644 @@ export interface OfficeParserConfig {
114
192
  * The URL/path to the PDF.js worker script.
115
193
  *
116
194
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
117
- * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
195
+ * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
118
196
  * You can override this with your own local path or a different CDN link.
119
197
  */
120
198
  pdfWorkerSrc?: string;
199
+ /**
200
+ * Flag to include break nodes in the AST.
201
+ * This is currently only supported for Word documents. (w:br nodes)
202
+ *
203
+ * Default is false
204
+ */
205
+ includeBreakNodes?: boolean;
206
+ /**
207
+ * Flag to ignore all internal (anchor) links during parsing.
208
+ * When true, all bookmarks, cross-references, and internal document jumps are stripped
209
+ * from the AST. Only external URLs will be preserved.
210
+ *
211
+ * Use this if you want a "flat" document without any internal interactivity.
212
+ *
213
+ * Default is false.
214
+ */
215
+ ignoreInternalLinks?: boolean;
216
+ /**
217
+ * Optional hint for the file format.
218
+ * When a Buffer or ArrayBuffer is passed, the parser relies on magic bytes to detect the file type.
219
+ * Text-based formats like 'md', 'html', and 'csv' lack reliable magic bytes.
220
+ * If you are parsing these formats from a Buffer, you must provide this fileType hint.
221
+ *
222
+ * This is authoritative and is used to determine the file type, so it should be accurate.
223
+ * If provided, this bypasses the magic bytes detection and the file extension-based detection either way.
224
+ *
225
+ * Default is null.
226
+ */
227
+ fileType?: SupportedFileType | null;
228
+ /**
229
+ * Custom delimiter for CSV files.
230
+ * Defaults to ',' but can be overridden (e.g., ';', '\t').
231
+ */
232
+ csvDelimiter?: string;
233
+ }
234
+ /**
235
+ * Represents a single issue (warning, error, or info) generated during document processing.
236
+ */
237
+ export interface OfficeIssue {
238
+ /** The severity of the issue. */
239
+ type: "warning" | "info" | "error";
240
+ /** Human-readable message text. */
241
+ message: string;
242
+ /** The specific AST node that triggered this issue, if applicable. */
243
+ node?: OfficeContentNode;
244
+ /** A unique error code for programmatic handling. */
245
+ code: OfficeWarningType | OfficeErrorType;
246
+ /** Optional additional context or original error object. */
247
+ details?: any;
248
+ }
249
+ /**
250
+ * The result of a document conversion operation.
251
+ */
252
+ export interface ConversionResult<D extends string = UniversalGeneratorFormat> {
253
+ /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
254
+ value: D extends "pdf" ? Uint8Array : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : D extends UniversalGeneratorFormat ? string : never;
255
+ /** A collection of issues (warnings/infos) generated during the process. */
256
+ messages: OfficeIssue[];
257
+ }
258
+ /**
259
+ * Universal formats supported by all source types for generation.
260
+ */
261
+ export type UniversalGeneratorFormat = "text" | "md" | "html" | "pdf" | "csv" | "rtf" | "chunks";
262
+ /**
263
+ * Allowed destination formats for a given source type.
264
+ * Currently, all generators are universal across all source formats.
265
+ */
266
+ export type SupportedDestination<_T extends SupportedFileType = SupportedFileType> = UniversalGeneratorFormat;
267
+ /**
268
+ * Configuration options for the OfficeGenerator.
269
+ */
270
+ /**
271
+ * Common configuration options for all generators.
272
+ */
273
+ export interface CommonGeneratorConfig {
274
+ /**
275
+ * Callback called for every node during generation.
276
+ * Allows users to modify nodes before processing, completely override rendering, or filter them out.
277
+ *
278
+ * #### Callback Capabilities:
279
+ * 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
280
+ * 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
281
+ * 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
282
+ * 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
283
+ */
284
+ onNode?: (node: OfficeContentNode) => string | false | Promise<string | false | void> | void;
285
+ /**
286
+ * Callback for warnings, non-fatal errors, or issues encountered during generation.
287
+ * Allows the process to continue while reporting skipping or approximation of content.
288
+ */
289
+ onWarning?: (issue: OfficeIssue) => void;
290
+ /**
291
+ * Map document styles (e.g., 'Heading 1', 'Intense Quote') to specific semantic elements.
292
+ *
293
+ * DESIGN PHILOSOPHY:
294
+ * This is the primary way to customize how the library interprets the visual
295
+ * structure of your source documents.
296
+ *
297
+ * To disable all semantic translation and use raw AST types only,
298
+ * set `ignoreDefaultStyleMap: true` and leave `styleMap` empty.
299
+ *
300
+ * It supports two formats:
301
+ *
302
+ * 1. LEGACY STRING DSL:
303
+ * Simple "selector => output" syntax. Highly compatible with mammoth.js style maps.
304
+ * @example ["p[style-name='Heading 1'] => h1"]
305
+ * @example ["p[style='Quote'] => blockquote"]
306
+ *
307
+ * 2. STRUCTURED OBJECTS (Recommended):
308
+ * More powerful and strictly typed. Ideal for complex logic or when you
309
+ * need to apply specific classes/attributes for the HTML generator.
310
+ * @example
311
+ * [
312
+ * {
313
+ * selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
314
+ * output: { tag: 'h1', classes: ['main-title'], attributes: { id: 'top' } }
315
+ * }
316
+ * ]
317
+ *
318
+ * Note: This property works in conjunction with `ignoreDefaultStyleMap`.
319
+ * Defaults to a robust built-in map that covers common standard Office styles.
320
+ */
321
+ styleMap?: string[] | StructuredStyleMapping[];
322
+ /**
323
+ * Whether to include visual formatting like font size, font family, and colors in the output.
324
+ * Set to false for clean, semantic output.
325
+ * Defaults to true.
326
+ */
327
+ includeFormatting?: boolean;
328
+ /**
329
+ * Whether to automatically generate unique slug-based IDs for headings.
330
+ * Useful for table-of-contents and anchor links.
331
+ * Defaults to true.
332
+ */
333
+ generateIds?: boolean;
334
+ /**
335
+ * Whether to render document metadata (title, author, etc.) as visible content
336
+ * in the generated output (e.g., a header block in HTML or plain text).
337
+ * Structural metadata (HTML <meta> tags, Markdown YAML frontmatter) is always included.
338
+ * Defaults to false.
339
+ */
340
+ renderMetadata?: boolean;
341
+ /**
342
+ * Whether to ignore the built-in default style mappings (e.g. "Heading 1" -> h1).
343
+ * Set to true if you want full control over style mapping.
344
+ * Defaults to false.
345
+ */
346
+ ignoreDefaultStyleMap?: boolean;
347
+ /**
348
+ * Whether to include images in the generated output.
349
+ * Defaults to true.
350
+ */
351
+ includeImages?: boolean;
352
+ /**
353
+ * Whether to include interactive charts in the generated output (HTML only).
354
+ * Defaults to true.
355
+ */
356
+ includeCharts?: boolean;
357
+ /**
358
+ * Whether to ignore all internal (anchor) links and anchor IDs during generation.
359
+ * When true, all bookmarks, cross-references, and internal document jumps are stripped.
360
+ * Specifically for Markdown, this removes the {#id} block from headings.
361
+ * Defaults to false.
362
+ */
363
+ ignoreInternalLinks?: boolean;
364
+ }
365
+ /**
366
+ * Destination-aware generator configuration.
367
+ * Restricts format-specific configurations to their respective destinations.
368
+ */
369
+ /**
370
+ * Mapping of destination formats to their specific configuration interfaces.
371
+ */
372
+ export interface GeneratorSubConfigMap {
373
+ html: HtmlGeneratorConfig;
374
+ md: MdGeneratorConfig;
375
+ pdf: PdfGeneratorConfig;
376
+ csv: CsvGeneratorConfig;
377
+ text: TextGeneratorConfig;
378
+ rtf: RtfGeneratorConfig;
379
+ chunks: ChunkingConfig;
380
+ }
381
+ /**
382
+ * Configuration options for document generators.
383
+ *
384
+ * This interface is designed to be format-aware. When you specify a destination format
385
+ * (e.g., `OfficeGenerator.generate(ast, 'html', config)`), the generic parameter `D`
386
+ * ensures that only the relevant sub-configuration (e.g., `htmlConfig`) is available
387
+ * for type checking.
388
+ *
389
+ * @template D The destination format string. Defaults to `string` for a general configuration.
390
+ */
391
+ export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & {
392
+ [K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
393
+ };
394
+ /**
395
+ * Configuration options for the OfficeConverter.
396
+ * Combines relevant parser and generator settings for a seamless one-step conversion.
397
+ *
398
+ * @template D The destination format string.
399
+ */
400
+ /**
401
+ * Configuration options for the OfficeConverter.
402
+ * Combines general generator settings with a specific subset of parser settings.
403
+ *
404
+ * @template D The destination format string.
405
+ * @template T The source file type.
406
+ */
407
+ export type OfficeConverterConfig<D extends string = string, T extends SupportedFileType = SupportedFileType> = {
408
+ /**
409
+ * Specific configuration for the source parsing phase.
410
+ */
411
+ parseConfig?: OfficeParserConfig & {
412
+ fileType?: T;
413
+ };
414
+ /**
415
+ * Specific configuration for the destination generation phase.
416
+ */
417
+ generatorConfig?: GeneratorConfig<D>;
418
+ /**
419
+ * Callback for warnings or non-fatal errors encountered during the entire conversion process.
420
+ * This is passed to both the parser and the generator.
421
+ * If provided, this takes precedence over callbacks inside parseConfig or generatorConfig.
422
+ */
423
+ onWarning?: (issue: OfficeIssue) => void;
424
+ };
425
+ /**
426
+ * Configuration options for HTML generation.
427
+ */
428
+ export interface HtmlGeneratorConfig {
429
+ /**
430
+ * Whether to wrap the output in a full HTML document structure (e.g., <html>, <head>, etc.).
431
+ * Defaults to true.
432
+ */
433
+ standalone?: boolean;
434
+ /**
435
+ * URL for the Chart.js library to use when 'includeCharts' is true.
436
+ * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
437
+ */
438
+ chartJsSrc?: string;
439
+ }
440
+ /**
441
+ * Configuration options for PDF generation.
442
+ * Maps closely to Puppeteer's PDF options.
443
+ */
444
+ export interface PdfGeneratorConfig {
445
+ /** Paper format. Defaults to 'A4'. */
446
+ format?: "letter" | "legal" | "tabloid" | "ledger" | "a0" | "a1" | "a2" | "a3" | "a4" | "a5" | "a6" | "Letter" | "Legal" | "Tabloid" | "Ledger" | "A0" | "A1" | "A2" | "A3" | "A4" | "A5" | "A6";
447
+ /** Paper width, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
448
+ width?: string | number;
449
+ /** Paper height, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
450
+ height?: string | number;
451
+ /** Whether to print in landscape orientation. Defaults to false. */
452
+ landscape?: boolean;
453
+ /** Whether to print background graphics. Defaults to true. */
454
+ printBackground?: boolean;
455
+ /** Scale of the webpage rendering. Defaults to 1. */
456
+ scale?: number;
457
+ /** Paper margins. */
458
+ margin?: {
459
+ top?: string | number;
460
+ right?: string | number;
461
+ bottom?: string | number;
462
+ left?: string | number;
463
+ };
464
+ /** Whether to display header and footer. Defaults to false. */
465
+ displayHeaderFooter?: boolean;
466
+ /** HTML template for the print header. */
467
+ headerTemplate?: string;
468
+ /** HTML template for the print footer. */
469
+ footerTemplate?: string;
470
+ /**
471
+ * Optional Puppeteer launch options for Node.js environment.
472
+ * Useful for setting custom executable paths or args in CI/CD.
473
+ */
474
+ launchOptions?: any;
475
+ }
476
+ /**
477
+ * Structured style mapping definition for the StyleMapper.
478
+ *
479
+ * DESIGN PHILOSOPHY: "Semantic Translation"
480
+ * -----------------------------------------
481
+ * Office documents (Word, RTF, PPTX) often use custom or localized style names
482
+ * (e.g., "Heading 1" in English vs "Titre 1" in French, or "MyCompany-Quote").
483
+ *
484
+ * This interface allows you to create a "semantic bridge" between these arbitrary
485
+ * source styles and a universal vocabulary of document elements.
486
+ *
487
+ * WHY USE HTML TAGS FOR NON-HTML OUTPUT?
488
+ * --------------------------------------
489
+ * We use HTML tags (`h1`, `blockquote`, `code`, `pre`) as a "Universal Intermediate
490
+ * Language". By mapping a custom Word style to `blockquote`, you are defining its
491
+ * SEMANTIC MEANING rather than its physical appearance.
492
+ *
493
+ * Each generator then interprets this meaning natively:
494
+ * - HTML Generator: Directly renders the `<blockquote>` tag with your classes.
495
+ * - Markdown Generator: Sees 'blockquote' and renders the standard `> ` prefix.
496
+ * - Text Generator: Sees 'blockquote' and applies appropriate structural indentation.
497
+ */
498
+ export interface StructuredStyleMapping {
499
+ /**
500
+ * The criteria used to identify which AST nodes should be transformed.
501
+ * Think of this as the "Source Filter".
502
+ */
503
+ selector: {
504
+ /**
505
+ * The structural type of the node (e.g., 'paragraph', 'heading', 'text').
506
+ * Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
507
+ */
508
+ nodeType?: string;
509
+ /**
510
+ * A dictionary of attributes to match on the node.
511
+ *
512
+ * The most common use case is matching the 'style' attribute from
513
+ * Word documents (e.g., { style: 'Intense Quote' }).
514
+ *
515
+ * Matchers:
516
+ * - Literal: `style: 'Heading 1'` matches exactly.
517
+ * - Operator: `{ value: 'Title', operator: '~=' }` matches if the word 'Title'
518
+ * is found within the style name.
519
+ */
520
+ attributes?: Record<string, string | number | boolean | {
521
+ value: string | number | boolean;
522
+ operator: "=" | "~=";
523
+ }>;
524
+ };
525
+ /**
526
+ * The target representation for the matched node.
527
+ * Think of this as the "Semantic Meaning" you want to assign to the match.
528
+ */
529
+ output: {
530
+ /**
531
+ * The universal semantic tag (e.g., 'h1', 'h2', 'blockquote', 'code', 'pre', 'u').
532
+ * All generators use this tag to decide their native output syntax.
533
+ */
534
+ tag: string;
535
+ /**
536
+ * CSS classes to apply to the output.
537
+ * This is utilized by the HTML generator to allow for downstream CSS styling.
538
+ */
539
+ classes?: string[];
540
+ /**
541
+ * Key-value pair of HTML attributes (like 'id', 'data-*', or 'style') to apply.
542
+ * Primarily used by the HTML generator for high-fidelity conversion.
543
+ */
544
+ attributes?: Record<string, string>;
545
+ /**
546
+ * If true, prevents the generator from collapsing this element into
547
+ * adjacent elements of the same type.
548
+ *
549
+ * For example, multiple paragraphs mapped to 'blockquote' normally merge into
550
+ * one big blockquote. Setting `fresh: true` forces them to be separate blocks.
551
+ */
552
+ fresh?: boolean;
553
+ };
554
+ }
555
+ /**
556
+ * Configuration options for RTF generation.
557
+ */
558
+ export interface RtfGeneratorConfig {
559
+ }
560
+ /**
561
+ * Configuration options for CSV generation.
562
+ */
563
+ export interface CsvGeneratorConfig {
564
+ /**
565
+ * Range of sheets to export.
566
+ * Supports formats like "1", "1-3", "1,2", "1,3-5,7".
567
+ * 1-based indexing.
568
+ * Default is '' (all sheets).
569
+ */
570
+ sheets?: string;
571
+ /**
572
+ * Whether to merge all selected sheets into a single CSV.
573
+ * If false, returns a ZIP archive containing individual CSV files.
574
+ * Defaults to false.
575
+ */
576
+ mergeSheets?: boolean;
577
+ /**
578
+ * Custom delimiter for CSV files.
579
+ * Defaults to ','.
580
+ */
581
+ columnDelimiter?: string;
582
+ }
583
+ /**
584
+ * Configuration options for Markdown generation.
585
+ */
586
+ export interface MdGeneratorConfig {
587
+ /**
588
+ * Whether to fallback to HTML tags for features not supported by standard Markdown.
589
+ *
590
+ * Markdown has limited support for complex document structures. This flag controls how
591
+ * the generator handles features that cannot be represented in pure Markdown:
592
+ *
593
+ * 1. If a feature is NOT supported natively by Markdown (e.g., nested tables, text alignment,
594
+ * underline, subscript/superscript):
595
+ * - If true: The generator will use HTML tags (<u>, <sub>, <div>, <table>, etc.) to
596
+ * maintain high fidelity.
597
+ * - If false: The generator will skip or simplify the feature (e.g., ignoring alignment,
598
+ * skipping underline, or hoisting nested tables out of their cells).
599
+ *
600
+ * 2. If a feature IS supported by Markdown but a higher quality version is possible
601
+ * via HTML (e.g., tables with merged cells):
602
+ * - If true: Use HTML for better fidelity.
603
+ * - If false: Use native Markdown syntax (e.g., a standard GFM table grid).
604
+ *
605
+ * Defaults to true.
606
+ */
607
+ fallbackToHtml?: boolean;
608
+ }
609
+ /**
610
+ * Configuration options for plain text generation.
611
+ */
612
+ export interface TextGeneratorConfig {
613
+ /**
614
+ * The delimiter used for every new line.
615
+ * Defaults to '\n'.
616
+ */
617
+ newlineDelimiter?: string;
618
+ /**
619
+ * Whether to attempt to preserve the original document layout.
620
+ * If true, tables will be rendered with separators and aligned columns.
621
+ * If false, output will be a flat stream of text nodes.
622
+ * Defaults to false.
623
+ */
624
+ preserveLayout?: boolean;
625
+ }
626
+ /**
627
+ * The strategy used for chunking a document for RAG pipelines.
628
+ * - 'fixed-size': Traditional character/token count based splitting.
629
+ * - 'document-structure': Leverages the AST to split at natural document boundaries.
630
+ * - 'semantic': Uses embedding similarity to find natural topic breakpoints.
631
+ */
632
+ export type ChunkingStrategy = "fixed-size" | "document-structure" | "semantic";
633
+ /**
634
+ * Base configuration applicable to all chunking strategies.
635
+ */
636
+ export interface BaseChunkingConfig {
637
+ /**
638
+ * The strategy used for chunking.
639
+ * Default is 'document-structure'.
640
+ */
641
+ strategy?: ChunkingStrategy;
642
+ /**
643
+ * A function that measures the size of a text string.
644
+ * Defaults to character count: `(text) => text.length`.
645
+ * Override with a token counter (e.g., `tiktoken`) for strict LLM context window adherence.
646
+ */
647
+ lengthFunction?: (text: string) => number;
648
+ /**
649
+ * Whether to strip leading/trailing whitespace from each chunk.
650
+ * Default is true.
651
+ */
652
+ stripWhitespace?: boolean;
653
+ /**
654
+ * Whether to include rich AST metadata (page number, slide number, heading, etc.)
655
+ * in the generated chunk objects.
656
+ * Default is true.
657
+ */
658
+ includeMetadata?: boolean;
659
+ /**
660
+ * Whether to include the starting character index of each chunk
661
+ * relative to the whole document. Useful for UI text highlighting.
662
+ * Default is false.
663
+ */
664
+ addStartIndex?: boolean;
665
+ /**
666
+ * Optional custom regex (as string or RegExp object) to identify sentence boundaries.
667
+ * Use this for languages or specific document types that require custom splitting logic.
668
+ * If provided, it overrides or augments the default segmenter.
669
+ * @example /[。?!]/
670
+ */
671
+ sentenceBoundaryRegex?: string | RegExp;
672
+ /**
673
+ * Optional list of abbreviations to ignore when splitting text into sentences.
674
+ * These words, if followed by a period, will not be treated as sentence boundaries.
675
+ * Use this to handle language-specific or domain-specific abbreviations.
676
+ * @example ["Inc", "Ltd", "approx"]
677
+ */
678
+ abbreviations?: string[];
679
+ }
680
+ /**
681
+ * Configuration for Fixed-Size Chunking.
682
+ * Cuts text based on a maximum size limit with an optional overlap.
683
+ * This is equivalent to LangChain's `RecursiveCharacterTextSplitter`.
684
+ */
685
+ export interface FixedSizeChunkingConfig extends BaseChunkingConfig {
686
+ strategy: "fixed-size";
687
+ /**
688
+ * Maximum size of the chunk, measured by `lengthFunction`.
689
+ * Default is 1000 characters.
690
+ */
691
+ chunkSize?: number;
692
+ /**
693
+ * Number of characters/tokens to overlap between consecutive chunks
694
+ * to avoid losing context at boundaries.
695
+ * Rule of thumb: ~10–20% of `chunkSize`.
696
+ * Default is 200.
697
+ */
698
+ chunkOverlap?: number;
699
+ /**
700
+ * Ordered list of separators to try when splitting.
701
+ * The chunker tries each in order; if a split would exceed `chunkSize`,
702
+ * it tries the next separator.
703
+ * Default is ['\n\n', '\n', ' ', ''].
704
+ */
705
+ separators?: string[];
706
+ }
707
+ /**
708
+ * Configuration for Document-Structure Chunking.
709
+ * Uses the officeParser AST to split at natural document boundaries like
710
+ * headings, paragraphs, slides, or pages. This is the recommended strategy
711
+ * as it preserves semantic context from the document's own structure.
712
+ */
713
+ export interface DocumentStructureChunkingConfig extends BaseChunkingConfig {
714
+ strategy: "document-structure";
715
+ /**
716
+ * The primary structural element at which to force a chunk boundary.
717
+ * - 'paragraph': Never cross a paragraph boundary (finest-grained, most precise).
718
+ * - 'heading': Split at every heading change.
719
+ * - 'page': Chunks never span multiple pages (PDF only).
720
+ * - 'slide': Chunks never span multiple slides (PPTX/ODP only).
721
+ * - 'sheet': Chunks never span multiple sheets (XLSX/ODS only).
722
+ * Default is 'paragraph'.
723
+ */
724
+ splitBy?: "page" | "slide" | "sheet" | "heading" | "paragraph";
725
+ /**
726
+ * Maximum size of a chunk (measured by `lengthFunction`).
727
+ * If a single structural unit (e.g., one paragraph) exceeds this limit,
728
+ * it will be further split using a recursive character splitter.
729
+ * Default is 1000 characters.
730
+ */
731
+ maxChunkSize?: number;
732
+ /**
733
+ * How to handle table nodes when splitting.
734
+ * - 'row': Split by rows, REPEATING the header row in every chunk so the LLM
735
+ * always understands what the columns mean. (Highly recommended for RAG)
736
+ * - 'flatten': Convert the table to plain text and split like a regular block.
737
+ * Default is 'row'.
738
+ */
739
+ tableSplitStrategy?: "row" | "flatten";
740
+ }
741
+ /**
742
+ * Configuration for Semantic Chunking.
743
+ * Uses an embedding model to detect topic shifts and create boundaries
744
+ * where content meaning naturally changes. Computationally expensive but
745
+ * produces the highest quality chunks.
746
+ */
747
+ export interface SemanticChunkingConfig extends BaseChunkingConfig {
748
+ strategy: "semantic";
749
+ /**
750
+ * A user-provided async function to generate vector embeddings for a text string.
751
+ * Required. Example: a wrapper around OpenAI's `text-embedding-3-small`.
752
+ * @example async (text) => await openai.embeddings.create({ input: text, model: 'text-embedding-3-small' }).then(r => r.data[0].embedding)
753
+ */
754
+ embeddingFunction: (text: string) => Promise<number[]>;
755
+ /**
756
+ * The cosine similarity threshold below which a chunk boundary is created.
757
+ * When the similarity between two adjacent sentences drops below this value,
758
+ * a new chunk starts. Higher = more splits, smaller chunks.
759
+ * Default is 0.8.
760
+ */
761
+ similarityThreshold?: number;
762
+ /**
763
+ * Maximum size of a chunk even if semantic similarity remains high.
764
+ * Prevents runaway chunks when an entire document is on one topic.
765
+ * Default is 2000 characters.
766
+ */
767
+ maxChunkSize?: number;
768
+ /**
769
+ * Number of surrounding sentences to include when computing similarity
770
+ * for a sentence. A larger window reduces noise from single odd sentences.
771
+ * Default is 1.
772
+ */
773
+ bufferSize?: number;
774
+ /**
775
+ * Number of sentences to process in a single batch when calling the embedding function.
776
+ * Higher values are faster but may trigger API rate limits.
777
+ * Default is 50.
778
+ */
779
+ embeddingBatchSize?: number;
780
+ }
781
+ /**
782
+ * Discriminated union of all chunking strategy configurations.
783
+ */
784
+ export type ChunkingConfig = FixedSizeChunkingConfig | DocumentStructureChunkingConfig | SemanticChunkingConfig;
785
+ /**
786
+ * Represents a single document chunk ready for a RAG (Retrieval-Augmented Generation) pipeline.
787
+ *
788
+ * Chunks are the result of splitting a document into smaller, semantically coherent
789
+ * pieces that fit within the context window of an LLM. Each chunk includes the
790
+ * extracted text and rich AST-derived metadata for citations and filtered retrieval.
791
+ */
792
+ export interface OfficeChunk {
793
+ /** The text content of this chunk. This is what gets embedded. */
794
+ text: string;
795
+ /**
796
+ * Rich contextual metadata extracted from the AST.
797
+ * Use this to populate vector DB metadata fields for filtered retrieval
798
+ * and for LLM citations.
799
+ */
800
+ metadata: {
801
+ /** The source file format (e.g., 'docx', 'pptx', 'pdf'). */
802
+ sourceType: SupportedFileType;
803
+ /** Page number (1-based), if available (PDF). */
804
+ pageNumber?: number;
805
+ /** Slide number (1-based), if available (PPTX/ODP). */
806
+ slideNumber?: number;
807
+ /** Sheet name, if available (XLSX/ODS). */
808
+ sheetName?: string;
809
+ /** The text of the nearest heading above this chunk in the document. */
810
+ closestHeading?: string;
811
+ /** True if this chunk is part of a table split. */
812
+ isTableChunk?: boolean;
813
+ /** Extensible for user-defined metadata. */
814
+ [key: string]: any;
815
+ };
816
+ /** The start character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
817
+ startIndex?: number;
818
+ /** The end character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
819
+ endIndex?: number;
121
820
  }
122
821
  /**
123
822
  * Supported file types for parsing.
124
823
  */
125
- export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf";
824
+ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf" | "md" | "html" | "csv";
126
825
  /**
127
826
  * Types of content nodes in the AST.
128
827
  */
129
- export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page";
828
+ export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment";
130
829
  /**
131
830
  * Supported MIME types for attachments.
132
831
  */
133
- export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation";
832
+ export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" | "application/vnd.openxmlformats-officedocument.presentationml.presentation" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation" | "application/rtf" | "text/csv" | "text/markdown" | "text/html";
134
833
  /**
135
834
  * Text formatting options available for text content.
136
835
  * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
@@ -219,6 +918,8 @@ export interface SlideMetadata {
219
918
  noteId?: string;
220
919
  /** The style of the slide. */
221
920
  style?: string;
921
+ /** Unique anchor IDs for internal linking. */
922
+ anchorIds?: string[];
222
923
  }
223
924
  /**
224
925
  * Metadata for a sheet in Excel.
@@ -228,6 +929,22 @@ export interface SheetMetadata {
228
929
  sheetName: string;
229
930
  /** The style of the sheet. */
230
931
  style?: string;
932
+ /** Unique anchor IDs for internal linking. */
933
+ anchorIds?: string[];
934
+ }
935
+ /**
936
+ * Detailed indentation information for paragraphs and headings.
937
+ * Values are typically in twentieths of a point (twips) in OOXML.
938
+ */
939
+ export interface IndentationMetadata {
940
+ /** Left indentation. */
941
+ left?: number;
942
+ /** Right indentation. */
943
+ right?: number;
944
+ /** First line indentation. */
945
+ firstLine?: number;
946
+ /** Hanging indentation. */
947
+ hanging?: number;
231
948
  }
232
949
  /**
233
950
  * Metadata for a heading.
@@ -239,6 +956,10 @@ export interface HeadingMetadata {
239
956
  alignment?: "left" | "center" | "right" | "justify";
240
957
  /** The style of the heading. */
241
958
  style?: string;
959
+ /** Detailed indentation information. */
960
+ paragraphIndentation?: IndentationMetadata;
961
+ /** Unique anchor IDs for internal linking. */
962
+ anchorIds?: string[];
242
963
  }
243
964
  /**
244
965
  * Metadata for a paragraph.
@@ -248,6 +969,10 @@ export interface ParagraphMetadata {
248
969
  alignment?: "left" | "center" | "right" | "justify";
249
970
  /** The style of the paragraph. */
250
971
  style?: string;
972
+ /** Detailed indentation information. */
973
+ paragraphIndentation?: IndentationMetadata;
974
+ /** Unique anchor IDs for internal linking. */
975
+ anchorIds?: string[];
251
976
  }
252
977
  /**
253
978
  * Metadata for a list item.
@@ -263,6 +988,8 @@ export interface ListMetadata {
263
988
  * @example 0 for top-level items, 1 for first nested level
264
989
  */
265
990
  indentation: number;
991
+ /** Detailed indentation information. */
992
+ paragraphIndentation?: IndentationMetadata;
266
993
  /**
267
994
  * Text alignment of the list item.
268
995
  * @example 'left', 'center', 'right', 'justify'
@@ -285,6 +1012,8 @@ export interface ListMetadata {
285
1012
  * @example "ListParagraph"
286
1013
  */
287
1014
  style?: string;
1015
+ /** Unique anchor IDs for internal linking. */
1016
+ anchorIds?: string[];
288
1017
  }
289
1018
  /**
290
1019
  * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
@@ -313,6 +1042,8 @@ export interface CellMetadata {
313
1042
  colSpan?: number;
314
1043
  /** The style of the cell. */
315
1044
  style?: string;
1045
+ /** Unique anchor IDs for internal linking. */
1046
+ anchorIds?: string[];
316
1047
  }
317
1048
  /**
318
1049
  * Metadata for a chart node in the document.
@@ -325,6 +1056,8 @@ export interface ChartMetadata {
325
1056
  * @example "chart1.xml"
326
1057
  */
327
1058
  attachmentName: string;
1059
+ /** Unique anchor IDs for internal linking. */
1060
+ anchorIds?: string[];
328
1061
  }
329
1062
  /**
330
1063
  * Metadata for an image node in the document.
@@ -343,6 +1076,14 @@ export interface ImageMetadata {
343
1076
  * @example "Company logo"
344
1077
  */
345
1078
  altText?: string;
1079
+ /**
1080
+ * URL of the image if it is an external link.
1081
+ * Typical for HTML or Markdown images that point to remote servers.
1082
+ * @example "https://example.com/image.png"
1083
+ */
1084
+ url?: string;
1085
+ /** Unique anchor IDs for internal linking. */
1086
+ anchorIds?: string[];
346
1087
  }
347
1088
  /**
348
1089
  * Metadata for PDF page nodes.
@@ -388,11 +1129,47 @@ export interface NoteMetadata {
388
1129
  * @example "1", "2"
389
1130
  */
390
1131
  noteId?: string;
1132
+ /** Unique anchor IDs for internal linking. */
1133
+ anchorIds?: string[];
1134
+ }
1135
+ /**
1136
+ * Metadata for break nodes.
1137
+ * Used in DOCX files to track line and page breaks.
1138
+ */
1139
+ export interface BreakMetadata {
1140
+ /**
1141
+ * Type of break. The break type determines the next location where
1142
+ * text shall be placed.
1143
+ * - 'column': The next text will be placed in the next column.
1144
+ * - 'page': The next text will be placed on the next page.
1145
+ * - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
1146
+ * - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
1147
+ * - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
1148
+ */
1149
+ breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn";
1150
+ /**
1151
+ * Specifies the location which shall be used as the next available line when breakType
1152
+ * has a value of 'textWrapping'. Should be ignored for other break types.
1153
+ * - 'all': text wrapping break shall advance the text to the next line which spans the full width of the line
1154
+ * - 'left': text wrapping break shall restart in next text region unblocked on the left
1155
+ * - 'none': text wrapping break shall advance the text to the next line regardless of any floating objects
1156
+ * - 'right': text wrapping break shall restart in next text region unblocked on the right
1157
+ */
1158
+ clear?: "all" | "left" | "none" | "right";
1159
+ }
1160
+ /**
1161
+ * Metadata for a code block.
1162
+ */
1163
+ export interface CodeMetadata {
1164
+ /** The programming language of the code block (e.g., 'typescript', 'python') */
1165
+ language?: string;
1166
+ /** Unique anchor IDs for internal linking. */
1167
+ anchorIds?: string[];
391
1168
  }
392
1169
  /**
393
1170
  * Union type for content metadata.
394
1171
  */
395
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | undefined;
1172
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
396
1173
  /**
397
1174
  * Represents a node in the document content tree.
398
1175
  * This is the core building block of the parsed document structure.
@@ -630,9 +1407,19 @@ export interface OfficeMetadata {
630
1407
  * console.log(ast.metadata.author); // 'John Doe'
631
1408
  * console.log(ast.content.length); // Number of top-level content nodes
632
1409
  * console.log(ast.toText()); // Plain text representation
1410
+ * console.log((await ast.to('md')).value); // Markdown representation
1411
+ * console.log((await ast.to('html')).value); // HTML representation
1412
+ * console.log((await ast.to('rtf')).value); // RTF representation
1413
+ * console.log((await ast.to('csv')).value); // CSV representation
1414
+ * console.log((await ast.to('chunks')).value); // Chunks representation
633
1415
  * ```
634
1416
  */
635
1417
  export interface OfficeParserAST {
1418
+ /**
1419
+ * The original configuration used to parse this document.
1420
+ * This includes options like OCR settings, delimiter choices, and filtering flags.
1421
+ */
1422
+ config: OfficeParserConfig;
636
1423
  /**
637
1424
  * The type of the parsed file.
638
1425
  * Indicates which parser was used and what format the input was in.
@@ -667,7 +1454,12 @@ export interface OfficeParserAST {
667
1454
  * @example [{ type: 'image', mimeType: 'image/png', data: 'base64...', name: 'image1.png' }]
668
1455
  */
669
1456
  attachments: OfficeAttachment[];
1457
+ /** Any warnings or non-fatal issues encountered during parsing. */
1458
+ warnings: OfficeIssue[];
670
1459
  /**
1460
+ * @deprecated Use `.to('text')` instead.
1461
+ * Note: This method is synchronous, while the new `.to()` method is asynchronous.
1462
+ *
671
1463
  * Converts the entire AST to plain text.
672
1464
  * This method flattens the document structure and returns just the text content,
673
1465
  * stripping out all formatting, metadata, and structure.
@@ -682,6 +1474,20 @@ export interface OfficeParserAST {
682
1474
  * ```
683
1475
  */
684
1476
  toText(): string;
1477
+ /**
1478
+ * Converts this AST to the specified destination format.
1479
+ * This is the recommended way to convert the AST to different formats.
1480
+ *
1481
+ * @param destination The target format (e.g., 'text', 'md', 'html', 'pdf').
1482
+ * @param config Optional configuration for the generator.
1483
+ * @returns A promise resolving to the generated content (string or Buffer).
1484
+ * @example
1485
+ * ```typescript
1486
+ * const html = await ast.to('html', { includeFormatting: false });
1487
+ * const md = await ast.to('md');
1488
+ * ```
1489
+ */
1490
+ to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
685
1491
  }
686
1492
  /**
687
1493
  * Main parser class providing office document parsing functionality.
@@ -710,6 +1516,9 @@ export declare class OfficeParser {
710
1516
  * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
711
1517
  * - `.pdf` → PdfParser (PDF.js)
712
1518
  * - `.rtf` → RtfParser (custom RTF parser)
1519
+ * - `.csv` → CsvParser
1520
+ * - `.md` → MarkdownParser
1521
+ * - `.html` → HtmlParser
713
1522
  *
714
1523
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
715
1524
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -746,8 +1555,72 @@ export declare class OfficeParser {
746
1555
  */
747
1556
  static terminateOcr(): Promise<void>;
748
1557
  }
1558
+ /**
1559
+ * Main generator class providing document conversion functionality.
1560
+ */
1561
+ export declare class OfficeGenerator {
1562
+ /**
1563
+ * Generates a file of the specified type from an AST.
1564
+ * This is the single source of truth for generation logic.
1565
+ *
1566
+ * @param ast - The OfficeParserAST to generate from
1567
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
1568
+ * @param config - Optional configuration for the generator
1569
+ * @returns A promise resolving to the ConversionResult containing the value and messages
1570
+ * @throws {Error} If the destination format is unsupported
1571
+ */
1572
+ static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
1573
+ type: T;
1574
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
1575
+ }
1576
+ /**
1577
+ * Utility type to infer the file type from a file path string literal.
1578
+ */
1579
+ export type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
1580
+ /**
1581
+ * Main converter class providing a streamlined one-step API for document conversion.
1582
+ *
1583
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
1584
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
1585
+ */
1586
+ export declare class OfficeConverter {
1587
+ /**
1588
+ * Converts an office document from its source format to a specified destination format.
1589
+ *
1590
+ * This method:
1591
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
1592
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
1593
+ * attachment extraction if images are requested in the output).
1594
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
1595
+ *
1596
+ * @template F The inferred type of the input file (path string or buffer).
1597
+ * @template T The authoritative source file type (inferred from path or config).
1598
+ *
1599
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
1600
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
1601
+ * @param config - Optional unified configuration for both the parser and generator phases.
1602
+ *
1603
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
1604
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
1605
+ *
1606
+ * @example
1607
+ * ```typescript
1608
+ * // Convert Word to Markdown with a single call
1609
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
1610
+ *
1611
+ * // Convert PDF to HTML with OCR enabled for images
1612
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
1613
+ * ocr: true,
1614
+ * includeImages: true
1615
+ * });
1616
+ * ```
1617
+ */
1618
+ static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
1619
+ }
749
1620
  export declare const parseOffice: typeof OfficeParser.parseOffice;
750
1621
  export declare const terminateOcr: typeof OfficeParser.terminateOcr;
1622
+ export declare const convert: typeof OfficeConverter.convert;
1623
+ export declare const generate: typeof OfficeGenerator.generate;
751
1624
 
752
1625
  export {
753
1626
  OfficeParser as default,