officeparser 7.1.0 → 7.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +152 -56
  2. package/dist/OfficeGenerator.d.ts +6 -2
  3. package/dist/OfficeGenerator.js +30 -9
  4. package/dist/OfficeParser.d.ts +1 -1
  5. package/dist/OfficeParser.js +1 -1
  6. package/dist/cli.d.ts +18 -12
  7. package/dist/cli.js +255 -81
  8. package/dist/defaults.js +12 -1
  9. package/dist/generators/BaseGenerator.d.ts +4 -3
  10. package/dist/generators/BaseGenerator.js +13 -1
  11. package/dist/generators/ChunkingGenerator.js +32 -5
  12. package/dist/generators/CsvGenerator.d.ts +1 -1
  13. package/dist/generators/HtmlGenerator.d.ts +2 -1
  14. package/dist/generators/HtmlGenerator.js +481 -42
  15. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  16. package/dist/generators/MarkdownGenerator.js +35 -2
  17. package/dist/generators/PdfGenerator.d.ts +1 -1
  18. package/dist/generators/PdfGenerator.js +0 -6
  19. package/dist/generators/RtfGenerator.d.ts +2 -1
  20. package/dist/generators/RtfGenerator.js +49 -6
  21. package/dist/generators/TextGenerator.d.ts +1 -1
  22. package/dist/generators/TextGenerator.js +6 -0
  23. package/dist/officeparser.browser.d.ts +267 -54
  24. package/dist/officeparser.browser.iife.js +599 -187
  25. package/dist/officeparser.browser.mjs +599 -187
  26. package/dist/parsers/CsvParser.js +1 -1
  27. package/dist/parsers/ExcelParser.js +63 -19
  28. package/dist/parsers/HtmlParser.js +10 -1
  29. package/dist/parsers/MarkdownParser.js +13 -10
  30. package/dist/parsers/OpenOfficeParser.js +57 -34
  31. package/dist/parsers/PdfParser.js +28 -3
  32. package/dist/parsers/PowerPointParser.js +164 -40
  33. package/dist/parsers/RtfParser.js +28 -24
  34. package/dist/parsers/WordParser.js +154 -11
  35. package/dist/sbom.cdx.json +100 -100
  36. package/dist/types.d.ts +268 -53
  37. package/dist/types.js +4 -0
  38. package/dist/utils/astUtils.d.ts +2 -2
  39. package/dist/utils/astUtils.js +2 -1
  40. package/dist/utils/configUtils.d.ts +5 -0
  41. package/dist/utils/configUtils.js +55 -1
  42. package/dist/utils/errorUtils.js +3 -1
  43. package/dist/utils/moduleLoader.js +55 -11
  44. package/dist/utils/xmlUtils.d.ts +9 -0
  45. package/dist/utils/xmlUtils.js +53 -1
  46. package/package.json +6 -3
package/README.md CHANGED
@@ -1,4 +1,4 @@
1
- # officeParser — Universal Office Document Parser & Generator
1
+ # officeParser: Universal Office Document Parser & Generator
2
2
 
3
3
  A robust, strictly-typed **Node.js and Browser** library for parsing office files into a rich **Abstract Syntax Tree (AST)** and generating high-fidelity output in multiple formats.
4
4
 
@@ -14,7 +14,7 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
14
14
  ---
15
15
 
16
16
  ### 🌟 [Live Interactive AST Visualizer & Documentation](https://harshankur.github.io/officeParser/) 🌟
17
- *Upload any office file in your browser — inspect the AST, tweak config, and preview generated output in real-time.*
17
+ *Upload any office file in your browser: inspect the AST, tweak config, and preview generated output in real-time.*
18
18
 
19
19
  - **AST Visualizer**: Inspect the hierarchical node tree, metadata, and raw content
20
20
  - **Config Configurator**: Tweak options (`ignoreNotes`, `ocr`, `newlineDelimiter`) and see results instantly
@@ -35,10 +35,10 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
35
35
  - [Async/Await](#asyncawait)
36
36
  - [Callback (Backward Compat)](#callback-backward-compat)
37
37
  - [File Buffers & ArrayBuffers](#file-buffers--arraybuffers)
38
- - [`ast.to()` — Generate from AST](#astto--generate-from-ast)
39
- - [`ast.toText()` — Quick Text Extraction](#asttotext--quick-text-extraction)
38
+ - [`ast.to()`: Generate from AST](#astto-generate-from-ast)
39
+ - [`ast.toText()`: Quick Text Extraction](#asttotext-quick-text-extraction)
40
40
  - [OfficeGenerator](#officegenerator)
41
- - [OfficeConverter — One-Step API](#officeconverter--one-step-api)
41
+ - [OfficeConverter: One-Step API](#officeconverter-one-step-api)
42
42
  - [Native RAG Chunking](#native-rag-chunking)
43
43
  - [The AST Structure](#the-ast-structure)
44
44
  - [Deep Dive: Document Components](#deep-dive-document-components)
@@ -47,8 +47,8 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
47
47
  - [Configuration Reference](#configuration-reference)
48
48
  - [OfficeParserConfig](#officeparserconfig)
49
49
  - [GeneratorConfig (Common)](#generatorconfig-common)
50
- - [onNode Callback](#onnode-callback--advanced-node-manipulation)
51
- - [styleMap — Semantic Style Mapping](#stylemap--semantic-style-mapping)
50
+ - [onNode Callback](#onnode-callback-advanced-node-manipulation)
51
+ - [styleMap: Semantic Style Mapping](#stylemap-semantic-style-mapping)
52
52
  - [HtmlGeneratorConfig](#htmlgeneratorconfig)
53
53
  - [MdGeneratorConfig](#mdgeneratorconfig)
54
54
  - [PdfGeneratorConfig](#pdfgeneratorconfig)
@@ -79,37 +79,58 @@ npm i officeparser
79
79
  npx officeparser /path/to/file.docx
80
80
 
81
81
  # Plain text output
82
- npx officeparser /path/to/file.docx --format=text
82
+ npx officeparser /path/to/file.docx --to=text
83
83
 
84
84
  # Convert DOCX to Markdown and save
85
- npx officeparser report.docx --format=md --output=report.md
85
+ npx officeparser report.docx --to=md --output=report.md
86
86
 
87
- # Convert PPTX to HTML
88
- npx officeparser presentation.pptx --format=html --output=preview.html
87
+ # Convert PPTX to HTML (using a bare flag for ocr)
88
+ npx officeparser presentation.pptx --to=html --output=preview.html --ocr
89
89
 
90
- # Convert XLSX to CSV
91
- npx officeparser data.xlsx --format=csv
90
+ # Convert XLSX to CSV with a custom delimiter
91
+ npx officeparser data.xlsx --to=csv --csvDelimiter=";"
92
92
 
93
93
  # Generate RAG chunks
94
- npx officeparser document.pdf --format=chunks
94
+ npx officeparser document.pdf --to=chunks
95
+
96
+ # Overriding file extension mapping
97
+ npx officeparser my_document --fileType=docx --to=json
95
98
  ```
96
99
 
100
+ ### CLI Syntax
101
+ - **Values:** Flags can be passed as `--flag=value` or `--flag value`.
102
+ - **Booleans:** Bare flags imply `true` (e.g. `--ocr` is equivalent to `--ocr=true`). Negation flags start with `no-` (e.g. `--no-ocr` is equivalent to `--ocr=false`).
103
+ - **Nested Objects:** You can pass nested properties directly using JSON dot-notation (e.g. `--ocrConfig.language=fra` or `--htmlConfig.containerWidth=900px`).
104
+
97
105
  ### CLI Options
98
106
 
99
107
  | Flag | Values | Default | Description |
100
108
  |------|--------|---------|-------------|
101
- | `--format` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
109
+ | `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
102
110
  | `--output` | path | — | Write output to a file |
103
- | `--toText` | `true\|false` | `false` | **Deprecated.** Use `--format=text` |
104
- | `--ignoreNotes` | `true\|false` | `false` | Ignore speaker notes (PPTX/ODP) |
105
- | `--putNotesAtLast` | `true\|false` | `false` | Collect notes at end of output |
106
- | `--newlineDelimiter` | string | `\n` | Delimiter between lines |
107
- | `--extractAttachments` | `true\|false` | `false` | Extract images/charts as Base64 |
108
- | `--ocr` | `true\|false` | `false` | Enable OCR for images |
109
- | `--includeRawContent` | `true\|false` | `false` | Include raw XML/RTF in nodes |
110
- | `--includeBreakNodes` | `true\|false` | `false` | Include break nodes (DOCX only) |
111
- | `--outputErrorToConsole` | `true\|false` | `false` | **Deprecated.** Use `onWarning` callback |
112
- | `--verbose` | `true\|false` | `false` | Show full error stack traces |
111
+ | `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html` | — | Explicitly override input file type detection |
112
+ | `--ocr` | boolean | `false` | Enable OCR for images |
113
+ | `--extractAttachments` | boolean | `false` | Extract images/charts as Base64 |
114
+ | `--ignoreNotes` | boolean | `false` | Ignore footnotes/endnotes/speaker notes |
115
+ | `--ignoreComments` | boolean | `false` | Ignore inline comments |
116
+ | `--ignoreHeadersAndFooters` | boolean | `false` | Ignore headers and footers |
117
+ | `--ignoreSlideMasters` | boolean | `false` | Ignore slide masters |
118
+ | `--ignoreInternalLinks` | boolean | `false` | Ignore internal links |
119
+ | `--newlineDelimiter` | string | `\n` | Delimiter between lines/blocks in plaintext outputs |
120
+ | `--csvDelimiter` | string | `,` | Custom delimiter for CSV files |
121
+ | `--includeRawContent` | boolean | `false` | Include raw XML/RTF in nodes |
122
+ | `--serializeRawContent` | boolean | `true` | Include stringified XML in metadata |
123
+ | `--preserveXmlWhitespace` | boolean | `false` | Keep raw formatting space |
124
+ | `--includeBreakNodes` | boolean | `false` | Include break nodes (DOCX only) |
125
+ | `--verbose` | boolean | `false` | Show full error stack traces and warning logs |
126
+ | `--includeFormatting` | boolean | `true` | Include formatting style map matching |
127
+ | `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
128
+ | `--htmlConfig.containerWidth` | string \| number | `auto` | HTML output container width (e.g. `900px`, `100%`) |
129
+ | ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | **Deprecated.** Use `--to` |
130
+ | ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text` |
131
+ | ~~`--ocrLanguage`~~ | string | `eng` | **Deprecated.** Use `--ocrConfig.language` |
132
+ | ~~`--putNotesAtLast`~~ | `true\|false` | `false` | **Deprecated and ignored.** Notes are attached structurally to their nodes. |
133
+ | ~~`--outputErrorToConsole`~~ | `true\|false` | `false` | **Deprecated.** Use `--verbose` |
113
134
 
114
135
  ---
115
136
 
@@ -232,7 +253,7 @@ const ast = await officeParser.parseOffice('scanned_document.pdf', {
232
253
  > **Non-Fatal Timeout Recovery**
233
254
  > If `workerLoad` or `recognition` timeouts are exceeded, the parser will log a warning in `ast.warnings` and **continue parsing the rest of the document**. The overall promise resolves successfully with the text extracted from the document layers (rather than failing the entire parse).
234
255
 
235
- ### `ast.to()` — Generate from AST
256
+ ### `ast.to()`: Generate from AST
236
257
 
237
258
  The preferred way to convert a parsed AST to another format. Returns a `ConversionResult`.
238
259
 
@@ -246,7 +267,7 @@ const { value: chunks } = await ast.to('chunks', { strategy: 'fixed-
246
267
  const { value: pdfBytes } = await ast.to('pdf'); // Uint8Array
247
268
  ```
248
269
 
249
- ### `ast.toText()` — Quick Text Extraction
270
+ ### `ast.toText()`: Quick Text Extraction
250
271
 
251
272
  > [!NOTE]
252
273
  > `toText()` is **synchronous** and deprecated in favour of the async `ast.to('text')`.
@@ -295,7 +316,7 @@ const { value: csv } = await OfficeGenerator.generate(ast, 'csv');
295
316
 
296
317
  ---
297
318
 
298
- ## OfficeConverter — One-Step API
319
+ ## OfficeConverter: One-Step API
299
320
 
300
321
  `OfficeConverter.convert()` combines parsing and generation in a single call. It automatically syncs parser options from generator config (e.g., enables `extractAttachments` when images are requested).
301
322
 
@@ -326,7 +347,7 @@ const { value: html, messages } = await OfficeConverter.convert('data.xlsx', 'ht
326
347
 
327
348
  > [!IMPORTANT]
328
349
  > The `OfficeConverterConfig` shape uses **nested** `parseConfig` and `generatorConfig` sub-objects.
329
- > Do **not** put parser or generator options at the top level — only `onWarning` lives there.
350
+ > Do **not** put parser or generator options at the top level; only `onWarning` lives there.
330
351
 
331
352
  ---
332
353
 
@@ -344,7 +365,7 @@ const { value: chunks } = await OfficeConverter.convert('report.docx', 'chunks',
344
365
  strategy: 'document-structure',
345
366
  splitBy: 'heading', // 'paragraph' | 'heading' | 'page' | 'slide' | 'sheet'
346
367
  maxChunkSize: 1500,
347
- tableSplitStrategy: 'row', // repeats header row in every chunk — ideal for RAG
368
+ tableSplitStrategy: 'row', // repeats header row in every chunk, ideal for RAG
348
369
  }
349
370
  }
350
371
  });
@@ -420,13 +441,19 @@ interface OfficeChunk {
420
441
  ```text
421
442
  OfficeParserAST
422
443
  ├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (11 formats)
423
- ├── metadata: { author, title, created, modified, customProperties, styleMap, ... }
444
+ ├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
424
445
  ├── content: [ OfficeContentNode ]
425
- │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | ...
446
+ │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
426
447
  │ ├── text: string (concatenated text of node + all descendants)
427
- │ ├── children: [ OfficeContentNode ] (recursive)
448
+ │ ├── children: [ OfficeContentNode ] (recursive structural children)
449
+ │ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
450
+ │ ├── comments: [ OfficeContentNode ] (inline comments attached to this node)
428
451
  │ ├── formatting: { bold, italic, underline, color, size, font, alignment, ... }
429
- │ └── metadata: { level, listId, row, col, rowSpan, colSpan, style, ... }
452
+ │ └── metadata: { level, listId, row, col, rowSpan, colSpan, backgroundColor, style, ... }
453
+ ├── auxiliary?: OfficeAuxiliaryContent (out-of-band layout elements)
454
+ │ ├── headers?: OfficeContentNode[] (DOCX headers)
455
+ │ ├── footers?: OfficeContentNode[] (DOCX footers)
456
+ │ └── slideMasters?: OfficeContentNode[] (PPTX slide masters)
430
457
  ├── attachments: [ OfficeAttachment ] (populated when extractAttachments: true)
431
458
  │ ├── type: 'image' | 'chart'
432
459
  │ ├── name: string
@@ -436,10 +463,10 @@ OfficeParserAST
436
463
  │ └── chartData?: { title, dataSets, labels }
437
464
  ├── warnings: OfficeIssue[] (non-fatal issues from the parsing phase)
438
465
  ├── to(format, config?) (format: 'html'|'md'|'text'|'csv'|'rtf'|'pdf'|'chunks', returns { value, messages })
439
- └── toText() (Deprecated: use .to('text') instead)
466
+ └── ~~toText()~~ (Deprecated: use .to('text') instead)
440
467
  ```
441
468
 
442
- ### `OfficeIssue` — Warning / Error Object
469
+ ### `OfficeIssue`: Warning / Error Object
443
470
 
444
471
  All warnings and errors (from both parsing and generation) use this shape:
445
472
 
@@ -552,17 +579,21 @@ ast.metadata = {
552
579
  created?: Date
553
580
  modified?: Date
554
581
  description?: string
555
- customProperties?: Record<string, any> // user-defined metadata from the document
556
- styleMap?: Record<string, TextFormatting> // named styles → formatting definitions
557
- formatting?: TextFormatting // document-wide defaults
582
+ keywords?: string // NEW: Keywords from document properties
583
+ customProperties?: Record<string, any> // User-defined metadata from the document
584
+ nativeProperties?: Record<string, any> // NEW: All format-specific raw metadata
585
+ styleMap?: Record<string, TextFormatting> // Named styles → formatting definitions
586
+ formatting?: TextFormatting // Document-wide defaults
558
587
  }
559
588
  ```
560
589
 
561
- **Accessing custom properties:**
590
+ **Accessing native properties (format-specific metadata):**
562
591
  ```js
563
592
  const ast = await officeParser.parseOffice('contract.docx');
564
- console.log(ast.metadata.customProperties);
565
- // { "ProjectID": "ABC-123", "InternalReview": true }
593
+ console.log(ast.metadata.nativeProperties);
594
+ // DOCX: { Pages: 5, Application: 'Microsoft Word' }
595
+ // HTML: { description: 'My page', 'og:title': 'Title' }
596
+ // PDF: { Title: 'Report', XMP: { ... } }
566
597
  ```
567
598
 
568
599
  ---
@@ -586,6 +617,61 @@ const headings = ast.content.filter(n => n.type === 'heading' && n.metadata?.lev
586
617
  console.log(headings.map(h => h.text));
587
618
  ```
588
619
 
620
+ ### Extract comments
621
+ ```ts
622
+ // Comments can be attached to any nested node, so we must traverse recursively
623
+ const printComments = (nodes: OfficeContentNode[]) => {
624
+ nodes.forEach(node => {
625
+ if (node.comments) {
626
+ node.comments.forEach(c => {
627
+ console.log(`Comment by ${c.metadata?.author}: ${c.text}`);
628
+ });
629
+ }
630
+ if (node.children) {
631
+ printComments(node.children);
632
+ }
633
+ });
634
+ };
635
+
636
+ printComments(ast.content);
637
+ ```
638
+
639
+ Set `ignoreComments: true` to skip extraction.
640
+
641
+ ### Extract footnotes, endnotes & slide notes
642
+ ```ts
643
+ // Slide speaker notes (PPTX) live on the slide node itself
644
+ const slide = ast.content.find(n => n.type === 'slide');
645
+ console.log(slide?.notes?.map(n => n.text));
646
+
647
+ // Footnotes and endnotes (DOCX/RTF) can be deeply nested, so we traverse recursively:
648
+ const printNotes = (nodes: OfficeContentNode[]) => {
649
+ nodes.forEach(node => {
650
+ if (node.notes) {
651
+ node.notes.forEach(note => console.log(note.text));
652
+ }
653
+ if (node.children) {
654
+ printNotes(node.children);
655
+ }
656
+ });
657
+ };
658
+
659
+ printNotes(ast.content);
660
+ ```
661
+
662
+ > [!IMPORTANT]
663
+ > `putNotesAtLast` is **deprecated**. Notes are always attached via `node.notes`; this flag has no effect and will be removed in a future major version.
664
+
665
+ ### Access headers, footers & slide masters
666
+ ```ts
667
+ // These are NOT in ast.content; use ast.auxiliary
668
+ console.log(ast.auxiliary?.headers?.map(h => h.text)); // DOCX headers
669
+ console.log(ast.auxiliary?.footers?.map(f => f.text)); // DOCX footers
670
+ console.log(ast.auxiliary?.slideMasters?.length); // PPTX slide masters
671
+ ```
672
+
673
+ Set `ignoreHeadersAndFooters: true` or `ignoreSlideMasters: true` to skip extraction.
674
+
589
675
  ### Extract images with OCR text
590
676
  ```js
591
677
  const ast = await officeParser.parseOffice('report.docx', { extractAttachments: true, ocr: true });
@@ -650,11 +736,14 @@ Pass as the second argument to `parseOffice(file, config)`.
650
736
  | Option | Type | Default | Description |
651
737
  |--------|------|---------|-------------|
652
738
  | `newlineDelimiter` | `string` | `'\n'` | Delimiter inserted between lines in text output |
653
- | `ignoreNotes` | `boolean` | `false` | Ignore speaker notes (PPTX/ODP) |
654
- | `putNotesAtLast` | `boolean` | `false` | Collect all notes at the end instead of inline |
739
+ | `ignoreNotes` | `boolean` | `false` | Ignore footnotes/endnotes (DOCX, RTF) and speaker notes (PPTX/ODP) |
740
+ | `ignoreComments` | `boolean` | `false` | **New**: Ignore inline comments/annotations (DOCX, XLSX, PPTX), attached by default via `node.comments[]` |
741
+ | `ignoreHeadersAndFooters` | `boolean` | `false` | **New**: Skip DOCX headers & footers (populated in `ast.auxiliary.headers/footers` by default) |
742
+ | `ignoreSlideMasters` | `boolean` | `false` | **New**: Skip PPTX slide masters (populated in `ast.auxiliary.slideMasters` by default) |
743
+ | ~~`putNotesAtLast`~~ | `boolean` | `false` | **Deprecated**: Notes are now attached via `node.notes[]`. This flag has no effect |
655
744
  | `extractAttachments` | `boolean` | `false` | Populate `ast.attachments` with Base64 images/charts |
656
745
  | `ocr` | `boolean` | `false` | Run Tesseract OCR on images (requires `extractAttachments: true`) |
657
- | `ocrConfig` | `OcrConfig` | `{}` | OCR worker pool settings — see [OCR section](#ocr-scheduler--resource-management) |
746
+ | `ocrConfig` | `OcrConfig` | `{}` | OCR worker pool settings (see [OCR section](#ocr-scheduler--resource-management)) |
658
747
  | `includeRawContent` | `boolean` | `false` | Attach raw XML/RTF source to each node |
659
748
  | `serializeRawContent` | `boolean` | `true` | Re-serialize XML to clean strings (only if `includeRawContent: true`) |
660
749
  | `preserveXmlWhitespace` | `boolean` | `false` | Preserve original XML whitespace during serialization |
@@ -665,7 +754,7 @@ Pass as the second argument to `parseOffice(file, config)`.
665
754
  | `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
666
755
  | `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
667
756
  | `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
668
- | `outputErrorToConsole` | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
757
+ | ~~`outputErrorToConsole`~~ | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
669
758
 
670
759
  ---
671
760
 
@@ -689,7 +778,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
689
778
 
690
779
  ---
691
780
 
692
- ### `onNode` Callback — Advanced Node Manipulation
781
+ ### `onNode` Callback: Advanced Node Manipulation
693
782
 
694
783
  Called for **every node** in the AST during generation. Can be `async`.
695
784
 
@@ -720,7 +809,7 @@ const { value: md } = await ast.to('md', {
720
809
 
721
810
  ---
722
811
 
723
- ### `styleMap` — Semantic Style Mapping
812
+ ### `styleMap`: Semantic Style Mapping
724
813
 
725
814
  Maps document style names to semantic output elements. Two formats supported:
726
815
 
@@ -764,6 +853,12 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
764
853
  |--------|------|---------|-------------|
765
854
  | `standalone` | `boolean` | `true` | Wrap output in a full `<html>` document with CSS |
766
855
  | `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
856
+ | `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
857
+ | `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block; use this to override built-in styles |
858
+ | `injections.headStart` | `string` | `''` | Raw HTML injected after `<head>` |
859
+ | `injections.headEnd` | `string` | `''` | Raw HTML injected before `</head>` |
860
+ | `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
861
+ | `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
767
862
 
768
863
  ### MdGeneratorConfig
769
864
 
@@ -780,6 +875,8 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
780
875
  | Option | Type | Default | Description |
781
876
  |--------|------|---------|-------------|
782
877
  | `format` | `string` | `'A4'` | Paper format (`'A4'`, `'Letter'`, `'Legal'`, etc.) |
878
+ | `width` | `string \| number` | `''` | Paper width (e.g., `'5in'`, `'3cm'`) or pixels |
879
+ | `height` | `string \| number` | `''` | Paper height (e.g., `'5in'`, `'3cm'`) or pixels |
783
880
  | `landscape` | `boolean` | `false` | Landscape page orientation |
784
881
  | `printBackground` | `boolean` | `true` | Print background graphics |
785
882
  | `margin` | `object` | `{0,0,0,0}` | Page margins (`top`, `right`, `bottom`, `left`) |
@@ -807,7 +904,7 @@ Pass as `textConfig` inside `GeneratorConfig`.
807
904
  | Option | Type | Default | Description |
808
905
  |--------|------|---------|-------------|
809
906
  | `newlineDelimiter` | `string` | `'\n'` | String inserted between structural blocks |
810
- | `preserveLayout` | `boolean` | `false` | Render tables with aligned columns using whitespace |
907
+ | `preserveLayout` | `boolean` | `true` | Render tables with aligned columns using whitespace |
811
908
 
812
909
  ---
813
910
 
@@ -825,7 +922,7 @@ Configuration for `OfficeConverter.convert(file, format, config)`.
825
922
 
826
923
  ### ChunkingConfig
827
924
 
828
- `ChunkingConfig` is a **discriminated union** — the available options depend on the `strategy` field.
925
+ `ChunkingConfig` is a **discriminated union**: the available options depend on the `strategy` field.
829
926
 
830
927
  #### Common Options (all strategies)
831
928
 
@@ -885,7 +982,7 @@ When `ocr: true` is set, `officeParser` maintains an intelligent **Smart Worker
885
982
  | `corePath` | `string` | `''` | Custom path to Tesseract core script |
886
983
  | `langPath` | `string` | `''` | Custom path for language data files |
887
984
  | `timeout` | `OcrTimeoutConfig` | `{}` | Consolidated timeouts: `autoTerminate`, `workerLoad`, `recognition` |
888
- | `autoTerminateTimeout` | `number` | `10000` | **Deprecated.** Use `timeout.autoTerminate` instead |
985
+ | ~~`autoTerminateTimeout`~~ | `number` | `10000` | **Deprecated.** Use `timeout.autoTerminate` instead |
889
986
 
890
987
  See all language codes at [tesseract-ocr.github.io](https://tesseract-ocr.github.io/tessdoc/Data-Files).
891
988
 
@@ -913,8 +1010,8 @@ Two bundles are available in the `dist/` directory:
913
1010
 
914
1011
  | Bundle | Usage |
915
1012
  |--------|-------|
916
- | `officeparser.browser.mjs` | ESM — use with `import` statements or modern bundlers (Vite, Webpack, Next.js) |
917
- | `officeparser.browser.iife.js` | IIFE — use with a `<script>` tag; exposes the global `officeParser` object |
1013
+ | `officeparser.browser.mjs` | ESM, use with `import` statements or modern bundlers (Vite, Webpack, Next.js) |
1014
+ | `officeparser.browser.iife.js` | IIFE, use with a `<script>` tag; exposes the global `officeParser` object |
918
1015
 
919
1016
  ### ESM (Vite / Webpack / Next.js)
920
1017
 
@@ -974,7 +1071,7 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer, {
974
1071
  | `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `5.6.205` |
975
1072
  | Low OCR accuracy | Verify `ocrConfig.language` matches the document language; quality depends on image resolution |
976
1073
  | Out of memory on large Excel files | Call `ast.toText()` early and discard the AST object to allow garbage collection |
977
- | `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config — these formats have no magic bytes |
1074
+ | `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config (these formats have no magic bytes) |
978
1075
  | `IMPROPER_BUFFERS` error | Usually means no file extension and no `fileType` hint was provided for a buffer input |
979
1076
  | PDF generation fails | Install the optional peer dependency: `npm install puppeteer` |
980
1077
 
@@ -986,7 +1083,6 @@ For a full debugging guide, visit the [Live Documentation](https://harshankur.gi
986
1083
 
987
1084
  1. **ODT/ODS Charts**: May show inaccurate data when the chart references external cell ranges or uses complex layout-based data.
988
1085
  2. **PDF Images (Browser)**: Extracted as BMP files for cross-platform compatibility. Conversion is automatic.
989
- 3. **RTF Notes**: `putNotesAtLast` has no effect for RTF files; footnotes and endnotes are always appended at the end.
990
1086
 
991
1087
  ---
992
1088
 
@@ -1011,4 +1107,4 @@ Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for det
1011
1107
 
1012
1108
  ## License
1013
1109
 
1014
- This project is licensed under the MIT License — see the [LICENSE](LICENSE) file for details.
1110
+ This project is licensed under the MIT License; see the [LICENSE](LICENSE) file for details.
@@ -1,8 +1,12 @@
1
- import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
1
+ import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType, UniversalGeneratorFormat } from './types.js';
2
2
  /**
3
3
  * Main generator class providing document conversion functionality.
4
4
  */
5
5
  export declare class OfficeGenerator {
6
+ /**
7
+ * Normalizes format aliases (e.g., 'txt' to 'text', 'markdown' to 'md') to standard internal formats.
8
+ */
9
+ static normalizeDestination(dest: string): UniversalGeneratorFormat;
6
10
  /**
7
11
  * Generates a file of the specified type from an AST.
8
12
  * This is the single source of truth for generation logic.
@@ -15,5 +19,5 @@ export declare class OfficeGenerator {
15
19
  */
16
20
  static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
17
21
  type: T;
18
- }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
22
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
19
23
  }
@@ -14,6 +14,17 @@ const errorUtils_js_1 = require("./utils/errorUtils.js");
14
14
  * Main generator class providing document conversion functionality.
15
15
  */
16
16
  class OfficeGenerator {
17
+ /**
18
+ * Normalizes format aliases (e.g., 'txt' to 'text', 'markdown' to 'md') to standard internal formats.
19
+ */
20
+ static normalizeDestination(dest) {
21
+ const d = dest?.toLowerCase();
22
+ if (d === 'txt')
23
+ return 'text';
24
+ if (d === 'markdown')
25
+ return 'md';
26
+ return d;
27
+ }
17
28
  /**
18
29
  * Generates a file of the specified type from an AST.
19
30
  * This is the single source of truth for generation logic.
@@ -25,24 +36,34 @@ class OfficeGenerator {
25
36
  * @throws {Error} If the destination format is unsupported
26
37
  */
27
38
  static async generate(ast, destination, config) {
28
- switch (destination.toLowerCase()) {
39
+ let generator;
40
+ const normalizedDestination = OfficeGenerator.normalizeDestination(destination);
41
+ switch (normalizedDestination) {
29
42
  case 'text':
30
- return new TextGenerator_js_1.TextGenerator(ast, config).generate();
43
+ generator = new TextGenerator_js_1.TextGenerator(ast, config);
44
+ break;
31
45
  case 'md':
32
- return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
46
+ generator = new MarkdownGenerator_js_1.MarkdownGenerator(ast, config);
47
+ break;
33
48
  case 'html':
34
- return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
49
+ generator = new HtmlGenerator_js_1.HtmlGenerator(ast, config);
50
+ break;
35
51
  case 'pdf':
36
- return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
52
+ generator = new PdfGenerator_js_1.PdfGenerator(ast, config);
53
+ break;
37
54
  case 'csv':
38
- return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
55
+ generator = new CsvGenerator_js_1.CsvGenerator(ast, config);
56
+ break;
39
57
  case 'rtf':
40
- return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
58
+ generator = new RtfGenerator_js_1.RtfGenerator(ast, config);
59
+ break;
41
60
  case 'chunks':
42
- return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
61
+ generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
62
+ break;
43
63
  default:
44
- throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
64
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED, undefined, destination);
45
65
  }
66
+ return generator.generate();
46
67
  }
47
68
  }
48
69
  exports.OfficeGenerator = OfficeGenerator;
@@ -81,7 +81,7 @@ export declare class OfficeParser {
81
81
  * });
82
82
  *
83
83
  * // Parse a Buffer with OCR enabled
84
- * const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
84
+ * const buffer = await retrieveData('document.pdf').then(r => r.arrayBuffer());
85
85
  * const ast = await OfficeParser.parseOffice(buffer, {
86
86
  * ocr: true,
87
87
  * ocrLanguage: 'eng+fra'
@@ -98,7 +98,7 @@ class OfficeParser {
98
98
  * });
99
99
  *
100
100
  * // Parse a Buffer with OCR enabled
101
- * const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
101
+ * const buffer = await retrieveData('document.pdf').then(r => r.arrayBuffer());
102
102
  * const ast = await OfficeParser.parseOffice(buffer, {
103
103
  * ocr: true,
104
104
  * ocrLanguage: 'eng+fra'
package/dist/cli.d.ts CHANGED
@@ -4,19 +4,25 @@
4
4
  *
5
5
  * Allows running officeparser from the command line:
6
6
  * npx officeparser file.docx
7
- * officeparser file.docx --toText=true
8
- * officeparser file.docx --ocr=true --extractAttachments=true
7
+ * officeparser file.docx --to=text
8
+ * officeparser file.docx --ocr --extractAttachments
9
9
  *
10
- * Options (--key=value):
11
- * --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
10
+ * Options (--key=value, --key value, or bare flags):
11
+ * --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
12
12
  * --output=path Save result to a file
13
- * --toText=true Legacy flag for plain text output
14
- * --ocr=true Enable OCR for images
15
- * --ocrLanguage=eng OCR language (default: eng)
16
- * --extractAttachments=true Extract embedded attachments
17
- * --ignoreNotes=true Ignore footnotes/endnotes
18
- * --putNotesAtLast=true Move notes to end of document
19
- * --includeRawContent=true Include raw content in AST
20
- * --outputErrorToConsole=true Log errors to console
13
+ * --fileType=docx|xlsx|... Override file type detection
14
+ * --ocr Enable OCR for images (default: false)
15
+ * --ocrConfig.language=eng OCR language (default: eng)
16
+ * --extractAttachments Extract embedded attachments (default: false)
17
+ * --ignoreNotes Ignore footnotes/endnotes/speaker notes (default: false)
18
+ * --ignoreComments Ignore inline comments (default: false)
19
+ * --ignoreHeadersAndFooters Ignore headers and footers (default: false)
20
+ * --ignoreSlideMasters Ignore slide masters (default: false)
21
+ * --ignoreInternalLinks Ignore internal links (default: false)
22
+ * --includeRawContent Include raw content in AST (default: false)
23
+ * --serializeRawContent Include stringified XML in metadata (default: true)
24
+ * --preserveXmlWhitespace Keep raw formatting space (default: false)
25
+ * --includeBreakNodes Include break nodes (DOCX only, default: false)
26
+ * --verbose Show full error stack traces and warning logs
21
27
  */
22
28
  export {};