officeparser 7.1.0 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +11 -0
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +8 -1
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +0 -6
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +259 -52
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +1 -1
- package/dist/parsers/ExcelParser.js +63 -19
- package/dist/parsers/HtmlParser.js +10 -1
- package/dist/parsers/MarkdownParser.js +13 -10
- package/dist/parsers/OpenOfficeParser.js +57 -34
- package/dist/parsers/PdfParser.js +23 -1
- package/dist/parsers/PowerPointParser.js +164 -40
- package/dist/parsers/RtfParser.js +28 -24
- package/dist/parsers/WordParser.js +154 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +265 -52
- package/dist/types.js +2 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +55 -1
- package/dist/utils/errorUtils.js +2 -1
- package/dist/utils/xmlUtils.d.ts +9 -0
- package/dist/utils/xmlUtils.js +53 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -100,7 +100,7 @@ npx officeparser document.pdf --format=chunks
|
|
|
100
100
|
|------|--------|---------|-------------|
|
|
101
101
|
| `--format` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
|
|
102
102
|
| `--output` | path | — | Write output to a file |
|
|
103
|
-
|
|
|
103
|
+
| ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--format=text` |
|
|
104
104
|
| `--ignoreNotes` | `true\|false` | `false` | Ignore speaker notes (PPTX/ODP) |
|
|
105
105
|
| `--putNotesAtLast` | `true\|false` | `false` | Collect notes at end of output |
|
|
106
106
|
| `--newlineDelimiter` | string | `\n` | Delimiter between lines |
|
|
@@ -108,7 +108,7 @@ npx officeparser document.pdf --format=chunks
|
|
|
108
108
|
| `--ocr` | `true\|false` | `false` | Enable OCR for images |
|
|
109
109
|
| `--includeRawContent` | `true\|false` | `false` | Include raw XML/RTF in nodes |
|
|
110
110
|
| `--includeBreakNodes` | `true\|false` | `false` | Include break nodes (DOCX only) |
|
|
111
|
-
|
|
|
111
|
+
| ~~`--outputErrorToConsole`~~ | `true\|false` | `false` | **Deprecated.** Use `onWarning` callback |
|
|
112
112
|
| `--verbose` | `true\|false` | `false` | Show full error stack traces |
|
|
113
113
|
|
|
114
114
|
---
|
|
@@ -420,13 +420,19 @@ interface OfficeChunk {
|
|
|
420
420
|
```text
|
|
421
421
|
OfficeParserAST
|
|
422
422
|
├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (11 formats)
|
|
423
|
-
├── metadata: { author, title, created, modified, customProperties, styleMap, ... }
|
|
423
|
+
├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
|
|
424
424
|
├── content: [ OfficeContentNode ]
|
|
425
|
-
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | ...
|
|
425
|
+
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
|
|
426
426
|
│ ├── text: string (concatenated text of node + all descendants)
|
|
427
|
-
│ ├── children: [ OfficeContentNode ] (recursive)
|
|
427
|
+
│ ├── children: [ OfficeContentNode ] (recursive structural children)
|
|
428
|
+
│ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
|
|
429
|
+
│ ├── comments: [ OfficeContentNode ] (inline comments attached to this node)
|
|
428
430
|
│ ├── formatting: { bold, italic, underline, color, size, font, alignment, ... }
|
|
429
|
-
│ └── metadata: { level, listId, row, col, rowSpan, colSpan, style, ... }
|
|
431
|
+
│ └── metadata: { level, listId, row, col, rowSpan, colSpan, backgroundColor, style, ... }
|
|
432
|
+
├── auxiliary?: OfficeAuxiliaryContent (out-of-band layout elements)
|
|
433
|
+
│ ├── headers?: OfficeContentNode[] (DOCX headers)
|
|
434
|
+
│ ├── footers?: OfficeContentNode[] (DOCX footers)
|
|
435
|
+
│ └── slideMasters?: OfficeContentNode[] (PPTX slide masters)
|
|
430
436
|
├── attachments: [ OfficeAttachment ] (populated when extractAttachments: true)
|
|
431
437
|
│ ├── type: 'image' | 'chart'
|
|
432
438
|
│ ├── name: string
|
|
@@ -436,7 +442,7 @@ OfficeParserAST
|
|
|
436
442
|
│ └── chartData?: { title, dataSets, labels }
|
|
437
443
|
├── warnings: OfficeIssue[] (non-fatal issues from the parsing phase)
|
|
438
444
|
├── to(format, config?) (format: 'html'|'md'|'text'|'csv'|'rtf'|'pdf'|'chunks', returns { value, messages })
|
|
439
|
-
└── toText() (Deprecated: use .to('text') instead)
|
|
445
|
+
└── ~~toText()~~ (Deprecated: use .to('text') instead)
|
|
440
446
|
```
|
|
441
447
|
|
|
442
448
|
### `OfficeIssue` — Warning / Error Object
|
|
@@ -552,17 +558,21 @@ ast.metadata = {
|
|
|
552
558
|
created?: Date
|
|
553
559
|
modified?: Date
|
|
554
560
|
description?: string
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
561
|
+
keywords?: string // NEW: Keywords from document properties
|
|
562
|
+
customProperties?: Record<string, any> // User-defined metadata from the document
|
|
563
|
+
nativeProperties?: Record<string, any> // NEW: All format-specific raw metadata
|
|
564
|
+
styleMap?: Record<string, TextFormatting> // Named styles → formatting definitions
|
|
565
|
+
formatting?: TextFormatting // Document-wide defaults
|
|
558
566
|
}
|
|
559
567
|
```
|
|
560
568
|
|
|
561
|
-
**Accessing
|
|
569
|
+
**Accessing native properties (format-specific metadata):**
|
|
562
570
|
```js
|
|
563
571
|
const ast = await officeParser.parseOffice('contract.docx');
|
|
564
|
-
console.log(ast.metadata.
|
|
565
|
-
// {
|
|
572
|
+
console.log(ast.metadata.nativeProperties);
|
|
573
|
+
// DOCX: { Pages: 5, Application: 'Microsoft Word' }
|
|
574
|
+
// HTML: { description: 'My page', 'og:title': 'Title' }
|
|
575
|
+
// PDF: { Title: 'Report', XMP: { ... } }
|
|
566
576
|
```
|
|
567
577
|
|
|
568
578
|
---
|
|
@@ -586,6 +596,61 @@ const headings = ast.content.filter(n => n.type === 'heading' && n.metadata?.lev
|
|
|
586
596
|
console.log(headings.map(h => h.text));
|
|
587
597
|
```
|
|
588
598
|
|
|
599
|
+
### Extract comments
|
|
600
|
+
```ts
|
|
601
|
+
// Comments can be attached to any nested node, so we must traverse recursively
|
|
602
|
+
const printComments = (nodes: OfficeContentNode[]) => {
|
|
603
|
+
nodes.forEach(node => {
|
|
604
|
+
if (node.comments) {
|
|
605
|
+
node.comments.forEach(c => {
|
|
606
|
+
console.log(`Comment by ${c.metadata?.author}: ${c.text}`);
|
|
607
|
+
});
|
|
608
|
+
}
|
|
609
|
+
if (node.children) {
|
|
610
|
+
printComments(node.children);
|
|
611
|
+
}
|
|
612
|
+
});
|
|
613
|
+
};
|
|
614
|
+
|
|
615
|
+
printComments(ast.content);
|
|
616
|
+
```
|
|
617
|
+
|
|
618
|
+
Set `ignoreComments: true` to skip extraction.
|
|
619
|
+
|
|
620
|
+
### Extract footnotes, endnotes & slide notes
|
|
621
|
+
```ts
|
|
622
|
+
// Slide speaker notes (PPTX) live on the slide node itself
|
|
623
|
+
const slide = ast.content.find(n => n.type === 'slide');
|
|
624
|
+
console.log(slide?.notes?.map(n => n.text));
|
|
625
|
+
|
|
626
|
+
// Footnotes and endnotes (DOCX/RTF) can be deeply nested, so we traverse recursively:
|
|
627
|
+
const printNotes = (nodes: OfficeContentNode[]) => {
|
|
628
|
+
nodes.forEach(node => {
|
|
629
|
+
if (node.notes) {
|
|
630
|
+
node.notes.forEach(note => console.log(note.text));
|
|
631
|
+
}
|
|
632
|
+
if (node.children) {
|
|
633
|
+
printNotes(node.children);
|
|
634
|
+
}
|
|
635
|
+
});
|
|
636
|
+
};
|
|
637
|
+
|
|
638
|
+
printNotes(ast.content);
|
|
639
|
+
```
|
|
640
|
+
|
|
641
|
+
> [!IMPORTANT]
|
|
642
|
+
> `putNotesAtLast` is **deprecated**. Notes are always attached via `node.notes` — this flag has no effect and will be removed in a future major version.
|
|
643
|
+
|
|
644
|
+
### Access headers, footers & slide masters
|
|
645
|
+
```ts
|
|
646
|
+
// These are NOT in ast.content — use ast.auxiliary
|
|
647
|
+
console.log(ast.auxiliary?.headers?.map(h => h.text)); // DOCX headers
|
|
648
|
+
console.log(ast.auxiliary?.footers?.map(f => f.text)); // DOCX footers
|
|
649
|
+
console.log(ast.auxiliary?.slideMasters?.length); // PPTX slide masters
|
|
650
|
+
```
|
|
651
|
+
|
|
652
|
+
Set `ignoreHeadersAndFooters: true` or `ignoreSlideMasters: true` to skip extraction.
|
|
653
|
+
|
|
589
654
|
### Extract images with OCR text
|
|
590
655
|
```js
|
|
591
656
|
const ast = await officeParser.parseOffice('report.docx', { extractAttachments: true, ocr: true });
|
|
@@ -650,8 +715,11 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
650
715
|
| Option | Type | Default | Description |
|
|
651
716
|
|--------|------|---------|-------------|
|
|
652
717
|
| `newlineDelimiter` | `string` | `'\n'` | Delimiter inserted between lines in text output |
|
|
653
|
-
| `ignoreNotes` | `boolean` | `false` | Ignore speaker notes (PPTX/ODP) |
|
|
654
|
-
| `
|
|
718
|
+
| `ignoreNotes` | `boolean` | `false` | Ignore footnotes/endnotes (DOCX, RTF) and speaker notes (PPTX/ODP) |
|
|
719
|
+
| `ignoreComments` | `boolean` | `false` | **New**: Ignore inline comments/annotations (DOCX, XLSX, PPTX) — by default attached via `node.comments[]` |
|
|
720
|
+
| `ignoreHeadersAndFooters` | `boolean` | `false` | **New**: Skip DOCX headers & footers (populated in `ast.auxiliary.headers/footers` by default) |
|
|
721
|
+
| `ignoreSlideMasters` | `boolean` | `false` | **New**: Skip PPTX slide masters (populated in `ast.auxiliary.slideMasters` by default) |
|
|
722
|
+
| ~~`putNotesAtLast`~~ | `boolean` | `false` | **Deprecated**: Notes are now attached via `node.notes[]`. This flag has no effect |
|
|
655
723
|
| `extractAttachments` | `boolean` | `false` | Populate `ast.attachments` with Base64 images/charts |
|
|
656
724
|
| `ocr` | `boolean` | `false` | Run Tesseract OCR on images (requires `extractAttachments: true`) |
|
|
657
725
|
| `ocrConfig` | `OcrConfig` | `{}` | OCR worker pool settings — see [OCR section](#ocr-scheduler--resource-management) |
|
|
@@ -665,7 +733,7 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
665
733
|
| `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
|
|
666
734
|
| `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
|
|
667
735
|
| `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
|
|
668
|
-
|
|
|
736
|
+
| ~~`outputErrorToConsole`~~ | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
|
|
669
737
|
|
|
670
738
|
---
|
|
671
739
|
|
|
@@ -764,6 +832,12 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
764
832
|
|--------|------|---------|-------------|
|
|
765
833
|
| `standalone` | `boolean` | `true` | Wrap output in a full `<html>` document with CSS |
|
|
766
834
|
| `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
|
|
835
|
+
| `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
|
|
836
|
+
| `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block — use to override built-in styles |
|
|
837
|
+
| `injections.headStart` | `string` | `''` | Raw HTML injected after `<head>` |
|
|
838
|
+
| `injections.headEnd` | `string` | `''` | Raw HTML injected before `</head>` |
|
|
839
|
+
| `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
|
|
840
|
+
| `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
|
|
767
841
|
|
|
768
842
|
### MdGeneratorConfig
|
|
769
843
|
|
|
@@ -780,6 +854,8 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
|
|
|
780
854
|
| Option | Type | Default | Description |
|
|
781
855
|
|--------|------|---------|-------------|
|
|
782
856
|
| `format` | `string` | `'A4'` | Paper format (`'A4'`, `'Letter'`, `'Legal'`, etc.) |
|
|
857
|
+
| `width` | `string \| number` | `''` | Paper width (e.g., `'5in'`, `'3cm'`) or pixels |
|
|
858
|
+
| `height` | `string \| number` | `''` | Paper height (e.g., `'5in'`, `'3cm'`) or pixels |
|
|
783
859
|
| `landscape` | `boolean` | `false` | Landscape page orientation |
|
|
784
860
|
| `printBackground` | `boolean` | `true` | Print background graphics |
|
|
785
861
|
| `margin` | `object` | `{0,0,0,0}` | Page margins (`top`, `right`, `bottom`, `left`) |
|
|
@@ -885,7 +961,7 @@ When `ocr: true` is set, `officeParser` maintains an intelligent **Smart Worker
|
|
|
885
961
|
| `corePath` | `string` | `''` | Custom path to Tesseract core script |
|
|
886
962
|
| `langPath` | `string` | `''` | Custom path for language data files |
|
|
887
963
|
| `timeout` | `OcrTimeoutConfig` | `{}` | Consolidated timeouts: `autoTerminate`, `workerLoad`, `recognition` |
|
|
888
|
-
|
|
|
964
|
+
| ~~`autoTerminateTimeout`~~ | `number` | `10000` | **Deprecated.** Use `timeout.autoTerminate` instead |
|
|
889
965
|
|
|
890
966
|
See all language codes at [tesseract-ocr.github.io](https://tesseract-ocr.github.io/tessdoc/Data-Files).
|
|
891
967
|
|
|
@@ -986,7 +1062,6 @@ For a full debugging guide, visit the [Live Documentation](https://harshankur.gi
|
|
|
986
1062
|
|
|
987
1063
|
1. **ODT/ODS Charts**: May show inaccurate data when the chart references external cell ranges or uses complex layout-based data.
|
|
988
1064
|
2. **PDF Images (Browser)**: Extracted as BMP files for cross-platform compatibility. Conversion is automatic.
|
|
989
|
-
3. **RTF Notes**: `putNotesAtLast` has no effect for RTF files; footnotes and endnotes are always appended at the end.
|
|
990
1065
|
|
|
991
1066
|
---
|
|
992
1067
|
|
|
@@ -15,5 +15,5 @@ export declare class OfficeGenerator {
|
|
|
15
15
|
*/
|
|
16
16
|
static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
|
|
17
17
|
type: T;
|
|
18
|
-
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult
|
|
18
|
+
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
19
19
|
}
|
package/dist/OfficeGenerator.js
CHANGED
|
@@ -25,24 +25,33 @@ class OfficeGenerator {
|
|
|
25
25
|
* @throws {Error} If the destination format is unsupported
|
|
26
26
|
*/
|
|
27
27
|
static async generate(ast, destination, config) {
|
|
28
|
+
let generator;
|
|
28
29
|
switch (destination.toLowerCase()) {
|
|
29
30
|
case 'text':
|
|
30
|
-
|
|
31
|
+
generator = new TextGenerator_js_1.TextGenerator(ast, config);
|
|
32
|
+
break;
|
|
31
33
|
case 'md':
|
|
32
|
-
|
|
34
|
+
generator = new MarkdownGenerator_js_1.MarkdownGenerator(ast, config);
|
|
35
|
+
break;
|
|
33
36
|
case 'html':
|
|
34
|
-
|
|
37
|
+
generator = new HtmlGenerator_js_1.HtmlGenerator(ast, config);
|
|
38
|
+
break;
|
|
35
39
|
case 'pdf':
|
|
36
|
-
|
|
40
|
+
generator = new PdfGenerator_js_1.PdfGenerator(ast, config);
|
|
41
|
+
break;
|
|
37
42
|
case 'csv':
|
|
38
|
-
|
|
43
|
+
generator = new CsvGenerator_js_1.CsvGenerator(ast, config);
|
|
44
|
+
break;
|
|
39
45
|
case 'rtf':
|
|
40
|
-
|
|
46
|
+
generator = new RtfGenerator_js_1.RtfGenerator(ast, config);
|
|
47
|
+
break;
|
|
41
48
|
case 'chunks':
|
|
42
|
-
|
|
49
|
+
generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
|
|
50
|
+
break;
|
|
43
51
|
default:
|
|
44
52
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
|
|
45
53
|
}
|
|
54
|
+
return generator.generate();
|
|
46
55
|
}
|
|
47
56
|
}
|
|
48
57
|
exports.OfficeGenerator = OfficeGenerator;
|
package/dist/cli.d.ts
CHANGED
|
@@ -15,6 +15,10 @@
|
|
|
15
15
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
16
16
|
* --extractAttachments=true Extract embedded attachments
|
|
17
17
|
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
18
|
+
* --ignoreComments=true Ignore inline comments
|
|
19
|
+
* --ignoreHeadersAndFooters=true Ignore headers and footers
|
|
20
|
+
* --ignoreSlideMasters=true Ignore slide masters
|
|
21
|
+
* --ignoreInternalLinks=true Ignore internal links
|
|
18
22
|
* --putNotesAtLast=true Move notes to end of document
|
|
19
23
|
* --includeRawContent=true Include raw content in AST
|
|
20
24
|
* --outputErrorToConsole=true Log errors to console
|
package/dist/cli.js
CHANGED
|
@@ -16,6 +16,10 @@
|
|
|
16
16
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
17
17
|
* --extractAttachments=true Extract embedded attachments
|
|
18
18
|
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
19
|
+
* --ignoreComments=true Ignore inline comments
|
|
20
|
+
* --ignoreHeadersAndFooters=true Ignore headers and footers
|
|
21
|
+
* --ignoreSlideMasters=true Ignore slide masters
|
|
22
|
+
* --ignoreInternalLinks=true Ignore internal links
|
|
19
23
|
* --putNotesAtLast=true Move notes to end of document
|
|
20
24
|
* --includeRawContent=true Include raw content in AST
|
|
21
25
|
* --outputErrorToConsole=true Log errors to console
|
|
@@ -83,9 +87,10 @@ if (fileArg) {
|
|
|
83
87
|
const lowerValue = value.toLowerCase();
|
|
84
88
|
const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
|
|
85
89
|
const knownBooleans = new Set([
|
|
86
|
-
'toText', 'ocr', 'extractAttachments', 'ignoreNotes', '
|
|
87
|
-
'
|
|
88
|
-
'
|
|
90
|
+
'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'ignoreComments',
|
|
91
|
+
'ignoreHeadersAndFooters', 'ignoreSlideMasters', 'ignoreInternalLinks',
|
|
92
|
+
'putNotesAtLast', 'includeRawContent', 'outputErrorToConsole',
|
|
93
|
+
'serializeRawContent', 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
|
|
89
94
|
]);
|
|
90
95
|
if (cleanKey === 'format') {
|
|
91
96
|
outputFormat = value;
|
|
@@ -184,6 +189,10 @@ else {
|
|
|
184
189
|
console.log(' --ocrLanguage=eng OCR language (default: eng)');
|
|
185
190
|
console.log(' --extractAttachments=true Extract embedded attachments');
|
|
186
191
|
console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
|
|
192
|
+
console.log(' --ignoreComments=true Ignore inline comments');
|
|
193
|
+
console.log(' --ignoreHeadersAndFooters=true Ignore headers and footers');
|
|
194
|
+
console.log(' --ignoreSlideMasters=true Ignore slide masters');
|
|
195
|
+
console.log(' --ignoreInternalLinks=true Ignore internal links');
|
|
187
196
|
console.log(' --putNotesAtLast=true Move notes to end of document');
|
|
188
197
|
console.log(' --includeRawContent=true Include raw content in AST');
|
|
189
198
|
console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
|
package/dist/defaults.js
CHANGED
|
@@ -42,6 +42,9 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
|
42
42
|
onWarning: () => { },
|
|
43
43
|
newlineDelimiter: '\n',
|
|
44
44
|
ignoreNotes: false,
|
|
45
|
+
ignoreComments: false,
|
|
46
|
+
ignoreHeadersAndFooters: false,
|
|
47
|
+
ignoreSlideMasters: false,
|
|
45
48
|
putNotesAtLast: false,
|
|
46
49
|
extractAttachments: false,
|
|
47
50
|
includeRawContent: false,
|
|
@@ -63,6 +66,14 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
|
63
66
|
const DEFAULT_HTML_GENERATOR_CONFIG = {
|
|
64
67
|
standalone: true,
|
|
65
68
|
chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
|
|
69
|
+
containerWidth: 'auto',
|
|
70
|
+
customCss: '',
|
|
71
|
+
injections: {
|
|
72
|
+
headStart: '',
|
|
73
|
+
headEnd: '',
|
|
74
|
+
bodyStart: '',
|
|
75
|
+
bodyEnd: '',
|
|
76
|
+
}
|
|
66
77
|
};
|
|
67
78
|
/**
|
|
68
79
|
* Default configuration for PDF generation.
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType } from '../types.js';
|
|
1
|
+
import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType, UniversalGeneratorFormat } from '../types.js';
|
|
2
2
|
import { StyleMapper } from '../utils/styleMapper.js';
|
|
3
3
|
/**
|
|
4
4
|
* Base class for all document generators.
|
|
5
5
|
* Provides common traversal logic and configuration handling.
|
|
6
6
|
*/
|
|
7
|
-
export declare abstract class BaseGenerator<D extends
|
|
7
|
+
export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat = UniversalGeneratorFormat> {
|
|
8
8
|
protected destination: D;
|
|
9
9
|
protected config: FullGeneratorConfig;
|
|
10
10
|
protected ast: OfficeParserAST;
|
|
@@ -24,7 +24,7 @@ export declare abstract class BaseGenerator<D extends string = string> {
|
|
|
24
24
|
/**
|
|
25
25
|
* Entry point for generation.
|
|
26
26
|
*/
|
|
27
|
-
abstract generate(): Promise<ConversionResult
|
|
27
|
+
abstract generate(): Promise<ConversionResult<D>>;
|
|
28
28
|
/**
|
|
29
29
|
* Centralized logic for handling the onNode callback.
|
|
30
30
|
* Evaluates the callback and returns a result that tells the generator how to proceed.
|
|
@@ -393,7 +393,14 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
393
393
|
renderedRows.push(row.text ?? '');
|
|
394
394
|
continue;
|
|
395
395
|
}
|
|
396
|
-
const
|
|
396
|
+
const getCellText = (cell) => {
|
|
397
|
+
if (cell.text)
|
|
398
|
+
return cell.text;
|
|
399
|
+
if (!cell.children || cell.children.length === 0)
|
|
400
|
+
return '';
|
|
401
|
+
return cell.children.map(c => getCellText(c)).join(' ');
|
|
402
|
+
};
|
|
403
|
+
const cells = row.children.map(cell => getCellText(cell).replace(/\n/g, ' ').trim());
|
|
397
404
|
renderedRows.push(`| ${cells.join(' | ')} |`);
|
|
398
405
|
}
|
|
399
406
|
return renderedRows.join('\n');
|
|
@@ -10,7 +10,7 @@ export declare class CsvGenerator extends BaseGenerator<'csv'> {
|
|
|
10
10
|
*
|
|
11
11
|
* @returns A CSV string or a ZIP archive containing multiple CSVs
|
|
12
12
|
*/
|
|
13
|
-
generate(): Promise<ConversionResult
|
|
13
|
+
generate(): Promise<ConversionResult<'csv'>>;
|
|
14
14
|
/**
|
|
15
15
|
* Recursively finds all nodes that can be treated as sheets (sheet or table).
|
|
16
16
|
*/
|
|
@@ -12,7 +12,7 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
|
|
|
12
12
|
*
|
|
13
13
|
* @returns An HTML string
|
|
14
14
|
*/
|
|
15
|
-
generate(): Promise<ConversionResult
|
|
15
|
+
generate(): Promise<ConversionResult<'html'>>;
|
|
16
16
|
private renderMetaTags;
|
|
17
17
|
private renderMetadataSummary;
|
|
18
18
|
/**
|
|
@@ -33,5 +33,6 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
|
|
|
33
33
|
private getInlineStyles;
|
|
34
34
|
private getPremiumStyles;
|
|
35
35
|
protected slugify(text: string): string;
|
|
36
|
+
private getColumnLetter;
|
|
36
37
|
private escape;
|
|
37
38
|
}
|