officeparser 7.4.0 → 7.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +85 -4
- package/dist/OfficeParser.js +54 -6
- package/dist/generators/BaseGenerator.d.ts +15 -0
- package/dist/generators/BaseGenerator.js +31 -0
- package/dist/generators/HtmlGenerator.d.ts +9 -0
- package/dist/generators/HtmlGenerator.js +34 -4
- package/dist/generators/MarkdownGenerator.d.ts +13 -0
- package/dist/generators/MarkdownGenerator.js +115 -41
- package/dist/generators/RtfGenerator.d.ts +13 -0
- package/dist/generators/RtfGenerator.js +23 -2
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +34 -1
- package/dist/officeparser.browser.iife.js +160 -160
- package/dist/officeparser.browser.mjs +198 -198
- package/dist/officeparser.browser.slim.d.ts +34 -1
- package/dist/officeparser.browser.slim.iife.js +186 -186
- package/dist/officeparser.browser.slim.mjs +186 -186
- package/dist/parsers/EpubParser.js +2 -2
- package/dist/parsers/ExcelParser.js +11 -7
- package/dist/parsers/HtmlParser.js +39 -0
- package/dist/parsers/OpenOfficeParser.js +139 -167
- package/dist/parsers/PowerPointParser.js +48 -11
- package/dist/parsers/WordParser.js +33 -7
- package/dist/sbom.cdx.json +92 -92
- package/dist/types.d.ts +34 -1
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.d.ts +15 -2
- package/dist/utils/configUtils.js +58 -13
- package/dist/utils/errorUtils.d.ts +8 -2
- package/dist/utils/errorUtils.js +23 -1
- package/dist/utils/mathUtils.d.ts +42 -0
- package/dist/utils/mathUtils.js +385 -0
- package/dist/utils/zipUtils.d.ts +64 -4
- package/dist/utils/zipUtils.js +188 -4
- package/package.json +9 -5
package/README.md
CHANGED
|
@@ -128,7 +128,7 @@ npx officeparser my_document --fileType=docx --to=json
|
|
|
128
128
|
| `--includeRawContent` | boolean | `false` | Include raw XML/RTF in nodes |
|
|
129
129
|
| `--serializeRawContent` | boolean | `true` | Include stringified XML in metadata |
|
|
130
130
|
| `--preserveXmlWhitespace` | boolean | `false` | Keep raw formatting space |
|
|
131
|
-
| `--includeBreakNodes` | boolean | `false` | Include break nodes (DOCX
|
|
131
|
+
| `--includeBreakNodes` | boolean | `false` | Include break nodes (DOCX and ODF) |
|
|
132
132
|
| `--verbose` | boolean | `false` | Show full error stack traces and warning logs |
|
|
133
133
|
| `--includeFormatting` | boolean | `true` | Include formatting style map matching |
|
|
134
134
|
| `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
|
|
@@ -206,6 +206,15 @@ const ast = await officeParser.parseOffice(buffer);
|
|
|
206
206
|
> const ast = await officeParser.parseOffice(markdownBuffer, { fileType: 'md' });
|
|
207
207
|
> ```
|
|
208
208
|
|
|
209
|
+
> [!NOTE]
|
|
210
|
+
> **ZIP-backed formats are identified from inside the archive.** DOCX, XLSX, PPTX, ODT, ODS, ODP
|
|
211
|
+
> and EPUB are all ZIP files, and telling them apart from the first bytes alone is unreliable for
|
|
212
|
+
> archives written by streaming producers or holding very many parts. When the byte signature is
|
|
213
|
+
> inconclusive, the archive is opened and the format is read from its own declaration
|
|
214
|
+
> (`[Content_Types].xml`, or the `mimetype` entry), so these parse from a buffer without a hint.
|
|
215
|
+
> Supplying `fileType` remains the fastest and most certain route: it decides which parser runs,
|
|
216
|
+
> and for these formats no archive inspection is done at all.
|
|
217
|
+
|
|
209
218
|
### Cancellation with AbortSignal
|
|
210
219
|
|
|
211
220
|
You can pass a standard `AbortSignal` (e.g. from an `AbortController`) to cancel an active parse operation. This is especially useful for setting request-level timeouts or canceling long-running parses (like large PDFs with OCR).
|
|
@@ -533,6 +542,34 @@ interface OfficeIssue {
|
|
|
533
542
|
}
|
|
534
543
|
```
|
|
535
544
|
|
|
545
|
+
Thrown errors carry the same object on `error.officeIssue`, so a failed parse is identified by
|
|
546
|
+
the same stable `code` you would branch on for a warning, rather than by matching message text:
|
|
547
|
+
|
|
548
|
+
```js
|
|
549
|
+
try {
|
|
550
|
+
const ast = await officeParser.parseOffice(buffer, { fileType: 'docx' });
|
|
551
|
+
} catch (err) {
|
|
552
|
+
switch (err.officeIssue?.code) {
|
|
553
|
+
case 'ZIP_NO_ENTRIES_FOUND': // not a ZIP archive at all
|
|
554
|
+
case 'ZIP_TRUNCATED': // cut off in transfer, entries incomplete
|
|
555
|
+
case 'REQUIRED_PART_MISSING': // readable ZIP, but not the format it claims
|
|
556
|
+
console.error('Unusable file:', err.officeIssue.message);
|
|
557
|
+
break;
|
|
558
|
+
default:
|
|
559
|
+
throw err;
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
```
|
|
563
|
+
|
|
564
|
+
> [!IMPORTANT]
|
|
565
|
+
> **A corrupt file throws; it does not parse as an empty document.** If an archive is not
|
|
566
|
+
> readable, is truncated, or is missing the part its format requires (`word/document.xml`,
|
|
567
|
+
> `xl/workbook.xml`, `ppt/presentation.xml`, ODF `content.xml`, the EPUB OPF), parsing rejects
|
|
568
|
+
> with one of the codes above. An empty result therefore means the document really is empty.
|
|
569
|
+
> Files that are legitimately empty still parse, and say so through `onWarning` /
|
|
570
|
+
> `ast.warnings` (`NO_WORKSHEETS_FOUND` for a chartsheet-only workbook, `NO_SLIDES_FOUND` for a
|
|
571
|
+
> presentation with no slides).
|
|
572
|
+
|
|
536
573
|
---
|
|
537
574
|
|
|
538
575
|
## Deep Dive: Document Components
|
|
@@ -608,7 +645,14 @@ formatting: {
|
|
|
608
645
|
}
|
|
609
646
|
```
|
|
610
647
|
|
|
611
|
-
|
|
648
|
+
> [!NOTE]
|
|
649
|
+
> On a **content node**, an absent flag and `false` mean the same thing — the flag is simply not
|
|
650
|
+
> applied. On **`ast.metadata.styleMap`**, they differ: an absent flag means the style says nothing
|
|
651
|
+
> about that property (so it inherits), while `false` means the style explicitly turns it off
|
|
652
|
+
> (ODF's `fo:font-weight="normal"`, DOCX's `<w:b w:val="0"/>`). Code resolving inheritance itself
|
|
653
|
+
> must test `=== undefined`, not truthiness, or it will treat "explicitly off" as "unspecified".
|
|
654
|
+
|
|
655
|
+
### 6. Break Nodes (DOCX and ODF)
|
|
612
656
|
|
|
613
657
|
When `includeBreakNodes: true`, break elements appear as nodes:
|
|
614
658
|
|
|
@@ -623,6 +667,43 @@ Break Node (type: 'break')
|
|
|
623
667
|
> [!NOTE]
|
|
624
668
|
> Break nodes have no `text` property, but `ast.toText()` and `ast.to('text')` automatically convert them to the configured newline delimiter.
|
|
625
669
|
|
|
670
|
+
> [!NOTE]
|
|
671
|
+
> DOCX writes breaks inline (`w:br`/`w:cr`), so they land as children of the paragraph. ODF instead
|
|
672
|
+
> carries page and column breaks on the paragraph *style* (`fo:break-before`/`fo:break-after`), so those
|
|
673
|
+
> are emitted as siblings around the paragraph rather than inside it. `<text:soft-page-break/>` maps onto
|
|
674
|
+
> `lastRenderedPage`, the same type as DOCX's `w:lastRenderedPageBreak`.
|
|
675
|
+
|
|
676
|
+
### 6b. Equations
|
|
677
|
+
|
|
678
|
+
Equations are extracted from every format that can carry them and normalized to **LaTeX**, so a
|
|
679
|
+
formula means the same thing whichever format it arrived in:
|
|
680
|
+
|
|
681
|
+
| Source format | Markup in the file |
|
|
682
|
+
|---|---|
|
|
683
|
+
| DOCX, PPTX | OOXML `<m:oMath>` / `<m:oMathPara>` |
|
|
684
|
+
| ODT, ODP, ODS | MathML inside the embedded formula object |
|
|
685
|
+
| HTML, EPUB | native MathML `<math>` |
|
|
686
|
+
| Markdown | `$inline$` / `$$block$$` |
|
|
687
|
+
|
|
688
|
+
They all land as the same node:
|
|
689
|
+
|
|
690
|
+
```text
|
|
691
|
+
Code Node (type: 'code')
|
|
692
|
+
├── text: '\\frac{1}{2}' // LaTeX, whatever the source markup was
|
|
693
|
+
└── metadata: { math: 'inline' | 'block' }
|
|
694
|
+
```
|
|
695
|
+
|
|
696
|
+
Fractions, sub/superscripts, radicals, delimiters, n-ary operators (sums, integrals), named
|
|
697
|
+
functions, accents, bars, matrices and math alphabets (`ℝ`, `𝒜`, …) are all preserved. When a
|
|
698
|
+
document supplies its own TeX source in an `<annotation encoding="application/x-tex">`, that is
|
|
699
|
+
used verbatim in preference to anything reconstructed from the presentation markup.
|
|
700
|
+
|
|
701
|
+
> [!NOTE]
|
|
702
|
+
> Equation text is *structure*, not prose: a fraction whose numerator and denominator are simply
|
|
703
|
+
> concatenated reads as a different number rather than as obviously-missing content. Consumers that
|
|
704
|
+
> index document text should treat `code` nodes carrying `math` as opaque LaTeX rather than
|
|
705
|
+
> splitting them as words.
|
|
706
|
+
|
|
626
707
|
### 7. Document Metadata
|
|
627
708
|
|
|
628
709
|
```ts
|
|
@@ -688,7 +769,7 @@ idempotent and `.md → AST → HTML → AST → .md` survives unchanged.
|
|
|
688
769
|
| Attribute lists | `{width=50% .centered}` | `ImageMetadata.width` / `.align`, `TableMetadata.align` |
|
|
689
770
|
| Citations | `[@smith2024]` | `TextMetadata.citationKey` |
|
|
690
771
|
| Wikilinks | `[[Page]]` / `[[Page\|Alias]]` | `TextMetadata.wikilink`, `.link`, `.linkType` |
|
|
691
|
-
| Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `
|
|
772
|
+
| Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `type: 'code'`, `CodeMetadata.math` (`'inline' \| 'block'`) |
|
|
692
773
|
| Frontmatter arrays | `tags: [a, b]` or `tags: ["a","b"]` | Real array in `metadata.customProperties`/`nativeProperties` |
|
|
693
774
|
| MDX components (import-only) | `<Component prop="x">...</Component>` | Stripped; inner Markdown is kept. Never generated back. |
|
|
694
775
|
|
|
@@ -888,7 +969,7 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
888
969
|
| `includeRawContent` | `boolean` | `false` | Attach raw XML/RTF source to each node |
|
|
889
970
|
| `serializeRawContent` | `boolean` | `true` | Re-serialize XML to clean strings (only if `includeRawContent: true`) |
|
|
890
971
|
| `preserveXmlWhitespace` | `boolean` | `false` | Preserve original XML whitespace during serialization |
|
|
891
|
-
| `includeBreakNodes` | `boolean` | `false` | Include `w:br
|
|
972
|
+
| `includeBreakNodes` | `boolean` | `false` | Include typed break nodes: DOCX `w:br`/`w:cr`, ODF `fo:break-before`/`fo:break-after` and `text:soft-page-break` |
|
|
892
973
|
| `ignoreInternalLinks` | `boolean` | `false` | Strip bookmarks and internal cross-references from AST |
|
|
893
974
|
| `fileType` | `SupportedFileType \| null` | `null` | **Required for text-based binary data** (`'md'`, `'html'`, `'csv'`) as these lack magic bytes. |
|
|
894
975
|
| `csvDelimiter` | `string` | `','` | Input delimiter when parsing CSV files |
|
package/dist/OfficeParser.js
CHANGED
|
@@ -55,6 +55,32 @@ const envUtils_js_1 = require("./utils/envUtils.js");
|
|
|
55
55
|
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
56
56
|
const moduleLoader_js_1 = require("./utils/moduleLoader.js");
|
|
57
57
|
const ocrUtils_js_1 = require("./utils/ocrUtils.js");
|
|
58
|
+
const zipUtils_js_1 = require("./utils/zipUtils.js");
|
|
59
|
+
/** What magic-byte sniffing reports for an archive it could not identify further. */
|
|
60
|
+
const GENERIC_ZIP_EXTENSION = 'zip';
|
|
61
|
+
/** The formats that are ZIP archives, and so cannot be contradicted by a bare `zip` result. */
|
|
62
|
+
const ZIP_BACKED_FILE_TYPES = new Set(['docx', 'xlsx', 'pptx', 'odt', 'ods', 'odp', 'epub']);
|
|
63
|
+
/**
|
|
64
|
+
* Upgrades a magic-byte result of `zip` (or none at all) into the specific office format the
|
|
65
|
+
* archive declares, by reading that declaration from inside the archive.
|
|
66
|
+
*
|
|
67
|
+
* Byte sniffing identifies an OOXML package by parsing `[Content_Types].xml`, but it walks the
|
|
68
|
+
* archive under fixed budgets and reports a plain `zip` when it runs out before finding that
|
|
69
|
+
* part. Since `zip` is not a format this library parses, a valid document then failed as an
|
|
70
|
+
* unsupported file type. Our own reader has no such budget, so it settles the question whenever
|
|
71
|
+
* sniffing is inconclusive.
|
|
72
|
+
*
|
|
73
|
+
* @param detected - What magic-byte sniffing reported, if anything
|
|
74
|
+
* @param buffer - The file content
|
|
75
|
+
* @param config - Resolved parser configuration, for its decompression limits
|
|
76
|
+
* @returns The resolved type, the original detection when nothing better is found, or undefined
|
|
77
|
+
*/
|
|
78
|
+
const resolveZipBackedType = async (detected, buffer, config) => {
|
|
79
|
+
if (detected && detected !== GENERIC_ZIP_EXTENSION)
|
|
80
|
+
return detected;
|
|
81
|
+
const resolved = await (0, zipUtils_js_1.detectOfficeTypeFromZip)(buffer, config.decompressionLimits ?? {});
|
|
82
|
+
return resolved ?? detected;
|
|
83
|
+
};
|
|
58
84
|
/**
|
|
59
85
|
* Main parser class providing office document parsing functionality.
|
|
60
86
|
*
|
|
@@ -168,15 +194,16 @@ class OfficeParser {
|
|
|
168
194
|
// This matches v6 behavior and prevents crashes in older Node environments
|
|
169
195
|
// where file-type 22.x might be incompatible.
|
|
170
196
|
if (buffer.length > 0 && !ext) {
|
|
197
|
+
let detected;
|
|
171
198
|
try {
|
|
172
199
|
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
173
200
|
const type = await fileTypeFromBuffer(buffer);
|
|
174
201
|
if (type) {
|
|
175
|
-
|
|
202
|
+
detected = type.ext;
|
|
176
203
|
}
|
|
177
204
|
else {
|
|
178
205
|
// If no extension could be detected and none was provided,
|
|
179
|
-
// it might be a text-based format (csv, md, html) which
|
|
206
|
+
// it might be a text-based format (csv, md, html) which
|
|
180
207
|
// lack magic bytes. We'll let the switch default handle it.
|
|
181
208
|
}
|
|
182
209
|
}
|
|
@@ -184,16 +211,28 @@ class OfficeParser {
|
|
|
184
211
|
// Log warning but don't crash; the switch below will handle unsupported/missing ext
|
|
185
212
|
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
|
|
186
213
|
}
|
|
214
|
+
ext = await resolveZipBackedType(detected, buffer, internalConfig) ?? '';
|
|
187
215
|
}
|
|
188
216
|
else if (buffer.length > 0 && ext) {
|
|
189
|
-
// If extension is known, we can optionally verify it, but we wrap it
|
|
217
|
+
// If extension is known, we can optionally verify it, but we wrap it
|
|
190
218
|
// in a try-catch to avoid breaking Node 18 if file-type fails to load.
|
|
191
219
|
try {
|
|
192
220
|
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
193
221
|
const type = await fileTypeFromBuffer(buffer);
|
|
194
|
-
|
|
222
|
+
// A bare `zip` cannot contradict a caller who already said "this is a
|
|
223
|
+
// docx", so there is nothing a closer look could add. Skipping it keeps an
|
|
224
|
+
// explicit fileType the cheapest route, rather than making it pay for an
|
|
225
|
+
// archive scan that exists only to decide whether to warn.
|
|
226
|
+
const worthResolving = !(type?.ext === GENERIC_ZIP_EXTENSION && ZIP_BACKED_FILE_TYPES.has(ext.toLowerCase()));
|
|
227
|
+
const detected = worthResolving
|
|
228
|
+
? await resolveZipBackedType(type?.ext, buffer, internalConfig)
|
|
229
|
+
: type?.ext;
|
|
230
|
+
// A bare `zip` says only that the bytes are an archive, which every format
|
|
231
|
+
// on this path already is. Reporting it as a mismatch against the caller's
|
|
232
|
+
// own extension is noise, so only a resolved format is worth comparing.
|
|
233
|
+
if (detected && detected !== GENERIC_ZIP_EXTENSION && detected.toLowerCase() !== ext.toLowerCase()) {
|
|
195
234
|
// Mismatch found between authoritative extension and detected content
|
|
196
|
-
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected
|
|
235
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected, expected: ext });
|
|
197
236
|
}
|
|
198
237
|
}
|
|
199
238
|
catch (error) {
|
|
@@ -218,7 +257,16 @@ class OfficeParser {
|
|
|
218
257
|
case 'odt':
|
|
219
258
|
case 'odp':
|
|
220
259
|
case 'ods':
|
|
221
|
-
|
|
260
|
+
// The three ODF types share one parser, which needs to know which of them
|
|
261
|
+
// it is looking at. It normally reads that from the archive's mimetype
|
|
262
|
+
// entry; passing the resolved type along gives it something accurate to
|
|
263
|
+
// fall back on when that entry is missing.
|
|
264
|
+
//
|
|
265
|
+
// Overridden on a copy rather than on internalConfig: resolveParserConfig
|
|
266
|
+
// returns an already-complete config by reference, so writing to it would
|
|
267
|
+
// pin the caller's own object to this file's type and misroute every later
|
|
268
|
+
// parse that reused it.
|
|
269
|
+
result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, { ...internalConfig, fileType: ext.toLowerCase() });
|
|
222
270
|
break;
|
|
223
271
|
case 'pdf':
|
|
224
272
|
result = await (0, PdfParser_js_1.parsePdf)(buffer, internalConfig);
|
|
@@ -81,6 +81,21 @@ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat =
|
|
|
81
81
|
* hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
|
|
82
82
|
*/
|
|
83
83
|
protected getFootnoteKey(note: OfficeContentNode): string;
|
|
84
|
+
/**
|
|
85
|
+
* True when every content-bearing text descendant satisfies `test` - i.e. the property is
|
|
86
|
+
* uniform across the whole node and therefore says nothing the node type does not already say.
|
|
87
|
+
*
|
|
88
|
+
* Used to decide whether a heading's or header row's inherited formatting can be dropped. The
|
|
89
|
+
* distinction matters: an ODF heading whose paragraph style is bold and 14pt yields a heading
|
|
90
|
+
* where *every* run is bold and 14pt, and re-emitting that gives `# **Heading**` in Markdown
|
|
91
|
+
* and, worse in RTF/HTML, an inner font-size that overrides the heading's own and visibly
|
|
92
|
+
* shrinks it. But `# Normal **Bold** Normal` is an author contrasting one word against the
|
|
93
|
+
* rest, and dropping that would discard real meaning. Only the uniform case is safe.
|
|
94
|
+
*
|
|
95
|
+
* Returns false when there is no text to judge, so an empty or image-only node never triggers
|
|
96
|
+
* suppression.
|
|
97
|
+
*/
|
|
98
|
+
protected hasUniformFormatting(node: OfficeContentNode, test: (formatting: OfficeContentNode['formatting']) => boolean): boolean;
|
|
84
99
|
/**
|
|
85
100
|
* Recursively extracts plain text from a node and its children.
|
|
86
101
|
*/
|
|
@@ -187,6 +187,37 @@ class BaseGenerator {
|
|
|
187
187
|
this.noteFootnoteKeys.set(note, key);
|
|
188
188
|
return key;
|
|
189
189
|
}
|
|
190
|
+
/**
|
|
191
|
+
* True when every content-bearing text descendant satisfies `test` - i.e. the property is
|
|
192
|
+
* uniform across the whole node and therefore says nothing the node type does not already say.
|
|
193
|
+
*
|
|
194
|
+
* Used to decide whether a heading's or header row's inherited formatting can be dropped. The
|
|
195
|
+
* distinction matters: an ODF heading whose paragraph style is bold and 14pt yields a heading
|
|
196
|
+
* where *every* run is bold and 14pt, and re-emitting that gives `# **Heading**` in Markdown
|
|
197
|
+
* and, worse in RTF/HTML, an inner font-size that overrides the heading's own and visibly
|
|
198
|
+
* shrinks it. But `# Normal **Bold** Normal` is an author contrasting one word against the
|
|
199
|
+
* rest, and dropping that would discard real meaning. Only the uniform case is safe.
|
|
200
|
+
*
|
|
201
|
+
* Returns false when there is no text to judge, so an empty or image-only node never triggers
|
|
202
|
+
* suppression.
|
|
203
|
+
*/
|
|
204
|
+
hasUniformFormatting(node, test) {
|
|
205
|
+
let sawText = false;
|
|
206
|
+
const walk = (n) => {
|
|
207
|
+
if (n.type === 'text') {
|
|
208
|
+
// Whitespace-only runs carry no visible formatting either way, so they neither
|
|
209
|
+
// count as evidence nor veto - otherwise a stray unformatted space between two
|
|
210
|
+
// bold runs would defeat the check on almost every real heading.
|
|
211
|
+
if (!(n.text || '').trim())
|
|
212
|
+
return true;
|
|
213
|
+
sawText = true;
|
|
214
|
+
return test(n.formatting);
|
|
215
|
+
}
|
|
216
|
+
return (n.children ?? []).every(walk);
|
|
217
|
+
};
|
|
218
|
+
const uniform = (node.children ?? []).every(walk);
|
|
219
|
+
return sawText && uniform;
|
|
220
|
+
}
|
|
190
221
|
/**
|
|
191
222
|
* Recursively extracts plain text from a node and its children.
|
|
192
223
|
*/
|
|
@@ -6,6 +6,14 @@ import { BaseGenerator } from './BaseGenerator.js';
|
|
|
6
6
|
export declare class HtmlGenerator extends BaseGenerator<'html'> {
|
|
7
7
|
private chartCounter;
|
|
8
8
|
private isSpreadsheetMode;
|
|
9
|
+
/**
|
|
10
|
+
* Set while rendering a heading's children, so `formatText` can drop the run-level bold and
|
|
11
|
+
* font-size the `<hN>` already establishes. See the note there, and the identical flag in
|
|
12
|
+
* `RtfGenerator`, where the same inherited size actively shrinks the heading.
|
|
13
|
+
*/
|
|
14
|
+
private inHeading;
|
|
15
|
+
/** As `inHeading`, but for the inherited font size - see `hasUniformFormatting`. */
|
|
16
|
+
private headingUniformSize;
|
|
9
17
|
constructor(ast: OfficeParserAST, config?: GeneratorConfig<'html'>);
|
|
10
18
|
/**
|
|
11
19
|
* Generates HTML string from the provided AST.
|
|
@@ -24,6 +32,7 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
|
|
|
24
32
|
*/
|
|
25
33
|
private tableNestingLevel;
|
|
26
34
|
protected processNodeRecursive(node: OfficeContentNode, processor: (node: OfficeContentNode, childrenOutput: string) => string | Promise<string>, override?: string | boolean | void): Promise<string>;
|
|
35
|
+
private processNodeRecursiveInner;
|
|
27
36
|
/**
|
|
28
37
|
* Internal processor for individual nodes.
|
|
29
38
|
*/
|
|
@@ -88,6 +88,14 @@ function resolveStandalone(standalone) {
|
|
|
88
88
|
class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
89
89
|
chartCounter = 0;
|
|
90
90
|
isSpreadsheetMode = false;
|
|
91
|
+
/**
|
|
92
|
+
* Set while rendering a heading's children, so `formatText` can drop the run-level bold and
|
|
93
|
+
* font-size the `<hN>` already establishes. See the note there, and the identical flag in
|
|
94
|
+
* `RtfGenerator`, where the same inherited size actively shrinks the heading.
|
|
95
|
+
*/
|
|
96
|
+
inHeading = false;
|
|
97
|
+
/** As `inHeading`, but for the inherited font size - see `hasUniformFormatting`. */
|
|
98
|
+
headingUniformSize = false;
|
|
91
99
|
constructor(ast, config) {
|
|
92
100
|
super('html', ast, config);
|
|
93
101
|
}
|
|
@@ -502,6 +510,21 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
502
510
|
// method entirely, so without repeating the check here the signal would be silently
|
|
503
511
|
// inert for this generator - which is exactly how it was missed.
|
|
504
512
|
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
513
|
+
const wasInHeading = this.inHeading;
|
|
514
|
+
const wasHeadingSize = this.headingUniformSize;
|
|
515
|
+
if (node.type === 'heading') {
|
|
516
|
+
this.inHeading = this.hasUniformFormatting(node, f => f?.bold === true);
|
|
517
|
+
this.headingUniformSize = this.hasUniformFormatting(node, f => !!f?.size);
|
|
518
|
+
}
|
|
519
|
+
try {
|
|
520
|
+
return await this.processNodeRecursiveInner(node, processor, override);
|
|
521
|
+
}
|
|
522
|
+
finally {
|
|
523
|
+
this.inHeading = wasInHeading;
|
|
524
|
+
this.headingUniformSize = wasHeadingSize;
|
|
525
|
+
}
|
|
526
|
+
}
|
|
527
|
+
async processNodeRecursiveInner(node, processor, override) {
|
|
505
528
|
// Use pre-evaluated override if provided, otherwise call handleOnNode
|
|
506
529
|
const actualOverride = override !== undefined ? override : await this.handleOnNode(node);
|
|
507
530
|
// Returning false skips the node and its children
|
|
@@ -1076,7 +1099,12 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1076
1099
|
let result = this.escape(text);
|
|
1077
1100
|
const f = node.formatting;
|
|
1078
1101
|
if (this.config.includeFormatting && f) {
|
|
1079
|
-
|
|
1102
|
+
// Inside an `<hN>`, the heading's own styling is authoritative. A run that also carries
|
|
1103
|
+
// bold and a font size - the normal case for ODF, where a heading's paragraph style is
|
|
1104
|
+
// inherited by its runs - would wrap the text in `<b>` the heading already implies and,
|
|
1105
|
+
// worse, in a `<span style="font-size: 14pt">` that *shrinks* the heading to the size
|
|
1106
|
+
// its paragraph style happened to name. See the same suppression in RtfGenerator.
|
|
1107
|
+
if (f.bold && !this.inHeading)
|
|
1080
1108
|
result = `<b>${result}</b>`;
|
|
1081
1109
|
if (f.italic)
|
|
1082
1110
|
result = `<i>${result}</i>`;
|
|
@@ -1088,7 +1116,9 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1088
1116
|
result = `<sub>${result}</sub>`;
|
|
1089
1117
|
if (f.superscript)
|
|
1090
1118
|
result = `<sup>${result}</sup>`;
|
|
1091
|
-
const styles = this.
|
|
1119
|
+
const styles = this.headingUniformSize
|
|
1120
|
+
? this.getInlineStyles(node, { skipFontSize: true })
|
|
1121
|
+
: this.getInlineStyles(node);
|
|
1092
1122
|
if (styles) {
|
|
1093
1123
|
result = `<span style="${styles}">${result}</span>`;
|
|
1094
1124
|
}
|
|
@@ -1120,7 +1150,7 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1120
1150
|
}
|
|
1121
1151
|
return result;
|
|
1122
1152
|
}
|
|
1123
|
-
getInlineStyles(node) {
|
|
1153
|
+
getInlineStyles(node, options = {}) {
|
|
1124
1154
|
const styles = [];
|
|
1125
1155
|
// Colors/sizes/fonts/alignments are free strings from an untrusted document;
|
|
1126
1156
|
// run each through sanitizeCssValue so it can't break out of the style="" attribute
|
|
@@ -1154,7 +1184,7 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1154
1184
|
pushSafe('color', f.color);
|
|
1155
1185
|
if (f.backgroundColor)
|
|
1156
1186
|
pushSafe('background-color', f.backgroundColor);
|
|
1157
|
-
if (f.size)
|
|
1187
|
+
if (f.size && !options.skipFontSize)
|
|
1158
1188
|
pushSafe('font-size', f.size);
|
|
1159
1189
|
if (f.font) {
|
|
1160
1190
|
const safeFont = (0, sanitize_js_1.sanitizeCssValue)(f.font);
|
|
@@ -31,6 +31,18 @@ import { BaseGenerator } from './BaseGenerator.js';
|
|
|
31
31
|
*/
|
|
32
32
|
export declare class MarkdownGenerator extends BaseGenerator<'md'> {
|
|
33
33
|
private isInsideTable;
|
|
34
|
+
/**
|
|
35
|
+
* Set while rendering the children of a heading, or the cells of a table's header row.
|
|
36
|
+
*
|
|
37
|
+
* Markdown already conveys "this is a heading" with `#` and "this is a header row" with the
|
|
38
|
+
* separator line, so a run inside one that also carries bold - the normal case for ODF, whose
|
|
39
|
+
* heading and header-row paragraph styles are bold and are now inherited by their runs - would
|
|
40
|
+
* render as `# **Heading**` and `| **ITEM** |`. That is redundant rather than wrong, but it
|
|
41
|
+
* also round-trips back into bold text nodes nested inside a heading, so the noise compounds
|
|
42
|
+
* on every parse/generate cycle. Emphasis the node type already implies is dropped; every
|
|
43
|
+
* other formatting flag still comes through.
|
|
44
|
+
*/
|
|
45
|
+
private inImplicitBold;
|
|
34
46
|
private hoistedContent;
|
|
35
47
|
private collectedAbbreviations;
|
|
36
48
|
private resolvedDialect;
|
|
@@ -72,6 +84,7 @@ export declare class MarkdownGenerator extends BaseGenerator<'md'> {
|
|
|
72
84
|
private optimizeNodes;
|
|
73
85
|
private areFormattingEqual;
|
|
74
86
|
private renderMarkdownTable;
|
|
87
|
+
private collectNotesFrom;
|
|
75
88
|
private renderMarkdownTableInternal;
|
|
76
89
|
private hasNestedTable;
|
|
77
90
|
private hasColspanOrRowspan;
|