officeparser 7.5.1 → 7.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -6
- package/dist/OfficeConverter.d.ts +2 -2
- package/dist/OfficeParser.d.ts +2 -2
- package/dist/OfficeParser.js +10 -0
- package/dist/cli.js +3 -0
- package/dist/defaults.js +3 -1
- package/dist/generators/ChunkingGenerator.d.ts +2 -1
- package/dist/generators/ChunkingGenerator.js +73 -6
- package/dist/generators/EpubGenerator.js +4 -1
- package/dist/generators/HtmlGenerator.js +117 -29
- package/dist/generators/MarkdownGenerator.js +120 -29
- package/dist/generators/PdfGenerator.js +4 -1
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +70 -12
- package/dist/officeparser.browser.iife.js +211 -202
- package/dist/officeparser.browser.mjs +211 -202
- package/dist/officeparser.browser.slim.d.ts +70 -12
- package/dist/officeparser.browser.slim.iife.js +213 -204
- package/dist/officeparser.browser.slim.mjs +213 -204
- package/dist/parsers/HtmlParser.js +206 -38
- package/dist/parsers/MarkdownParser.js +135 -18
- package/dist/parsers/WordParser.js +7 -17
- package/dist/sbom.cdx.json +95 -95
- package/dist/types.d.ts +68 -10
- package/dist/utils/configUtils.js +3 -1
- package/dist/utils/sanitize.d.ts +9 -0
- package/dist/utils/sanitize.js +26 -0
- package/package.json +5 -5
package/README.md
CHANGED
|
@@ -188,9 +188,9 @@ officeParser.parseOffice('/path/to/file.docx', function(ast, err) {
|
|
|
188
188
|
});
|
|
189
189
|
```
|
|
190
190
|
|
|
191
|
-
### File Buffers &
|
|
191
|
+
### File Buffers, ArrayBuffers & Blobs
|
|
192
192
|
|
|
193
|
-
Pass a `Buffer`, `ArrayBuffer`, or `
|
|
193
|
+
Pass a `Buffer`, `ArrayBuffer`, `Uint8Array`, or a web `Blob`/`File` instead of a file path:
|
|
194
194
|
|
|
195
195
|
```js
|
|
196
196
|
const fs = require('fs');
|
|
@@ -198,6 +198,15 @@ const buffer = fs.readFileSync('/path/to/file.pdf');
|
|
|
198
198
|
const ast = await officeParser.parseOffice(buffer);
|
|
199
199
|
```
|
|
200
200
|
|
|
201
|
+
In the browser you can hand a `File`/`Blob` straight from an `<input type="file">` — no need to
|
|
202
|
+
read it into a buffer first. A `File`'s name drives type detection, so no `fileType` hint is
|
|
203
|
+
needed when the name has a recognizable extension:
|
|
204
|
+
|
|
205
|
+
```js
|
|
206
|
+
// input.files[0] is a File (e.g. "report.docx")
|
|
207
|
+
const ast = await officeParser.parseOffice(input.files[0]);
|
|
208
|
+
```
|
|
209
|
+
|
|
201
210
|
> [!IMPORTANT]
|
|
202
211
|
> **Text-based formats from buffers need a `fileType` hint.**
|
|
203
212
|
> Formats like `md`, `html`, and `csv` have no magic bytes, so the parser cannot
|
|
@@ -475,6 +484,8 @@ const { value: chunks } = await OfficeConverter.convert('report.docx', 'chunks',
|
|
|
475
484
|
|
|
476
485
|
### The `OfficeChunk` Object
|
|
477
486
|
|
|
487
|
+
`generate(ast, 'chunks')` (and `ast.to('chunks')`) resolves to a real `OfficeChunk[]` **array**, not a JSON string - serialize it to JSON/JSONL yourself if your pipeline needs that.
|
|
488
|
+
|
|
478
489
|
Every chunk contains text and rich metadata for citations and filtered retrieval:
|
|
479
490
|
|
|
480
491
|
```ts
|
|
@@ -738,7 +749,7 @@ Admonition Node (type: 'admonition')
|
|
|
738
749
|
└── children: [ Paragraph | List | ... ] (block content)
|
|
739
750
|
|
|
740
751
|
Embed Node (type: 'embed')
|
|
741
|
-
└── metadata: { embedType: 'youtube', videoId
|
|
752
|
+
└── metadata: { embedType: 'youtube' | 'iframe', videoId?: string, url?: string, width?: string, height?: string, align?: string }
|
|
742
753
|
|
|
743
754
|
Definition List Node (type: 'definitionList')
|
|
744
755
|
└── children:
|
|
@@ -973,7 +984,8 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
973
984
|
| `ignoreInternalLinks` | `boolean` | `false` | Strip bookmarks and internal cross-references from AST |
|
|
974
985
|
| `fileType` | `SupportedFileType \| null` | `null` | **Required for text-based binary data** (`'md'`, `'html'`, `'csv'`) as these lack magic bytes. |
|
|
975
986
|
| `csvDelimiter` | `string` | `','` | Input delimiter when parsing CSV files |
|
|
976
|
-
| `decompressionLimits` | `DecompressionLimits` | `{ maxUncompressedBytes: 512MB, maxZipEntries: 10000 }` | **New**: Limits applied during ZIP extraction to protect against excessive memory and resource usage |
|
|
987
|
+
| `decompressionLimits` | `DecompressionLimits` | `{ maxUncompressedBytes: 512MB, maxZipEntries: 10000, maxTableCells: 1000000 }` | **New**: Limits applied during ZIP extraction (and ODF cell expansion) to protect against excessive memory and resource usage |
|
|
988
|
+
| `htmlParserConfig` | `HtmlParserConfig` | `{}` | HTML/XHTML/EPUB parsing options. `preserveAttributes` (`boolean`, default `false`): keep generic source attributes no typed field consumed on `node.htmlAttributes`. `preserveIframes` (`boolean \| string[]`, default `false`): preserve non-YouTube `<iframe>` embeds (otherwise dropped) as `embed` nodes — `true` for any, or a hostname allowlist; the src is scheme-checked on generation |
|
|
977
989
|
| `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
|
|
978
990
|
| `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
|
|
979
991
|
| `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
|
|
@@ -988,7 +1000,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
|
|
|
988
1000
|
| Option | Type | Default | Description |
|
|
989
1001
|
|--------|------|---------|-------------|
|
|
990
1002
|
| `includeFormatting` | `boolean` | `true` | Include bold/italic/colors/sizes in output |
|
|
991
|
-
| `generateIds` | `boolean` | `true` |
|
|
1003
|
+
| `generateIds` | `boolean` | `true` | Slug-based heading anchors: `id` attributes on HTML headings, and a `{#slug}` suffix on Markdown headings (`# Title {#title}`, kramdown/Pandoc). Set `false` to omit both — useful when the Markdown is rendered by GFM/CommonMark, which show `{#slug}` as literal text. Applies to all generator formats (it is a top-level option, not under `mdConfig`/`htmlConfig`). |
|
|
992
1004
|
| `renderMetadata` | `boolean` | `false` | Render title/author as visible header block |
|
|
993
1005
|
| `metadataOverrides` | `MetadataOverrides` | `{}` | Override the metadata embedded in the output, merged per field over `ast.metadata` |
|
|
994
1006
|
| `includeImages` | `boolean` | `true` | Include image nodes in output |
|
|
@@ -1083,6 +1095,7 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
1083
1095
|
| `injections.headEnd` | `string` | `''` | Raw HTML injected before `</head>` |
|
|
1084
1096
|
| `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
|
|
1085
1097
|
| `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
|
|
1098
|
+
| `sourceAttributes` | `boolean` | `false` | Carry each rich node's raw source in a `data-*` attribute (undelimited text), so attribute-driven consumers can rehydrate it: `data-wikilink`/`data-target`/`data-alias` on wikilinks, a `<span class="citation" data-key>` for citations, the LaTeX in `data-math`, and a `<div class="mermaid" data-mermaid>` for mermaid. Off = byte-identical to before; the parser reads every shape it emits. Forced off for PDF/EPUB |
|
|
1086
1099
|
|
|
1087
1100
|
#### `standalone`: granular envelope control
|
|
1088
1101
|
|
|
@@ -1133,7 +1146,7 @@ Pass as `mdConfig` inside `GeneratorConfig`.
|
|
|
1133
1146
|
|
|
1134
1147
|
| Option | Type | Default | Description |
|
|
1135
1148
|
|--------|------|---------|-------------|
|
|
1136
|
-
| `fallbackToHtml` | `boolean` | `true` | Use HTML tags for features Markdown cannot represent (underlines, merged table cells, etc.) |
|
|
1149
|
+
| `fallbackToHtml` | `boolean \| FallbackToHtmlConfig` | `true` | Use HTML tags for features Markdown cannot represent (underlines, merged table cells, embeds, etc.). Pass an object for per-feature control. `inlineFormatting` (default `false`, opt-in even when the boolean is `true`) additionally round-trips inline color/highlight/font-size as `<span style="...">` runs. |
|
|
1137
1150
|
|
|
1138
1151
|
### PdfGeneratorConfig
|
|
1139
1152
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
|
|
1
|
+
import { BlobLike, ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
|
|
2
2
|
/**
|
|
3
3
|
* Utility type to infer the file type from a file path string literal.
|
|
4
4
|
*/
|
|
@@ -42,6 +42,6 @@ export declare class OfficeConverter {
|
|
|
42
42
|
* });
|
|
43
43
|
* ```
|
|
44
44
|
*/
|
|
45
|
-
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
|
|
45
|
+
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array | BlobLike, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
|
|
46
46
|
}
|
|
47
47
|
export {};
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -36,7 +36,7 @@
|
|
|
36
36
|
*
|
|
37
37
|
* @module OfficeParser
|
|
38
38
|
*/
|
|
39
|
-
import { OfficeParserAST, OfficeParserConfig } from './types.js';
|
|
39
|
+
import { BlobLike, OfficeParserAST, OfficeParserConfig } from './types.js';
|
|
40
40
|
/**
|
|
41
41
|
* Main parser class providing office document parsing functionality.
|
|
42
42
|
*
|
|
@@ -93,7 +93,7 @@ export declare class OfficeParser {
|
|
|
93
93
|
* const text = ast.toText();
|
|
94
94
|
* ```
|
|
95
95
|
*/
|
|
96
|
-
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
96
|
+
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array | BlobLike, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
97
97
|
/**
|
|
98
98
|
* Terminates all active OCR workers and cleans up resources.
|
|
99
99
|
*
|
package/dist/OfficeParser.js
CHANGED
|
@@ -187,6 +187,16 @@ class OfficeParser {
|
|
|
187
187
|
buffer = fs.readFileSync(file);
|
|
188
188
|
ext = ext || file.split('.').pop() || '';
|
|
189
189
|
}
|
|
190
|
+
else if (file && typeof file.arrayBuffer === 'function') {
|
|
191
|
+
// Web Blob/File (or any BlobLike). Read its bytes; if it carries a filename, use
|
|
192
|
+
// the extension for type detection - never as a filesystem path. A nameless blob
|
|
193
|
+
// still resolves through the magic-byte sniffing below.
|
|
194
|
+
buffer = Buffer.from(await file.arrayBuffer());
|
|
195
|
+
const name = file.name;
|
|
196
|
+
if (!ext && typeof name === 'string' && name.includes('.')) {
|
|
197
|
+
ext = name.split('.').pop() || '';
|
|
198
|
+
}
|
|
199
|
+
}
|
|
190
200
|
else {
|
|
191
201
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
|
|
192
202
|
}
|
package/dist/cli.js
CHANGED
|
@@ -351,17 +351,20 @@ else {
|
|
|
351
351
|
console.log(' --verbose Show full error stack traces and warning logs');
|
|
352
352
|
console.log(' --newlineDelimiter=string Delimiter string between blocks/lines (default: \\n)');
|
|
353
353
|
console.log(' --csvDelimiter=char Custom CSV delimiter (default: ,)');
|
|
354
|
+
console.log(' --htmlParserConfig.preserveIframes Keep non-YouTube <iframe> embeds (dropped by default)');
|
|
354
355
|
console.log('');
|
|
355
356
|
console.log('High-Value Generator Options:');
|
|
356
357
|
console.log(' --includeFormatting Include font formatting like bold/italic (default: true)');
|
|
357
358
|
console.log(' --renderMetadata Render metadata in output content (default: false)');
|
|
358
359
|
console.log(' --htmlConfig.containerWidth=value HTML container width (auto | px | % | vw etc., default: auto)');
|
|
360
|
+
console.log(' --htmlConfig.sourceAttributes Carry each rich node\'s source in a data-* attribute (default: false)');
|
|
359
361
|
console.log('');
|
|
360
362
|
console.log('Advanced Nested Config Examples:');
|
|
361
363
|
console.log(' --pdfConfig.format=Letter Configure Puppeteer PDF format (A4 | Letter | Legal etc.)');
|
|
362
364
|
console.log(' --chunksConfig.strategy=fixed-size Chunking strategy (fixed-size | document-structure | semantic)');
|
|
363
365
|
console.log(' --mdConfig.dialect=github Markdown dialect (extended | github | gitlab | obsidian | pandoc | commonmark)');
|
|
364
366
|
console.log(' --mdConfig.fallbackToHtml=false Disable HTML fallback for unsupported Markdown features (default: true)');
|
|
367
|
+
console.log(' --mdConfig.fallbackToHtml.inlineFormatting Round-trip inline color/highlight/font-size as <span style> (opt-in, default: false)');
|
|
365
368
|
console.log('');
|
|
366
369
|
console.log('Format Syntax:');
|
|
367
370
|
console.log(' Flags can be written as --flag (presence implies true), --no-flag (negation),');
|
package/dist/defaults.js
CHANGED
|
@@ -40,6 +40,7 @@ const DEFAULT_OCR_CONFIG = {
|
|
|
40
40
|
*/
|
|
41
41
|
const DEFAULT_HTML_PARSER_CONFIG = {
|
|
42
42
|
preserveAttributes: false,
|
|
43
|
+
preserveIframes: false,
|
|
43
44
|
};
|
|
44
45
|
/**
|
|
45
46
|
* Default configuration for the OfficeParser.
|
|
@@ -86,7 +87,8 @@ const DEFAULT_HTML_GENERATOR_CONFIG = {
|
|
|
86
87
|
headEnd: '',
|
|
87
88
|
bodyStart: '',
|
|
88
89
|
bodyEnd: '',
|
|
89
|
-
}
|
|
90
|
+
},
|
|
91
|
+
sourceAttributes: false,
|
|
90
92
|
};
|
|
91
93
|
/**
|
|
92
94
|
* Default configuration for PDF generation.
|
|
@@ -16,7 +16,8 @@ export declare class ChunkingGenerator extends BaseGenerator<'chunks'> {
|
|
|
16
16
|
private resolveChunkingConfig;
|
|
17
17
|
/**
|
|
18
18
|
* Main entry point. Routes to the correct strategy implementation.
|
|
19
|
-
* Note: ConversionResult.value is a
|
|
19
|
+
* Note: ConversionResult.value is a real OfficeChunk[] array for the 'chunks' destination, not
|
|
20
|
+
* a JSON string. Consumers serialize it to JSON/JSONL themselves.
|
|
20
21
|
*/
|
|
21
22
|
generate(): Promise<ConversionResult<'chunks'>>;
|
|
22
23
|
/**
|
|
@@ -5,6 +5,72 @@ const defaults_js_1 = require("../defaults.js");
|
|
|
5
5
|
const types_js_1 = require("../types.js");
|
|
6
6
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
7
7
|
const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
8
|
+
/** Node types whose text is block-level, so a boundary between two of them is a real break. */
|
|
9
|
+
const BLOCK_NODE_TYPES = new Set([
|
|
10
|
+
'paragraph', 'heading', 'list', 'table', 'row', 'cell', 'code', 'note', 'admonition',
|
|
11
|
+
'definitionList', 'definitionTerm', 'definitionDescription', 'sheet', 'slide', 'page', 'embed',
|
|
12
|
+
]);
|
|
13
|
+
/**
|
|
14
|
+
* The visible text of a content node: its own `.text` when set, otherwise its descendants' text,
|
|
15
|
+
* plus any footnote/endnote bodies hanging off `node.notes`.
|
|
16
|
+
* HTML- and Markdown-origin parsers build paragraphs as `{ children: [...] }` with no `.text`, so
|
|
17
|
+
* reading `node.text` alone dropped their content from every chunk. Block-level children are joined
|
|
18
|
+
* with a newline so words don't merge across paragraphs/list items (matching the text generator);
|
|
19
|
+
* inline runs join with no separator. Notes live on `node.notes`, off the children tree, and a
|
|
20
|
+
* content node is emitted as a chunk without recursing into them - so their text was silently
|
|
21
|
+
* absent from the RAG index. Fold them in here (joined as block content) so a footnote's body is
|
|
22
|
+
* searchable alongside the paragraph that references it.
|
|
23
|
+
*/
|
|
24
|
+
function collectNodeText(node) {
|
|
25
|
+
let out = '';
|
|
26
|
+
if (typeof node.text === 'string' && node.text.length > 0) {
|
|
27
|
+
out = node.text;
|
|
28
|
+
// The `.text` fast-path above skips the children walk, but DOCX/ODT/RTF set `.text` on the
|
|
29
|
+
// paragraph while the footnote hangs off a nested text child - so its body would be missed.
|
|
30
|
+
// Fold in descendant note bodies (visible text already covered by `.text`, not re-added).
|
|
31
|
+
const descendantNotes = collectDescendantNoteText(node);
|
|
32
|
+
if (descendantNotes)
|
|
33
|
+
out += '\n' + descendantNotes;
|
|
34
|
+
}
|
|
35
|
+
else if (node.children && node.children.length > 0) {
|
|
36
|
+
for (const child of node.children) {
|
|
37
|
+
if (out && BLOCK_NODE_TYPES.has(child.type))
|
|
38
|
+
out += '\n';
|
|
39
|
+
out += collectNodeText(child);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
if (node.notes && node.notes.length > 0) {
|
|
43
|
+
for (const note of node.notes) {
|
|
44
|
+
const noteText = collectNodeText(note);
|
|
45
|
+
if (noteText)
|
|
46
|
+
out += (out ? '\n' : '') + noteText;
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
return out;
|
|
50
|
+
}
|
|
51
|
+
/**
|
|
52
|
+
* Footnote/endnote bodies hanging off a node's descendants, without the descendants' own visible
|
|
53
|
+
* text (the caller already has that via `.text`). Only reached from the `.text` fast-path above, to
|
|
54
|
+
* recover notes that office-origin parsers attach to a nested text child of a `.text`-bearing node.
|
|
55
|
+
*/
|
|
56
|
+
function collectDescendantNoteText(node) {
|
|
57
|
+
if (!node.children || node.children.length === 0)
|
|
58
|
+
return '';
|
|
59
|
+
let out = '';
|
|
60
|
+
for (const child of node.children) {
|
|
61
|
+
if (child.notes) {
|
|
62
|
+
for (const note of child.notes) {
|
|
63
|
+
const t = collectNodeText(note);
|
|
64
|
+
if (t)
|
|
65
|
+
out += (out ? '\n' : '') + t;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
const deeper = collectDescendantNoteText(child);
|
|
69
|
+
if (deeper)
|
|
70
|
+
out += (out ? '\n' : '') + deeper;
|
|
71
|
+
}
|
|
72
|
+
return out;
|
|
73
|
+
}
|
|
8
74
|
/**
|
|
9
75
|
* Generates a list of OfficeChunk objects from an AST for use in RAG pipelines.
|
|
10
76
|
* Supports three strategies: 'fixed-size', 'document-structure', and 'semantic'.
|
|
@@ -36,7 +102,8 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
36
102
|
}
|
|
37
103
|
/**
|
|
38
104
|
* Main entry point. Routes to the correct strategy implementation.
|
|
39
|
-
* Note: ConversionResult.value is a
|
|
105
|
+
* Note: ConversionResult.value is a real OfficeChunk[] array for the 'chunks' destination, not
|
|
106
|
+
* a JSON string. Consumers serialize it to JSON/JSONL themselves.
|
|
40
107
|
*/
|
|
41
108
|
async generate() {
|
|
42
109
|
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
@@ -241,7 +308,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
241
308
|
}
|
|
242
309
|
const isContentNode = node.type === 'paragraph' || node.type === 'heading' || node.type === 'list' || node.type === 'code' || node.type === 'cell' || (node.text && (!node.children || node.children.length === 0));
|
|
243
310
|
if (isStructuralBoundary || isContentNode) {
|
|
244
|
-
const text = typeof override === 'string' ? override : (node
|
|
311
|
+
const text = typeof override === 'string' ? override : collectNodeText(node);
|
|
245
312
|
const isWhitespaceOnly = !text.trim() && !text.includes('\u00A0');
|
|
246
313
|
if (isWhitespaceOnly && text.length > 0) {
|
|
247
314
|
// Log skipped empty nodes if debugging
|
|
@@ -302,8 +369,8 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
302
369
|
async processTableNode(node, config, maxChunkSize, measure, chunks, contextStack) {
|
|
303
370
|
const strategy = config.tableSplitStrategy;
|
|
304
371
|
if (strategy === 'flatten' || !node.children || node.children.length === 0) {
|
|
305
|
-
// Flatten: treat as plain text
|
|
306
|
-
const text = node
|
|
372
|
+
// Flatten: treat as plain text (collect from children for HTML/MD-origin tables).
|
|
373
|
+
const text = collectNodeText(node);
|
|
307
374
|
if (!text.trim())
|
|
308
375
|
return;
|
|
309
376
|
chunks.push({
|
|
@@ -514,7 +581,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
514
581
|
currentSheet = node.metadata?.sheetName;
|
|
515
582
|
const isContentNode = node.type === 'paragraph' || node.type === 'heading' || node.type === 'list' || node.type === 'cell' || (node.text && (!node.children || node.children.length === 0));
|
|
516
583
|
if (isContentNode) {
|
|
517
|
-
const text = (typeof override === 'string' ? override : (node
|
|
584
|
+
const text = (typeof override === 'string' ? override : collectNodeText(node)).trim();
|
|
518
585
|
if (!text)
|
|
519
586
|
return;
|
|
520
587
|
// Split paragraph text into individual sentences for finer-grained similarity
|
|
@@ -573,7 +640,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
573
640
|
currentSheet = node.metadata?.sheetName;
|
|
574
641
|
const isContentNode = node.type === 'paragraph' || node.type === 'heading' || node.type === 'list' || node.type === 'code' || node.type === 'cell' || (node.text && (!node.children || node.children.length === 0));
|
|
575
642
|
if (isContentNode) {
|
|
576
|
-
const nodeText = typeof override === 'string' ? override : (node
|
|
643
|
+
const nodeText = typeof override === 'string' ? override : collectNodeText(node);
|
|
577
644
|
const txt = nodeText + '\n';
|
|
578
645
|
nodeMap.push({ offset, heading: currentHeading, slideNumber: currentSlide, pageNumber: currentPage, sheetName: currentSheet });
|
|
579
646
|
parts.push(txt);
|
|
@@ -182,7 +182,10 @@ class EpubGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
182
182
|
async generate() {
|
|
183
183
|
const htmlGenerator = new HtmlGenerator_js_1.HtmlGenerator(this.ast, {
|
|
184
184
|
...this.config,
|
|
185
|
-
|
|
185
|
+
// Force sourceAttributes off: those data-* attributes are wire-format plumbing for
|
|
186
|
+
// structured consumers and change the mermaid shape's rendered appearance, neither of
|
|
187
|
+
// which belongs in a packaged EPUB.
|
|
188
|
+
htmlConfig: { ...this.config.htmlConfig, standalone: false, sourceAttributes: false },
|
|
186
189
|
});
|
|
187
190
|
const htmlResult = await htmlGenerator.generate();
|
|
188
191
|
let bodyHtml = typeof htmlResult.value === 'string' ? htmlResult.value : '';
|
|
@@ -117,14 +117,20 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
117
117
|
containerClass = 'pdf-container';
|
|
118
118
|
let bodyContent = await this.processNodeArray(this.ast.content);
|
|
119
119
|
if (this.collectedNotes.length > 0) {
|
|
120
|
+
// De-duplicate by node identity first. A table row with sparse column metadata
|
|
121
|
+
// re-processes its cells in `case 'row'` after they were already processed for
|
|
122
|
+
// `childrenOutput`, so a footnote referenced inside a cell gets pushed here twice -
|
|
123
|
+
// the same object reference both times, which a Set collapses back to one. Genuinely
|
|
124
|
+
// distinct notes (even two references to the same id) are different objects and stay.
|
|
125
|
+
const collectedNotes = [...new Set(this.collectedNotes)];
|
|
120
126
|
// Footnotes/endnotes get their own <section data-footnotes> (the agreed
|
|
121
|
-
// contract with
|
|
127
|
+
// contract with attribute-driven editors' footnote nodes); other note types (e.g.
|
|
122
128
|
// slide speaker notes) keep the existing generic notes wrapper.
|
|
123
|
-
const footnotes =
|
|
129
|
+
const footnotes = collectedNotes.filter(n => {
|
|
124
130
|
const t = n.metadata?.noteType;
|
|
125
131
|
return t === 'footnote' || t === 'endnote';
|
|
126
132
|
});
|
|
127
|
-
const otherNotes =
|
|
133
|
+
const otherNotes = collectedNotes.filter(n => !footnotes.includes(n));
|
|
128
134
|
if (footnotes.length > 0) {
|
|
129
135
|
let footnotesHtml = '';
|
|
130
136
|
for (const note of footnotes) {
|
|
@@ -460,6 +466,23 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
460
466
|
if (override === false) {
|
|
461
467
|
continue;
|
|
462
468
|
}
|
|
469
|
+
// A top-level footnote/endnote note is an orphan definition (the MarkdownParser
|
|
470
|
+
// recovers unreferenced `[^id]: ...` defs as trailing note nodes). Route it into the
|
|
471
|
+
// collected footnotes so it renders inside `<section data-footnotes>` - where HtmlParser
|
|
472
|
+
// reads it back on import - instead of inline outside the section with a dead back-link.
|
|
473
|
+
const orphanMeta = node.metadata;
|
|
474
|
+
const orphanNoteType = orphanMeta?.noteType;
|
|
475
|
+
if (node.type === 'note' && orphanMeta?.unreferenced && (orphanNoteType === 'footnote' || orphanNoteType === 'endnote')) {
|
|
476
|
+
// Only an unreferenced (orphan) definition is hoisted into <section data-footnotes>.
|
|
477
|
+
// The `unreferenced` guard also keeps the generators in agreement at depth: this
|
|
478
|
+
// routing runs in every processNodeArray call, so without it a `note` sitting as a
|
|
479
|
+
// CHILD of a container (a consumer-built AST; no shipped parser emits this) would be
|
|
480
|
+
// hoisted here and then given an `<a href="#footnote-ref-N">` back-link with no
|
|
481
|
+
// anchor - the exact dangling link the orphan handling removes. MarkdownGenerator's
|
|
482
|
+
// equivalent routing is top-level only, so gating on the flag matches it.
|
|
483
|
+
this.collectedNotes.push(node);
|
|
484
|
+
continue;
|
|
485
|
+
}
|
|
463
486
|
if (node.type === 'list') {
|
|
464
487
|
const meta = node.metadata;
|
|
465
488
|
const type = meta?.listType === 'ordered' ? 'ordered' : 'unordered';
|
|
@@ -772,19 +795,40 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
772
795
|
}
|
|
773
796
|
return '';
|
|
774
797
|
}
|
|
775
|
-
case 'break':
|
|
776
|
-
|
|
798
|
+
case 'break': {
|
|
799
|
+
const breakType = node.metadata?.breakType;
|
|
800
|
+
if (breakType === 'page')
|
|
801
|
+
return '<hr class="page-break">';
|
|
802
|
+
// A thematic break is a plain rule; the parser reads a bare <hr> back as one.
|
|
803
|
+
if (breakType === 'thematic')
|
|
804
|
+
return '<hr>';
|
|
805
|
+
return '<br>';
|
|
806
|
+
}
|
|
777
807
|
case 'code': {
|
|
778
808
|
const meta = node.metadata;
|
|
779
809
|
if (meta?.math) {
|
|
780
|
-
// No pinned editor contract yet (inscript-editor's math node is v1.2,
|
|
781
|
-
// not built) - this is the proposed shape: data-math signals the
|
|
782
|
-
// display mode, and the visible text keeps its $ delimiters so the
|
|
783
|
-
// raw LaTeX degrades gracefully without a KaTeX renderer.
|
|
784
|
-
const delimited = meta.math === 'block' ? `$$${node.text || ''}$$` : `$${node.text || ''}$`;
|
|
785
810
|
const tag = meta.math === 'block' ? 'div' : 'span';
|
|
811
|
+
if (this.config.htmlConfig.sourceAttributes) {
|
|
812
|
+
// Attribute-driven emission: the raw (undelimited) LaTeX lives in both
|
|
813
|
+
// data-math and the text content, and the class token carries the mode.
|
|
814
|
+
// The widened HtmlParser reads this back (a data-math value other than
|
|
815
|
+
// inline/block is treated as LaTeX).
|
|
816
|
+
const latex = node.text || '';
|
|
817
|
+
return `${extraAnchors}<${tag} class="math math-${this.escape(meta.math)}" data-math="${this.escape(latex)}"${idAttr}${mappedAttrs}${styleAttr}>${this.escape(latex)}</${tag}>`;
|
|
818
|
+
}
|
|
819
|
+
// Default emission: data-math names the display mode, and the visible text
|
|
820
|
+
// keeps its $ delimiters so the raw LaTeX degrades gracefully without a
|
|
821
|
+
// KaTeX renderer.
|
|
822
|
+
const delimited = meta.math === 'block' ? `$$${node.text || ''}$$` : `$${node.text || ''}$`;
|
|
786
823
|
return `${extraAnchors}<${tag} class="math math-${this.escape(meta.math)}" data-math="${this.escape(meta.math)}"${idAttr}${mappedAttrs}${styleAttr}>${this.escape(delimited)}</${tag}>`;
|
|
787
824
|
}
|
|
825
|
+
if (this.config.htmlConfig.sourceAttributes && meta?.language === 'mermaid') {
|
|
826
|
+
// Attribute-driven emission: a <div class="mermaid" data-mermaid> the widened
|
|
827
|
+
// parser maps back to a mermaid code node. escape() encodes '>' so diagram
|
|
828
|
+
// arrows (-->), plus newlines and quotes, stay inside the tag and attribute.
|
|
829
|
+
const code = node.text || '';
|
|
830
|
+
return `${extraAnchors}<div class="mermaid" data-mermaid="${this.escape(code)}"${idAttr}${mappedAttrs}${styleAttr}>${this.escape(code)}</div>`;
|
|
831
|
+
}
|
|
788
832
|
const lang = meta?.language ? ` class="language-${this.escape(meta.language)}"` : '';
|
|
789
833
|
const codeHtml = `<code${lang}>${this.escape(node.text || '')}</code>`;
|
|
790
834
|
if (node.text && node.text.includes('\n')) {
|
|
@@ -820,9 +864,13 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
820
864
|
const isHeaderStyle = firstRow.metadata?.style?.toLowerCase().includes('header');
|
|
821
865
|
const allBold = firstRowCells.length > 0 && firstRowCells.every(c => c.children?.every(child => child.formatting?.bold === true));
|
|
822
866
|
if (isHeaderStyle || allBold) {
|
|
823
|
-
// Re-process
|
|
867
|
+
// Re-process the first row as header cells, wrapped in a <tr>. Without the
|
|
868
|
+
// <tr>, the header cells sit directly under <thead> (`<thead><th>…`), which
|
|
869
|
+
// is invalid HTML that HtmlParser does not read back as a table row - so a
|
|
870
|
+
// md -> HTML -> md round trip lost the header content. `<thead><tr><th>…` is
|
|
871
|
+
// valid and self-idempotent.
|
|
824
872
|
const headOutput = await this.processNodeRecursive(firstRow, async (n, children) => {
|
|
825
|
-
return children.replace(/<td/g, '<th').replace(/<\/td>/g, '</th>')
|
|
873
|
+
return `<tr>${children.replace(/<td/g, '<th').replace(/<\/td>/g, '</th>')}</tr>`;
|
|
826
874
|
});
|
|
827
875
|
const bodyRows = rows.slice(1);
|
|
828
876
|
const bodyOutput = await this.processNodeArray(bodyRows);
|
|
@@ -1048,15 +1096,34 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1048
1096
|
const meta = node.metadata;
|
|
1049
1097
|
if (meta?.noteType === 'footnote' || meta?.noteType === 'endnote') {
|
|
1050
1098
|
const key = this.escape(this.getFootnoteKey(node));
|
|
1051
|
-
|
|
1099
|
+
// A <div> wrapper, not <p>: childrenOutput is block content, so a <p> wrapper
|
|
1100
|
+
// nests <p> inside <p> (every DOM parser splits it, leaving the wrapper empty),
|
|
1101
|
+
// and attribute-driven editors match `div[data-footnote-id]`. This changes the
|
|
1102
|
+
// default footnote-definition markup, which was broken-by-construction before.
|
|
1103
|
+
// An unreferenced (orphan) note has no citation anchor, so the back-link would
|
|
1104
|
+
// dangle - omit it.
|
|
1105
|
+
const backLink = meta?.unreferenced ? '' : ` <a href="#footnote-ref-${key}">↩</a>`;
|
|
1106
|
+
return `<div id="footnote-${key}" data-footnote-id="${key}">${childrenOutput}${backLink}</div>`;
|
|
1052
1107
|
}
|
|
1053
1108
|
const noteClass = meta?.noteType ? ` note-${this.escape(meta.noteType)}` : '';
|
|
1054
1109
|
return `${extraAnchors}<div class="slide-note${noteClass}"${idAttr}${className}${mappedAttrs}${styleAttr}>${childrenOutput}</div>`;
|
|
1055
1110
|
}
|
|
1056
1111
|
case 'embed': {
|
|
1057
|
-
// Match the Youtube extension's exact wrapper shape so a loaded embed
|
|
1058
|
-
// re-hydrates the editor's Youtube node.
|
|
1059
1112
|
const meta = node.metadata;
|
|
1113
|
+
if (meta?.embedType === 'iframe') {
|
|
1114
|
+
// Generic preserved iframe. sanitizeUrl scheme-checks the src (only http/https
|
|
1115
|
+
// and the other non-executing schemes survive), so a javascript:/data: src is
|
|
1116
|
+
// dropped even with preservation on. The node only exists via opt-in parsing or
|
|
1117
|
+
// a programmatic AST, so this guard is unconditional.
|
|
1118
|
+
const src = (0, sanitize_js_1.sanitizeUrl)(meta?.url || '');
|
|
1119
|
+
if (!src)
|
|
1120
|
+
return '';
|
|
1121
|
+
const w = meta?.width ? ` width="${this.escape(meta.width)}"` : '';
|
|
1122
|
+
const h = meta?.height ? ` height="${this.escape(meta.height)}"` : '';
|
|
1123
|
+
return `${extraAnchors}<iframe src="${src}"${w}${h}${idAttr}${mappedAttrs}${styleAttr}></iframe>`;
|
|
1124
|
+
}
|
|
1125
|
+
// Match the attribute-driven Youtube wrapper shape so a loaded embed re-hydrates
|
|
1126
|
+
// an editor's Youtube node.
|
|
1060
1127
|
const id = meta?.videoId || '';
|
|
1061
1128
|
const width = meta?.width || '100%';
|
|
1062
1129
|
const align = meta?.align || 'center';
|
|
@@ -1068,8 +1135,8 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1068
1135
|
return `${extraAnchors}<div data-youtube-video="${this.escape(id)}" data-width="${this.escape(width)}" data-align="${this.escape(align)}" class="youtube-embed"${idAttr}${mappedAttrs} style="width: ${(0, sanitize_js_1.sanitizeCssValue)(width)}; margin-left: ${ml}; margin-right: ${mr};">${iframe}</div>`;
|
|
1069
1136
|
}
|
|
1070
1137
|
case 'admonition': {
|
|
1071
|
-
// Match
|
|
1072
|
-
// reaches
|
|
1138
|
+
// Match the attribute-driven admonition wrapper so a loaded admonition
|
|
1139
|
+
// reaches an editor as that node instead of a plain blockquote.
|
|
1073
1140
|
const meta = node.metadata;
|
|
1074
1141
|
const admonitionType = this.escape(meta?.admonitionType || 'note');
|
|
1075
1142
|
return `${extraAnchors}<div class="admonition admonition-${admonitionType}" data-type="${admonitionType}"${idAttr}${mappedAttrs}${styleAttr}>${childrenOutput}</div>`;
|
|
@@ -1117,19 +1184,37 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1117
1184
|
if (f.superscript)
|
|
1118
1185
|
result = `<sup>${result}</sup>`;
|
|
1119
1186
|
const styles = this.headingUniformSize
|
|
1120
|
-
? this.getInlineStyles(node, { skipFontSize: true })
|
|
1121
|
-
: this.getInlineStyles(node);
|
|
1187
|
+
? this.getInlineStyles(node, { skipFontSize: true, skipBackgroundColor: true })
|
|
1188
|
+
: this.getInlineStyles(node, { skipBackgroundColor: true });
|
|
1122
1189
|
if (styles) {
|
|
1123
1190
|
result = `<span style="${styles}">${result}</span>`;
|
|
1124
1191
|
}
|
|
1192
|
+
// Highlight -> <mark>, not a <span style="background-color">. Tiptap's Highlight
|
|
1193
|
+
// extension parseHTML matches exactly `mark`, so an editor round trip only rehydrates
|
|
1194
|
+
// the highlight from a <mark>; a bare span comes back as unhighlighted text. Kept
|
|
1195
|
+
// outside the colour/size span above so a run carrying both still rehydrates both. The
|
|
1196
|
+
// widened HtmlParser reads this shape back (style wins over data-color). Behaviour
|
|
1197
|
+
// change (was a span), noted in the changelog.
|
|
1198
|
+
if (f.backgroundColor) {
|
|
1199
|
+
const safeBg = (0, sanitize_js_1.sanitizeCssValue)(f.backgroundColor);
|
|
1200
|
+
if (safeBg) {
|
|
1201
|
+
result = `<mark data-color="${this.escape(safeBg)}" style="background-color: ${safeBg}">${result}</mark>`;
|
|
1202
|
+
}
|
|
1203
|
+
}
|
|
1125
1204
|
}
|
|
1126
1205
|
const meta = node.metadata;
|
|
1127
1206
|
if (meta?.wikilink) {
|
|
1128
|
-
//
|
|
1129
|
-
//
|
|
1130
|
-
// the host app's resolver may rewrite to a real URL).
|
|
1207
|
+
// data-wikilink-page preserves the exact page name separately from the display
|
|
1208
|
+
// text/alias and from href (which a host resolver may rewrite to a real URL).
|
|
1131
1209
|
if (!this.config.ignoreInternalLinks) {
|
|
1132
|
-
|
|
1210
|
+
// Attribute-driven emission adds data-wikilink/data-target (and data-alias when the
|
|
1211
|
+
// display text differs from the page) alongside the default attributes. It is a
|
|
1212
|
+
// superset - the widened parser still resolves via data-wikilink-page first.
|
|
1213
|
+
const extra = this.config.htmlConfig.sourceAttributes
|
|
1214
|
+
? ` data-wikilink="true" data-target="${this.escape(meta.link || '')}"`
|
|
1215
|
+
+ (node.text && node.text !== meta.link ? ` data-alias="${this.escape(node.text)}"` : '')
|
|
1216
|
+
: '';
|
|
1217
|
+
result = `<a href="#${this.escape(this.slugify(meta.link || ''))}" data-wikilink-page="${this.escape(meta.link || '')}"${extra}>${result}</a>`;
|
|
1133
1218
|
}
|
|
1134
1219
|
}
|
|
1135
1220
|
else if (meta?.link) {
|
|
@@ -1142,11 +1227,12 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1142
1227
|
result = `<abbr title="${this.escape(meta.abbreviationTitle)}">${result}</abbr>`;
|
|
1143
1228
|
}
|
|
1144
1229
|
if (meta?.citationKey) {
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1230
|
+
result = this.config.htmlConfig.sourceAttributes
|
|
1231
|
+
// Attribute-driven emission: a <span class="citation"> carrying data-key, which the
|
|
1232
|
+
// widened parser reads back (the default <cite> shape is not attribute-keyed).
|
|
1233
|
+
? `<span class="citation" data-key="${this.escape(meta.citationKey)}">[@${this.escape(meta.citationKey)}]</span>`
|
|
1234
|
+
// Default emission: a <cite> carrying the bare key, matching Pandoc's [@citekey].
|
|
1235
|
+
: `<cite data-citation-key="${this.escape(meta.citationKey)}">[@${this.escape(meta.citationKey)}]</cite>`;
|
|
1150
1236
|
}
|
|
1151
1237
|
return result;
|
|
1152
1238
|
}
|
|
@@ -1182,7 +1268,9 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
1182
1268
|
const f = node.formatting;
|
|
1183
1269
|
if (f.color)
|
|
1184
1270
|
pushSafe('color', f.color);
|
|
1185
|
-
|
|
1271
|
+
// Highlights are emitted as <mark> by formatText (see there); when that path owns the
|
|
1272
|
+
// background it passes skipBackgroundColor so the colour is not also duplicated here.
|
|
1273
|
+
if (f.backgroundColor && !options.skipBackgroundColor)
|
|
1186
1274
|
pushSafe('background-color', f.backgroundColor);
|
|
1187
1275
|
if (f.size && !options.skipFontSize)
|
|
1188
1276
|
pushSafe('font-size', f.size);
|