officeparser 7.5.1 → 7.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -6
- package/dist/OfficeConverter.d.ts +2 -2
- package/dist/OfficeParser.d.ts +2 -2
- package/dist/OfficeParser.js +10 -0
- package/dist/cli.js +3 -0
- package/dist/defaults.js +3 -1
- package/dist/generators/ChunkingGenerator.d.ts +2 -1
- package/dist/generators/ChunkingGenerator.js +73 -6
- package/dist/generators/EpubGenerator.js +4 -1
- package/dist/generators/HtmlGenerator.js +117 -29
- package/dist/generators/MarkdownGenerator.js +120 -29
- package/dist/generators/PdfGenerator.js +4 -1
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +70 -12
- package/dist/officeparser.browser.iife.js +211 -202
- package/dist/officeparser.browser.mjs +211 -202
- package/dist/officeparser.browser.slim.d.ts +70 -12
- package/dist/officeparser.browser.slim.iife.js +213 -204
- package/dist/officeparser.browser.slim.mjs +213 -204
- package/dist/parsers/HtmlParser.js +206 -38
- package/dist/parsers/MarkdownParser.js +135 -18
- package/dist/parsers/WordParser.js +7 -17
- package/dist/sbom.cdx.json +95 -95
- package/dist/types.d.ts +68 -10
- package/dist/utils/configUtils.js +3 -1
- package/dist/utils/sanitize.d.ts +9 -0
- package/dist/utils/sanitize.js +26 -0
- package/package.json +5 -5
|
@@ -68,7 +68,10 @@ function resolveDialect(dialect) {
|
|
|
68
68
|
*/
|
|
69
69
|
function resolveFallbackToHtml(fallbackToHtml) {
|
|
70
70
|
const uniform = (on) => ({
|
|
71
|
+
// inlineFormatting is opt-in only: it is never enabled by the boolean form, since it changes
|
|
72
|
+
// default output. Every other field follows the boolean.
|
|
71
73
|
textFormatting: on, alignment: on, anchors: on, tables: on, embeds: on, cellLineBreaks: on,
|
|
74
|
+
inlineFormatting: false,
|
|
72
75
|
});
|
|
73
76
|
if (fallbackToHtml === undefined || typeof fallbackToHtml === 'boolean')
|
|
74
77
|
return uniform(fallbackToHtml ?? true);
|
|
@@ -80,6 +83,7 @@ function resolveFallbackToHtml(fallbackToHtml) {
|
|
|
80
83
|
tables: fallbackToHtml.tables ?? on.tables,
|
|
81
84
|
embeds: fallbackToHtml.embeds ?? on.embeds,
|
|
82
85
|
cellLineBreaks: fallbackToHtml.cellLineBreaks ?? on.cellLineBreaks,
|
|
86
|
+
inlineFormatting: fallbackToHtml.inlineFormatting ?? on.inlineFormatting,
|
|
83
87
|
};
|
|
84
88
|
}
|
|
85
89
|
/**
|
|
@@ -203,35 +207,40 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
203
207
|
// Add Metadata (YAML Front Matter)
|
|
204
208
|
const meta = this.effectiveMetadata;
|
|
205
209
|
if (meta) {
|
|
206
|
-
|
|
207
|
-
//
|
|
208
|
-
//
|
|
209
|
-
//
|
|
210
|
-
|
|
210
|
+
// Build the field lines first. JSON-encode scalar values so a title/author/description
|
|
211
|
+
// containing a quote or newline can't break out of the YAML string and inject arbitrary
|
|
212
|
+
// front-matter keys. (JSON.stringify of a benign value yields the same `"..."` form as
|
|
213
|
+
// before, so normal output is unchanged.)
|
|
214
|
+
let fields = '';
|
|
211
215
|
if (meta.title)
|
|
212
|
-
|
|
216
|
+
fields += `title: ${JSON.stringify(meta.title)}\n`;
|
|
213
217
|
if (meta.author)
|
|
214
|
-
|
|
218
|
+
fields += `author: ${JSON.stringify(meta.author)}\n`;
|
|
215
219
|
const createdIso = this.toIsoDate(meta.created);
|
|
216
220
|
if (createdIso)
|
|
217
|
-
|
|
221
|
+
fields += `created: ${createdIso}\n`;
|
|
218
222
|
const modifiedIso = this.toIsoDate(meta.modified);
|
|
219
223
|
if (modifiedIso)
|
|
220
|
-
|
|
224
|
+
fields += `modified: ${modifiedIso}\n`;
|
|
221
225
|
if (meta.description)
|
|
222
|
-
|
|
226
|
+
fields += `description: ${JSON.stringify(meta.description)}\n`;
|
|
223
227
|
if (meta.subject)
|
|
224
|
-
|
|
228
|
+
fields += `subject: ${JSON.stringify(meta.subject)}\n`;
|
|
225
229
|
if (meta.keywords)
|
|
226
|
-
|
|
230
|
+
fields += `keywords: ${JSON.stringify(meta.keywords)}\n`;
|
|
227
231
|
if (meta.customProperties) {
|
|
228
232
|
for (const [key, val] of Object.entries(meta.customProperties)) {
|
|
229
233
|
// Strip newlines/colons from the key so it can't inject a new mapping.
|
|
230
234
|
const safeKey = String(key).replace(/[\r\n:]+/g, ' ').trim();
|
|
231
|
-
|
|
235
|
+
fields += `${safeKey}: ${Array.isArray(val) ? this.serializeFrontmatterArray(val) : JSON.stringify(val)}\n`;
|
|
232
236
|
}
|
|
233
237
|
}
|
|
234
|
-
|
|
238
|
+
// Only emit the frontmatter fence when at least one field is present. Empty metadata
|
|
239
|
+
// (a bare Tiptap/HTML fragment with no <head>) would otherwise emit `---\n---`, which
|
|
240
|
+
// reparses as a setext `## ---` heading and corrupts the document on every save/reload.
|
|
241
|
+
if (fields) {
|
|
242
|
+
output += `---\n${fields}---\n\n`;
|
|
243
|
+
}
|
|
235
244
|
}
|
|
236
245
|
const processor = async (node, childrenOutput) => {
|
|
237
246
|
// Handle Style Mapping for Markdown using the semantic mapping helper
|
|
@@ -256,6 +265,19 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
256
265
|
// HTML tag (e.g. <script>) when the Markdown is rendered to HTML.
|
|
257
266
|
let text = (0, sanitize_js_1.markdownEscapeText)(node.text || '');
|
|
258
267
|
if (this.config.includeFormatting && node.formatting) {
|
|
268
|
+
// Inline code: re-wrap the RAW text in backticks. The content is literal
|
|
269
|
+
// inside a code span, so the entity-escaped form above must not show through.
|
|
270
|
+
// The fence is one backtick longer than the longest embedded run so an inner
|
|
271
|
+
// backtick can't close the span early, padded when the content touches a
|
|
272
|
+
// backtick. Done before emphasis so bold/italic wrap the span (`**`code`**`).
|
|
273
|
+
// Previously a monospace text node emitted its bare text, dropping the code.
|
|
274
|
+
if (node.formatting.font === 'monospace') {
|
|
275
|
+
const raw = node.text || '';
|
|
276
|
+
const longestRun = Math.max(0, ...(raw.match(/`+/g) || []).map(s => s.length));
|
|
277
|
+
const fence = '`'.repeat(longestRun + 1);
|
|
278
|
+
const pad = (raw.startsWith('`') || raw.endsWith('`')) ? ' ' : '';
|
|
279
|
+
text = `${fence}${pad}${raw}${pad}${fence}`;
|
|
280
|
+
}
|
|
259
281
|
const emphasisAsterisk = this.resolvedDialect.emphasisMarker === 'asterisk';
|
|
260
282
|
if (node.formatting.bold && !this.inImplicitBold)
|
|
261
283
|
text = emphasisAsterisk ? `**${text}**` : `__${text}__`;
|
|
@@ -272,6 +294,24 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
272
294
|
if (node.formatting.superscript)
|
|
273
295
|
text = `<sup>${text}</sup>`;
|
|
274
296
|
}
|
|
297
|
+
// Inline color / highlight / font size have no Markdown syntax; emit a styled
|
|
298
|
+
// <span> (outermost, so the inner Markdown markers survive) only when opted in,
|
|
299
|
+
// so default output is unchanged. Values are CSS-sanitized against injection.
|
|
300
|
+
if (this.resolvedFallbackToHtml.inlineFormatting) {
|
|
301
|
+
const styles = [];
|
|
302
|
+
const pushStyle = (prop, val) => {
|
|
303
|
+
if (!val)
|
|
304
|
+
return;
|
|
305
|
+
const safe = (0, sanitize_js_1.sanitizeCssValue)(val); // drops url()/expression()/<>/quotes
|
|
306
|
+
if (safe)
|
|
307
|
+
styles.push(`${prop}: ${safe}`);
|
|
308
|
+
};
|
|
309
|
+
pushStyle('color', node.formatting.color);
|
|
310
|
+
pushStyle('background-color', node.formatting.backgroundColor);
|
|
311
|
+
pushStyle('font-size', node.formatting.size);
|
|
312
|
+
if (styles.length)
|
|
313
|
+
text = `<span style="${styles.join('; ')}">${text}</span>`;
|
|
314
|
+
}
|
|
275
315
|
}
|
|
276
316
|
const meta = node.metadata;
|
|
277
317
|
if (meta?.wikilink && this.resolvedDialect.wikilinks) {
|
|
@@ -411,13 +451,17 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
411
451
|
return childrenOutput;
|
|
412
452
|
}
|
|
413
453
|
case 'break': {
|
|
414
|
-
// A hard line break (CommonMark: two trailing spaces before the
|
|
415
|
-
//
|
|
416
|
-
//
|
|
417
|
-
//
|
|
454
|
+
// A hard line break (CommonMark: two trailing spaces before the newline)
|
|
455
|
+
// round-trips back to a distinct 'break' node on reparse. A thematic break
|
|
456
|
+
// emits `---` as its own block (the top-level loop supplies the surrounding
|
|
457
|
+
// blank lines), so a Markdown `---` / HTML `<hr>` survives a save instead of
|
|
458
|
+
// collapsing to whitespace. Every other breakType - notably 'page', which
|
|
459
|
+
// Markdown has no syntax for - keeps emitting a bare newline, unchanged.
|
|
418
460
|
const meta = node.metadata;
|
|
419
461
|
if (meta?.breakType === 'carriageReturn')
|
|
420
462
|
return ' \n';
|
|
463
|
+
if (meta?.breakType === 'thematic')
|
|
464
|
+
return '---';
|
|
421
465
|
return '\n';
|
|
422
466
|
}
|
|
423
467
|
case 'code': {
|
|
@@ -447,17 +491,20 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
447
491
|
return this.resolvedDialect.math === 'dollar' ? `$${mathInline}$` : mathInline;
|
|
448
492
|
}
|
|
449
493
|
const lang = (meta?.language || '').replace(/[\r\n`]+/g, '');
|
|
450
|
-
//
|
|
451
|
-
//
|
|
452
|
-
//
|
|
453
|
-
//
|
|
454
|
-
//
|
|
455
|
-
|
|
494
|
+
// A `code` node is always block-level: genuinely inline code is a monospace
|
|
495
|
+
// text node, never a `code` node. So emit a fenced block whenever the node
|
|
496
|
+
// carries a language OR spans multiple lines. Previously the decision keyed only
|
|
497
|
+
// off a line break, so a single-line code node with a language - `const x = 1;`
|
|
498
|
+
// tagged `js`, or a one-line `mermaid` diagram - collapsed to an inline span,
|
|
499
|
+
// silently dropping both its language and its block-ness. (Testing `[\r\n]`, not
|
|
500
|
+
// just `\n`, still routes a CR-only body to the fenced branch, where a renderer
|
|
501
|
+
// that normalizes `\r` to a line ending would otherwise kill an inline span.)
|
|
502
|
+
if (lang || (node.text && /[\r\n]/.test(node.text))) {
|
|
456
503
|
// Fence with one more backtick than the longest run inside the content
|
|
457
504
|
// so an embedded ``` can't close the block early and inject markup.
|
|
458
|
-
const longestRun = Math.max(0, ...(node.text.match(/`+/g) || []).map(s => s.length));
|
|
505
|
+
const longestRun = Math.max(0, ...((node.text || '').match(/`+/g) || []).map(s => s.length));
|
|
459
506
|
const fence = '`'.repeat(Math.max(3, longestRun + 1));
|
|
460
|
-
return `\n${fence}${lang}\n${node.text}\n${fence}\n\n`;
|
|
507
|
+
return `\n${fence}${lang}\n${node.text || ''}\n${fence}\n\n`;
|
|
461
508
|
}
|
|
462
509
|
else {
|
|
463
510
|
const t = node.text || '';
|
|
@@ -489,7 +536,10 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
489
536
|
// into an end-of-document "### Notes" section under a [^id] marker.
|
|
490
537
|
return childrenOutput.trim();
|
|
491
538
|
}
|
|
492
|
-
|
|
539
|
+
// Indent continuation lines one level so a multi-line body re-parses as a
|
|
540
|
+
// single definition (a bare newline would end it). Single-line bodies, the
|
|
541
|
+
// common case, are unaffected.
|
|
542
|
+
return `[^${this.getFootnoteKey(node)}]: ${childrenOutput.trim().replace(/\n/g, '\n ')}\n\n`;
|
|
493
543
|
}
|
|
494
544
|
return `> **Note:** ${childrenOutput.trim()}\n\n`;
|
|
495
545
|
}
|
|
@@ -498,6 +548,20 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
498
548
|
// save default), emit the exact single-line div MarkdownParser recognises on
|
|
499
549
|
// reimport; otherwise degrade to a plain link.
|
|
500
550
|
const meta = node.metadata;
|
|
551
|
+
if (meta?.embedType === 'iframe') {
|
|
552
|
+
// sanitizeUrl scheme-checks and HTML-escapes the src (hostile schemes drop
|
|
553
|
+
// the node). The single-line <iframe> is what MarkdownParser recognises on
|
|
554
|
+
// reimport, gated there on preserveIframes.
|
|
555
|
+
const safe = (0, sanitize_js_1.sanitizeUrl)(meta?.url || '');
|
|
556
|
+
if (!safe)
|
|
557
|
+
return '';
|
|
558
|
+
if (this.resolvedFallbackToHtml.embeds) {
|
|
559
|
+
const w = meta?.width ? ` width="${(0, sanitize_js_1.escapeHtml)(meta.width)}"` : '';
|
|
560
|
+
const h = meta?.height ? ` height="${(0, sanitize_js_1.escapeHtml)(meta.height)}"` : '';
|
|
561
|
+
return `\n<iframe src="${safe}"${w}${h}></iframe>\n\n`;
|
|
562
|
+
}
|
|
563
|
+
return `[Embed](${(0, sanitize_js_1.sanitizeMarkdownUrl)(meta?.url || '')})\n\n`;
|
|
564
|
+
}
|
|
501
565
|
const id = meta?.videoId || '';
|
|
502
566
|
if (this.resolvedFallbackToHtml.embeds) {
|
|
503
567
|
const width = meta?.width ? ` data-width="${(0, sanitize_js_1.escapeHtml)(meta.width)}"` : '';
|
|
@@ -569,6 +633,14 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
569
633
|
for (let i = 0; i < optimizedContent.length; i++) {
|
|
570
634
|
const node = optimizedContent[i];
|
|
571
635
|
const nextNode = optimizedContent[i + 1];
|
|
636
|
+
// A top-level footnote/endnote note is an orphan definition (unreferenced `[^id]: ...`
|
|
637
|
+
// the MarkdownParser recovered). Collect it so it's emitted with the other definitions
|
|
638
|
+
// at the document end rather than inline before them.
|
|
639
|
+
const orphanNoteType = node.metadata?.noteType;
|
|
640
|
+
if (node.type === 'note' && (orphanNoteType === 'footnote' || orphanNoteType === 'endnote')) {
|
|
641
|
+
this.collectedNotes.push(node);
|
|
642
|
+
continue;
|
|
643
|
+
}
|
|
572
644
|
let result = await this.processNodeRecursive(node, processor);
|
|
573
645
|
// Ensure lists and other block elements are separated from non-similar content by a blank line
|
|
574
646
|
if (nextNode) {
|
|
@@ -585,8 +657,21 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
585
657
|
output += result;
|
|
586
658
|
}
|
|
587
659
|
if (this.collectedNotes.length > 0) {
|
|
588
|
-
|
|
589
|
-
|
|
660
|
+
// No decorative `---\n\n### Notes` preamble: `[^id]:` definitions are valid on their own
|
|
661
|
+
// (GitHub/Pandoc render the footnotes section and its rule automatically), and the
|
|
662
|
+
// literal heading round-tripped as a real `###` node - so every save/reload re-emitted
|
|
663
|
+
// the parsed heading AND a fresh one, growing the document unbounded. Emitting the bare
|
|
664
|
+
// definitions makes the cycle byte-stable. Behaviour change, noted in the changelog.
|
|
665
|
+
// Collapse the preceding block's trailing blank lines so exactly one blank line separates
|
|
666
|
+
// the body from the definitions (rather than the doubled `\n\n\n\n` the concatenation
|
|
667
|
+
// would otherwise leave).
|
|
668
|
+
output = output.replace(/\n+$/, '');
|
|
669
|
+
let notesMd = '\n\n';
|
|
670
|
+
// De-duplicate by node identity: a footnote referenced more than once shares a single
|
|
671
|
+
// note object (see MarkdownParser), pushed here once per reference. Emit its definition
|
|
672
|
+
// just once. Distinct notes - even two office notes that happen to share a numeric id -
|
|
673
|
+
// are separate objects and are all kept.
|
|
674
|
+
for (const note of [...new Set(this.collectedNotes)]) {
|
|
590
675
|
notesMd += await this.processNodeRecursive(note, processor);
|
|
591
676
|
}
|
|
592
677
|
output += notesMd;
|
|
@@ -689,6 +774,12 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
689
774
|
let current = null;
|
|
690
775
|
for (const node of nodes) {
|
|
691
776
|
if (node.type === 'text' && current && current.type === 'text' &&
|
|
777
|
+
// A note anchors to the end of its text run and its `[^id]` marker is emitted there;
|
|
778
|
+
// merging a following run onto a note-carrying run would slide the marker past it
|
|
779
|
+
// (`Body[^1].` -> `Body.[^1]`). Keep such runs separate so the marker stays put and
|
|
780
|
+
// matches where HtmlGenerator emits it.
|
|
781
|
+
(!current.notes || current.notes.length === 0) &&
|
|
782
|
+
(!node.notes || node.notes.length === 0) &&
|
|
692
783
|
this.areFormattingEqual(node.formatting, current.formatting) &&
|
|
693
784
|
JSON.stringify(node.metadata) === JSON.stringify(current.metadata)) {
|
|
694
785
|
current.text = (current.text || '') + (node.text || '');
|
|
@@ -22,7 +22,10 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
22
22
|
// We reuse the current configuration but ensure standalone mode is on for HTML
|
|
23
23
|
const htmlGenerator = new HtmlGenerator_js_1.HtmlGenerator(this.ast, {
|
|
24
24
|
...this.config,
|
|
25
|
-
|
|
25
|
+
// Force sourceAttributes off: those data-* attributes are wire-format plumbing for
|
|
26
|
+
// structured consumers and change the mermaid shape's rendered appearance, neither of
|
|
27
|
+
// which belongs in a printed PDF.
|
|
28
|
+
htmlConfig: { ...this.config.htmlConfig, standalone: true, sourceAttributes: false },
|
|
26
29
|
});
|
|
27
30
|
const htmlResult = await htmlGenerator.generate();
|
|
28
31
|
const html = typeof htmlResult.value === 'string' ? htmlResult.value : '';
|
package/dist/index.d.ts
CHANGED
|
@@ -51,10 +51,10 @@
|
|
|
51
51
|
import { OfficeParser } from './OfficeParser.js';
|
|
52
52
|
import { OfficeGenerator } from './OfficeGenerator.js';
|
|
53
53
|
import { OfficeConverter } from './OfficeConverter.js';
|
|
54
|
-
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult } from './types.js';
|
|
54
|
+
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, BlobLike, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, HtmlGeneratorConfig, HtmlParserConfig, MdGeneratorConfig, CsvGeneratorConfig, PdfGeneratorConfig, RtfGeneratorConfig, TextGeneratorConfig, MarkdownDialectConfig, MarkdownDialectPreset, StandaloneConfig, MetadataOverrides, FallbackToHtmlConfig, HtmlInjectionConfig, OcrConfig, DecompressionLimits, TextMetadata, TableMetadata, CodeMetadata, NoteMetadata, AdmonitionMetadata, EmbedMetadata, ChartMetadata, CommentMetadata, HeaderFooterMetadata, IndentationMetadata, ParagraphMetadata } from './types.js';
|
|
55
55
|
declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
56
56
|
declare const terminateOcr: typeof OfficeParser.terminateOcr;
|
|
57
57
|
declare const convert: typeof OfficeConverter.convert;
|
|
58
58
|
declare const generate: typeof OfficeGenerator.generate;
|
|
59
|
-
export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, };
|
|
59
|
+
export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, BlobLike, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, HtmlGeneratorConfig, HtmlParserConfig, MdGeneratorConfig, CsvGeneratorConfig, PdfGeneratorConfig, RtfGeneratorConfig, TextGeneratorConfig, MarkdownDialectConfig, MarkdownDialectPreset, StandaloneConfig, MetadataOverrides, FallbackToHtmlConfig, HtmlInjectionConfig, OcrConfig, DecompressionLimits, TextMetadata, TableMetadata, CodeMetadata, NoteMetadata, AdmonitionMetadata, EmbedMetadata, ChartMetadata, CommentMetadata, HeaderFooterMetadata, IndentationMetadata, ParagraphMetadata, };
|
|
60
60
|
export default OfficeParser;
|
|
@@ -403,6 +403,18 @@ export interface HtmlParserConfig {
|
|
|
403
403
|
* Defaults to false.
|
|
404
404
|
*/
|
|
405
405
|
preserveAttributes?: boolean;
|
|
406
|
+
/**
|
|
407
|
+
* Preserve `<iframe>` embeds that aren't recognized as a known provider (YouTube is always
|
|
408
|
+
* recognized). By default every non-YouTube iframe is dropped, which is a deliberate security
|
|
409
|
+
* posture other consumers rely on; set this to opt back in. `true` preserves any iframe; an
|
|
410
|
+
* array is a hostname allowlist (an entry matches the src's host exactly or as a `.`-suffix,
|
|
411
|
+
* so `"vimeo.com"` also matches `player.vimeo.com`). Preserved iframes become `embed` nodes
|
|
412
|
+
* with `embedType: 'iframe'`; on generation the `src` is still scheme-checked (only http/https
|
|
413
|
+
* survive). This also governs a raw `<iframe>` block encountered in Markdown input.
|
|
414
|
+
*
|
|
415
|
+
* Defaults to false.
|
|
416
|
+
*/
|
|
417
|
+
preserveIframes?: boolean | string[];
|
|
406
418
|
}
|
|
407
419
|
/**
|
|
408
420
|
* Maps an input format string to its corresponding format-specific parser configuration, mirroring
|
|
@@ -859,6 +871,18 @@ export interface HtmlGeneratorConfig {
|
|
|
859
871
|
* Granular injection points for custom HTML, scripts, and styles.
|
|
860
872
|
*/
|
|
861
873
|
injections?: HtmlInjectionConfig;
|
|
874
|
+
/**
|
|
875
|
+
* Carry each rich node's raw source in a `data-*` attribute, with undelimited text content,
|
|
876
|
+
* so attribute-driven structured consumers (rich-text editors, custom viewers) can rehydrate
|
|
877
|
+
* the node from the markup rather than re-parsing the display text. Affects wikilinks
|
|
878
|
+
* (adds `data-wikilink`/`data-target`/`data-alias`), citations (a `<span class="citation">`
|
|
879
|
+
* carrying `data-key` instead of `<cite>`), math (the LaTeX in `data-math`, undelimited) and
|
|
880
|
+
* mermaid (a `<div class="mermaid" data-mermaid>` instead of `<pre><code>`).
|
|
881
|
+
*
|
|
882
|
+
* Off by default; the default output is byte-identical to previous releases. The widened
|
|
883
|
+
* `HtmlParser` reads every shape this emits, so output stays self-round-trippable.
|
|
884
|
+
*/
|
|
885
|
+
sourceAttributes?: boolean;
|
|
862
886
|
}
|
|
863
887
|
/**
|
|
864
888
|
* Configuration options for PDF generation.
|
|
@@ -1074,6 +1098,14 @@ export interface FallbackToHtmlConfig {
|
|
|
1074
1098
|
embeds?: boolean;
|
|
1075
1099
|
/** Multi-line table cell content joined with `<br>` instead of a space. */
|
|
1076
1100
|
cellLineBreaks?: boolean;
|
|
1101
|
+
/**
|
|
1102
|
+
* Inline text color, highlight, and font size via a `<span style="color:...;background-color:...;
|
|
1103
|
+
* font-size:...">` run, which the Markdown parser reads back. These have no Markdown syntax and
|
|
1104
|
+
* are silently lost otherwise. Unlike the other fields this is **off by default even when
|
|
1105
|
+
* `fallbackToHtml` is `true`**, because it changes default output; enable it explicitly with
|
|
1106
|
+
* `fallbackToHtml: { inlineFormatting: true }`.
|
|
1107
|
+
*/
|
|
1108
|
+
inlineFormatting?: boolean;
|
|
1077
1109
|
}
|
|
1078
1110
|
/**
|
|
1079
1111
|
* Configuration options for Markdown generation.
|
|
@@ -1345,6 +1377,16 @@ export interface OfficeChunk {
|
|
|
1345
1377
|
* Supported file types for parsing.
|
|
1346
1378
|
*/
|
|
1347
1379
|
export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf" | "md" | "html" | "csv" | "epub";
|
|
1380
|
+
/**
|
|
1381
|
+
* A structural stand-in for the web `Blob`/`File` so `parseOffice`/`convert` accept them in the
|
|
1382
|
+
* browser without pulling the DOM lib into this package's types. Any object with an
|
|
1383
|
+
* `arrayBuffer()` method qualifies. When `name` is present (as on a `File`) it is used only for
|
|
1384
|
+
* extension-based type detection, never as a filesystem path.
|
|
1385
|
+
*/
|
|
1386
|
+
export interface BlobLike {
|
|
1387
|
+
arrayBuffer(): Promise<ArrayBuffer>;
|
|
1388
|
+
name?: string;
|
|
1389
|
+
}
|
|
1348
1390
|
/**
|
|
1349
1391
|
* Types of content nodes in the AST.
|
|
1350
1392
|
*/
|
|
@@ -1586,7 +1628,7 @@ export interface TableMetadata {
|
|
|
1586
1628
|
/** Unique anchor IDs for internal linking. */
|
|
1587
1629
|
anchorIds?: string[];
|
|
1588
1630
|
/**
|
|
1589
|
-
* Layout alignment of the table on the page (e.g.
|
|
1631
|
+
* Layout alignment of the table on the page (e.g. an editor's custom table node).
|
|
1590
1632
|
* @example 'center'
|
|
1591
1633
|
*/
|
|
1592
1634
|
align?: "left" | "center" | "right";
|
|
@@ -1631,12 +1673,12 @@ export interface ImageMetadata {
|
|
|
1631
1673
|
/** Unique anchor IDs for internal linking. */
|
|
1632
1674
|
anchorIds?: string[];
|
|
1633
1675
|
/**
|
|
1634
|
-
* Display width of the image (e.g.
|
|
1676
|
+
* Display width of the image (e.g. an editor's custom image node), as a CSS length or percentage.
|
|
1635
1677
|
* @example "50%"
|
|
1636
1678
|
*/
|
|
1637
1679
|
width?: string;
|
|
1638
1680
|
/**
|
|
1639
|
-
* Layout alignment of the image (e.g.
|
|
1681
|
+
* Layout alignment of the image (e.g. an editor's custom image node).
|
|
1640
1682
|
* @example 'center'
|
|
1641
1683
|
*/
|
|
1642
1684
|
align?: "left" | "center" | "right";
|
|
@@ -1646,14 +1688,19 @@ export interface ImageMetadata {
|
|
|
1646
1688
|
* Markdown has no native syntax for this - see `MarkdownGenerator`'s `embed` case.
|
|
1647
1689
|
*/
|
|
1648
1690
|
export interface EmbedMetadata {
|
|
1649
|
-
/**
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1691
|
+
/**
|
|
1692
|
+
* The kind of embed. 'youtube' is recognized from a `data-youtube-video` wrapper or a YouTube
|
|
1693
|
+
* iframe; 'iframe' is a generic preserved iframe (opt-in via `HtmlParserConfig.preserveIframes`).
|
|
1694
|
+
*/
|
|
1695
|
+
embedType: "youtube" | "iframe";
|
|
1696
|
+
/** The provider-specific video ID (e.g. the 11-character YouTube video ID). Absent for generic iframes. */
|
|
1697
|
+
videoId?: string;
|
|
1698
|
+
/** The original/canonical URL of the embedded media, if known. For a generic iframe, its `src`. */
|
|
1654
1699
|
url?: string;
|
|
1655
1700
|
/** Display width, as a CSS length or percentage. */
|
|
1656
1701
|
width?: string;
|
|
1702
|
+
/** Display height, as a CSS length or percentage (generic iframes). */
|
|
1703
|
+
height?: string;
|
|
1657
1704
|
/** Layout alignment of the embed. */
|
|
1658
1705
|
align?: "left" | "center" | "right";
|
|
1659
1706
|
}
|
|
@@ -1737,6 +1784,14 @@ export interface NoteMetadata {
|
|
|
1737
1784
|
anchorIds?: string[];
|
|
1738
1785
|
/** The slide number this note is associated with (used in PowerPoint). */
|
|
1739
1786
|
slideNumber?: number;
|
|
1787
|
+
/**
|
|
1788
|
+
* True for a footnote/endnote definition that no reference points at (an "orphan").
|
|
1789
|
+
* The Markdown parser sets this when it recovers a `[^id]: ...` definition with no matching
|
|
1790
|
+
* `[^id]` reference so the definition is preserved rather than dropped. Generators route such
|
|
1791
|
+
* notes into their footnotes section without a citation marker, and the HTML generator omits
|
|
1792
|
+
* the (otherwise dangling) back-link.
|
|
1793
|
+
*/
|
|
1794
|
+
unreferenced?: boolean;
|
|
1740
1795
|
}
|
|
1741
1796
|
/**
|
|
1742
1797
|
* Metadata for break nodes.
|
|
@@ -1751,8 +1806,11 @@ export interface BreakMetadata {
|
|
|
1751
1806
|
* - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
|
|
1752
1807
|
* - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
|
|
1753
1808
|
* - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
|
|
1809
|
+
* - 'thematic': A thematic break (Markdown `---`/`***`/`___`, HTML `<hr>`) - a horizontal
|
|
1810
|
+
* rule separating sections, distinct from a page break. Emitted as `---` in Markdown and
|
|
1811
|
+
* `<hr>` in HTML.
|
|
1754
1812
|
*/
|
|
1755
|
-
breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn";
|
|
1813
|
+
breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn" | "thematic";
|
|
1756
1814
|
/**
|
|
1757
1815
|
* Specifies the location which shall be used as the next available line when breakType
|
|
1758
1816
|
* has a value of 'textWrapping'. Should be ignored for other break types.
|
|
@@ -1774,7 +1832,7 @@ export interface CodeMetadata {
|
|
|
1774
1832
|
/**
|
|
1775
1833
|
* When set, this node is a LaTeX math expression rather than a code block. `node.text`
|
|
1776
1834
|
* holds the bare LaTeX (delimiters excluded); 'inline' round-trips as `$...$`,
|
|
1777
|
-
* 'block' as `$$...$$`. Matches
|
|
1835
|
+
* 'block' as `$$...$$`. Matches attribute-driven editors' math nodes.
|
|
1778
1836
|
*/
|
|
1779
1837
|
math?: "inline" | "block";
|
|
1780
1838
|
}
|
|
@@ -2342,7 +2400,7 @@ export declare class OfficeParser {
|
|
|
2342
2400
|
* const text = ast.toText();
|
|
2343
2401
|
* ```
|
|
2344
2402
|
*/
|
|
2345
|
-
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
2403
|
+
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array | BlobLike, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
2346
2404
|
/**
|
|
2347
2405
|
* Terminates all active OCR workers and cleans up resources.
|
|
2348
2406
|
*
|
|
@@ -2418,7 +2476,7 @@ export declare class OfficeConverter {
|
|
|
2418
2476
|
* });
|
|
2419
2477
|
* ```
|
|
2420
2478
|
*/
|
|
2421
|
-
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
|
|
2479
|
+
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array | BlobLike, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
|
|
2422
2480
|
}
|
|
2423
2481
|
export declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
2424
2482
|
export declare const terminateOcr: typeof OfficeParser.terminateOcr;
|