officeparser 7.2.2 → 7.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/README.md +161 -17
  2. package/dist/OfficeGenerator.js +4 -0
  3. package/dist/OfficeParser.d.ts +2 -0
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +1 -1
  6. package/dist/cli.js +3 -2
  7. package/dist/defaults.js +3 -3
  8. package/dist/generators/BaseGenerator.d.ts +11 -0
  9. package/dist/generators/BaseGenerator.js +29 -0
  10. package/dist/generators/CsvGenerator.d.ts +9 -1
  11. package/dist/generators/CsvGenerator.js +24 -14
  12. package/dist/generators/EpubGenerator.d.ts +18 -0
  13. package/dist/generators/EpubGenerator.js +242 -0
  14. package/dist/generators/HtmlGenerator.d.ts +12 -0
  15. package/dist/generators/HtmlGenerator.js +266 -51
  16. package/dist/generators/MarkdownGenerator.d.ts +16 -0
  17. package/dist/generators/MarkdownGenerator.js +173 -24
  18. package/dist/generators/PdfGenerator.js +32 -0
  19. package/dist/generators/RtfGenerator.js +12 -15
  20. package/dist/generators/TextGenerator.js +11 -0
  21. package/dist/index.d.ts +1 -0
  22. package/dist/index.js +1 -0
  23. package/dist/officeparser.browser.d.ts +144 -7
  24. package/dist/officeparser.browser.iife.js +289 -193
  25. package/dist/officeparser.browser.mjs +289 -193
  26. package/dist/officeparser.browser.slim.d.ts +2129 -0
  27. package/dist/officeparser.browser.slim.iife.js +1278 -0
  28. package/dist/officeparser.browser.slim.mjs +1277 -0
  29. package/dist/parsers/EpubParser.d.ts +8 -0
  30. package/dist/parsers/EpubParser.js +217 -0
  31. package/dist/parsers/HtmlParser.js +284 -20
  32. package/dist/parsers/MarkdownParser.js +424 -33
  33. package/dist/parsers/OpenOfficeParser.js +241 -54
  34. package/dist/parsers/PdfParser.js +4 -1
  35. package/dist/parsers/WordParser.js +2 -2
  36. package/dist/sbom.cdx.json +111 -223
  37. package/dist/types.d.ts +146 -7
  38. package/dist/types.js +2 -0
  39. package/dist/utils/errorUtils.js +3 -2
  40. package/dist/utils/sanitize.d.ts +99 -0
  41. package/dist/utils/sanitize.js +228 -0
  42. package/dist/utils/zipUtils.js +76 -26
  43. package/package.json +19 -10
package/dist/types.d.ts CHANGED
@@ -39,6 +39,8 @@ export declare enum OfficeErrorType {
39
39
  ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
40
40
  /** ZIP uncompressed size limit exceeded */
41
41
  ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
42
+ /** Document element/structure nesting exceeded the safe recursion depth */
43
+ MAX_NESTING_DEPTH_EXCEEDED = "MAX_NESTING_DEPTH_EXCEEDED",
42
44
  /** Embedding call timed out */
43
45
  EMBEDDING_TIMEOUT = "EMBEDDING_TIMEOUT"
44
46
  }
@@ -317,7 +319,7 @@ export interface OfficeParserConfig {
317
319
  * The URL/path to the PDF.js worker script.
318
320
  *
319
321
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
320
- * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
322
+ * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs`.
321
323
  * You can override this with your own local path or a different CDN link.
322
324
  */
323
325
  pdfWorkerSrc?: string;
@@ -401,7 +403,7 @@ export interface OfficeIssue {
401
403
  /**
402
404
  * The result of a document conversion operation.
403
405
  */
404
- type ConversionValue<D extends UniversalGeneratorFormat> = D extends 'pdf' ? Uint8Array | string : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : string;
406
+ type ConversionValue<D extends UniversalGeneratorFormat> = D extends 'pdf' ? Uint8Array | string : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : D extends 'epub' ? Uint8Array : string;
405
407
  export interface ConversionResult<D extends UniversalGeneratorFormat> {
406
408
  /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
407
409
  value: ConversionValue<D>;
@@ -411,7 +413,7 @@ export interface ConversionResult<D extends UniversalGeneratorFormat> {
411
413
  /**
412
414
  * Universal formats supported by all source types for generation.
413
415
  */
414
- export type UniversalGeneratorFormat = 'text' | 'md' | 'html' | 'pdf' | 'csv' | 'rtf' | 'chunks';
416
+ export type UniversalGeneratorFormat = 'text' | 'md' | 'html' | 'pdf' | 'csv' | 'rtf' | 'chunks' | 'epub';
415
417
  /**
416
418
  * Allowed destination formats for a given source type.
417
419
  * Currently, all generators are universal across all source formats.
@@ -628,15 +630,64 @@ export interface HtmlInjectionConfig {
628
630
  /** Raw HTML injected immediately before the closing </body> tag */
629
631
  bodyEnd?: string;
630
632
  }
633
+ /**
634
+ * Granular control over which parts of the full HTML "document envelope" are emitted.
635
+ * Shorthand: `standalone: true` == every part on (a complete document); `standalone: false` ==
636
+ * every part off (a bare content fragment). When an object is passed, any field you omit
637
+ * defaults to its "on" (standalone) value.
638
+ */
639
+ export interface StandaloneConfig {
640
+ /**
641
+ * Wrap the output in `<!DOCTYPE html><html><head>…</head><body>…</body></html>`.
642
+ * When false, only the inner content fragment is emitted. Defaults to true.
643
+ */
644
+ document?: boolean;
645
+ /**
646
+ * Emit `<title>` and `<meta>` tags (author, description, dates, custom properties) in the head.
647
+ * Only meaningful when `document` is true. Defaults to true.
648
+ */
649
+ metaTags?: boolean;
650
+ /**
651
+ * How the library's built-in CSS is delivered:
652
+ * - `'full'` — the complete premium stylesheet using global selectors (`body`, `h1`, `table`, …).
653
+ * This is what `standalone: true` has always emitted.
654
+ * - `'scoped'` — the same styling, scoped under the fragment's container via CSS `@scope` so it
655
+ * cannot leak into a host page's own styles. Requires a modern browser engine (Chrome 118+,
656
+ * Safari 17.4+, Firefox 128+).
657
+ * - `'none'` — no stylesheet is emitted; the host page (or EPUB reader, or rich-text editor)
658
+ * supplies its own styling.
659
+ * The boolean shorthand for `standalone` maps `true` → `'full'`, `false` → `'none'`.
660
+ * Defaults to `'full'`.
661
+ */
662
+ styles?: 'full' | 'scoped' | 'none';
663
+ /**
664
+ * Emit injected `<script>` tags: the Chart.js loader (when `includeCharts` is true and charts
665
+ * are present) and the spreadsheet interactivity script. Defaults to true.
666
+ */
667
+ scripts?: boolean;
668
+ /**
669
+ * Apply `injections.headStart` / `injections.headEnd`. Only meaningful when `document` is true
670
+ * (there is no `<head>` to inject into otherwise). Defaults to true.
671
+ */
672
+ headInjections?: boolean;
673
+ /**
674
+ * Apply `injections.bodyStart` / `injections.bodyEnd`. Applies even when generating a bare
675
+ * fragment (`document: false`), since these wrap body *content*, not the document shell.
676
+ * Defaults to true.
677
+ */
678
+ bodyInjections?: boolean;
679
+ }
631
680
  /**
632
681
  * Configuration options for HTML generation.
633
682
  */
634
683
  export interface HtmlGeneratorConfig {
635
684
  /**
636
685
  * Whether to wrap the output in a full HTML document structure (e.g., <html>, <head>, etc.).
686
+ * Pass an object instead of a boolean for granular control over individual parts of the
687
+ * envelope (document shell, meta tags, styles, scripts, injections) - see `StandaloneConfig`.
637
688
  * Defaults to true.
638
689
  */
639
- standalone?: boolean;
690
+ standalone?: boolean | StandaloneConfig;
640
691
  /**
641
692
  * URL for the Chart.js library to use when 'includeCharts' is true.
642
693
  * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
@@ -1057,11 +1108,11 @@ export interface OfficeChunk {
1057
1108
  /**
1058
1109
  * Supported file types for parsing.
1059
1110
  */
1060
- export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf' | 'md' | 'html' | 'csv';
1111
+ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf' | 'md' | 'html' | 'csv' | 'epub';
1061
1112
  /**
1062
1113
  * Types of content nodes in the AST.
1063
1114
  */
1064
- export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment' | 'header' | 'footer' | 'slideMaster';
1115
+ export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment' | 'header' | 'footer' | 'slideMaster' | 'embed' | 'admonition' | 'definitionList' | 'definitionTerm' | 'definitionDescription';
1065
1116
  /**
1066
1117
  * Supported MIME types for attachments.
1067
1118
  */
@@ -1255,6 +1306,10 @@ export interface ListMetadata {
1255
1306
  style?: string;
1256
1307
  /** Unique anchor IDs for internal linking. */
1257
1308
  anchorIds?: string[];
1309
+ /** True when this list item is a GFM task-list item (checkbox), regardless of checked state. */
1310
+ isTask?: boolean;
1311
+ /** Checked state for a task-list item. Only meaningful when isTask is true. */
1312
+ checked?: boolean;
1258
1313
  }
1259
1314
  /**
1260
1315
  * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
@@ -1294,6 +1349,11 @@ export interface CellMetadata {
1294
1349
  export interface TableMetadata {
1295
1350
  /** Unique anchor IDs for internal linking. */
1296
1351
  anchorIds?: string[];
1352
+ /**
1353
+ * Layout alignment of the table on the page (e.g. inscript-editor's `CustomTable`).
1354
+ * @example 'center'
1355
+ */
1356
+ align?: 'left' | 'center' | 'right';
1297
1357
  }
1298
1358
  /**
1299
1359
  * Metadata for a chart node in the document.
@@ -1334,6 +1394,42 @@ export interface ImageMetadata {
1334
1394
  url?: string;
1335
1395
  /** Unique anchor IDs for internal linking. */
1336
1396
  anchorIds?: string[];
1397
+ /**
1398
+ * Display width of the image (e.g. inscript-editor's `CustomImage`), as a CSS length or percentage.
1399
+ * @example "50%"
1400
+ */
1401
+ width?: string;
1402
+ /**
1403
+ * Layout alignment of the image (e.g. inscript-editor's `CustomImage`).
1404
+ * @example 'center'
1405
+ */
1406
+ align?: 'left' | 'center' | 'right';
1407
+ }
1408
+ /**
1409
+ * Metadata for an embedded external media node (e.g. a YouTube video).
1410
+ * Markdown has no native syntax for this - see `MarkdownGenerator`'s `embed` case.
1411
+ */
1412
+ export interface EmbedMetadata {
1413
+ /** The kind of embed. Only 'youtube' is supported today; the shape is generic for future providers. */
1414
+ embedType: 'youtube';
1415
+ /** The provider-specific video ID (e.g. the 11-character YouTube video ID). */
1416
+ videoId: string;
1417
+ /** The original/canonical URL of the embedded media, if known. */
1418
+ url?: string;
1419
+ /** Display width, as a CSS length or percentage. */
1420
+ width?: string;
1421
+ /** Layout alignment of the embed. */
1422
+ align?: 'left' | 'center' | 'right';
1423
+ }
1424
+ /**
1425
+ * Metadata for an admonition/alert node (e.g. GitHub's `> [!NOTE]` or GLFM's `:::note`).
1426
+ * `MarkdownParser` accepts both syntaxes; `MarkdownGenerator` only ever writes the
1427
+ * blockquote form. Children are block content (paragraphs) wrapped by the admonition.
1428
+ */
1429
+ export interface AdmonitionMetadata {
1430
+ admonitionType: 'note' | 'tip' | 'important' | 'warning' | 'caution';
1431
+ /** Optional custom title; falls back to the type label. */
1432
+ title?: string;
1337
1433
  }
1338
1434
  /**
1339
1435
  * Metadata for PDF page nodes.
@@ -1364,6 +1460,25 @@ export interface TextMetadata {
1364
1460
  * - 'external': Link to an external URL
1365
1461
  */
1366
1462
  linkType?: 'internal' | 'external';
1463
+ /**
1464
+ * When set, this text is an abbreviation and this is its full-form expansion,
1465
+ * rendered as `<abbr title="...">`. Populated from Markdown Extra's
1466
+ * `*[HTML]: Hypertext Markup Language` syntax or an HTML `<abbr>` tag.
1467
+ */
1468
+ abbreviationTitle?: string;
1469
+ /**
1470
+ * When set, this text is a Pandoc/MultiMarkdown-style citation reference
1471
+ * (`[@citekey]`), and this is the bare citekey (e.g. "smith2024"). Bibliography
1472
+ * resolution (author/year display, .bib management) is left to the consuming app.
1473
+ */
1474
+ citationKey?: string;
1475
+ /**
1476
+ * True when this is an Obsidian-style wikilink (`[[page]]` / `[[page|alias]]`).
1477
+ * `link` holds the bare page name and `linkType` is always 'internal'; the
1478
+ * per-workspace enable/disable toggle lives in markdownwriter, not here -
1479
+ * officeParser always parses/generates the syntax.
1480
+ */
1481
+ wikilink?: boolean;
1367
1482
  }
1368
1483
  /**
1369
1484
  * Metadata for note nodes (footnotes/endnotes).
@@ -1417,6 +1532,12 @@ export interface CodeMetadata {
1417
1532
  language?: string;
1418
1533
  /** Unique anchor IDs for internal linking. */
1419
1534
  anchorIds?: string[];
1535
+ /**
1536
+ * When set, this node is a LaTeX math expression rather than a code block. `node.text`
1537
+ * holds the bare LaTeX (delimiters excluded); 'inline' round-trips as `$...$`,
1538
+ * 'block' as `$$...$$`. Matches inscript-editor's math node (Roadmap Step 11.5).
1539
+ */
1540
+ math?: 'inline' | 'block';
1420
1541
  }
1421
1542
  /**
1422
1543
  * Metadata for a comment/annotation.
@@ -1436,7 +1557,7 @@ export interface HeaderFooterMetadata {
1436
1557
  /**
1437
1558
  * Union type for content metadata.
1438
1559
  */
1439
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
1560
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | EmbedMetadata | AdmonitionMetadata | undefined;
1440
1561
  /**
1441
1562
  * Represents a node in the document content tree.
1442
1563
  * This is the core building block of the parsed document structure.
@@ -1598,6 +1719,21 @@ export type OfficeContentNode = BaseContentNode & ({
1598
1719
  } | {
1599
1720
  type: 'slideMaster';
1600
1721
  metadata?: SlideMetadata;
1722
+ } | {
1723
+ type: 'embed';
1724
+ metadata?: EmbedMetadata;
1725
+ } | {
1726
+ type: 'admonition';
1727
+ metadata?: AdmonitionMetadata;
1728
+ } | {
1729
+ type: 'definitionList';
1730
+ metadata?: undefined;
1731
+ } | {
1732
+ type: 'definitionTerm';
1733
+ metadata?: undefined;
1734
+ } | {
1735
+ type: 'definitionDescription';
1736
+ metadata?: undefined;
1601
1737
  });
1602
1738
  /**
1603
1739
  * Structured information extracted from a chart.
@@ -1872,4 +2008,7 @@ export interface OfficeParserAST {
1872
2008
  */
1873
2009
  to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1874
2010
  }
2011
+ declare global {
2012
+ const __SLIM__: boolean | undefined;
2013
+ }
1875
2014
  export {};
package/dist/types.js CHANGED
@@ -43,6 +43,8 @@ var OfficeErrorType;
43
43
  OfficeErrorType["ZIP_ENTRY_INVALID_SIZE"] = "ZIP_ENTRY_INVALID_SIZE";
44
44
  /** ZIP uncompressed size limit exceeded */
45
45
  OfficeErrorType["ZIP_SIZE_LIMIT_EXCEEDED"] = "ZIP_SIZE_LIMIT_EXCEEDED";
46
+ /** Document element/structure nesting exceeded the safe recursion depth */
47
+ OfficeErrorType["MAX_NESTING_DEPTH_EXCEEDED"] = "MAX_NESTING_DEPTH_EXCEEDED";
46
48
  /** Embedding call timed out */
47
49
  OfficeErrorType["EMBEDDING_TIMEOUT"] = "EMBEDDING_TIMEOUT";
48
50
  })(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
@@ -16,8 +16,8 @@ const ERRORHEADER = "[OfficeParser]: ";
16
16
  * Some entries are functions that take parameters to build dynamic messages.
17
17
  */
18
18
  const ERROR_MESSAGES = {
19
- [types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
20
- [types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks.`,
19
+ [types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv, epub files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
20
+ [types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks, epub.`,
21
21
  [types_js_1.OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
22
22
  [types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
23
23
  [types_js_1.OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
@@ -34,6 +34,7 @@ const ERROR_MESSAGES = {
34
34
  [types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
35
35
  [types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
36
36
  [types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
37
+ [types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED]: `Document nesting depth exceeded the safe limit (possible denial-of-service input)`,
37
38
  [types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
38
39
  };
39
40
  /**
@@ -0,0 +1,99 @@
1
+ /**
2
+ * Shared output-sanitization helpers.
3
+ *
4
+ * Every string in the parsed AST originates from an untrusted document, so any
5
+ * value interpolated into generated output (HTML, XHTML, CSS, URLs, inline
6
+ * scripts, CSV, RTF, Markdown) must be escaped for its destination context.
7
+ * These are the single source of truth — each generator delegates to them so
8
+ * escaping stays consistent and a gap fixed here is fixed everywhere.
9
+ */
10
+ /**
11
+ * Escapes text for an HTML text node or a double-quoted attribute value.
12
+ * Includes the single quote so the result is also safe inside single-quoted
13
+ * attributes.
14
+ */
15
+ export declare function escapeHtml(text: string): string;
16
+ /**
17
+ * Escapes text for an XML text node or attribute (XHTML/OPF/NCX). Same as
18
+ * escapeHtml but emits the XML-canonical `&apos;` for the single quote.
19
+ */
20
+ export declare function escapeXml(text: string): string;
21
+ /**
22
+ * Sanitizes a single CSS value (e.g. a color/size/font/alignment pulled from a
23
+ * document) for placement inside a `style="prop: VALUE"` attribute.
24
+ *
25
+ * - Drops the whole value if it contains a resource-fetching or executing
26
+ * construct (`url()`, `expression()`, `@import`, `image-set()`, `javascript:`)
27
+ * or angle brackets that could break out of the attribute/tag.
28
+ * - Strips characters that break out of `prop: value` (`;`, quotes), out of a
29
+ * `<style>` rule (`{}`), CSS escapes (`\`), and control characters.
30
+ *
31
+ * `rgb()/hsl()` and hex/named colors, lengths, and (unquoted) font names all
32
+ * survive; the trade-off is that legitimately quoted font names lose their
33
+ * quotes, which browsers tolerate.
34
+ */
35
+ export declare function sanitizeCssValue(value: string): string;
36
+ /**
37
+ * Escapes a document-supplied URL for use in an href/src attribute. Beyond the
38
+ * usual attribute escaping, this rejects script-executing schemes (javascript:,
39
+ * vbscript:, data:, etc.) so a hyperlink extracted from an untrusted document
40
+ * can't run code when clicked — only http(s)/mailto/tel and relative/fragment
41
+ * URLs are passed through.
42
+ */
43
+ export declare function sanitizeUrl(url: string): string;
44
+ /**
45
+ * Like sanitizeUrl but for an <img>/<source> src: additionally permits
46
+ * `data:image/*` URIs (embedded document images) while still rejecting
47
+ * script-executing schemes and non-image data URIs (e.g. data:text/html).
48
+ */
49
+ export declare function sanitizeImageUrl(url: string): string;
50
+ /**
51
+ * Serializes data for embedding inside an inline <script> block. JSON.stringify
52
+ * alone doesn't escape "<", so a value containing "</script>" (e.g. a chart
53
+ * label from attacker-controlled document XML) would close the script early and
54
+ * inject markup. Also escapes the U+2028/U+2029 line separators, which are
55
+ * invalid in JS string literals.
56
+ */
57
+ export declare function serializeForInlineScript(data: unknown): string;
58
+ /**
59
+ * Formats a value for a CSV field: guards against spreadsheet formula/DDE
60
+ * injection (CWE-1236) and applies RFC 4180 quoting.
61
+ *
62
+ * A cell beginning with `= + - @` (or a tab/CR that some apps treat as a
63
+ * formula start) is prefixed with a single quote so Excel/Sheets render it as
64
+ * literal text rather than executing it. Genuine numbers (including negatives)
65
+ * are exempt so numeric columns are preserved.
66
+ */
67
+ export declare function csvSafeCell(value: string, delimiter: string): string;
68
+ /**
69
+ * Escapes text for RTF: neutralizes the control/group metacharacters `\ { }`
70
+ * (which would otherwise inject RTF control words or groups), encodes the double
71
+ * quote (so a hyperlink field argument can't be terminated early), and hex/unicode
72
+ * encodes non-ASCII characters.
73
+ */
74
+ export declare function escapeRtf(text: string): string;
75
+ /**
76
+ * Escapes document text for a Markdown text position. Markdown passes raw HTML
77
+ * through to the renderer, so a `<` that begins an HTML tag or comment must be
78
+ * neutralized to prevent `<script>`/`<img onerror>` injection when the Markdown
79
+ * is later rendered to HTML.
80
+ *
81
+ * Deliberately narrow — only a `<` immediately followed by a letter, `/`, `!` or
82
+ * `?` (i.e. one that actually opens a tag/comment/PI, matching how browsers
83
+ * detect tags) is encoded. A bare `<` (e.g. `a < b`), `>`, `&`, `[]` and other
84
+ * Markdown metacharacters are left untouched: they can't start a tag, and
85
+ * MarkdownParser round-trips this output without decoding entities, so encoding
86
+ * them would corrupt re-parsed content. URL schemes are handled by
87
+ * sanitizeMarkdownUrl.
88
+ */
89
+ export declare function markdownEscapeText(text: string): string;
90
+ /**
91
+ * Sanitizes a document-supplied URL for a Markdown `[text](url)` / `![alt](url)`
92
+ * target. Rejects script-executing schemes (returning '' → a dead link) and
93
+ * percent-encodes the characters that would break out of the `(...)` or inject
94
+ * markup. `&` is preserved so query strings survive; set `allowDataImage` for
95
+ * image targets so embedded `data:image/*` URIs are permitted.
96
+ */
97
+ export declare function sanitizeMarkdownUrl(url: string, opts?: {
98
+ allowDataImage?: boolean;
99
+ }): string;
@@ -0,0 +1,228 @@
1
+ "use strict";
2
+ /**
3
+ * Shared output-sanitization helpers.
4
+ *
5
+ * Every string in the parsed AST originates from an untrusted document, so any
6
+ * value interpolated into generated output (HTML, XHTML, CSS, URLs, inline
7
+ * scripts, CSV, RTF, Markdown) must be escaped for its destination context.
8
+ * These are the single source of truth — each generator delegates to them so
9
+ * escaping stays consistent and a gap fixed here is fixed everywhere.
10
+ */
11
+ Object.defineProperty(exports, "__esModule", { value: true });
12
+ exports.escapeHtml = escapeHtml;
13
+ exports.escapeXml = escapeXml;
14
+ exports.sanitizeCssValue = sanitizeCssValue;
15
+ exports.sanitizeUrl = sanitizeUrl;
16
+ exports.sanitizeImageUrl = sanitizeImageUrl;
17
+ exports.serializeForInlineScript = serializeForInlineScript;
18
+ exports.csvSafeCell = csvSafeCell;
19
+ exports.escapeRtf = escapeRtf;
20
+ exports.markdownEscapeText = markdownEscapeText;
21
+ exports.sanitizeMarkdownUrl = sanitizeMarkdownUrl;
22
+ /**
23
+ * Escapes text for an HTML text node or a double-quoted attribute value.
24
+ * Includes the single quote so the result is also safe inside single-quoted
25
+ * attributes.
26
+ */
27
+ function escapeHtml(text) {
28
+ if (typeof text !== 'string')
29
+ return text;
30
+ return text
31
+ .replace(/&/g, '&amp;')
32
+ .replace(/</g, '&lt;')
33
+ .replace(/>/g, '&gt;')
34
+ .replace(/"/g, '&quot;')
35
+ .replace(/'/g, '&#39;');
36
+ }
37
+ /**
38
+ * Escapes text for an XML text node or attribute (XHTML/OPF/NCX). Same as
39
+ * escapeHtml but emits the XML-canonical `&apos;` for the single quote.
40
+ */
41
+ function escapeXml(text) {
42
+ if (typeof text !== 'string')
43
+ return '';
44
+ return text
45
+ .replace(/&/g, '&amp;')
46
+ .replace(/</g, '&lt;')
47
+ .replace(/>/g, '&gt;')
48
+ .replace(/"/g, '&quot;')
49
+ .replace(/'/g, '&apos;');
50
+ }
51
+ /**
52
+ * Sanitizes a single CSS value (e.g. a color/size/font/alignment pulled from a
53
+ * document) for placement inside a `style="prop: VALUE"` attribute.
54
+ *
55
+ * - Drops the whole value if it contains a resource-fetching or executing
56
+ * construct (`url()`, `expression()`, `@import`, `image-set()`, `javascript:`)
57
+ * or angle brackets that could break out of the attribute/tag.
58
+ * - Strips characters that break out of `prop: value` (`;`, quotes), out of a
59
+ * `<style>` rule (`{}`), CSS escapes (`\`), and control characters.
60
+ *
61
+ * `rgb()/hsl()` and hex/named colors, lengths, and (unquoted) font names all
62
+ * survive; the trade-off is that legitimately quoted font names lose their
63
+ * quotes, which browsers tolerate.
64
+ */
65
+ function sanitizeCssValue(value) {
66
+ if (typeof value !== 'string')
67
+ return '';
68
+ // Strip control chars and CSS comments FIRST, then test for dangerous constructs.
69
+ // Order matters: a payload like "u\nrl(" or "url/*x*/(" would survive the test if
70
+ // tested before removal, then reassemble into "url(" once the noise is stripped.
71
+ const cleaned = value
72
+ .replace(/[\x00-\x1F\x7F]/g, '') // control chars (incl. newlines/tabs)
73
+ .replace(/\/\*[\s\S]*?\*\//g, ''); // CSS comments used to obfuscate
74
+ if (/(?:url|expression|image-set|element|-moz-binding)\s*\(|@import|javascript:|[<>]/i.test(cleaned)) {
75
+ return '';
76
+ }
77
+ return cleaned.replace(/[;{}"'`\\]/g, '').trim();
78
+ }
79
+ /**
80
+ * Escapes a document-supplied URL for use in an href/src attribute. Beyond the
81
+ * usual attribute escaping, this rejects script-executing schemes (javascript:,
82
+ * vbscript:, data:, etc.) so a hyperlink extracted from an untrusted document
83
+ * can't run code when clicked — only http(s)/mailto/tel and relative/fragment
84
+ * URLs are passed through.
85
+ */
86
+ function sanitizeUrl(url) {
87
+ if (typeof url !== 'string')
88
+ return '';
89
+ const trimmed = url.trim();
90
+ // Browsers ignore control characters when parsing a URL scheme, so strip them
91
+ // first to catch obfuscated payloads like "java\tscript:alert(1)".
92
+ const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
93
+ const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
94
+ if (schemeMatch && !/^(https?|mailto|tel)$/i.test(schemeMatch[1])) {
95
+ return '';
96
+ }
97
+ // Emit the same normalized string that was validated.
98
+ return escapeHtml(stripped);
99
+ }
100
+ /**
101
+ * Like sanitizeUrl but for an <img>/<source> src: additionally permits
102
+ * `data:image/*` URIs (embedded document images) while still rejecting
103
+ * script-executing schemes and non-image data URIs (e.g. data:text/html).
104
+ */
105
+ function sanitizeImageUrl(url) {
106
+ if (typeof url !== 'string')
107
+ return '';
108
+ const trimmed = url.trim();
109
+ const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
110
+ const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
111
+ if (schemeMatch) {
112
+ const scheme = schemeMatch[1].toLowerCase();
113
+ if (scheme === 'data') {
114
+ if (!/^data:image\//i.test(stripped))
115
+ return '';
116
+ }
117
+ else if (scheme !== 'http' && scheme !== 'https') {
118
+ return '';
119
+ }
120
+ }
121
+ return escapeHtml(stripped);
122
+ }
123
+ /**
124
+ * Serializes data for embedding inside an inline <script> block. JSON.stringify
125
+ * alone doesn't escape "<", so a value containing "</script>" (e.g. a chart
126
+ * label from attacker-controlled document XML) would close the script early and
127
+ * inject markup. Also escapes the U+2028/U+2029 line separators, which are
128
+ * invalid in JS string literals.
129
+ */
130
+ function serializeForInlineScript(data) {
131
+ // U+2028/U+2029 (line/paragraph separators) are valid in JSON but break
132
+ // JS string literals; reference them by code point to keep the source ASCII.
133
+ const lineSep = String.fromCharCode(0x2028);
134
+ const paraSep = String.fromCharCode(0x2029);
135
+ return JSON.stringify(data)
136
+ .replace(/</g, '\\u003C')
137
+ .replace(/>/g, '\\u003E')
138
+ .split(lineSep).join('\\u2028')
139
+ .split(paraSep).join('\\u2029');
140
+ }
141
+ /**
142
+ * Formats a value for a CSV field: guards against spreadsheet formula/DDE
143
+ * injection (CWE-1236) and applies RFC 4180 quoting.
144
+ *
145
+ * A cell beginning with `= + - @` (or a tab/CR that some apps treat as a
146
+ * formula start) is prefixed with a single quote so Excel/Sheets render it as
147
+ * literal text rather than executing it. Genuine numbers (including negatives)
148
+ * are exempt so numeric columns are preserved.
149
+ */
150
+ function csvSafeCell(value, delimiter) {
151
+ let v = typeof value === 'string' ? value : String(value ?? '');
152
+ // A plain signed number (e.g. "-8", "+7", "-5.3") can't be a formula, so exempt it —
153
+ // otherwise numeric columns get quoted as text. Anything else starting with a formula
154
+ // trigger (including "+1+1", "-1+cmd", "=", "@") is prefixed with a quote.
155
+ const isNumber = /^[+-]?(?:\d+\.?\d*|\.\d+)(?:[eE][+-]?\d+)?$/.test(v.trim());
156
+ if (!isNumber && /^[=+\-@\t\r]/.test(v)) {
157
+ v = `'${v}`;
158
+ }
159
+ if (v.includes(delimiter) || v.includes('"') || v.includes('\n') || v.includes('\r')) {
160
+ return `"${v.replace(/"/g, '""')}"`;
161
+ }
162
+ return v;
163
+ }
164
+ /**
165
+ * Escapes text for RTF: neutralizes the control/group metacharacters `\ { }`
166
+ * (which would otherwise inject RTF control words or groups), encodes the double
167
+ * quote (so a hyperlink field argument can't be terminated early), and hex/unicode
168
+ * encodes non-ASCII characters.
169
+ */
170
+ function escapeRtf(text) {
171
+ if (typeof text !== 'string')
172
+ return '';
173
+ return text
174
+ .replace(/\\/g, '\\\\')
175
+ .replace(/{/g, '\\{')
176
+ .replace(/}/g, '\\}')
177
+ .replace(/"/g, "\\'22")
178
+ .replace(/[^\x00-\x7F]/g, (match) => {
179
+ let code = match.charCodeAt(0);
180
+ if (code < 256) {
181
+ return `\\'${code.toString(16).padStart(2, '0')}`;
182
+ }
183
+ if (code > 32767) {
184
+ code -= 65536;
185
+ }
186
+ return `{\\uc0\\u${code}}`;
187
+ });
188
+ }
189
+ /**
190
+ * Escapes document text for a Markdown text position. Markdown passes raw HTML
191
+ * through to the renderer, so a `<` that begins an HTML tag or comment must be
192
+ * neutralized to prevent `<script>`/`<img onerror>` injection when the Markdown
193
+ * is later rendered to HTML.
194
+ *
195
+ * Deliberately narrow — only a `<` immediately followed by a letter, `/`, `!` or
196
+ * `?` (i.e. one that actually opens a tag/comment/PI, matching how browsers
197
+ * detect tags) is encoded. A bare `<` (e.g. `a < b`), `>`, `&`, `[]` and other
198
+ * Markdown metacharacters are left untouched: they can't start a tag, and
199
+ * MarkdownParser round-trips this output without decoding entities, so encoding
200
+ * them would corrupt re-parsed content. URL schemes are handled by
201
+ * sanitizeMarkdownUrl.
202
+ */
203
+ function markdownEscapeText(text) {
204
+ if (typeof text !== 'string')
205
+ return '';
206
+ return text.replace(/<(?=[a-zA-Z/!?])/g, '&lt;');
207
+ }
208
+ /**
209
+ * Sanitizes a document-supplied URL for a Markdown `[text](url)` / `![alt](url)`
210
+ * target. Rejects script-executing schemes (returning '' → a dead link) and
211
+ * percent-encodes the characters that would break out of the `(...)` or inject
212
+ * markup. `&` is preserved so query strings survive; set `allowDataImage` for
213
+ * image targets so embedded `data:image/*` URIs are permitted.
214
+ */
215
+ function sanitizeMarkdownUrl(url, opts) {
216
+ if (typeof url !== 'string')
217
+ return '';
218
+ const stripped = url.trim().replace(/[\x00-\x1F\x7F]+/g, '');
219
+ const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
220
+ if (schemeMatch) {
221
+ const scheme = schemeMatch[1].toLowerCase();
222
+ const ok = /^(?:https?|mailto|tel)$/.test(scheme)
223
+ || (opts?.allowDataImage === true && /^data:image\//i.test(stripped));
224
+ if (!ok)
225
+ return '';
226
+ }
227
+ return stripped.replace(/[\s()<>"`\\]/g, (c) => '%' + c.charCodeAt(0).toString(16).toUpperCase().padStart(2, '0'));
228
+ }