officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -8,6 +8,37 @@ exports.isValidContainerWidth = isValidContainerWidth;
8
8
  const defaults_js_1 = require("../defaults.js");
9
9
  const types_js_1 = require("../types.js");
10
10
  const errorUtils_js_1 = require("./errorUtils.js");
11
+ /**
12
+ * Keys that must never be copied from a caller-supplied config onto one of our objects.
13
+ *
14
+ * A config that arrived via `JSON.parse` can carry `__proto__` as a genuine **own enumerable**
15
+ * property (an object *literal* cannot - there `__proto__` invokes the setter at parse time),
16
+ * which is exactly the shape of a host application accepting a JSON config blob. Copying that
17
+ * key reaches `Object.prototype` and corrupts every object in the process.
18
+ *
19
+ * `constructor` and `prototype` are included because they are the other two names that reach a
20
+ * prototype through an ordinary property write.
21
+ */
22
+ const PROTOTYPE_POLLUTION_KEYS = new Set(['__proto__', 'constructor', 'prototype']);
23
+ /**
24
+ * Returns a copy of `source` with prototype-reaching keys removed.
25
+ *
26
+ * Needed before `Object.assign`, which does **not** pollute `Object.prototype` (it writes via
27
+ * `[[Set]]`, so `__proto__` invokes the inherited setter rather than creating an own property) -
28
+ * but that setter is not inert: it **replaces the target's prototype**, so the returned config
29
+ * silently inherits attacker-chosen properties for every field the defaults don't set as an own
30
+ * property. Narrower than global pollution, still wrong. Do not "simplify" this away on the
31
+ * grounds that `Object.assign` is safe; it is safe only against the *global* variant.
32
+ */
33
+ function withoutPrototypeKeys(source) {
34
+ const safe = {};
35
+ for (const key of Object.keys(source)) {
36
+ if (PROTOTYPE_POLLUTION_KEYS.has(key))
37
+ continue;
38
+ safe[key] = source[key];
39
+ }
40
+ return safe;
41
+ }
11
42
  /**
12
43
  * Deep clones an object, specifically handling arrays and plain objects.
13
44
  */
@@ -61,6 +92,7 @@ function resolveParserConfig(userConfig) {
61
92
  userConfig.decompressionLimits = {
62
93
  maxUncompressedBytes: 512 * 1024 * 1024,
63
94
  maxZipEntries: 10000,
95
+ maxTableCells: 1000000,
64
96
  };
65
97
  }
66
98
  return userConfig;
@@ -71,15 +103,22 @@ function resolveParserConfig(userConfig) {
71
103
  return config;
72
104
  }
73
105
  // 2. Merge user config
74
- // We handle ocrConfig and decompressionLimits specially to avoid shallow-overwriting the whole objects
75
- const { ocrConfig, decompressionLimits, ...rest } = userConfig;
76
- Object.assign(config, rest);
106
+ // We handle ocrConfig, decompressionLimits, and htmlParserConfig specially to avoid
107
+ // shallow-overwriting the whole nested objects
108
+ const { ocrConfig, decompressionLimits, htmlParserConfig, ...rest } = userConfig;
109
+ Object.assign(config, withoutPrototypeKeys(rest));
77
110
  if (decompressionLimits) {
78
111
  config.decompressionLimits = {
79
112
  ...config.decompressionLimits,
80
113
  ...decompressionLimits,
81
114
  };
82
115
  }
116
+ if (htmlParserConfig) {
117
+ config.htmlParserConfig = {
118
+ ...config.htmlParserConfig,
119
+ ...htmlParserConfig,
120
+ };
121
+ }
83
122
  if (ocrConfig) {
84
123
  const { timeout, ...ocrRest } = ocrConfig;
85
124
  config.ocrConfig = {
@@ -124,12 +163,22 @@ function resolveGeneratorConfig(destination, astConfig, userConfig) {
124
163
  if (userConfig) {
125
164
  // Extract sub-configs to avoid shallow-overwriting the whole sub-config objects
126
165
  const { htmlConfig, mdConfig, pdfConfig, csvConfig, textConfig, chunksConfig, ...commonProps } = userConfig;
127
- Object.assign(config, commonProps);
166
+ Object.assign(config, withoutPrototypeKeys(commonProps));
128
167
  // Merge sub-configs individually, ignoring undefined properties to preserve defaults
129
168
  const mergeSubConfig = (target, source) => {
130
169
  if (!source)
131
170
  return;
132
171
  for (const key in source) {
172
+ // Both guards are load-bearing and neither subsumes the other. The own-property
173
+ // check (matching deepClone above) stops inherited enumerable properties, which
174
+ // matters once anything else in the process has already polluted a prototype. It
175
+ // does NOT stop this attack on its own: `JSON.parse('{"__proto__":{...}}')` yields
176
+ // `__proto__` as an own enumerable key, so it passes hasOwnProperty and would be
177
+ // written straight through to Object.prototype by the recursion below.
178
+ if (!Object.prototype.hasOwnProperty.call(source, key))
179
+ continue;
180
+ if (PROTOTYPE_POLLUTION_KEYS.has(key))
181
+ continue;
133
182
  if (source[key] !== undefined) {
134
183
  // Deep merge plain objects (like injections or margin)
135
184
  if (typeof source[key] === 'object' &&
@@ -16,8 +16,8 @@ const ERRORHEADER = "[OfficeParser]: ";
16
16
  * Some entries are functions that take parameters to build dynamic messages.
17
17
  */
18
18
  const ERROR_MESSAGES = {
19
- [types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
20
- [types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks.`,
19
+ [types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv, epub files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
20
+ [types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks, epub.`,
21
21
  [types_js_1.OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
22
22
  [types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
23
23
  [types_js_1.OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
@@ -34,6 +34,7 @@ const ERROR_MESSAGES = {
34
34
  [types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
35
35
  [types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
36
36
  [types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
37
+ [types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED]: `Document nesting depth exceeded the safe limit (possible denial-of-service input)`,
37
38
  [types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
38
39
  };
39
40
  /**
@@ -56,7 +57,10 @@ const WARNING_MESSAGES = {
56
57
  [types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED]: `Auto-detection of file type failed. This can happen on older Node.js versions with modern file-type versions. Please provide the 'fileType' hint in the configuration if parsing fails.`,
57
58
  [types_js_1.OfficeWarningType.EMPTY_CHUNK_GENERATED]: (strategy) => `No chunks generated for document. Check if the document content is compatible with the '${strategy}' strategy.`,
58
59
  [types_js_1.OfficeWarningType.WHITESPACE_NODE_SKIPPED]: (nodeType) => `Skipped whitespace-only node of type: ${nodeType}`,
59
- [types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH]: (val) => `Invalid HTML containerWidth: ${JSON.stringify(val)}. Falling back to "auto". Width must be a positive number, a valid CSS length string (e.g., "900px", "100%", "50vw"), or "auto".`
60
+ [types_js_1.OfficeWarningType.TABLE_CELL_LIMIT_EXCEEDED]: (limit) => `Table cell limit (${limit}) reached while expanding repeated ODF cells/rows; the remaining cells were not materialized. A few hundred bytes of XML can request an unbounded number of cells via table:number-columns-repeated / table:number-rows-repeated, so this is capped. Raise decompressionLimits.maxTableCells if your documents legitimately exceed it.`,
61
+ [types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH]: (val) => `Invalid HTML containerWidth: ${JSON.stringify(val)}. Falling back to "auto". Width must be a positive number, a valid CSS length string (e.g., "900px", "100%", "50vw"), or "auto".`,
62
+ [types_js_1.OfficeWarningType.METADATA_NOT_REPRESENTABLE]: (info) => `Custom metadata ${info.keys.map(k => `'${k}'`).join(', ')} could not be written to ${info.format} output: the format has a fixed metadata vocabulary with no place for caller-defined keys. The named metadata fields (title, author, etc.) were still applied.`,
63
+ [types_js_1.OfficeWarningType.INVALID_STYLE_MAP_TAG]: (tag) => `styleMap output.tag ${JSON.stringify(tag)} is not an allowed element name and was ignored; the node's default tag was used instead. A tag name is written into both the opening and closing tag, so only a known-safe set of block, heading and inline elements is accepted.`
60
64
  };
61
65
  /**
62
66
  * Creates a formatted warning message for a specific warning type.
@@ -0,0 +1,139 @@
1
+ /**
2
+ * Shared output-sanitization helpers.
3
+ *
4
+ * Every string in the parsed AST originates from an untrusted document, so any
5
+ * value interpolated into generated output (HTML, XHTML, CSS, URLs, inline
6
+ * scripts, CSV, RTF, Markdown) must be escaped for its destination context.
7
+ * These are the single source of truth — each generator delegates to them so
8
+ * escaping stays consistent and a gap fixed here is fixed everywhere.
9
+ */
10
+ /**
11
+ * Whether a string is a plain HTML attribute *name*, safe to interpolate before `="..."`.
12
+ *
13
+ * Escaping the value is not enough on its own: a key containing a quote or `=` closes the
14
+ * attribute and opens another, so `x" onmouseover="alert(1)" z` yields a real event handler no
15
+ * matter how carefully the value is escaped. This is the shape an attribute-injection payload
16
+ * takes, and rejecting it outright is simpler and safer than trying to escape a name.
17
+ *
18
+ * The predicate is shared rather than restated because it is now applied at four independent
19
+ * points (the parser's attribute collection, the generator's attribute bag, and two
20
+ * styleMap-driven paths). Each of those still keeps its own skip-list inline: the lists are the
21
+ * same policy expressed for different layers, and collapsing them would erase the defence in
22
+ * depth the surrounding comments describe.
23
+ */
24
+ export declare function isSafeHtmlAttributeName(name: string): boolean;
25
+ /**
26
+ * Whether a `styleMap` `output.tag` may be emitted as an element name.
27
+ *
28
+ * Callers must fall back to their default tag when this returns false, never emit the value.
29
+ */
30
+ export declare function isSafeStyleMapTag(tag: unknown): tag is string;
31
+ /**
32
+ * Escapes text for an HTML text node or a double-quoted attribute value.
33
+ * Includes the single quote so the result is also safe inside single-quoted
34
+ * attributes.
35
+ */
36
+ export declare function escapeHtml(text: string): string;
37
+ /**
38
+ * Escapes text for an XML text node or attribute (XHTML/OPF/NCX). Same as
39
+ * escapeHtml but emits the XML-canonical `'` for the single quote.
40
+ */
41
+ export declare function escapeXml(text: string): string;
42
+ /**
43
+ * Sanitizes a single CSS value (e.g. a color/size/font/alignment pulled from a
44
+ * document) for placement inside a `style="prop: VALUE"` attribute.
45
+ *
46
+ * - Drops the whole value if it contains a resource-fetching or executing
47
+ * construct (`url()`, `expression()`, `@import`, `image-set()`, `javascript:`)
48
+ * or angle brackets that could break out of the attribute/tag.
49
+ * - Strips characters that break out of `prop: value` (`;`, quotes), out of a
50
+ * `<style>` rule (`{}`), CSS escapes (`\`), and control characters.
51
+ *
52
+ * `rgb()/hsl()` and hex/named colors, lengths, and (unquoted) font names all
53
+ * survive; the trade-off is that legitimately quoted font names lose their
54
+ * quotes, which browsers tolerate.
55
+ */
56
+ export declare function sanitizeCssValue(value: string): string;
57
+ /**
58
+ * Escapes a document-supplied URL for use in an href/src attribute. Beyond the
59
+ * usual attribute escaping, this rejects script-executing schemes (javascript:,
60
+ * vbscript:, data:, etc.) so a hyperlink extracted from an untrusted document
61
+ * can't run code when clicked — only http(s)/mailto/tel and relative/fragment
62
+ * URLs are passed through.
63
+ */
64
+ export declare function sanitizeUrl(url: string): string;
65
+ /**
66
+ * Like sanitizeUrl but for an <img>/<source> src: additionally permits
67
+ * `data:image/*` URIs (embedded document images) while still rejecting
68
+ * script-executing schemes and non-image data URIs (e.g. data:text/html).
69
+ */
70
+ export declare function sanitizeImageUrl(url: string): string;
71
+ /**
72
+ * Serializes data for embedding inside an inline <script> block. JSON.stringify
73
+ * alone doesn't escape "<", so a value containing "</script>" (e.g. a chart
74
+ * label from attacker-controlled document XML) would close the script early and
75
+ * inject markup. Also escapes the U+2028/U+2029 line separators, which are
76
+ * invalid in JS string literals.
77
+ */
78
+ export declare function serializeForInlineScript(data: unknown): string;
79
+ /**
80
+ * Formats a value for a CSV field: guards against spreadsheet formula/DDE
81
+ * injection (CWE-1236) and applies RFC 4180 quoting.
82
+ *
83
+ * A cell beginning with `= + - @` (or a tab/CR that some apps treat as a
84
+ * formula start) is prefixed with a single quote so Excel/Sheets render it as
85
+ * literal text rather than executing it. Genuine numbers (including negatives)
86
+ * are exempt so numeric columns are preserved.
87
+ */
88
+ export declare function csvSafeCell(value: string, delimiter: string): string;
89
+ /**
90
+ * Validates and escapes a document-supplied URL for an RTF `HYPERLINK` field argument.
91
+ *
92
+ * Mirrors `sanitizeUrl`'s contract (validate the scheme, then encode for the destination, else
93
+ * return `''`) but cannot reuse it: `sanitizeUrl` returns `escapeHtml(...)`, which would emit
94
+ * `&amp;` into an RTF field. The scheme allowlist is deliberately identical to `sanitizeUrl`'s
95
+ * and `sanitizeMarkdownUrl`'s, so all three text generators agree on what a hyperlink may point at.
96
+ *
97
+ * **Additionally rejects UNC paths (`\\host\share`), which the HTML allowlist does not.** In a
98
+ * browser `\\evil.com\share` is an inert relative path; in Word it is a live UNC reference that
99
+ * triggers an SMB fetch and an NTLM handshake on click, which is a credential-leak vector rather
100
+ * than a rendering quirk. That asymmetry is why this is a separate function and not a flag on
101
+ * `sanitizeUrl` - the HTML helper must NOT gain this behaviour, since there the path is harmless
102
+ * and rejecting it would break legitimate relative links.
103
+ *
104
+ * Returns `''` for a rejected URL; callers emit the link text without the field wrapper, matching
105
+ * how HTML degrades to `href=""` and Markdown to `[text]()`.
106
+ */
107
+ export declare function sanitizeRtfUrl(url: string): string;
108
+ /**
109
+ * Escapes text for RTF: neutralizes the control/group metacharacters `\ { }`
110
+ * (which would otherwise inject RTF control words or groups), encodes the double
111
+ * quote (so a hyperlink field argument can't be terminated early), and hex/unicode
112
+ * encodes non-ASCII characters.
113
+ */
114
+ export declare function escapeRtf(text: string): string;
115
+ /**
116
+ * Escapes document text for a Markdown text position. Markdown passes raw HTML
117
+ * through to the renderer, so a `<` that begins an HTML tag or comment must be
118
+ * neutralized to prevent `<script>`/`<img onerror>` injection when the Markdown
119
+ * is later rendered to HTML.
120
+ *
121
+ * Deliberately narrow — only a `<` immediately followed by a letter, `/`, `!` or
122
+ * `?` (i.e. one that actually opens a tag/comment/PI, matching how browsers
123
+ * detect tags) is encoded. A bare `<` (e.g. `a < b`), `>`, `&`, `[]` and other
124
+ * Markdown metacharacters are left untouched: they can't start a tag, and
125
+ * MarkdownParser round-trips this output without decoding entities, so encoding
126
+ * them would corrupt re-parsed content. URL schemes are handled by
127
+ * sanitizeMarkdownUrl.
128
+ */
129
+ export declare function markdownEscapeText(text: string): string;
130
+ /**
131
+ * Sanitizes a document-supplied URL for a Markdown `[text](url)` / `![alt](url)`
132
+ * target. Rejects script-executing schemes (returning '' → a dead link) and
133
+ * percent-encodes the characters that would break out of the `(...)` or inject
134
+ * markup. `&` is preserved so query strings survive; set `allowDataImage` for
135
+ * image targets so embedded `data:image/*` URIs are permitted.
136
+ */
137
+ export declare function sanitizeMarkdownUrl(url: string, opts?: {
138
+ allowDataImage?: boolean;
139
+ }): string;
@@ -0,0 +1,318 @@
1
+ "use strict";
2
+ /**
3
+ * Shared output-sanitization helpers.
4
+ *
5
+ * Every string in the parsed AST originates from an untrusted document, so any
6
+ * value interpolated into generated output (HTML, XHTML, CSS, URLs, inline
7
+ * scripts, CSV, RTF, Markdown) must be escaped for its destination context.
8
+ * These are the single source of truth — each generator delegates to them so
9
+ * escaping stays consistent and a gap fixed here is fixed everywhere.
10
+ */
11
+ Object.defineProperty(exports, "__esModule", { value: true });
12
+ exports.isSafeHtmlAttributeName = isSafeHtmlAttributeName;
13
+ exports.isSafeStyleMapTag = isSafeStyleMapTag;
14
+ exports.escapeHtml = escapeHtml;
15
+ exports.escapeXml = escapeXml;
16
+ exports.sanitizeCssValue = sanitizeCssValue;
17
+ exports.sanitizeUrl = sanitizeUrl;
18
+ exports.sanitizeImageUrl = sanitizeImageUrl;
19
+ exports.serializeForInlineScript = serializeForInlineScript;
20
+ exports.csvSafeCell = csvSafeCell;
21
+ exports.sanitizeRtfUrl = sanitizeRtfUrl;
22
+ exports.escapeRtf = escapeRtf;
23
+ exports.markdownEscapeText = markdownEscapeText;
24
+ exports.sanitizeMarkdownUrl = sanitizeMarkdownUrl;
25
+ /**
26
+ * Whether a string is a plain HTML attribute *name*, safe to interpolate before `="..."`.
27
+ *
28
+ * Escaping the value is not enough on its own: a key containing a quote or `=` closes the
29
+ * attribute and opens another, so `x" onmouseover="alert(1)" z` yields a real event handler no
30
+ * matter how carefully the value is escaped. This is the shape an attribute-injection payload
31
+ * takes, and rejecting it outright is simpler and safer than trying to escape a name.
32
+ *
33
+ * The predicate is shared rather than restated because it is now applied at four independent
34
+ * points (the parser's attribute collection, the generator's attribute bag, and two
35
+ * styleMap-driven paths). Each of those still keeps its own skip-list inline: the lists are the
36
+ * same policy expressed for different layers, and collapsing them would erase the defence in
37
+ * depth the surrounding comments describe.
38
+ */
39
+ function isSafeHtmlAttributeName(name) {
40
+ return typeof name === 'string' && /^[a-zA-Z][a-zA-Z0-9-]*$/.test(name);
41
+ }
42
+ /**
43
+ * Element names a `styleMap` may map a node onto.
44
+ *
45
+ * An allowlist rather than a pattern: a tag name is interpolated into both `<TAG …>` and
46
+ * `</TAG>`, so it is not enough for it to *look* like a name - `script`, `style`, `iframe` and
47
+ * friends are perfectly well-formed names that would introduce an active context the rest of the
48
+ * generator's escaping assumes does not exist. This is the semantic set a style mapping is for:
49
+ * block containers, headings, and the inline emphasis elements.
50
+ */
51
+ const SAFE_STYLE_MAP_TAGS = new Set([
52
+ 'p', 'div', 'span', 'section', 'article', 'aside', 'header', 'footer', 'main', 'figure', 'figcaption',
53
+ 'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
54
+ 'blockquote', 'pre', 'code', 'q', 'cite', 'address',
55
+ 'ul', 'ol', 'li', 'dl', 'dt', 'dd',
56
+ 'b', 'strong', 'i', 'em', 'u', 'ins', 'del', 's', 'strike', 'mark', 'small', 'sub', 'sup', 'kbd', 'samp', 'var', 'abbr',
57
+ ]);
58
+ /**
59
+ * Whether a `styleMap` `output.tag` may be emitted as an element name.
60
+ *
61
+ * Callers must fall back to their default tag when this returns false, never emit the value.
62
+ */
63
+ function isSafeStyleMapTag(tag) {
64
+ return typeof tag === 'string' && SAFE_STYLE_MAP_TAGS.has(tag.toLowerCase());
65
+ }
66
+ /**
67
+ * Escapes text for an HTML text node or a double-quoted attribute value.
68
+ * Includes the single quote so the result is also safe inside single-quoted
69
+ * attributes.
70
+ */
71
+ function escapeHtml(text) {
72
+ if (typeof text !== 'string')
73
+ return text;
74
+ return text
75
+ .replace(/&/g, '&amp;')
76
+ .replace(/</g, '&lt;')
77
+ .replace(/>/g, '&gt;')
78
+ .replace(/"/g, '&quot;')
79
+ .replace(/'/g, '&#39;');
80
+ }
81
+ /**
82
+ * Escapes text for an XML text node or attribute (XHTML/OPF/NCX). Same as
83
+ * escapeHtml but emits the XML-canonical `&apos;` for the single quote.
84
+ */
85
+ function escapeXml(text) {
86
+ if (typeof text !== 'string')
87
+ return '';
88
+ return text
89
+ .replace(/&/g, '&amp;')
90
+ .replace(/</g, '&lt;')
91
+ .replace(/>/g, '&gt;')
92
+ .replace(/"/g, '&quot;')
93
+ .replace(/'/g, '&apos;');
94
+ }
95
+ /**
96
+ * Sanitizes a single CSS value (e.g. a color/size/font/alignment pulled from a
97
+ * document) for placement inside a `style="prop: VALUE"` attribute.
98
+ *
99
+ * - Drops the whole value if it contains a resource-fetching or executing
100
+ * construct (`url()`, `expression()`, `@import`, `image-set()`, `javascript:`)
101
+ * or angle brackets that could break out of the attribute/tag.
102
+ * - Strips characters that break out of `prop: value` (`;`, quotes), out of a
103
+ * `<style>` rule (`{}`), CSS escapes (`\`), and control characters.
104
+ *
105
+ * `rgb()/hsl()` and hex/named colors, lengths, and (unquoted) font names all
106
+ * survive; the trade-off is that legitimately quoted font names lose their
107
+ * quotes, which browsers tolerate.
108
+ */
109
+ function sanitizeCssValue(value) {
110
+ if (typeof value !== 'string')
111
+ return '';
112
+ // Strip every form of intra-token noise FIRST, then test for dangerous constructs.
113
+ // Order matters: a payload like "u\nrl(", "url/*x*/(" or "u\rl(" would survive the test if
114
+ // tested before removal, then reassemble into "url(" once the noise is stripped.
115
+ //
116
+ // The backslash strip belongs here, not after the test. CSS treats `\` as an escape a
117
+ // browser resolves away, so `u\rl(http://evil)` IS `url(http://evil)` to a renderer -
118
+ // stripping it downstream of the test meant the sanitizer handed back a live `url()` it
119
+ // had just declared safe. Every construct in the denylist is reachable this way
120
+ // (`expr\ession(`, `image\-set(`), so the fix is the ordering, not another pattern.
121
+ const cleaned = value
122
+ .replace(/[\x00-\x1F\x7F]/g, '') // control chars (incl. newlines/tabs)
123
+ .replace(/\/\*[\s\S]*?\*\//g, '') // CSS comments used to obfuscate
124
+ .replace(/\\/g, ''); // CSS escapes; see above
125
+ if (/(?:url|expression|image-set|element|-moz-binding)\s*\(|@import|javascript:|[<>]/i.test(cleaned)) {
126
+ return '';
127
+ }
128
+ // Backslash is already gone above; the rest still have work to do here.
129
+ return cleaned.replace(/[;{}"'`]/g, '').trim();
130
+ }
131
+ /**
132
+ * Escapes a document-supplied URL for use in an href/src attribute. Beyond the
133
+ * usual attribute escaping, this rejects script-executing schemes (javascript:,
134
+ * vbscript:, data:, etc.) so a hyperlink extracted from an untrusted document
135
+ * can't run code when clicked — only http(s)/mailto/tel and relative/fragment
136
+ * URLs are passed through.
137
+ */
138
+ function sanitizeUrl(url) {
139
+ if (typeof url !== 'string')
140
+ return '';
141
+ const trimmed = url.trim();
142
+ // Browsers ignore control characters when parsing a URL scheme, so strip them
143
+ // first to catch obfuscated payloads like "java\tscript:alert(1)".
144
+ const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
145
+ const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
146
+ if (schemeMatch && !/^(https?|mailto|tel)$/i.test(schemeMatch[1])) {
147
+ return '';
148
+ }
149
+ // Emit the same normalized string that was validated.
150
+ return escapeHtml(stripped);
151
+ }
152
+ /**
153
+ * Like sanitizeUrl but for an <img>/<source> src: additionally permits
154
+ * `data:image/*` URIs (embedded document images) while still rejecting
155
+ * script-executing schemes and non-image data URIs (e.g. data:text/html).
156
+ */
157
+ function sanitizeImageUrl(url) {
158
+ if (typeof url !== 'string')
159
+ return '';
160
+ const trimmed = url.trim();
161
+ const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
162
+ const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
163
+ if (schemeMatch) {
164
+ const scheme = schemeMatch[1].toLowerCase();
165
+ if (scheme === 'data') {
166
+ if (!/^data:image\//i.test(stripped))
167
+ return '';
168
+ }
169
+ else if (scheme !== 'http' && scheme !== 'https') {
170
+ return '';
171
+ }
172
+ }
173
+ return escapeHtml(stripped);
174
+ }
175
+ /**
176
+ * Serializes data for embedding inside an inline <script> block. JSON.stringify
177
+ * alone doesn't escape "<", so a value containing "</script>" (e.g. a chart
178
+ * label from attacker-controlled document XML) would close the script early and
179
+ * inject markup. Also escapes the U+2028/U+2029 line separators, which are
180
+ * invalid in JS string literals.
181
+ */
182
+ function serializeForInlineScript(data) {
183
+ // U+2028/U+2029 (line/paragraph separators) are valid in JSON but break
184
+ // JS string literals; reference them by code point to keep the source ASCII.
185
+ const lineSep = String.fromCharCode(0x2028);
186
+ const paraSep = String.fromCharCode(0x2029);
187
+ return JSON.stringify(data)
188
+ .replace(/</g, '\\u003C')
189
+ .replace(/>/g, '\\u003E')
190
+ .split(lineSep).join('\\u2028')
191
+ .split(paraSep).join('\\u2029');
192
+ }
193
+ /**
194
+ * Formats a value for a CSV field: guards against spreadsheet formula/DDE
195
+ * injection (CWE-1236) and applies RFC 4180 quoting.
196
+ *
197
+ * A cell beginning with `= + - @` (or a tab/CR that some apps treat as a
198
+ * formula start) is prefixed with a single quote so Excel/Sheets render it as
199
+ * literal text rather than executing it. Genuine numbers (including negatives)
200
+ * are exempt so numeric columns are preserved.
201
+ */
202
+ function csvSafeCell(value, delimiter) {
203
+ let v = typeof value === 'string' ? value : String(value ?? '');
204
+ // A plain signed number (e.g. "-8", "+7", "-5.3") can't be a formula, so exempt it —
205
+ // otherwise numeric columns get quoted as text. Anything else starting with a formula
206
+ // trigger (including "+1+1", "-1+cmd", "=", "@") is prefixed with a quote.
207
+ const isNumber = /^[+-]?(?:\d+\.?\d*|\.\d+)(?:[eE][+-]?\d+)?$/.test(v.trim());
208
+ // Test the trimmed value: the numeric exemption above already trims, so testing the raw
209
+ // string here meant a leading space slipped a trigger past the guard (" =1+1" was emitted
210
+ // unprefixed). Most spreadsheet apps treat a leading-space cell as text and would not
211
+ // evaluate it, so this is defence in depth rather than a demonstrated bypass - but the
212
+ // asymmetry between the two tests was an accident, not a decision.
213
+ if (!isNumber && /^[=+\-@\t\r]/.test(v.trim())) {
214
+ v = `'${v}`;
215
+ }
216
+ if (v.includes(delimiter) || v.includes('"') || v.includes('\n') || v.includes('\r')) {
217
+ return `"${v.replace(/"/g, '""')}"`;
218
+ }
219
+ return v;
220
+ }
221
+ /**
222
+ * Validates and escapes a document-supplied URL for an RTF `HYPERLINK` field argument.
223
+ *
224
+ * Mirrors `sanitizeUrl`'s contract (validate the scheme, then encode for the destination, else
225
+ * return `''`) but cannot reuse it: `sanitizeUrl` returns `escapeHtml(...)`, which would emit
226
+ * `&amp;` into an RTF field. The scheme allowlist is deliberately identical to `sanitizeUrl`'s
227
+ * and `sanitizeMarkdownUrl`'s, so all three text generators agree on what a hyperlink may point at.
228
+ *
229
+ * **Additionally rejects UNC paths (`\\host\share`), which the HTML allowlist does not.** In a
230
+ * browser `\\evil.com\share` is an inert relative path; in Word it is a live UNC reference that
231
+ * triggers an SMB fetch and an NTLM handshake on click, which is a credential-leak vector rather
232
+ * than a rendering quirk. That asymmetry is why this is a separate function and not a flag on
233
+ * `sanitizeUrl` - the HTML helper must NOT gain this behaviour, since there the path is harmless
234
+ * and rejecting it would break legitimate relative links.
235
+ *
236
+ * Returns `''` for a rejected URL; callers emit the link text without the field wrapper, matching
237
+ * how HTML degrades to `href=""` and Markdown to `[text]()`.
238
+ */
239
+ function sanitizeRtfUrl(url) {
240
+ if (typeof url !== 'string')
241
+ return '';
242
+ const trimmed = url.trim();
243
+ // Control characters are stripped before scheme matching for the same reason as sanitizeUrl:
244
+ // they are ignored when the target application parses the scheme.
245
+ const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
246
+ if (/^[\\/]{2}[^\\/]/.test(stripped))
247
+ return ''; // UNC (\\host\share, //host\share)
248
+ const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
249
+ if (schemeMatch && !/^(https?|mailto|tel)$/i.test(schemeMatch[1])) {
250
+ return '';
251
+ }
252
+ return escapeRtf(stripped);
253
+ }
254
+ /**
255
+ * Escapes text for RTF: neutralizes the control/group metacharacters `\ { }`
256
+ * (which would otherwise inject RTF control words or groups), encodes the double
257
+ * quote (so a hyperlink field argument can't be terminated early), and hex/unicode
258
+ * encodes non-ASCII characters.
259
+ */
260
+ function escapeRtf(text) {
261
+ if (typeof text !== 'string')
262
+ return '';
263
+ return text
264
+ .replace(/\\/g, '\\\\')
265
+ .replace(/{/g, '\\{')
266
+ .replace(/}/g, '\\}')
267
+ .replace(/"/g, "\\'22")
268
+ .replace(/[^\x00-\x7F]/g, (match) => {
269
+ let code = match.charCodeAt(0);
270
+ if (code < 256) {
271
+ return `\\'${code.toString(16).padStart(2, '0')}`;
272
+ }
273
+ if (code > 32767) {
274
+ code -= 65536;
275
+ }
276
+ return `{\\uc0\\u${code}}`;
277
+ });
278
+ }
279
+ /**
280
+ * Escapes document text for a Markdown text position. Markdown passes raw HTML
281
+ * through to the renderer, so a `<` that begins an HTML tag or comment must be
282
+ * neutralized to prevent `<script>`/`<img onerror>` injection when the Markdown
283
+ * is later rendered to HTML.
284
+ *
285
+ * Deliberately narrow — only a `<` immediately followed by a letter, `/`, `!` or
286
+ * `?` (i.e. one that actually opens a tag/comment/PI, matching how browsers
287
+ * detect tags) is encoded. A bare `<` (e.g. `a < b`), `>`, `&`, `[]` and other
288
+ * Markdown metacharacters are left untouched: they can't start a tag, and
289
+ * MarkdownParser round-trips this output without decoding entities, so encoding
290
+ * them would corrupt re-parsed content. URL schemes are handled by
291
+ * sanitizeMarkdownUrl.
292
+ */
293
+ function markdownEscapeText(text) {
294
+ if (typeof text !== 'string')
295
+ return '';
296
+ return text.replace(/<(?=[a-zA-Z/!?])/g, '&lt;');
297
+ }
298
+ /**
299
+ * Sanitizes a document-supplied URL for a Markdown `[text](url)` / `![alt](url)`
300
+ * target. Rejects script-executing schemes (returning '' → a dead link) and
301
+ * percent-encodes the characters that would break out of the `(...)` or inject
302
+ * markup. `&` is preserved so query strings survive; set `allowDataImage` for
303
+ * image targets so embedded `data:image/*` URIs are permitted.
304
+ */
305
+ function sanitizeMarkdownUrl(url, opts) {
306
+ if (typeof url !== 'string')
307
+ return '';
308
+ const stripped = url.trim().replace(/[\x00-\x1F\x7F]+/g, '');
309
+ const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
310
+ if (schemeMatch) {
311
+ const scheme = schemeMatch[1].toLowerCase();
312
+ const ok = /^(?:https?|mailto|tel)$/.test(scheme)
313
+ || (opts?.allowDataImage === true && /^data:image\//i.test(stripped));
314
+ if (!ok)
315
+ return '';
316
+ }
317
+ return stripped.replace(/[\s()<>"`\\]/g, (c) => '%' + c.charCodeAt(0).toString(16).toUpperCase().padStart(2, '0'));
318
+ }
@@ -440,12 +440,12 @@ const decodeXmlEntities = (text) => {
440
440
  if (entity[1] === 'x' || entity[1] === 'X') {
441
441
  const hex = entity.slice(2);
442
442
  const code = parseInt(hex, 16);
443
- return !isNaN(code) ? String.fromCodePoint(code) : match;
443
+ return (!isNaN(code) && code >= 0 && code <= 0x10FFFF) ? String.fromCodePoint(code) : match;
444
444
  }
445
445
  else {
446
446
  const dec = entity.slice(1);
447
447
  const code = parseInt(dec, 10);
448
- return !isNaN(code) ? String.fromCodePoint(code) : match;
448
+ return (!isNaN(code) && code >= 0 && code <= 0x10FFFF) ? String.fromCodePoint(code) : match;
449
449
  }
450
450
  }
451
451
  switch (entity) {