officeparser 7.2.3 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +277 -17
- package/dist/OfficeConverter.d.ts +1 -1
- package/dist/OfficeConverter.js +3 -0
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +5 -2
- package/dist/defaults.js +12 -0
- package/dist/generators/BaseGenerator.d.ts +34 -1
- package/dist/generators/BaseGenerator.js +98 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +28 -16
- package/dist/generators/EpubGenerator.d.ts +43 -0
- package/dist/generators/EpubGenerator.js +312 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +378 -61
- package/dist/generators/MarkdownGenerator.d.ts +28 -5
- package/dist/generators/MarkdownGenerator.js +432 -51
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +47 -22
- package/dist/generators/TextGenerator.js +98 -11
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +427 -20
- package/dist/officeparser.browser.iife.js +338 -206
- package/dist/officeparser.browser.mjs +346 -214
- package/dist/officeparser.browser.slim.d.ts +427 -20
- package/dist/officeparser.browser.slim.iife.js +346 -214
- package/dist/officeparser.browser.slim.mjs +346 -214
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/ExcelParser.js +2 -0
- package/dist/parsers/HtmlParser.js +507 -48
- package/dist/parsers/MarkdownParser.js +704 -92
- package/dist/parsers/OpenOfficeParser.js +128 -20
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/PowerPointParser.js +1 -0
- package/dist/parsers/WordParser.js +1 -0
- package/dist/sbom.cdx.json +1695 -0
- package/dist/types.d.ts +427 -20
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +53 -4
- package/dist/utils/errorUtils.js +7 -3
- package/dist/utils/sanitize.d.ts +139 -0
- package/dist/utils/sanitize.js +318 -0
- package/dist/utils/xmlUtils.js +2 -2
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +16 -12
|
@@ -8,6 +8,37 @@ exports.isValidContainerWidth = isValidContainerWidth;
|
|
|
8
8
|
const defaults_js_1 = require("../defaults.js");
|
|
9
9
|
const types_js_1 = require("../types.js");
|
|
10
10
|
const errorUtils_js_1 = require("./errorUtils.js");
|
|
11
|
+
/**
|
|
12
|
+
* Keys that must never be copied from a caller-supplied config onto one of our objects.
|
|
13
|
+
*
|
|
14
|
+
* A config that arrived via `JSON.parse` can carry `__proto__` as a genuine **own enumerable**
|
|
15
|
+
* property (an object *literal* cannot - there `__proto__` invokes the setter at parse time),
|
|
16
|
+
* which is exactly the shape of a host application accepting a JSON config blob. Copying that
|
|
17
|
+
* key reaches `Object.prototype` and corrupts every object in the process.
|
|
18
|
+
*
|
|
19
|
+
* `constructor` and `prototype` are included because they are the other two names that reach a
|
|
20
|
+
* prototype through an ordinary property write.
|
|
21
|
+
*/
|
|
22
|
+
const PROTOTYPE_POLLUTION_KEYS = new Set(['__proto__', 'constructor', 'prototype']);
|
|
23
|
+
/**
|
|
24
|
+
* Returns a copy of `source` with prototype-reaching keys removed.
|
|
25
|
+
*
|
|
26
|
+
* Needed before `Object.assign`, which does **not** pollute `Object.prototype` (it writes via
|
|
27
|
+
* `[[Set]]`, so `__proto__` invokes the inherited setter rather than creating an own property) -
|
|
28
|
+
* but that setter is not inert: it **replaces the target's prototype**, so the returned config
|
|
29
|
+
* silently inherits attacker-chosen properties for every field the defaults don't set as an own
|
|
30
|
+
* property. Narrower than global pollution, still wrong. Do not "simplify" this away on the
|
|
31
|
+
* grounds that `Object.assign` is safe; it is safe only against the *global* variant.
|
|
32
|
+
*/
|
|
33
|
+
function withoutPrototypeKeys(source) {
|
|
34
|
+
const safe = {};
|
|
35
|
+
for (const key of Object.keys(source)) {
|
|
36
|
+
if (PROTOTYPE_POLLUTION_KEYS.has(key))
|
|
37
|
+
continue;
|
|
38
|
+
safe[key] = source[key];
|
|
39
|
+
}
|
|
40
|
+
return safe;
|
|
41
|
+
}
|
|
11
42
|
/**
|
|
12
43
|
* Deep clones an object, specifically handling arrays and plain objects.
|
|
13
44
|
*/
|
|
@@ -61,6 +92,7 @@ function resolveParserConfig(userConfig) {
|
|
|
61
92
|
userConfig.decompressionLimits = {
|
|
62
93
|
maxUncompressedBytes: 512 * 1024 * 1024,
|
|
63
94
|
maxZipEntries: 10000,
|
|
95
|
+
maxTableCells: 1000000,
|
|
64
96
|
};
|
|
65
97
|
}
|
|
66
98
|
return userConfig;
|
|
@@ -71,15 +103,22 @@ function resolveParserConfig(userConfig) {
|
|
|
71
103
|
return config;
|
|
72
104
|
}
|
|
73
105
|
// 2. Merge user config
|
|
74
|
-
// We handle ocrConfig and
|
|
75
|
-
|
|
76
|
-
|
|
106
|
+
// We handle ocrConfig, decompressionLimits, and htmlParserConfig specially to avoid
|
|
107
|
+
// shallow-overwriting the whole nested objects
|
|
108
|
+
const { ocrConfig, decompressionLimits, htmlParserConfig, ...rest } = userConfig;
|
|
109
|
+
Object.assign(config, withoutPrototypeKeys(rest));
|
|
77
110
|
if (decompressionLimits) {
|
|
78
111
|
config.decompressionLimits = {
|
|
79
112
|
...config.decompressionLimits,
|
|
80
113
|
...decompressionLimits,
|
|
81
114
|
};
|
|
82
115
|
}
|
|
116
|
+
if (htmlParserConfig) {
|
|
117
|
+
config.htmlParserConfig = {
|
|
118
|
+
...config.htmlParserConfig,
|
|
119
|
+
...htmlParserConfig,
|
|
120
|
+
};
|
|
121
|
+
}
|
|
83
122
|
if (ocrConfig) {
|
|
84
123
|
const { timeout, ...ocrRest } = ocrConfig;
|
|
85
124
|
config.ocrConfig = {
|
|
@@ -124,12 +163,22 @@ function resolveGeneratorConfig(destination, astConfig, userConfig) {
|
|
|
124
163
|
if (userConfig) {
|
|
125
164
|
// Extract sub-configs to avoid shallow-overwriting the whole sub-config objects
|
|
126
165
|
const { htmlConfig, mdConfig, pdfConfig, csvConfig, textConfig, chunksConfig, ...commonProps } = userConfig;
|
|
127
|
-
Object.assign(config, commonProps);
|
|
166
|
+
Object.assign(config, withoutPrototypeKeys(commonProps));
|
|
128
167
|
// Merge sub-configs individually, ignoring undefined properties to preserve defaults
|
|
129
168
|
const mergeSubConfig = (target, source) => {
|
|
130
169
|
if (!source)
|
|
131
170
|
return;
|
|
132
171
|
for (const key in source) {
|
|
172
|
+
// Both guards are load-bearing and neither subsumes the other. The own-property
|
|
173
|
+
// check (matching deepClone above) stops inherited enumerable properties, which
|
|
174
|
+
// matters once anything else in the process has already polluted a prototype. It
|
|
175
|
+
// does NOT stop this attack on its own: `JSON.parse('{"__proto__":{...}}')` yields
|
|
176
|
+
// `__proto__` as an own enumerable key, so it passes hasOwnProperty and would be
|
|
177
|
+
// written straight through to Object.prototype by the recursion below.
|
|
178
|
+
if (!Object.prototype.hasOwnProperty.call(source, key))
|
|
179
|
+
continue;
|
|
180
|
+
if (PROTOTYPE_POLLUTION_KEYS.has(key))
|
|
181
|
+
continue;
|
|
133
182
|
if (source[key] !== undefined) {
|
|
134
183
|
// Deep merge plain objects (like injections or margin)
|
|
135
184
|
if (typeof source[key] === 'object' &&
|
package/dist/utils/errorUtils.js
CHANGED
|
@@ -16,8 +16,8 @@ const ERRORHEADER = "[OfficeParser]: ";
|
|
|
16
16
|
* Some entries are functions that take parameters to build dynamic messages.
|
|
17
17
|
*/
|
|
18
18
|
const ERROR_MESSAGES = {
|
|
19
|
-
[types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
20
|
-
[types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks.`,
|
|
19
|
+
[types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv, epub files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
20
|
+
[types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks, epub.`,
|
|
21
21
|
[types_js_1.OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
22
22
|
[types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
23
23
|
[types_js_1.OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
|
|
@@ -34,6 +34,7 @@ const ERROR_MESSAGES = {
|
|
|
34
34
|
[types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
|
|
35
35
|
[types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
|
|
36
36
|
[types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
|
|
37
|
+
[types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED]: `Document nesting depth exceeded the safe limit (possible denial-of-service input)`,
|
|
37
38
|
[types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
|
|
38
39
|
};
|
|
39
40
|
/**
|
|
@@ -56,7 +57,10 @@ const WARNING_MESSAGES = {
|
|
|
56
57
|
[types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED]: `Auto-detection of file type failed. This can happen on older Node.js versions with modern file-type versions. Please provide the 'fileType' hint in the configuration if parsing fails.`,
|
|
57
58
|
[types_js_1.OfficeWarningType.EMPTY_CHUNK_GENERATED]: (strategy) => `No chunks generated for document. Check if the document content is compatible with the '${strategy}' strategy.`,
|
|
58
59
|
[types_js_1.OfficeWarningType.WHITESPACE_NODE_SKIPPED]: (nodeType) => `Skipped whitespace-only node of type: ${nodeType}`,
|
|
59
|
-
[types_js_1.OfficeWarningType.
|
|
60
|
+
[types_js_1.OfficeWarningType.TABLE_CELL_LIMIT_EXCEEDED]: (limit) => `Table cell limit (${limit}) reached while expanding repeated ODF cells/rows; the remaining cells were not materialized. A few hundred bytes of XML can request an unbounded number of cells via table:number-columns-repeated / table:number-rows-repeated, so this is capped. Raise decompressionLimits.maxTableCells if your documents legitimately exceed it.`,
|
|
61
|
+
[types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH]: (val) => `Invalid HTML containerWidth: ${JSON.stringify(val)}. Falling back to "auto". Width must be a positive number, a valid CSS length string (e.g., "900px", "100%", "50vw"), or "auto".`,
|
|
62
|
+
[types_js_1.OfficeWarningType.METADATA_NOT_REPRESENTABLE]: (info) => `Custom metadata ${info.keys.map(k => `'${k}'`).join(', ')} could not be written to ${info.format} output: the format has a fixed metadata vocabulary with no place for caller-defined keys. The named metadata fields (title, author, etc.) were still applied.`,
|
|
63
|
+
[types_js_1.OfficeWarningType.INVALID_STYLE_MAP_TAG]: (tag) => `styleMap output.tag ${JSON.stringify(tag)} is not an allowed element name and was ignored; the node's default tag was used instead. A tag name is written into both the opening and closing tag, so only a known-safe set of block, heading and inline elements is accepted.`
|
|
60
64
|
};
|
|
61
65
|
/**
|
|
62
66
|
* Creates a formatted warning message for a specific warning type.
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared output-sanitization helpers.
|
|
3
|
+
*
|
|
4
|
+
* Every string in the parsed AST originates from an untrusted document, so any
|
|
5
|
+
* value interpolated into generated output (HTML, XHTML, CSS, URLs, inline
|
|
6
|
+
* scripts, CSV, RTF, Markdown) must be escaped for its destination context.
|
|
7
|
+
* These are the single source of truth — each generator delegates to them so
|
|
8
|
+
* escaping stays consistent and a gap fixed here is fixed everywhere.
|
|
9
|
+
*/
|
|
10
|
+
/**
|
|
11
|
+
* Whether a string is a plain HTML attribute *name*, safe to interpolate before `="..."`.
|
|
12
|
+
*
|
|
13
|
+
* Escaping the value is not enough on its own: a key containing a quote or `=` closes the
|
|
14
|
+
* attribute and opens another, so `x" onmouseover="alert(1)" z` yields a real event handler no
|
|
15
|
+
* matter how carefully the value is escaped. This is the shape an attribute-injection payload
|
|
16
|
+
* takes, and rejecting it outright is simpler and safer than trying to escape a name.
|
|
17
|
+
*
|
|
18
|
+
* The predicate is shared rather than restated because it is now applied at four independent
|
|
19
|
+
* points (the parser's attribute collection, the generator's attribute bag, and two
|
|
20
|
+
* styleMap-driven paths). Each of those still keeps its own skip-list inline: the lists are the
|
|
21
|
+
* same policy expressed for different layers, and collapsing them would erase the defence in
|
|
22
|
+
* depth the surrounding comments describe.
|
|
23
|
+
*/
|
|
24
|
+
export declare function isSafeHtmlAttributeName(name: string): boolean;
|
|
25
|
+
/**
|
|
26
|
+
* Whether a `styleMap` `output.tag` may be emitted as an element name.
|
|
27
|
+
*
|
|
28
|
+
* Callers must fall back to their default tag when this returns false, never emit the value.
|
|
29
|
+
*/
|
|
30
|
+
export declare function isSafeStyleMapTag(tag: unknown): tag is string;
|
|
31
|
+
/**
|
|
32
|
+
* Escapes text for an HTML text node or a double-quoted attribute value.
|
|
33
|
+
* Includes the single quote so the result is also safe inside single-quoted
|
|
34
|
+
* attributes.
|
|
35
|
+
*/
|
|
36
|
+
export declare function escapeHtml(text: string): string;
|
|
37
|
+
/**
|
|
38
|
+
* Escapes text for an XML text node or attribute (XHTML/OPF/NCX). Same as
|
|
39
|
+
* escapeHtml but emits the XML-canonical `'` for the single quote.
|
|
40
|
+
*/
|
|
41
|
+
export declare function escapeXml(text: string): string;
|
|
42
|
+
/**
|
|
43
|
+
* Sanitizes a single CSS value (e.g. a color/size/font/alignment pulled from a
|
|
44
|
+
* document) for placement inside a `style="prop: VALUE"` attribute.
|
|
45
|
+
*
|
|
46
|
+
* - Drops the whole value if it contains a resource-fetching or executing
|
|
47
|
+
* construct (`url()`, `expression()`, `@import`, `image-set()`, `javascript:`)
|
|
48
|
+
* or angle brackets that could break out of the attribute/tag.
|
|
49
|
+
* - Strips characters that break out of `prop: value` (`;`, quotes), out of a
|
|
50
|
+
* `<style>` rule (`{}`), CSS escapes (`\`), and control characters.
|
|
51
|
+
*
|
|
52
|
+
* `rgb()/hsl()` and hex/named colors, lengths, and (unquoted) font names all
|
|
53
|
+
* survive; the trade-off is that legitimately quoted font names lose their
|
|
54
|
+
* quotes, which browsers tolerate.
|
|
55
|
+
*/
|
|
56
|
+
export declare function sanitizeCssValue(value: string): string;
|
|
57
|
+
/**
|
|
58
|
+
* Escapes a document-supplied URL for use in an href/src attribute. Beyond the
|
|
59
|
+
* usual attribute escaping, this rejects script-executing schemes (javascript:,
|
|
60
|
+
* vbscript:, data:, etc.) so a hyperlink extracted from an untrusted document
|
|
61
|
+
* can't run code when clicked — only http(s)/mailto/tel and relative/fragment
|
|
62
|
+
* URLs are passed through.
|
|
63
|
+
*/
|
|
64
|
+
export declare function sanitizeUrl(url: string): string;
|
|
65
|
+
/**
|
|
66
|
+
* Like sanitizeUrl but for an <img>/<source> src: additionally permits
|
|
67
|
+
* `data:image/*` URIs (embedded document images) while still rejecting
|
|
68
|
+
* script-executing schemes and non-image data URIs (e.g. data:text/html).
|
|
69
|
+
*/
|
|
70
|
+
export declare function sanitizeImageUrl(url: string): string;
|
|
71
|
+
/**
|
|
72
|
+
* Serializes data for embedding inside an inline <script> block. JSON.stringify
|
|
73
|
+
* alone doesn't escape "<", so a value containing "</script>" (e.g. a chart
|
|
74
|
+
* label from attacker-controlled document XML) would close the script early and
|
|
75
|
+
* inject markup. Also escapes the U+2028/U+2029 line separators, which are
|
|
76
|
+
* invalid in JS string literals.
|
|
77
|
+
*/
|
|
78
|
+
export declare function serializeForInlineScript(data: unknown): string;
|
|
79
|
+
/**
|
|
80
|
+
* Formats a value for a CSV field: guards against spreadsheet formula/DDE
|
|
81
|
+
* injection (CWE-1236) and applies RFC 4180 quoting.
|
|
82
|
+
*
|
|
83
|
+
* A cell beginning with `= + - @` (or a tab/CR that some apps treat as a
|
|
84
|
+
* formula start) is prefixed with a single quote so Excel/Sheets render it as
|
|
85
|
+
* literal text rather than executing it. Genuine numbers (including negatives)
|
|
86
|
+
* are exempt so numeric columns are preserved.
|
|
87
|
+
*/
|
|
88
|
+
export declare function csvSafeCell(value: string, delimiter: string): string;
|
|
89
|
+
/**
|
|
90
|
+
* Validates and escapes a document-supplied URL for an RTF `HYPERLINK` field argument.
|
|
91
|
+
*
|
|
92
|
+
* Mirrors `sanitizeUrl`'s contract (validate the scheme, then encode for the destination, else
|
|
93
|
+
* return `''`) but cannot reuse it: `sanitizeUrl` returns `escapeHtml(...)`, which would emit
|
|
94
|
+
* `&` into an RTF field. The scheme allowlist is deliberately identical to `sanitizeUrl`'s
|
|
95
|
+
* and `sanitizeMarkdownUrl`'s, so all three text generators agree on what a hyperlink may point at.
|
|
96
|
+
*
|
|
97
|
+
* **Additionally rejects UNC paths (`\\host\share`), which the HTML allowlist does not.** In a
|
|
98
|
+
* browser `\\evil.com\share` is an inert relative path; in Word it is a live UNC reference that
|
|
99
|
+
* triggers an SMB fetch and an NTLM handshake on click, which is a credential-leak vector rather
|
|
100
|
+
* than a rendering quirk. That asymmetry is why this is a separate function and not a flag on
|
|
101
|
+
* `sanitizeUrl` - the HTML helper must NOT gain this behaviour, since there the path is harmless
|
|
102
|
+
* and rejecting it would break legitimate relative links.
|
|
103
|
+
*
|
|
104
|
+
* Returns `''` for a rejected URL; callers emit the link text without the field wrapper, matching
|
|
105
|
+
* how HTML degrades to `href=""` and Markdown to `[text]()`.
|
|
106
|
+
*/
|
|
107
|
+
export declare function sanitizeRtfUrl(url: string): string;
|
|
108
|
+
/**
|
|
109
|
+
* Escapes text for RTF: neutralizes the control/group metacharacters `\ { }`
|
|
110
|
+
* (which would otherwise inject RTF control words or groups), encodes the double
|
|
111
|
+
* quote (so a hyperlink field argument can't be terminated early), and hex/unicode
|
|
112
|
+
* encodes non-ASCII characters.
|
|
113
|
+
*/
|
|
114
|
+
export declare function escapeRtf(text: string): string;
|
|
115
|
+
/**
|
|
116
|
+
* Escapes document text for a Markdown text position. Markdown passes raw HTML
|
|
117
|
+
* through to the renderer, so a `<` that begins an HTML tag or comment must be
|
|
118
|
+
* neutralized to prevent `<script>`/`<img onerror>` injection when the Markdown
|
|
119
|
+
* is later rendered to HTML.
|
|
120
|
+
*
|
|
121
|
+
* Deliberately narrow — only a `<` immediately followed by a letter, `/`, `!` or
|
|
122
|
+
* `?` (i.e. one that actually opens a tag/comment/PI, matching how browsers
|
|
123
|
+
* detect tags) is encoded. A bare `<` (e.g. `a < b`), `>`, `&`, `[]` and other
|
|
124
|
+
* Markdown metacharacters are left untouched: they can't start a tag, and
|
|
125
|
+
* MarkdownParser round-trips this output without decoding entities, so encoding
|
|
126
|
+
* them would corrupt re-parsed content. URL schemes are handled by
|
|
127
|
+
* sanitizeMarkdownUrl.
|
|
128
|
+
*/
|
|
129
|
+
export declare function markdownEscapeText(text: string): string;
|
|
130
|
+
/**
|
|
131
|
+
* Sanitizes a document-supplied URL for a Markdown `[text](url)` / ``
|
|
132
|
+
* target. Rejects script-executing schemes (returning '' → a dead link) and
|
|
133
|
+
* percent-encodes the characters that would break out of the `(...)` or inject
|
|
134
|
+
* markup. `&` is preserved so query strings survive; set `allowDataImage` for
|
|
135
|
+
* image targets so embedded `data:image/*` URIs are permitted.
|
|
136
|
+
*/
|
|
137
|
+
export declare function sanitizeMarkdownUrl(url: string, opts?: {
|
|
138
|
+
allowDataImage?: boolean;
|
|
139
|
+
}): string;
|
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Shared output-sanitization helpers.
|
|
4
|
+
*
|
|
5
|
+
* Every string in the parsed AST originates from an untrusted document, so any
|
|
6
|
+
* value interpolated into generated output (HTML, XHTML, CSS, URLs, inline
|
|
7
|
+
* scripts, CSV, RTF, Markdown) must be escaped for its destination context.
|
|
8
|
+
* These are the single source of truth — each generator delegates to them so
|
|
9
|
+
* escaping stays consistent and a gap fixed here is fixed everywhere.
|
|
10
|
+
*/
|
|
11
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
12
|
+
exports.isSafeHtmlAttributeName = isSafeHtmlAttributeName;
|
|
13
|
+
exports.isSafeStyleMapTag = isSafeStyleMapTag;
|
|
14
|
+
exports.escapeHtml = escapeHtml;
|
|
15
|
+
exports.escapeXml = escapeXml;
|
|
16
|
+
exports.sanitizeCssValue = sanitizeCssValue;
|
|
17
|
+
exports.sanitizeUrl = sanitizeUrl;
|
|
18
|
+
exports.sanitizeImageUrl = sanitizeImageUrl;
|
|
19
|
+
exports.serializeForInlineScript = serializeForInlineScript;
|
|
20
|
+
exports.csvSafeCell = csvSafeCell;
|
|
21
|
+
exports.sanitizeRtfUrl = sanitizeRtfUrl;
|
|
22
|
+
exports.escapeRtf = escapeRtf;
|
|
23
|
+
exports.markdownEscapeText = markdownEscapeText;
|
|
24
|
+
exports.sanitizeMarkdownUrl = sanitizeMarkdownUrl;
|
|
25
|
+
/**
|
|
26
|
+
* Whether a string is a plain HTML attribute *name*, safe to interpolate before `="..."`.
|
|
27
|
+
*
|
|
28
|
+
* Escaping the value is not enough on its own: a key containing a quote or `=` closes the
|
|
29
|
+
* attribute and opens another, so `x" onmouseover="alert(1)" z` yields a real event handler no
|
|
30
|
+
* matter how carefully the value is escaped. This is the shape an attribute-injection payload
|
|
31
|
+
* takes, and rejecting it outright is simpler and safer than trying to escape a name.
|
|
32
|
+
*
|
|
33
|
+
* The predicate is shared rather than restated because it is now applied at four independent
|
|
34
|
+
* points (the parser's attribute collection, the generator's attribute bag, and two
|
|
35
|
+
* styleMap-driven paths). Each of those still keeps its own skip-list inline: the lists are the
|
|
36
|
+
* same policy expressed for different layers, and collapsing them would erase the defence in
|
|
37
|
+
* depth the surrounding comments describe.
|
|
38
|
+
*/
|
|
39
|
+
function isSafeHtmlAttributeName(name) {
|
|
40
|
+
return typeof name === 'string' && /^[a-zA-Z][a-zA-Z0-9-]*$/.test(name);
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Element names a `styleMap` may map a node onto.
|
|
44
|
+
*
|
|
45
|
+
* An allowlist rather than a pattern: a tag name is interpolated into both `<TAG …>` and
|
|
46
|
+
* `</TAG>`, so it is not enough for it to *look* like a name - `script`, `style`, `iframe` and
|
|
47
|
+
* friends are perfectly well-formed names that would introduce an active context the rest of the
|
|
48
|
+
* generator's escaping assumes does not exist. This is the semantic set a style mapping is for:
|
|
49
|
+
* block containers, headings, and the inline emphasis elements.
|
|
50
|
+
*/
|
|
51
|
+
const SAFE_STYLE_MAP_TAGS = new Set([
|
|
52
|
+
'p', 'div', 'span', 'section', 'article', 'aside', 'header', 'footer', 'main', 'figure', 'figcaption',
|
|
53
|
+
'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
|
|
54
|
+
'blockquote', 'pre', 'code', 'q', 'cite', 'address',
|
|
55
|
+
'ul', 'ol', 'li', 'dl', 'dt', 'dd',
|
|
56
|
+
'b', 'strong', 'i', 'em', 'u', 'ins', 'del', 's', 'strike', 'mark', 'small', 'sub', 'sup', 'kbd', 'samp', 'var', 'abbr',
|
|
57
|
+
]);
|
|
58
|
+
/**
|
|
59
|
+
* Whether a `styleMap` `output.tag` may be emitted as an element name.
|
|
60
|
+
*
|
|
61
|
+
* Callers must fall back to their default tag when this returns false, never emit the value.
|
|
62
|
+
*/
|
|
63
|
+
function isSafeStyleMapTag(tag) {
|
|
64
|
+
return typeof tag === 'string' && SAFE_STYLE_MAP_TAGS.has(tag.toLowerCase());
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Escapes text for an HTML text node or a double-quoted attribute value.
|
|
68
|
+
* Includes the single quote so the result is also safe inside single-quoted
|
|
69
|
+
* attributes.
|
|
70
|
+
*/
|
|
71
|
+
function escapeHtml(text) {
|
|
72
|
+
if (typeof text !== 'string')
|
|
73
|
+
return text;
|
|
74
|
+
return text
|
|
75
|
+
.replace(/&/g, '&')
|
|
76
|
+
.replace(/</g, '<')
|
|
77
|
+
.replace(/>/g, '>')
|
|
78
|
+
.replace(/"/g, '"')
|
|
79
|
+
.replace(/'/g, ''');
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Escapes text for an XML text node or attribute (XHTML/OPF/NCX). Same as
|
|
83
|
+
* escapeHtml but emits the XML-canonical `'` for the single quote.
|
|
84
|
+
*/
|
|
85
|
+
function escapeXml(text) {
|
|
86
|
+
if (typeof text !== 'string')
|
|
87
|
+
return '';
|
|
88
|
+
return text
|
|
89
|
+
.replace(/&/g, '&')
|
|
90
|
+
.replace(/</g, '<')
|
|
91
|
+
.replace(/>/g, '>')
|
|
92
|
+
.replace(/"/g, '"')
|
|
93
|
+
.replace(/'/g, ''');
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* Sanitizes a single CSS value (e.g. a color/size/font/alignment pulled from a
|
|
97
|
+
* document) for placement inside a `style="prop: VALUE"` attribute.
|
|
98
|
+
*
|
|
99
|
+
* - Drops the whole value if it contains a resource-fetching or executing
|
|
100
|
+
* construct (`url()`, `expression()`, `@import`, `image-set()`, `javascript:`)
|
|
101
|
+
* or angle brackets that could break out of the attribute/tag.
|
|
102
|
+
* - Strips characters that break out of `prop: value` (`;`, quotes), out of a
|
|
103
|
+
* `<style>` rule (`{}`), CSS escapes (`\`), and control characters.
|
|
104
|
+
*
|
|
105
|
+
* `rgb()/hsl()` and hex/named colors, lengths, and (unquoted) font names all
|
|
106
|
+
* survive; the trade-off is that legitimately quoted font names lose their
|
|
107
|
+
* quotes, which browsers tolerate.
|
|
108
|
+
*/
|
|
109
|
+
function sanitizeCssValue(value) {
|
|
110
|
+
if (typeof value !== 'string')
|
|
111
|
+
return '';
|
|
112
|
+
// Strip every form of intra-token noise FIRST, then test for dangerous constructs.
|
|
113
|
+
// Order matters: a payload like "u\nrl(", "url/*x*/(" or "u\rl(" would survive the test if
|
|
114
|
+
// tested before removal, then reassemble into "url(" once the noise is stripped.
|
|
115
|
+
//
|
|
116
|
+
// The backslash strip belongs here, not after the test. CSS treats `\` as an escape a
|
|
117
|
+
// browser resolves away, so `u\rl(http://evil)` IS `url(http://evil)` to a renderer -
|
|
118
|
+
// stripping it downstream of the test meant the sanitizer handed back a live `url()` it
|
|
119
|
+
// had just declared safe. Every construct in the denylist is reachable this way
|
|
120
|
+
// (`expr\ession(`, `image\-set(`), so the fix is the ordering, not another pattern.
|
|
121
|
+
const cleaned = value
|
|
122
|
+
.replace(/[\x00-\x1F\x7F]/g, '') // control chars (incl. newlines/tabs)
|
|
123
|
+
.replace(/\/\*[\s\S]*?\*\//g, '') // CSS comments used to obfuscate
|
|
124
|
+
.replace(/\\/g, ''); // CSS escapes; see above
|
|
125
|
+
if (/(?:url|expression|image-set|element|-moz-binding)\s*\(|@import|javascript:|[<>]/i.test(cleaned)) {
|
|
126
|
+
return '';
|
|
127
|
+
}
|
|
128
|
+
// Backslash is already gone above; the rest still have work to do here.
|
|
129
|
+
return cleaned.replace(/[;{}"'`]/g, '').trim();
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* Escapes a document-supplied URL for use in an href/src attribute. Beyond the
|
|
133
|
+
* usual attribute escaping, this rejects script-executing schemes (javascript:,
|
|
134
|
+
* vbscript:, data:, etc.) so a hyperlink extracted from an untrusted document
|
|
135
|
+
* can't run code when clicked — only http(s)/mailto/tel and relative/fragment
|
|
136
|
+
* URLs are passed through.
|
|
137
|
+
*/
|
|
138
|
+
function sanitizeUrl(url) {
|
|
139
|
+
if (typeof url !== 'string')
|
|
140
|
+
return '';
|
|
141
|
+
const trimmed = url.trim();
|
|
142
|
+
// Browsers ignore control characters when parsing a URL scheme, so strip them
|
|
143
|
+
// first to catch obfuscated payloads like "java\tscript:alert(1)".
|
|
144
|
+
const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
|
|
145
|
+
const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
|
|
146
|
+
if (schemeMatch && !/^(https?|mailto|tel)$/i.test(schemeMatch[1])) {
|
|
147
|
+
return '';
|
|
148
|
+
}
|
|
149
|
+
// Emit the same normalized string that was validated.
|
|
150
|
+
return escapeHtml(stripped);
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* Like sanitizeUrl but for an <img>/<source> src: additionally permits
|
|
154
|
+
* `data:image/*` URIs (embedded document images) while still rejecting
|
|
155
|
+
* script-executing schemes and non-image data URIs (e.g. data:text/html).
|
|
156
|
+
*/
|
|
157
|
+
function sanitizeImageUrl(url) {
|
|
158
|
+
if (typeof url !== 'string')
|
|
159
|
+
return '';
|
|
160
|
+
const trimmed = url.trim();
|
|
161
|
+
const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
|
|
162
|
+
const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
|
|
163
|
+
if (schemeMatch) {
|
|
164
|
+
const scheme = schemeMatch[1].toLowerCase();
|
|
165
|
+
if (scheme === 'data') {
|
|
166
|
+
if (!/^data:image\//i.test(stripped))
|
|
167
|
+
return '';
|
|
168
|
+
}
|
|
169
|
+
else if (scheme !== 'http' && scheme !== 'https') {
|
|
170
|
+
return '';
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
return escapeHtml(stripped);
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* Serializes data for embedding inside an inline <script> block. JSON.stringify
|
|
177
|
+
* alone doesn't escape "<", so a value containing "</script>" (e.g. a chart
|
|
178
|
+
* label from attacker-controlled document XML) would close the script early and
|
|
179
|
+
* inject markup. Also escapes the U+2028/U+2029 line separators, which are
|
|
180
|
+
* invalid in JS string literals.
|
|
181
|
+
*/
|
|
182
|
+
function serializeForInlineScript(data) {
|
|
183
|
+
// U+2028/U+2029 (line/paragraph separators) are valid in JSON but break
|
|
184
|
+
// JS string literals; reference them by code point to keep the source ASCII.
|
|
185
|
+
const lineSep = String.fromCharCode(0x2028);
|
|
186
|
+
const paraSep = String.fromCharCode(0x2029);
|
|
187
|
+
return JSON.stringify(data)
|
|
188
|
+
.replace(/</g, '\\u003C')
|
|
189
|
+
.replace(/>/g, '\\u003E')
|
|
190
|
+
.split(lineSep).join('\\u2028')
|
|
191
|
+
.split(paraSep).join('\\u2029');
|
|
192
|
+
}
|
|
193
|
+
/**
|
|
194
|
+
* Formats a value for a CSV field: guards against spreadsheet formula/DDE
|
|
195
|
+
* injection (CWE-1236) and applies RFC 4180 quoting.
|
|
196
|
+
*
|
|
197
|
+
* A cell beginning with `= + - @` (or a tab/CR that some apps treat as a
|
|
198
|
+
* formula start) is prefixed with a single quote so Excel/Sheets render it as
|
|
199
|
+
* literal text rather than executing it. Genuine numbers (including negatives)
|
|
200
|
+
* are exempt so numeric columns are preserved.
|
|
201
|
+
*/
|
|
202
|
+
function csvSafeCell(value, delimiter) {
|
|
203
|
+
let v = typeof value === 'string' ? value : String(value ?? '');
|
|
204
|
+
// A plain signed number (e.g. "-8", "+7", "-5.3") can't be a formula, so exempt it —
|
|
205
|
+
// otherwise numeric columns get quoted as text. Anything else starting with a formula
|
|
206
|
+
// trigger (including "+1+1", "-1+cmd", "=", "@") is prefixed with a quote.
|
|
207
|
+
const isNumber = /^[+-]?(?:\d+\.?\d*|\.\d+)(?:[eE][+-]?\d+)?$/.test(v.trim());
|
|
208
|
+
// Test the trimmed value: the numeric exemption above already trims, so testing the raw
|
|
209
|
+
// string here meant a leading space slipped a trigger past the guard (" =1+1" was emitted
|
|
210
|
+
// unprefixed). Most spreadsheet apps treat a leading-space cell as text and would not
|
|
211
|
+
// evaluate it, so this is defence in depth rather than a demonstrated bypass - but the
|
|
212
|
+
// asymmetry between the two tests was an accident, not a decision.
|
|
213
|
+
if (!isNumber && /^[=+\-@\t\r]/.test(v.trim())) {
|
|
214
|
+
v = `'${v}`;
|
|
215
|
+
}
|
|
216
|
+
if (v.includes(delimiter) || v.includes('"') || v.includes('\n') || v.includes('\r')) {
|
|
217
|
+
return `"${v.replace(/"/g, '""')}"`;
|
|
218
|
+
}
|
|
219
|
+
return v;
|
|
220
|
+
}
|
|
221
|
+
/**
|
|
222
|
+
* Validates and escapes a document-supplied URL for an RTF `HYPERLINK` field argument.
|
|
223
|
+
*
|
|
224
|
+
* Mirrors `sanitizeUrl`'s contract (validate the scheme, then encode for the destination, else
|
|
225
|
+
* return `''`) but cannot reuse it: `sanitizeUrl` returns `escapeHtml(...)`, which would emit
|
|
226
|
+
* `&` into an RTF field. The scheme allowlist is deliberately identical to `sanitizeUrl`'s
|
|
227
|
+
* and `sanitizeMarkdownUrl`'s, so all three text generators agree on what a hyperlink may point at.
|
|
228
|
+
*
|
|
229
|
+
* **Additionally rejects UNC paths (`\\host\share`), which the HTML allowlist does not.** In a
|
|
230
|
+
* browser `\\evil.com\share` is an inert relative path; in Word it is a live UNC reference that
|
|
231
|
+
* triggers an SMB fetch and an NTLM handshake on click, which is a credential-leak vector rather
|
|
232
|
+
* than a rendering quirk. That asymmetry is why this is a separate function and not a flag on
|
|
233
|
+
* `sanitizeUrl` - the HTML helper must NOT gain this behaviour, since there the path is harmless
|
|
234
|
+
* and rejecting it would break legitimate relative links.
|
|
235
|
+
*
|
|
236
|
+
* Returns `''` for a rejected URL; callers emit the link text without the field wrapper, matching
|
|
237
|
+
* how HTML degrades to `href=""` and Markdown to `[text]()`.
|
|
238
|
+
*/
|
|
239
|
+
function sanitizeRtfUrl(url) {
|
|
240
|
+
if (typeof url !== 'string')
|
|
241
|
+
return '';
|
|
242
|
+
const trimmed = url.trim();
|
|
243
|
+
// Control characters are stripped before scheme matching for the same reason as sanitizeUrl:
|
|
244
|
+
// they are ignored when the target application parses the scheme.
|
|
245
|
+
const stripped = trimmed.replace(/[\x00-\x1F\x7F]+/g, '');
|
|
246
|
+
if (/^[\\/]{2}[^\\/]/.test(stripped))
|
|
247
|
+
return ''; // UNC (\\host\share, //host\share)
|
|
248
|
+
const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
|
|
249
|
+
if (schemeMatch && !/^(https?|mailto|tel)$/i.test(schemeMatch[1])) {
|
|
250
|
+
return '';
|
|
251
|
+
}
|
|
252
|
+
return escapeRtf(stripped);
|
|
253
|
+
}
|
|
254
|
+
/**
|
|
255
|
+
* Escapes text for RTF: neutralizes the control/group metacharacters `\ { }`
|
|
256
|
+
* (which would otherwise inject RTF control words or groups), encodes the double
|
|
257
|
+
* quote (so a hyperlink field argument can't be terminated early), and hex/unicode
|
|
258
|
+
* encodes non-ASCII characters.
|
|
259
|
+
*/
|
|
260
|
+
function escapeRtf(text) {
|
|
261
|
+
if (typeof text !== 'string')
|
|
262
|
+
return '';
|
|
263
|
+
return text
|
|
264
|
+
.replace(/\\/g, '\\\\')
|
|
265
|
+
.replace(/{/g, '\\{')
|
|
266
|
+
.replace(/}/g, '\\}')
|
|
267
|
+
.replace(/"/g, "\\'22")
|
|
268
|
+
.replace(/[^\x00-\x7F]/g, (match) => {
|
|
269
|
+
let code = match.charCodeAt(0);
|
|
270
|
+
if (code < 256) {
|
|
271
|
+
return `\\'${code.toString(16).padStart(2, '0')}`;
|
|
272
|
+
}
|
|
273
|
+
if (code > 32767) {
|
|
274
|
+
code -= 65536;
|
|
275
|
+
}
|
|
276
|
+
return `{\\uc0\\u${code}}`;
|
|
277
|
+
});
|
|
278
|
+
}
|
|
279
|
+
/**
|
|
280
|
+
* Escapes document text for a Markdown text position. Markdown passes raw HTML
|
|
281
|
+
* through to the renderer, so a `<` that begins an HTML tag or comment must be
|
|
282
|
+
* neutralized to prevent `<script>`/`<img onerror>` injection when the Markdown
|
|
283
|
+
* is later rendered to HTML.
|
|
284
|
+
*
|
|
285
|
+
* Deliberately narrow — only a `<` immediately followed by a letter, `/`, `!` or
|
|
286
|
+
* `?` (i.e. one that actually opens a tag/comment/PI, matching how browsers
|
|
287
|
+
* detect tags) is encoded. A bare `<` (e.g. `a < b`), `>`, `&`, `[]` and other
|
|
288
|
+
* Markdown metacharacters are left untouched: they can't start a tag, and
|
|
289
|
+
* MarkdownParser round-trips this output without decoding entities, so encoding
|
|
290
|
+
* them would corrupt re-parsed content. URL schemes are handled by
|
|
291
|
+
* sanitizeMarkdownUrl.
|
|
292
|
+
*/
|
|
293
|
+
function markdownEscapeText(text) {
|
|
294
|
+
if (typeof text !== 'string')
|
|
295
|
+
return '';
|
|
296
|
+
return text.replace(/<(?=[a-zA-Z/!?])/g, '<');
|
|
297
|
+
}
|
|
298
|
+
/**
|
|
299
|
+
* Sanitizes a document-supplied URL for a Markdown `[text](url)` / ``
|
|
300
|
+
* target. Rejects script-executing schemes (returning '' → a dead link) and
|
|
301
|
+
* percent-encodes the characters that would break out of the `(...)` or inject
|
|
302
|
+
* markup. `&` is preserved so query strings survive; set `allowDataImage` for
|
|
303
|
+
* image targets so embedded `data:image/*` URIs are permitted.
|
|
304
|
+
*/
|
|
305
|
+
function sanitizeMarkdownUrl(url, opts) {
|
|
306
|
+
if (typeof url !== 'string')
|
|
307
|
+
return '';
|
|
308
|
+
const stripped = url.trim().replace(/[\x00-\x1F\x7F]+/g, '');
|
|
309
|
+
const schemeMatch = /^([a-z][a-z0-9+.-]*):/i.exec(stripped);
|
|
310
|
+
if (schemeMatch) {
|
|
311
|
+
const scheme = schemeMatch[1].toLowerCase();
|
|
312
|
+
const ok = /^(?:https?|mailto|tel)$/.test(scheme)
|
|
313
|
+
|| (opts?.allowDataImage === true && /^data:image\//i.test(stripped));
|
|
314
|
+
if (!ok)
|
|
315
|
+
return '';
|
|
316
|
+
}
|
|
317
|
+
return stripped.replace(/[\s()<>"`\\]/g, (c) => '%' + c.charCodeAt(0).toString(16).toUpperCase().padStart(2, '0'));
|
|
318
|
+
}
|
package/dist/utils/xmlUtils.js
CHANGED
|
@@ -440,12 +440,12 @@ const decodeXmlEntities = (text) => {
|
|
|
440
440
|
if (entity[1] === 'x' || entity[1] === 'X') {
|
|
441
441
|
const hex = entity.slice(2);
|
|
442
442
|
const code = parseInt(hex, 16);
|
|
443
|
-
return !isNaN(code) ? String.fromCodePoint(code) : match;
|
|
443
|
+
return (!isNaN(code) && code >= 0 && code <= 0x10FFFF) ? String.fromCodePoint(code) : match;
|
|
444
444
|
}
|
|
445
445
|
else {
|
|
446
446
|
const dec = entity.slice(1);
|
|
447
447
|
const code = parseInt(dec, 10);
|
|
448
|
-
return !isNaN(code) ? String.fromCodePoint(code) : match;
|
|
448
|
+
return (!isNaN(code) && code >= 0 && code <= 0x10FFFF) ? String.fromCodePoint(code) : match;
|
|
449
449
|
}
|
|
450
450
|
}
|
|
451
451
|
switch (entity) {
|