officeparser 7.2.3 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +277 -17
- package/dist/OfficeConverter.d.ts +1 -1
- package/dist/OfficeConverter.js +3 -0
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +5 -2
- package/dist/defaults.js +12 -0
- package/dist/generators/BaseGenerator.d.ts +34 -1
- package/dist/generators/BaseGenerator.js +98 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +28 -16
- package/dist/generators/EpubGenerator.d.ts +43 -0
- package/dist/generators/EpubGenerator.js +312 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +378 -61
- package/dist/generators/MarkdownGenerator.d.ts +28 -5
- package/dist/generators/MarkdownGenerator.js +432 -51
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +47 -22
- package/dist/generators/TextGenerator.js +98 -11
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +427 -20
- package/dist/officeparser.browser.iife.js +338 -206
- package/dist/officeparser.browser.mjs +346 -214
- package/dist/officeparser.browser.slim.d.ts +427 -20
- package/dist/officeparser.browser.slim.iife.js +346 -214
- package/dist/officeparser.browser.slim.mjs +346 -214
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/ExcelParser.js +2 -0
- package/dist/parsers/HtmlParser.js +507 -48
- package/dist/parsers/MarkdownParser.js +704 -92
- package/dist/parsers/OpenOfficeParser.js +128 -20
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/PowerPointParser.js +1 -0
- package/dist/parsers/WordParser.js +1 -0
- package/dist/sbom.cdx.json +1695 -0
- package/dist/types.d.ts +427 -20
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +53 -4
- package/dist/utils/errorUtils.js +7 -3
- package/dist/utils/sanitize.d.ts +139 -0
- package/dist/utils/sanitize.js +318 -0
- package/dist/utils/xmlUtils.js +2 -2
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +16 -12
|
@@ -85,6 +85,35 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
85
85
|
// a system-installed Chrome, both of which are too intrusive for a library.
|
|
86
86
|
browser = await puppeteer.launch(launchOptions);
|
|
87
87
|
const page = await browser.newPage();
|
|
88
|
+
// Harden against SSRF: the HTML being rendered is derived from an untrusted
|
|
89
|
+
// document, and `networkidle0` would otherwise fetch every URL it references
|
|
90
|
+
// (external images, stylesheets, etc.) from this host — reaching internal
|
|
91
|
+
// services or a cloud metadata endpoint (169.254.169.254). Intercept requests
|
|
92
|
+
// and allow only inline data/blob URIs and the configured chart CDN; abort every
|
|
93
|
+
// other remote fetch.
|
|
94
|
+
const allowedHosts = new Set();
|
|
95
|
+
try {
|
|
96
|
+
const chartSrc = this.config.htmlConfig?.chartJsSrc;
|
|
97
|
+
if (chartSrc)
|
|
98
|
+
allowedHosts.add(new URL(chartSrc).host);
|
|
99
|
+
}
|
|
100
|
+
catch { /* no or invalid chart CDN configured */ }
|
|
101
|
+
let blockedRemoteResource = false;
|
|
102
|
+
await page.setRequestInterception(true);
|
|
103
|
+
page.on('request', (req) => {
|
|
104
|
+
const url = req.url();
|
|
105
|
+
if (url.startsWith('data:') || url.startsWith('blob:') || url.startsWith('about:')) {
|
|
106
|
+
return req.continue().catch(() => { });
|
|
107
|
+
}
|
|
108
|
+
try {
|
|
109
|
+
if (allowedHosts.has(new URL(url).host)) {
|
|
110
|
+
return req.continue().catch(() => { });
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
catch { /* unparseable URL — fall through and block */ }
|
|
114
|
+
blockedRemoteResource = true;
|
|
115
|
+
return req.abort().catch(() => { });
|
|
116
|
+
});
|
|
88
117
|
const pdfConfig = this.config.pdfConfig;
|
|
89
118
|
const timeout = pdfConfig.timeout;
|
|
90
119
|
if (timeout !== undefined && timeout > 0) {
|
|
@@ -107,6 +136,9 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
107
136
|
});
|
|
108
137
|
await browser.close();
|
|
109
138
|
browser = null;
|
|
139
|
+
if (blockedRemoteResource) {
|
|
140
|
+
this.warn(types_js_1.OfficeWarningType.BROWSER_GENERATION_LIMITATION, 'One or more remote resources referenced by the document were blocked during PDF rendering to prevent server-side request forgery (SSRF). Only inline images and the configured chart CDN are loaded.');
|
|
141
|
+
}
|
|
110
142
|
return {
|
|
111
143
|
value: new Uint8Array(pdfBuffer),
|
|
112
144
|
messages: this.messages
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.RtfGenerator = void 0;
|
|
4
|
+
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
4
5
|
const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
6
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
5
7
|
/**
|
|
6
8
|
* Generates high-fidelity RTF (Rich Text Format) from an AST.
|
|
7
9
|
*/
|
|
@@ -17,14 +19,23 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
17
19
|
const bodyContent = await this.renderBody(this.ast);
|
|
18
20
|
let output = '{\\rtf1\\ansi\\uc1\\deff0\n';
|
|
19
21
|
// 1. Info Group (Metadata)
|
|
20
|
-
|
|
22
|
+
const meta = this.effectiveMetadata;
|
|
23
|
+
if (this.config.renderMetadata && meta) {
|
|
24
|
+
// RTF's \info group is a fixed set of control words with no slot for caller-defined keys.
|
|
25
|
+
this.warnUnrepresentableCustomMetadata('RTF');
|
|
21
26
|
output += '{\\info';
|
|
22
|
-
if (
|
|
23
|
-
output += `{\\title ${this.escapeRtf(
|
|
24
|
-
if (
|
|
25
|
-
output += `{\\author ${this.escapeRtf(
|
|
26
|
-
if (
|
|
27
|
-
output += `{\\comm ${this.escapeRtf(
|
|
27
|
+
if (meta.title)
|
|
28
|
+
output += `{\\title ${this.escapeRtf(meta.title)}}`;
|
|
29
|
+
if (meta.author)
|
|
30
|
+
output += `{\\author ${this.escapeRtf(meta.author)}}`;
|
|
31
|
+
if (meta.description)
|
|
32
|
+
output += `{\\comm ${this.escapeRtf(meta.description)}}`;
|
|
33
|
+
// \subject and \keywords are standard \info destinations, so unlike a caller's
|
|
34
|
+
// arbitrary custom keys these DO have a home in RTF.
|
|
35
|
+
if (meta.subject)
|
|
36
|
+
output += `{\\subject ${this.escapeRtf(meta.subject)}}`;
|
|
37
|
+
if (meta.keywords)
|
|
38
|
+
output += `{\\keywords ${this.escapeRtf(meta.keywords)}}`;
|
|
28
39
|
output += '}\n';
|
|
29
40
|
}
|
|
30
41
|
// 2. Font Table
|
|
@@ -50,6 +61,10 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
50
61
|
};
|
|
51
62
|
}
|
|
52
63
|
async processNodeRecursive(node, processor) {
|
|
64
|
+
// Mirrors the check in BaseGenerator.processNodeRecursive. This override replaces that
|
|
65
|
+
// method entirely, so without repeating the check here the signal would be silently
|
|
66
|
+
// inert for this generator - which is exactly how it was missed.
|
|
67
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
53
68
|
const wasInTable = this.inTable;
|
|
54
69
|
if (node.type === 'table')
|
|
55
70
|
this.inTable = true;
|
|
@@ -134,7 +149,16 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
134
149
|
if (meta?.link) {
|
|
135
150
|
const isInternal = meta.linkType !== 'external';
|
|
136
151
|
if (!this.config.ignoreInternalLinks || !isInternal) {
|
|
137
|
-
|
|
152
|
+
// Scheme-checked, not merely escaped. escapeRtf neutralizes the field
|
|
153
|
+
// metacharacters but says nothing about where the link points, so RTF
|
|
154
|
+
// was the one generator that would emit `javascript:` or a `file://`
|
|
155
|
+
// /UNC target that HTML and Markdown both reject. On rejection, fall
|
|
156
|
+
// through to the bare link text - the same degradation as HTML's
|
|
157
|
+
// href="" and Markdown's [text]().
|
|
158
|
+
const safeLink = (0, sanitize_js_1.sanitizeRtfUrl)(meta.link);
|
|
159
|
+
if (safeLink) {
|
|
160
|
+
return `{\\field{\\*\\fldinst{HYPERLINK "${safeLink}"}}{\\fldrslt ${text}}}`;
|
|
161
|
+
}
|
|
138
162
|
}
|
|
139
163
|
}
|
|
140
164
|
return text;
|
|
@@ -214,6 +238,20 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
214
238
|
case 'break': {
|
|
215
239
|
return node.metadata?.breakType === 'page' ? '\\page\n' : '\\line\n';
|
|
216
240
|
}
|
|
241
|
+
case 'embed': {
|
|
242
|
+
// RTF has no embed concept - degrade to the URL as plain text rather than
|
|
243
|
+
// silently dropping the node (it has no children to fall back to).
|
|
244
|
+
const meta = node.metadata;
|
|
245
|
+
if (!meta?.url)
|
|
246
|
+
return '';
|
|
247
|
+
// Rendered as visible text rather than a field, but still a URL a reader may
|
|
248
|
+
// copy, so it gets the same scheme policy.
|
|
249
|
+
const safeUrl = (0, sanitize_js_1.sanitizeRtfUrl)(meta.url);
|
|
250
|
+
if (!safeUrl)
|
|
251
|
+
return '';
|
|
252
|
+
const pPr = this.inTable ? '\\pard\\intbl' : '\\pard';
|
|
253
|
+
return `${pPr}\\sa120 ${safeUrl}\\par\n`;
|
|
254
|
+
}
|
|
217
255
|
default:
|
|
218
256
|
return childrenOutput;
|
|
219
257
|
}
|
|
@@ -251,20 +289,7 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
251
289
|
return luminance > 0.8;
|
|
252
290
|
}
|
|
253
291
|
escapeRtf(text) {
|
|
254
|
-
return text
|
|
255
|
-
.replace(/\\/g, '\\\\')
|
|
256
|
-
.replace(/{/g, '\\{')
|
|
257
|
-
.replace(/}/g, '\\}')
|
|
258
|
-
.replace(/[^\x00-\x7F]/g, (match) => {
|
|
259
|
-
let code = match.charCodeAt(0);
|
|
260
|
-
if (code < 256) {
|
|
261
|
-
return `\\'${code.toString(16).padStart(2, '0')}`;
|
|
262
|
-
}
|
|
263
|
-
if (code > 32767) {
|
|
264
|
-
code -= 65536;
|
|
265
|
-
}
|
|
266
|
-
return `{\\uc0\\u${code}}`;
|
|
267
|
-
});
|
|
292
|
+
return (0, sanitize_js_1.escapeRtf)(text);
|
|
268
293
|
}
|
|
269
294
|
}
|
|
270
295
|
exports.RtfGenerator = RtfGenerator;
|
|
@@ -2,6 +2,14 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.TextGenerator = void 0;
|
|
4
4
|
const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
5
|
+
const escapeRegExpChars = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
6
|
+
/**
|
|
7
|
+
* Separator between table/sheet cells on the paths that don't render an aligned grid (a `table`
|
|
8
|
+
* with `preserveLayout` off, and `sheet`/`row`/`cell`, which never go through `renderTable`).
|
|
9
|
+
* A tab is the conventional plain-text column delimiter and, unlike a space, survives a value that
|
|
10
|
+
* already contains spaces.
|
|
11
|
+
*/
|
|
12
|
+
const CELL_SEPARATOR = '\t';
|
|
5
13
|
/**
|
|
6
14
|
* Generates plain text from an AST.
|
|
7
15
|
*/
|
|
@@ -16,13 +24,26 @@ class TextGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
16
24
|
let output = '';
|
|
17
25
|
const newline = this.config.textConfig.newlineDelimiter;
|
|
18
26
|
// Add Metadata Header
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
27
|
+
const meta = this.effectiveMetadata;
|
|
28
|
+
if (this.config.renderMetadata && meta) {
|
|
29
|
+
// The header is a structured `Key: value` block terminated by a rule, and consumers
|
|
30
|
+
// parse it as such. A value containing a line break would forge extra fields - a title
|
|
31
|
+
// of "Real\nAuthor: Attacker" renders an Author line the document never had - so line
|
|
32
|
+
// breaks are folded to spaces, matching how CsvGenerator guards its `#` comment block.
|
|
33
|
+
// Plain text has no code-execution context, but fabricated structure is still a lie
|
|
34
|
+
// about the document, and every AST string is treated as attacker-controlled.
|
|
35
|
+
const oneLine = (value) => String(value ?? '').replace(/[\r\n]+/g, ' ');
|
|
36
|
+
if (meta.title)
|
|
37
|
+
output += `Title: ${oneLine(meta.title)}${newline}`;
|
|
38
|
+
if (meta.author)
|
|
39
|
+
output += `Author: ${oneLine(meta.author)}${newline}`;
|
|
40
|
+
// Guarded like HtmlGenerator's toIsoDate: a malformed date must not render the literal
|
|
41
|
+
// "Invalid Date" into the header as if it were the document's creation time.
|
|
42
|
+
if (meta.created) {
|
|
43
|
+
const created = new Date(meta.created);
|
|
44
|
+
if (!isNaN(created.getTime()))
|
|
45
|
+
output += `Created: ${oneLine(created.toLocaleString())}${newline}`;
|
|
46
|
+
}
|
|
26
47
|
output += `-------------------${newline}${newline}`;
|
|
27
48
|
}
|
|
28
49
|
const processor = async (node, childrenOutput) => {
|
|
@@ -40,6 +61,17 @@ class TextGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
40
61
|
const meta = node.metadata;
|
|
41
62
|
return `[Image: ${meta?.altText || meta?.attachmentName || 'Untitled'}]${newline}`;
|
|
42
63
|
}
|
|
64
|
+
if (node.type === 'embed') {
|
|
65
|
+
const meta = node.metadata;
|
|
66
|
+
return meta?.url ? `[${meta.embedType === 'youtube' ? 'YouTube' : 'Embed'}: ${meta.url}]${newline}` : '';
|
|
67
|
+
}
|
|
68
|
+
if (node.type === 'admonition') {
|
|
69
|
+
const meta = node.metadata;
|
|
70
|
+
if (childrenOutput.trim() === '')
|
|
71
|
+
return '';
|
|
72
|
+
const label = (meta?.admonitionType || 'note').toUpperCase();
|
|
73
|
+
return `[${label}] ${childrenOutput.trim()}${newline}`;
|
|
74
|
+
}
|
|
43
75
|
if (node.type === 'table' && this.config.textConfig.preserveLayout) {
|
|
44
76
|
return await this.renderTable(node, processor, newline);
|
|
45
77
|
}
|
|
@@ -50,26 +82,81 @@ class TextGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
50
82
|
const marker = meta?.listType === 'ordered' ? `${(meta.itemIndex ?? 0) + 1}. ` : '- ';
|
|
51
83
|
return `${indent}${marker}${childrenOutput.trimStart()}` + (childrenOutput.endsWith(newline) ? '' : newline);
|
|
52
84
|
}
|
|
53
|
-
//
|
|
85
|
+
// Cells need an explicit separator between them. Without one they concatenate into a
|
|
86
|
+
// single run - "ITEM" + "NEEDED" becomes "ITEMNEEDED", which is unreadable and loses the
|
|
87
|
+
// cell boundary entirely. This was previously masked for spreadsheets only because
|
|
88
|
+
// XLSX/ODS cell values happen to carry a trailing non-breaking space in the source data,
|
|
89
|
+
// so they *appeared* separated; formats whose cell text has no trailing whitespace (MD,
|
|
90
|
+
// HTML) collided outright. Relying on the data to supply a delimiter isn't a separator.
|
|
91
|
+
//
|
|
92
|
+
// Applies to the paths that don't go through renderTable: a `table` when preserveLayout
|
|
93
|
+
// is off, and `sheet`/`row`/`cell` always, since a sheet is not a `table` node.
|
|
94
|
+
if (node.type === 'cell') {
|
|
95
|
+
// Only when the cell doesn't already end in a line break. Cell content varies by
|
|
96
|
+
// source: DOCX/ODT/RTF wrap it in a paragraph, which emits its own newline and so
|
|
97
|
+
// already separates cells, while MD/HTML/XLSX hold bare text that would otherwise
|
|
98
|
+
// collide. Appending unconditionally would push a stray tab onto the formats that
|
|
99
|
+
// were already correct - measured, not assumed: doing so moved DOCX from 6 to 178
|
|
100
|
+
// line-edits away from toText().
|
|
101
|
+
if (childrenOutput === '' || childrenOutput.endsWith(newline))
|
|
102
|
+
return childrenOutput;
|
|
103
|
+
return childrenOutput + CELL_SEPARATOR;
|
|
104
|
+
}
|
|
105
|
+
// Append newline for block-level elements to maintain structure.
|
|
54
106
|
const blockTypes = ['paragraph', 'heading', 'row', 'sheet', 'slide', 'note', 'list', 'table', 'code'];
|
|
107
|
+
if (node.type === 'row') {
|
|
108
|
+
// Trailing separator on the final cell is an artifact of appending one per cell, not
|
|
109
|
+
// content, so drop it rather than leaving every row ending in a stray tab.
|
|
110
|
+
const row = childrenOutput.endsWith(CELL_SEPARATOR)
|
|
111
|
+
? childrenOutput.slice(0, -CELL_SEPARATOR.length)
|
|
112
|
+
: childrenOutput;
|
|
113
|
+
if (row === '')
|
|
114
|
+
return '';
|
|
115
|
+
return row + (row.endsWith(newline) ? '' : newline);
|
|
116
|
+
}
|
|
55
117
|
if (blockTypes.includes(node.type)) {
|
|
56
|
-
|
|
118
|
+
// Drop a block only when it is genuinely empty, not merely whitespace. A paragraph
|
|
119
|
+
// containing spaces is content the document actually holds - discarding it silently
|
|
120
|
+
// deletes an author's blank-but-not-empty line, and disagreed with every parser's
|
|
121
|
+
// own toTextSync, which filters on `!== ''` rather than on trimmed emptiness.
|
|
122
|
+
if (childrenOutput === '')
|
|
57
123
|
return '';
|
|
58
124
|
return childrenOutput + (childrenOutput.endsWith(newline) ? '' : newline);
|
|
59
125
|
}
|
|
126
|
+
// Fallback for node types with no explicit handling above. Prefer rendered children,
|
|
127
|
+
// but fall back to the node's own text when it has none: a `chart` carries its whole
|
|
128
|
+
// data series in `text` with zero child nodes, and a CSV `comment` likewise, so
|
|
129
|
+
// returning only `childrenOutput` silently dropped both. Every parser's own toTextSync
|
|
130
|
+
// reads `node.text`, so this is what the rest of the library already does - and it
|
|
131
|
+
// covers any future node type of the same shape rather than just the two known today.
|
|
132
|
+
if (!childrenOutput && node.text) {
|
|
133
|
+
return node.text + (node.text.endsWith(newline) ? '' : newline);
|
|
134
|
+
}
|
|
60
135
|
return childrenOutput;
|
|
61
136
|
};
|
|
62
137
|
for (const node of this.ast.content) {
|
|
63
138
|
output += await this.processNodeRecursive(node, processor);
|
|
64
139
|
}
|
|
65
|
-
if (this.collectedNotes.length > 0) {
|
|
140
|
+
if (this.collectedNotes.length > 0 && this.config.textConfig.renderNotes) {
|
|
66
141
|
output += `${newline}${newline}--- Notes ---${newline}`;
|
|
67
142
|
for (const note of this.collectedNotes) {
|
|
68
143
|
output += await this.processNodeRecursive(note, processor);
|
|
69
144
|
}
|
|
70
145
|
}
|
|
146
|
+
// Every block-level node above unconditionally appends its own trailing `newline` as a
|
|
147
|
+
// separator from whatever sibling follows - including, unavoidably, the very last one,
|
|
148
|
+
// which has no sibling to separate from - and renderTable below unconditionally *prepends*
|
|
149
|
+
// one too, as a separator from whatever precedes it (also unavoidably applied when a table
|
|
150
|
+
// is the very first/only node). Both are pure generator artifacts, never part of the
|
|
151
|
+
// document's actual content, so a run of exactly this delimiter at either end is the only
|
|
152
|
+
// thing safe to strip. Nothing else is: not leading/trailing spaces or tabs (e.g. an
|
|
153
|
+
// intentionally-indented opening line, or trailing spaces on the last line - both real
|
|
154
|
+
// content), and not any whitespace that isn't composed of this exact repeated delimiter. A
|
|
155
|
+
// blanket trim()/trimEnd() would silently destroy all of those.
|
|
156
|
+
const d = escapeRegExpChars(newline);
|
|
157
|
+
const leadingOrTrailingArtifact = new RegExp(`^(?:${d})+|(?:${d})+$`, 'g');
|
|
71
158
|
return {
|
|
72
|
-
value: output.
|
|
159
|
+
value: output.replace(leadingOrTrailingArtifact, ''),
|
|
73
160
|
messages: this.messages
|
|
74
161
|
};
|
|
75
162
|
}
|
package/dist/index.d.ts
CHANGED