officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -85,6 +85,35 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
85
85
  // a system-installed Chrome, both of which are too intrusive for a library.
86
86
  browser = await puppeteer.launch(launchOptions);
87
87
  const page = await browser.newPage();
88
+ // Harden against SSRF: the HTML being rendered is derived from an untrusted
89
+ // document, and `networkidle0` would otherwise fetch every URL it references
90
+ // (external images, stylesheets, etc.) from this host — reaching internal
91
+ // services or a cloud metadata endpoint (169.254.169.254). Intercept requests
92
+ // and allow only inline data/blob URIs and the configured chart CDN; abort every
93
+ // other remote fetch.
94
+ const allowedHosts = new Set();
95
+ try {
96
+ const chartSrc = this.config.htmlConfig?.chartJsSrc;
97
+ if (chartSrc)
98
+ allowedHosts.add(new URL(chartSrc).host);
99
+ }
100
+ catch { /* no or invalid chart CDN configured */ }
101
+ let blockedRemoteResource = false;
102
+ await page.setRequestInterception(true);
103
+ page.on('request', (req) => {
104
+ const url = req.url();
105
+ if (url.startsWith('data:') || url.startsWith('blob:') || url.startsWith('about:')) {
106
+ return req.continue().catch(() => { });
107
+ }
108
+ try {
109
+ if (allowedHosts.has(new URL(url).host)) {
110
+ return req.continue().catch(() => { });
111
+ }
112
+ }
113
+ catch { /* unparseable URL — fall through and block */ }
114
+ blockedRemoteResource = true;
115
+ return req.abort().catch(() => { });
116
+ });
88
117
  const pdfConfig = this.config.pdfConfig;
89
118
  const timeout = pdfConfig.timeout;
90
119
  if (timeout !== undefined && timeout > 0) {
@@ -107,6 +136,9 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
107
136
  });
108
137
  await browser.close();
109
138
  browser = null;
139
+ if (blockedRemoteResource) {
140
+ this.warn(types_js_1.OfficeWarningType.BROWSER_GENERATION_LIMITATION, 'One or more remote resources referenced by the document were blocked during PDF rendering to prevent server-side request forgery (SSRF). Only inline images and the configured chart CDN are loaded.');
141
+ }
110
142
  return {
111
143
  value: new Uint8Array(pdfBuffer),
112
144
  messages: this.messages
@@ -1,7 +1,9 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.RtfGenerator = void 0;
4
+ const sanitize_js_1 = require("../utils/sanitize.js");
4
5
  const BaseGenerator_js_1 = require("./BaseGenerator.js");
6
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
5
7
  /**
6
8
  * Generates high-fidelity RTF (Rich Text Format) from an AST.
7
9
  */
@@ -17,14 +19,23 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
17
19
  const bodyContent = await this.renderBody(this.ast);
18
20
  let output = '{\\rtf1\\ansi\\uc1\\deff0\n';
19
21
  // 1. Info Group (Metadata)
20
- if (this.config.renderMetadata && this.ast.metadata) {
22
+ const meta = this.effectiveMetadata;
23
+ if (this.config.renderMetadata && meta) {
24
+ // RTF's \info group is a fixed set of control words with no slot for caller-defined keys.
25
+ this.warnUnrepresentableCustomMetadata('RTF');
21
26
  output += '{\\info';
22
- if (this.ast.metadata.title)
23
- output += `{\\title ${this.escapeRtf(this.ast.metadata.title)}}`;
24
- if (this.ast.metadata.author)
25
- output += `{\\author ${this.escapeRtf(this.ast.metadata.author)}}`;
26
- if (this.ast.metadata.description)
27
- output += `{\\comm ${this.escapeRtf(this.ast.metadata.description)}}`;
27
+ if (meta.title)
28
+ output += `{\\title ${this.escapeRtf(meta.title)}}`;
29
+ if (meta.author)
30
+ output += `{\\author ${this.escapeRtf(meta.author)}}`;
31
+ if (meta.description)
32
+ output += `{\\comm ${this.escapeRtf(meta.description)}}`;
33
+ // \subject and \keywords are standard \info destinations, so unlike a caller's
34
+ // arbitrary custom keys these DO have a home in RTF.
35
+ if (meta.subject)
36
+ output += `{\\subject ${this.escapeRtf(meta.subject)}}`;
37
+ if (meta.keywords)
38
+ output += `{\\keywords ${this.escapeRtf(meta.keywords)}}`;
28
39
  output += '}\n';
29
40
  }
30
41
  // 2. Font Table
@@ -50,6 +61,10 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
50
61
  };
51
62
  }
52
63
  async processNodeRecursive(node, processor) {
64
+ // Mirrors the check in BaseGenerator.processNodeRecursive. This override replaces that
65
+ // method entirely, so without repeating the check here the signal would be silently
66
+ // inert for this generator - which is exactly how it was missed.
67
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
53
68
  const wasInTable = this.inTable;
54
69
  if (node.type === 'table')
55
70
  this.inTable = true;
@@ -134,7 +149,16 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
134
149
  if (meta?.link) {
135
150
  const isInternal = meta.linkType !== 'external';
136
151
  if (!this.config.ignoreInternalLinks || !isInternal) {
137
- return `{\\field{\\*\\fldinst{HYPERLINK "${meta.link}"}}{\\fldrslt ${text}}}`;
152
+ // Scheme-checked, not merely escaped. escapeRtf neutralizes the field
153
+ // metacharacters but says nothing about where the link points, so RTF
154
+ // was the one generator that would emit `javascript:` or a `file://`
155
+ // /UNC target that HTML and Markdown both reject. On rejection, fall
156
+ // through to the bare link text - the same degradation as HTML's
157
+ // href="" and Markdown's [text]().
158
+ const safeLink = (0, sanitize_js_1.sanitizeRtfUrl)(meta.link);
159
+ if (safeLink) {
160
+ return `{\\field{\\*\\fldinst{HYPERLINK "${safeLink}"}}{\\fldrslt ${text}}}`;
161
+ }
138
162
  }
139
163
  }
140
164
  return text;
@@ -214,6 +238,20 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
214
238
  case 'break': {
215
239
  return node.metadata?.breakType === 'page' ? '\\page\n' : '\\line\n';
216
240
  }
241
+ case 'embed': {
242
+ // RTF has no embed concept - degrade to the URL as plain text rather than
243
+ // silently dropping the node (it has no children to fall back to).
244
+ const meta = node.metadata;
245
+ if (!meta?.url)
246
+ return '';
247
+ // Rendered as visible text rather than a field, but still a URL a reader may
248
+ // copy, so it gets the same scheme policy.
249
+ const safeUrl = (0, sanitize_js_1.sanitizeRtfUrl)(meta.url);
250
+ if (!safeUrl)
251
+ return '';
252
+ const pPr = this.inTable ? '\\pard\\intbl' : '\\pard';
253
+ return `${pPr}\\sa120 ${safeUrl}\\par\n`;
254
+ }
217
255
  default:
218
256
  return childrenOutput;
219
257
  }
@@ -251,20 +289,7 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
251
289
  return luminance > 0.8;
252
290
  }
253
291
  escapeRtf(text) {
254
- return text
255
- .replace(/\\/g, '\\\\')
256
- .replace(/{/g, '\\{')
257
- .replace(/}/g, '\\}')
258
- .replace(/[^\x00-\x7F]/g, (match) => {
259
- let code = match.charCodeAt(0);
260
- if (code < 256) {
261
- return `\\'${code.toString(16).padStart(2, '0')}`;
262
- }
263
- if (code > 32767) {
264
- code -= 65536;
265
- }
266
- return `{\\uc0\\u${code}}`;
267
- });
292
+ return (0, sanitize_js_1.escapeRtf)(text);
268
293
  }
269
294
  }
270
295
  exports.RtfGenerator = RtfGenerator;
@@ -2,6 +2,14 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.TextGenerator = void 0;
4
4
  const BaseGenerator_js_1 = require("./BaseGenerator.js");
5
+ const escapeRegExpChars = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
6
+ /**
7
+ * Separator between table/sheet cells on the paths that don't render an aligned grid (a `table`
8
+ * with `preserveLayout` off, and `sheet`/`row`/`cell`, which never go through `renderTable`).
9
+ * A tab is the conventional plain-text column delimiter and, unlike a space, survives a value that
10
+ * already contains spaces.
11
+ */
12
+ const CELL_SEPARATOR = '\t';
5
13
  /**
6
14
  * Generates plain text from an AST.
7
15
  */
@@ -16,13 +24,26 @@ class TextGenerator extends BaseGenerator_js_1.BaseGenerator {
16
24
  let output = '';
17
25
  const newline = this.config.textConfig.newlineDelimiter;
18
26
  // Add Metadata Header
19
- if (this.config.renderMetadata && this.ast.metadata) {
20
- if (this.ast.metadata.title)
21
- output += `Title: ${this.ast.metadata.title}${newline}`;
22
- if (this.ast.metadata.author)
23
- output += `Author: ${this.ast.metadata.author}${newline}`;
24
- if (this.ast.metadata.created)
25
- output += `Created: ${new Date(this.ast.metadata.created).toLocaleString()}${newline}`;
27
+ const meta = this.effectiveMetadata;
28
+ if (this.config.renderMetadata && meta) {
29
+ // The header is a structured `Key: value` block terminated by a rule, and consumers
30
+ // parse it as such. A value containing a line break would forge extra fields - a title
31
+ // of "Real\nAuthor: Attacker" renders an Author line the document never had - so line
32
+ // breaks are folded to spaces, matching how CsvGenerator guards its `#` comment block.
33
+ // Plain text has no code-execution context, but fabricated structure is still a lie
34
+ // about the document, and every AST string is treated as attacker-controlled.
35
+ const oneLine = (value) => String(value ?? '').replace(/[\r\n]+/g, ' ');
36
+ if (meta.title)
37
+ output += `Title: ${oneLine(meta.title)}${newline}`;
38
+ if (meta.author)
39
+ output += `Author: ${oneLine(meta.author)}${newline}`;
40
+ // Guarded like HtmlGenerator's toIsoDate: a malformed date must not render the literal
41
+ // "Invalid Date" into the header as if it were the document's creation time.
42
+ if (meta.created) {
43
+ const created = new Date(meta.created);
44
+ if (!isNaN(created.getTime()))
45
+ output += `Created: ${oneLine(created.toLocaleString())}${newline}`;
46
+ }
26
47
  output += `-------------------${newline}${newline}`;
27
48
  }
28
49
  const processor = async (node, childrenOutput) => {
@@ -40,6 +61,17 @@ class TextGenerator extends BaseGenerator_js_1.BaseGenerator {
40
61
  const meta = node.metadata;
41
62
  return `[Image: ${meta?.altText || meta?.attachmentName || 'Untitled'}]${newline}`;
42
63
  }
64
+ if (node.type === 'embed') {
65
+ const meta = node.metadata;
66
+ return meta?.url ? `[${meta.embedType === 'youtube' ? 'YouTube' : 'Embed'}: ${meta.url}]${newline}` : '';
67
+ }
68
+ if (node.type === 'admonition') {
69
+ const meta = node.metadata;
70
+ if (childrenOutput.trim() === '')
71
+ return '';
72
+ const label = (meta?.admonitionType || 'note').toUpperCase();
73
+ return `[${label}] ${childrenOutput.trim()}${newline}`;
74
+ }
43
75
  if (node.type === 'table' && this.config.textConfig.preserveLayout) {
44
76
  return await this.renderTable(node, processor, newline);
45
77
  }
@@ -50,26 +82,81 @@ class TextGenerator extends BaseGenerator_js_1.BaseGenerator {
50
82
  const marker = meta?.listType === 'ordered' ? `${(meta.itemIndex ?? 0) + 1}. ` : '- ';
51
83
  return `${indent}${marker}${childrenOutput.trimStart()}` + (childrenOutput.endsWith(newline) ? '' : newline);
52
84
  }
53
- // Append newline for block-level elements to maintain structure
85
+ // Cells need an explicit separator between them. Without one they concatenate into a
86
+ // single run - "ITEM" + "NEEDED" becomes "ITEMNEEDED", which is unreadable and loses the
87
+ // cell boundary entirely. This was previously masked for spreadsheets only because
88
+ // XLSX/ODS cell values happen to carry a trailing non-breaking space in the source data,
89
+ // so they *appeared* separated; formats whose cell text has no trailing whitespace (MD,
90
+ // HTML) collided outright. Relying on the data to supply a delimiter isn't a separator.
91
+ //
92
+ // Applies to the paths that don't go through renderTable: a `table` when preserveLayout
93
+ // is off, and `sheet`/`row`/`cell` always, since a sheet is not a `table` node.
94
+ if (node.type === 'cell') {
95
+ // Only when the cell doesn't already end in a line break. Cell content varies by
96
+ // source: DOCX/ODT/RTF wrap it in a paragraph, which emits its own newline and so
97
+ // already separates cells, while MD/HTML/XLSX hold bare text that would otherwise
98
+ // collide. Appending unconditionally would push a stray tab onto the formats that
99
+ // were already correct - measured, not assumed: doing so moved DOCX from 6 to 178
100
+ // line-edits away from toText().
101
+ if (childrenOutput === '' || childrenOutput.endsWith(newline))
102
+ return childrenOutput;
103
+ return childrenOutput + CELL_SEPARATOR;
104
+ }
105
+ // Append newline for block-level elements to maintain structure.
54
106
  const blockTypes = ['paragraph', 'heading', 'row', 'sheet', 'slide', 'note', 'list', 'table', 'code'];
107
+ if (node.type === 'row') {
108
+ // Trailing separator on the final cell is an artifact of appending one per cell, not
109
+ // content, so drop it rather than leaving every row ending in a stray tab.
110
+ const row = childrenOutput.endsWith(CELL_SEPARATOR)
111
+ ? childrenOutput.slice(0, -CELL_SEPARATOR.length)
112
+ : childrenOutput;
113
+ if (row === '')
114
+ return '';
115
+ return row + (row.endsWith(newline) ? '' : newline);
116
+ }
55
117
  if (blockTypes.includes(node.type)) {
56
- if (childrenOutput.trim() === '')
118
+ // Drop a block only when it is genuinely empty, not merely whitespace. A paragraph
119
+ // containing spaces is content the document actually holds - discarding it silently
120
+ // deletes an author's blank-but-not-empty line, and disagreed with every parser's
121
+ // own toTextSync, which filters on `!== ''` rather than on trimmed emptiness.
122
+ if (childrenOutput === '')
57
123
  return '';
58
124
  return childrenOutput + (childrenOutput.endsWith(newline) ? '' : newline);
59
125
  }
126
+ // Fallback for node types with no explicit handling above. Prefer rendered children,
127
+ // but fall back to the node's own text when it has none: a `chart` carries its whole
128
+ // data series in `text` with zero child nodes, and a CSV `comment` likewise, so
129
+ // returning only `childrenOutput` silently dropped both. Every parser's own toTextSync
130
+ // reads `node.text`, so this is what the rest of the library already does - and it
131
+ // covers any future node type of the same shape rather than just the two known today.
132
+ if (!childrenOutput && node.text) {
133
+ return node.text + (node.text.endsWith(newline) ? '' : newline);
134
+ }
60
135
  return childrenOutput;
61
136
  };
62
137
  for (const node of this.ast.content) {
63
138
  output += await this.processNodeRecursive(node, processor);
64
139
  }
65
- if (this.collectedNotes.length > 0) {
140
+ if (this.collectedNotes.length > 0 && this.config.textConfig.renderNotes) {
66
141
  output += `${newline}${newline}--- Notes ---${newline}`;
67
142
  for (const note of this.collectedNotes) {
68
143
  output += await this.processNodeRecursive(note, processor);
69
144
  }
70
145
  }
146
+ // Every block-level node above unconditionally appends its own trailing `newline` as a
147
+ // separator from whatever sibling follows - including, unavoidably, the very last one,
148
+ // which has no sibling to separate from - and renderTable below unconditionally *prepends*
149
+ // one too, as a separator from whatever precedes it (also unavoidably applied when a table
150
+ // is the very first/only node). Both are pure generator artifacts, never part of the
151
+ // document's actual content, so a run of exactly this delimiter at either end is the only
152
+ // thing safe to strip. Nothing else is: not leading/trailing spaces or tabs (e.g. an
153
+ // intentionally-indented opening line, or trailing spaces on the last line - both real
154
+ // content), and not any whitespace that isn't composed of this exact repeated delimiter. A
155
+ // blanket trim()/trimEnd() would silently destroy all of those.
156
+ const d = escapeRegExpChars(newline);
157
+ const leadingOrTrailingArtifact = new RegExp(`^(?:${d})+|(?:${d})+$`, 'g');
71
158
  return {
72
- value: output.trim(),
159
+ value: output.replace(leadingOrTrailingArtifact, ''),
73
160
  messages: this.messages
74
161
  };
75
162
  }
package/dist/index.d.ts CHANGED
@@ -10,6 +10,7 @@
10
10
  * - Portable: PDF
11
11
  * - Legacy: RTF (Rich Text Format)
12
12
  * - Web/Plain: CSV, MD, HTML
13
+ * - E-book: EPUB
13
14
  *
14
15
  * **Key Features:**
15
16
  * - Unified AST output across all formats
package/dist/index.js CHANGED
@@ -11,6 +11,7 @@
11
11
  * - Portable: PDF
12
12
  * - Legacy: RTF (Rich Text Format)
13
13
  * - Web/Plain: CSV, MD, HTML
14
+ * - E-book: EPUB
14
15
  *
15
16
  * **Key Features:**
16
17
  * - Unified AST output across all formats