officeparser 7.2.2 → 7.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/README.md +161 -17
  2. package/dist/OfficeGenerator.js +4 -0
  3. package/dist/OfficeParser.d.ts +2 -0
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +1 -1
  6. package/dist/cli.js +3 -2
  7. package/dist/defaults.js +3 -3
  8. package/dist/generators/BaseGenerator.d.ts +11 -0
  9. package/dist/generators/BaseGenerator.js +29 -0
  10. package/dist/generators/CsvGenerator.d.ts +9 -1
  11. package/dist/generators/CsvGenerator.js +24 -14
  12. package/dist/generators/EpubGenerator.d.ts +18 -0
  13. package/dist/generators/EpubGenerator.js +242 -0
  14. package/dist/generators/HtmlGenerator.d.ts +12 -0
  15. package/dist/generators/HtmlGenerator.js +266 -51
  16. package/dist/generators/MarkdownGenerator.d.ts +16 -0
  17. package/dist/generators/MarkdownGenerator.js +173 -24
  18. package/dist/generators/PdfGenerator.js +32 -0
  19. package/dist/generators/RtfGenerator.js +12 -15
  20. package/dist/generators/TextGenerator.js +11 -0
  21. package/dist/index.d.ts +1 -0
  22. package/dist/index.js +1 -0
  23. package/dist/officeparser.browser.d.ts +144 -7
  24. package/dist/officeparser.browser.iife.js +289 -193
  25. package/dist/officeparser.browser.mjs +289 -193
  26. package/dist/officeparser.browser.slim.d.ts +2129 -0
  27. package/dist/officeparser.browser.slim.iife.js +1278 -0
  28. package/dist/officeparser.browser.slim.mjs +1277 -0
  29. package/dist/parsers/EpubParser.d.ts +8 -0
  30. package/dist/parsers/EpubParser.js +217 -0
  31. package/dist/parsers/HtmlParser.js +284 -20
  32. package/dist/parsers/MarkdownParser.js +424 -33
  33. package/dist/parsers/OpenOfficeParser.js +241 -54
  34. package/dist/parsers/PdfParser.js +4 -1
  35. package/dist/parsers/WordParser.js +2 -2
  36. package/dist/sbom.cdx.json +111 -223
  37. package/dist/types.d.ts +146 -7
  38. package/dist/types.js +2 -0
  39. package/dist/utils/errorUtils.js +3 -2
  40. package/dist/utils/sanitize.d.ts +99 -0
  41. package/dist/utils/sanitize.js +228 -0
  42. package/dist/utils/zipUtils.js +76 -26
  43. package/package.json +19 -10
@@ -0,0 +1,242 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.EpubGenerator = void 0;
4
+ const fflate_1 = require("fflate");
5
+ const BaseGenerator_js_1 = require("./BaseGenerator.js");
6
+ const HtmlGenerator_js_1 = require("./HtmlGenerator.js");
7
+ const sanitize_js_1 = require("../utils/sanitize.js");
8
+ const VOID_TAGS = ['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr'];
9
+ /** Block-level tags whose HTML5 content model does not permit them inside a <p>. */
10
+ const BLOCK_TAGS_INVALID_IN_P = ['div', 'table', 'ul', 'ol', 'dl', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'blockquote', 'pre', 'section', 'figure'];
11
+ /**
12
+ * HTML5's content model forbids block elements inside <p> (a <p> can only hold "phrasing
13
+ * content"), but browsers silently fix this via the HTML5 parsing algorithm's
14
+ * auto-closing rule: seeing a block start tag implicitly closes the open <p> first.
15
+ * XML parsers have no such rule - they just build the tree exactly as written. Since
16
+ * HtmlGenerator always wraps images in `<div class="image-container">`, a paragraph
17
+ * whose only child is an image becomes `<p><div>...</div></p>`: well-formed XML, but
18
+ * many EPUB rendering engines refuse to lay out a block box found inside a paragraph and
19
+ * simply drop it - silently, with no parse error, which is why the image vanishes.
20
+ *
21
+ * Fixes this by promoting any `<p ...>` that contains a nested block tag to a `<div ...>`
22
+ * instead, matching what a browser's auto-correction effectively produces. Paragraphs
23
+ * don't nest, so the first `</p>` after each `<p>` is always its match.
24
+ */
25
+ const promoteParagraphsWithBlockContent = (html) => {
26
+ const blockTagPattern = new RegExp(`<(?:${BLOCK_TAGS_INVALID_IN_P.join('|')})\\b`, 'i');
27
+ let result = '';
28
+ let cursor = 0;
29
+ const pOpenRegex = /<p(\s[^>]*)?>/gi;
30
+ let match;
31
+ while ((match = pOpenRegex.exec(html)) !== null) {
32
+ if (match.index < cursor)
33
+ continue; // inside content already emitted by a prior promotion
34
+ result += html.slice(cursor, match.index);
35
+ const contentStart = match.index + match[0].length;
36
+ const closeMatch = /<\/p>/i.exec(html.slice(contentStart));
37
+ if (!closeMatch) {
38
+ // No closing tag found (shouldn't happen with well-formed generator output) -
39
+ // leave as-is rather than risk corrupting the rest of the document.
40
+ result += match[0];
41
+ cursor = contentStart;
42
+ pOpenRegex.lastIndex = cursor;
43
+ continue;
44
+ }
45
+ const inner = html.slice(contentStart, contentStart + closeMatch.index);
46
+ const attrs = match[1] || '';
47
+ result += blockTagPattern.test(inner) ? `<div${attrs}>${inner}</div>` : `<p${attrs}>${inner}</p>`;
48
+ cursor = contentStart + closeMatch.index + closeMatch[0].length;
49
+ pOpenRegex.lastIndex = cursor;
50
+ }
51
+ result += html.slice(cursor);
52
+ return result;
53
+ };
54
+ /**
55
+ * Converts HtmlGenerator's HTML output into well-formed XHTML, which EPUB reading
56
+ * systems parse as strict XML (unlike browsers, which tolerate HTML's looseness).
57
+ *
58
+ * This is more than cosmetic: a single raw `&` or unclosed tag makes the whole content
59
+ * document fail to open. The conversion:
60
+ * - strips `<script>` blocks — EpubGenerator renders through HtmlGenerator with
61
+ * `standalone: false`, which already omits the envelope-level stylesheet and Chart.js/
62
+ * spreadsheet scripts entirely, but a chart *node* still emits its own inline
63
+ * `<script>` (chart-init JS) regardless of that flag, since it's content, not envelope.
64
+ * A reading system can't execute it anyway, and its JS operators can contain raw `&`/`<`
65
+ * that are illegal as XML character data, so it's stripped here;
66
+ * - promotes `<p>` tags that contain nested block content (see
67
+ * promoteParagraphsWithBlockContent above) to `<div>`, since XML readers don't apply
68
+ * HTML5's auto-closing correction that hides this in a browser;
69
+ * - normalises HTML named entities (`&nbsp;`) to numeric references, since XML predefines
70
+ * only `&amp;`/`&lt;`/`&gt;`/`&quot;`/`&apos;`;
71
+ * - escapes stray ampersands (e.g. in `href` query strings) not already part of a valid
72
+ * reference;
73
+ * - gives HTML boolean attributes an explicit value (`checked` -> `checked="checked"`);
74
+ * - self-closes void elements (`<br>` -> `<br/>`).
75
+ */
76
+ const toXhtml = (html) => {
77
+ let out = html.replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, '');
78
+ out = promoteParagraphsWithBlockContent(out);
79
+ // Named -> numeric entities (nbsp is the only named entity HtmlGenerator emits).
80
+ out = out.replace(/&nbsp;/g, '&#160;');
81
+ // Escape ampersands that don't already open a valid XML entity reference.
82
+ out = out.replace(/&(?!(?:amp|lt|gt|quot|apos|#\d+|#x[0-9a-fA-F]+);)/g, '&amp;');
83
+ // Give bare boolean attributes an explicit value. Scoped to the specific tags that
84
+ // emit them (checkbox task-list items, media iframes) so body text like "the selected
85
+ // option" is never rewritten.
86
+ out = out.replace(/<input\b([^>]*?)\schecked(\s*\/?>)/gi, '<input$1 checked="checked"$2');
87
+ out = out.replace(/<(iframe|video|audio)\b([^>]*?)\s(allowfullscreen|autoplay|controls|loop|muted)(\s*\/?>|\s)/gi, '<$1$2 $3="$3"$4');
88
+ // Self-close void elements (non-greedy attr capture so an already-present trailing `/`
89
+ // isn't duplicated, e.g. `<meta .../>` must not become `<meta ...//>`).
90
+ const voidTagPattern = new RegExp(`<(${VOID_TAGS.join('|')})((?:\\s[^>]*?)?)\\s*/?>`, 'gi');
91
+ out = out.replace(voidTagPattern, (_m, tag, attrs) => `<${tag}${attrs}/>`);
92
+ return out;
93
+ };
94
+ /**
95
+ * Minimal, book-friendly CSS injected into every EPUB content document. Kept static and
96
+ * free of `&`/`<` so it is XML-safe inline; the reading system supplies typography, so
97
+ * this only covers structural essentials the stripped page-chrome would otherwise lose.
98
+ */
99
+ const EPUB_STYLESHEET = `img { max-width: 100%; height: auto; }
100
+ table { border-collapse: collapse; margin: 1em 0; }
101
+ td, th { border: 1px solid #ccc; padding: 4px 8px; }`;
102
+ /** Maps an image MIME type to a file extension for the packaged resource. */
103
+ const MIME_EXT = {
104
+ 'image/jpeg': 'jpg', 'image/jpg': 'jpg', 'image/png': 'png', 'image/gif': 'gif',
105
+ 'image/svg+xml': 'svg', 'image/webp': 'webp', 'image/bmp': 'bmp', 'image/tiff': 'tiff'
106
+ };
107
+ /** Decodes a base64 string to raw bytes, cross-env (atob exists in Node 16+ and browsers). */
108
+ const decodeBase64 = (b64) => {
109
+ const bin = atob(b64);
110
+ const bytes = new Uint8Array(bin.length);
111
+ for (let i = 0; i < bin.length; i++)
112
+ bytes[i] = bin.charCodeAt(i);
113
+ return bytes;
114
+ };
115
+ /**
116
+ * Generates a minimal, valid EPUB 3 file from an AST.
117
+ *
118
+ * Every AST node is rendered as a single XHTML content document (reusing `HtmlGenerator`
119
+ * for the actual markup, since EPUB content documents are XHTML) and packaged with the
120
+ * required `mimetype`, `META-INF/container.xml`, OPF manifest, and navigation document.
121
+ *
122
+ * `HtmlGenerator` embeds images as base64 `data:` URIs, but EPUB reading systems do not
123
+ * render `data:` URIs - images must be packaged as separate resources referenced by a
124
+ * relative path. So each data-URI image is extracted into `OEBPS/images/`, declared in
125
+ * the manifest, and its `<img src>` rewritten to point at the packaged file.
126
+ */
127
+ class EpubGenerator extends BaseGenerator_js_1.BaseGenerator {
128
+ constructor(ast, config) {
129
+ super('epub', ast, config);
130
+ }
131
+ async generate() {
132
+ const htmlGenerator = new HtmlGenerator_js_1.HtmlGenerator(this.ast, {
133
+ ...this.config,
134
+ htmlConfig: { ...this.config.htmlConfig, standalone: false },
135
+ });
136
+ const htmlResult = await htmlGenerator.generate();
137
+ let bodyHtml = typeof htmlResult.value === 'string' ? htmlResult.value : '';
138
+ // Extract base64 data-URI images into packaged files (EPUB readers don't render
139
+ // `data:` URIs). Each distinct image becomes one OEBPS/images/imageN.ext resource,
140
+ // a manifest <item>, and a rewritten relative `src`. Deduped so a repeated image
141
+ // is packaged once.
142
+ const imageResources = {};
143
+ const imageManifestItems = [];
144
+ const dataUriToHref = new Map();
145
+ let imageCounter = 0;
146
+ bodyHtml = bodyHtml.replace(/(<img\b[^>]*\bsrc=")(data:(image\/[a-zA-Z0-9.+-]+);base64,([^"]+))(")/gi, (_full, pre, dataUri, mime, b64, post) => {
147
+ let href = dataUriToHref.get(dataUri);
148
+ if (!href) {
149
+ imageCounter++;
150
+ const ext = MIME_EXT[mime.toLowerCase()] || 'img';
151
+ href = `images/image${imageCounter}.${ext}`;
152
+ dataUriToHref.set(dataUri, href);
153
+ try {
154
+ imageResources[`OEBPS/${href}`] = decodeBase64(b64);
155
+ imageManifestItems.push(`<item id="img${imageCounter}" href="${href}" media-type="${mime}"/>`);
156
+ }
157
+ catch {
158
+ // Undecodable data - leave the original src untouched rather than
159
+ // emit a manifest entry for a resource we couldn't write.
160
+ dataUriToHref.delete(dataUri);
161
+ return `${pre}${dataUri}${post}`;
162
+ }
163
+ }
164
+ return `${pre}${href}${post}`;
165
+ });
166
+ const xhtmlBody = toXhtml(bodyHtml);
167
+ const title = this.ast.metadata?.title || 'Untitled';
168
+ const author = this.ast.metadata?.author;
169
+ const description = this.ast.metadata?.description;
170
+ const nativeProps = (this.ast.metadata?.nativeProperties || {});
171
+ const language = nativeProps.language || 'en';
172
+ const identifier = nativeProps.identifier || `urn:x-officeparser:${this.slugify(title)}-${xhtmlBody.length}`;
173
+ const modified = new Date().toISOString().replace(/\.\d+Z$/, 'Z');
174
+ const chapterXhtml = `<?xml version="1.0" encoding="UTF-8"?>
175
+ <!DOCTYPE html>
176
+ <html xmlns="http://www.w3.org/1999/xhtml" xml:lang="${(0, sanitize_js_1.escapeXml)(language)}">
177
+ <head>
178
+ <meta charset="utf-8"/>
179
+ <title>${(0, sanitize_js_1.escapeXml)(title)}</title>
180
+ <style type="text/css">
181
+ ${EPUB_STYLESHEET}
182
+ </style>
183
+ </head>
184
+ <body>
185
+ ${xhtmlBody}
186
+ </body>
187
+ </html>`;
188
+ const opf = `<?xml version="1.0" encoding="UTF-8"?>
189
+ <package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="pub-id">
190
+ <metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
191
+ <dc:identifier id="pub-id">${(0, sanitize_js_1.escapeXml)(identifier)}</dc:identifier>
192
+ <dc:title>${(0, sanitize_js_1.escapeXml)(title)}</dc:title>
193
+ ${author ? `<dc:creator>${(0, sanitize_js_1.escapeXml)(author)}</dc:creator>` : ''}
194
+ ${description ? `<dc:description>${(0, sanitize_js_1.escapeXml)(description)}</dc:description>` : ''}
195
+ <dc:language>${(0, sanitize_js_1.escapeXml)(language)}</dc:language>
196
+ <meta property="dcterms:modified">${modified}</meta>
197
+ </metadata>
198
+ <manifest>
199
+ <item id="chapter1" href="chapter1.xhtml" media-type="application/xhtml+xml"/>
200
+ <item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" properties="nav"/>${imageManifestItems.length ? '\n ' + imageManifestItems.join('\n ') : ''}
201
+ </manifest>
202
+ <spine>
203
+ <itemref idref="chapter1"/>
204
+ </spine>
205
+ </package>`;
206
+ const navXhtml = `<?xml version="1.0" encoding="UTF-8"?>
207
+ <!DOCTYPE html>
208
+ <html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops">
209
+ <head><meta charset="utf-8"/><title>Navigation</title></head>
210
+ <body>
211
+ <nav epub:type="toc" id="toc">
212
+ <h1>${(0, sanitize_js_1.escapeXml)(title)}</h1>
213
+ <ol>
214
+ <li><a href="chapter1.xhtml">${(0, sanitize_js_1.escapeXml)(title)}</a></li>
215
+ </ol>
216
+ </nav>
217
+ </body>
218
+ </html>`;
219
+ const containerXml = `<?xml version="1.0" encoding="UTF-8"?>
220
+ <container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
221
+ <rootfiles>
222
+ <rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/>
223
+ </rootfiles>
224
+ </container>`;
225
+ const encoder = new TextEncoder();
226
+ // EPUB requires the mimetype entry to be the first file in the archive, stored
227
+ // uncompressed (level 0) - readers use it to sniff the format before parsing any XML.
228
+ const zipFiles = {
229
+ mimetype: [encoder.encode('application/epub+zip'), { level: 0 }],
230
+ 'META-INF/container.xml': encoder.encode(containerXml),
231
+ 'OEBPS/content.opf': encoder.encode(opf),
232
+ 'OEBPS/nav.xhtml': encoder.encode(navXhtml),
233
+ 'OEBPS/chapter1.xhtml': encoder.encode(chapterXhtml),
234
+ ...imageResources,
235
+ };
236
+ return {
237
+ value: (0, fflate_1.zipSync)(zipFiles),
238
+ messages: this.messages,
239
+ };
240
+ }
241
+ }
242
+ exports.EpubGenerator = EpubGenerator;
@@ -32,7 +32,19 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
32
32
  private formatText;
33
33
  private getInlineStyles;
34
34
  private getPremiumStyles;
35
+ /**
36
+ * Same as `getPremiumStyles()`, but wrapped in a CSS `@scope` block anchored to the
37
+ * `.op-html-scope` wrapper so the rules only apply within the generated fragment - they
38
+ * cannot leak onto a host page's own elements. `:root` and `body` selectors specifically
39
+ * target the real page root/body, so they're remapped to `:scope` (the scope root, i.e. the
40
+ * `.op-html-scope` wrapper) first; every other selector is naturally confined by `@scope`
41
+ * without needing per-selector rewriting. `customCss` is included in this scoping too.
42
+ */
43
+ private getScopedPremiumStyles;
35
44
  protected slugify(text: string): string;
36
45
  private getColumnLetter;
37
46
  private escape;
47
+ /** Converts a document-supplied date to an ISO string, or '' if it is invalid
48
+ * (a malformed date would otherwise throw a RangeError and abort generation). */
49
+ private toIsoDate;
38
50
  }