officeparser 7.2.2 → 7.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +161 -17
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +3 -2
- package/dist/defaults.js +3 -3
- package/dist/generators/BaseGenerator.d.ts +11 -0
- package/dist/generators/BaseGenerator.js +29 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +24 -14
- package/dist/generators/EpubGenerator.d.ts +18 -0
- package/dist/generators/EpubGenerator.js +242 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +266 -51
- package/dist/generators/MarkdownGenerator.d.ts +16 -0
- package/dist/generators/MarkdownGenerator.js +173 -24
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +12 -15
- package/dist/generators/TextGenerator.js +11 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +144 -7
- package/dist/officeparser.browser.iife.js +289 -193
- package/dist/officeparser.browser.mjs +289 -193
- package/dist/officeparser.browser.slim.d.ts +2129 -0
- package/dist/officeparser.browser.slim.iife.js +1278 -0
- package/dist/officeparser.browser.slim.mjs +1277 -0
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/HtmlParser.js +284 -20
- package/dist/parsers/MarkdownParser.js +424 -33
- package/dist/parsers/OpenOfficeParser.js +241 -54
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/WordParser.js +2 -2
- package/dist/sbom.cdx.json +111 -223
- package/dist/types.d.ts +146 -7
- package/dist/types.js +2 -0
- package/dist/utils/errorUtils.js +3 -2
- package/dist/utils/sanitize.d.ts +99 -0
- package/dist/utils/sanitize.js +228 -0
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +19 -10
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.EpubGenerator = void 0;
|
|
4
|
+
const fflate_1 = require("fflate");
|
|
5
|
+
const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
6
|
+
const HtmlGenerator_js_1 = require("./HtmlGenerator.js");
|
|
7
|
+
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
8
|
+
const VOID_TAGS = ['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr'];
|
|
9
|
+
/** Block-level tags whose HTML5 content model does not permit them inside a <p>. */
|
|
10
|
+
const BLOCK_TAGS_INVALID_IN_P = ['div', 'table', 'ul', 'ol', 'dl', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'blockquote', 'pre', 'section', 'figure'];
|
|
11
|
+
/**
|
|
12
|
+
* HTML5's content model forbids block elements inside <p> (a <p> can only hold "phrasing
|
|
13
|
+
* content"), but browsers silently fix this via the HTML5 parsing algorithm's
|
|
14
|
+
* auto-closing rule: seeing a block start tag implicitly closes the open <p> first.
|
|
15
|
+
* XML parsers have no such rule - they just build the tree exactly as written. Since
|
|
16
|
+
* HtmlGenerator always wraps images in `<div class="image-container">`, a paragraph
|
|
17
|
+
* whose only child is an image becomes `<p><div>...</div></p>`: well-formed XML, but
|
|
18
|
+
* many EPUB rendering engines refuse to lay out a block box found inside a paragraph and
|
|
19
|
+
* simply drop it - silently, with no parse error, which is why the image vanishes.
|
|
20
|
+
*
|
|
21
|
+
* Fixes this by promoting any `<p ...>` that contains a nested block tag to a `<div ...>`
|
|
22
|
+
* instead, matching what a browser's auto-correction effectively produces. Paragraphs
|
|
23
|
+
* don't nest, so the first `</p>` after each `<p>` is always its match.
|
|
24
|
+
*/
|
|
25
|
+
const promoteParagraphsWithBlockContent = (html) => {
|
|
26
|
+
const blockTagPattern = new RegExp(`<(?:${BLOCK_TAGS_INVALID_IN_P.join('|')})\\b`, 'i');
|
|
27
|
+
let result = '';
|
|
28
|
+
let cursor = 0;
|
|
29
|
+
const pOpenRegex = /<p(\s[^>]*)?>/gi;
|
|
30
|
+
let match;
|
|
31
|
+
while ((match = pOpenRegex.exec(html)) !== null) {
|
|
32
|
+
if (match.index < cursor)
|
|
33
|
+
continue; // inside content already emitted by a prior promotion
|
|
34
|
+
result += html.slice(cursor, match.index);
|
|
35
|
+
const contentStart = match.index + match[0].length;
|
|
36
|
+
const closeMatch = /<\/p>/i.exec(html.slice(contentStart));
|
|
37
|
+
if (!closeMatch) {
|
|
38
|
+
// No closing tag found (shouldn't happen with well-formed generator output) -
|
|
39
|
+
// leave as-is rather than risk corrupting the rest of the document.
|
|
40
|
+
result += match[0];
|
|
41
|
+
cursor = contentStart;
|
|
42
|
+
pOpenRegex.lastIndex = cursor;
|
|
43
|
+
continue;
|
|
44
|
+
}
|
|
45
|
+
const inner = html.slice(contentStart, contentStart + closeMatch.index);
|
|
46
|
+
const attrs = match[1] || '';
|
|
47
|
+
result += blockTagPattern.test(inner) ? `<div${attrs}>${inner}</div>` : `<p${attrs}>${inner}</p>`;
|
|
48
|
+
cursor = contentStart + closeMatch.index + closeMatch[0].length;
|
|
49
|
+
pOpenRegex.lastIndex = cursor;
|
|
50
|
+
}
|
|
51
|
+
result += html.slice(cursor);
|
|
52
|
+
return result;
|
|
53
|
+
};
|
|
54
|
+
/**
|
|
55
|
+
* Converts HtmlGenerator's HTML output into well-formed XHTML, which EPUB reading
|
|
56
|
+
* systems parse as strict XML (unlike browsers, which tolerate HTML's looseness).
|
|
57
|
+
*
|
|
58
|
+
* This is more than cosmetic: a single raw `&` or unclosed tag makes the whole content
|
|
59
|
+
* document fail to open. The conversion:
|
|
60
|
+
* - strips `<script>` blocks — EpubGenerator renders through HtmlGenerator with
|
|
61
|
+
* `standalone: false`, which already omits the envelope-level stylesheet and Chart.js/
|
|
62
|
+
* spreadsheet scripts entirely, but a chart *node* still emits its own inline
|
|
63
|
+
* `<script>` (chart-init JS) regardless of that flag, since it's content, not envelope.
|
|
64
|
+
* A reading system can't execute it anyway, and its JS operators can contain raw `&`/`<`
|
|
65
|
+
* that are illegal as XML character data, so it's stripped here;
|
|
66
|
+
* - promotes `<p>` tags that contain nested block content (see
|
|
67
|
+
* promoteParagraphsWithBlockContent above) to `<div>`, since XML readers don't apply
|
|
68
|
+
* HTML5's auto-closing correction that hides this in a browser;
|
|
69
|
+
* - normalises HTML named entities (` `) to numeric references, since XML predefines
|
|
70
|
+
* only `&`/`<`/`>`/`"`/`'`;
|
|
71
|
+
* - escapes stray ampersands (e.g. in `href` query strings) not already part of a valid
|
|
72
|
+
* reference;
|
|
73
|
+
* - gives HTML boolean attributes an explicit value (`checked` -> `checked="checked"`);
|
|
74
|
+
* - self-closes void elements (`<br>` -> `<br/>`).
|
|
75
|
+
*/
|
|
76
|
+
const toXhtml = (html) => {
|
|
77
|
+
let out = html.replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, '');
|
|
78
|
+
out = promoteParagraphsWithBlockContent(out);
|
|
79
|
+
// Named -> numeric entities (nbsp is the only named entity HtmlGenerator emits).
|
|
80
|
+
out = out.replace(/ /g, ' ');
|
|
81
|
+
// Escape ampersands that don't already open a valid XML entity reference.
|
|
82
|
+
out = out.replace(/&(?!(?:amp|lt|gt|quot|apos|#\d+|#x[0-9a-fA-F]+);)/g, '&');
|
|
83
|
+
// Give bare boolean attributes an explicit value. Scoped to the specific tags that
|
|
84
|
+
// emit them (checkbox task-list items, media iframes) so body text like "the selected
|
|
85
|
+
// option" is never rewritten.
|
|
86
|
+
out = out.replace(/<input\b([^>]*?)\schecked(\s*\/?>)/gi, '<input$1 checked="checked"$2');
|
|
87
|
+
out = out.replace(/<(iframe|video|audio)\b([^>]*?)\s(allowfullscreen|autoplay|controls|loop|muted)(\s*\/?>|\s)/gi, '<$1$2 $3="$3"$4');
|
|
88
|
+
// Self-close void elements (non-greedy attr capture so an already-present trailing `/`
|
|
89
|
+
// isn't duplicated, e.g. `<meta .../>` must not become `<meta ...//>`).
|
|
90
|
+
const voidTagPattern = new RegExp(`<(${VOID_TAGS.join('|')})((?:\\s[^>]*?)?)\\s*/?>`, 'gi');
|
|
91
|
+
out = out.replace(voidTagPattern, (_m, tag, attrs) => `<${tag}${attrs}/>`);
|
|
92
|
+
return out;
|
|
93
|
+
};
|
|
94
|
+
/**
|
|
95
|
+
* Minimal, book-friendly CSS injected into every EPUB content document. Kept static and
|
|
96
|
+
* free of `&`/`<` so it is XML-safe inline; the reading system supplies typography, so
|
|
97
|
+
* this only covers structural essentials the stripped page-chrome would otherwise lose.
|
|
98
|
+
*/
|
|
99
|
+
const EPUB_STYLESHEET = `img { max-width: 100%; height: auto; }
|
|
100
|
+
table { border-collapse: collapse; margin: 1em 0; }
|
|
101
|
+
td, th { border: 1px solid #ccc; padding: 4px 8px; }`;
|
|
102
|
+
/** Maps an image MIME type to a file extension for the packaged resource. */
|
|
103
|
+
const MIME_EXT = {
|
|
104
|
+
'image/jpeg': 'jpg', 'image/jpg': 'jpg', 'image/png': 'png', 'image/gif': 'gif',
|
|
105
|
+
'image/svg+xml': 'svg', 'image/webp': 'webp', 'image/bmp': 'bmp', 'image/tiff': 'tiff'
|
|
106
|
+
};
|
|
107
|
+
/** Decodes a base64 string to raw bytes, cross-env (atob exists in Node 16+ and browsers). */
|
|
108
|
+
const decodeBase64 = (b64) => {
|
|
109
|
+
const bin = atob(b64);
|
|
110
|
+
const bytes = new Uint8Array(bin.length);
|
|
111
|
+
for (let i = 0; i < bin.length; i++)
|
|
112
|
+
bytes[i] = bin.charCodeAt(i);
|
|
113
|
+
return bytes;
|
|
114
|
+
};
|
|
115
|
+
/**
|
|
116
|
+
* Generates a minimal, valid EPUB 3 file from an AST.
|
|
117
|
+
*
|
|
118
|
+
* Every AST node is rendered as a single XHTML content document (reusing `HtmlGenerator`
|
|
119
|
+
* for the actual markup, since EPUB content documents are XHTML) and packaged with the
|
|
120
|
+
* required `mimetype`, `META-INF/container.xml`, OPF manifest, and navigation document.
|
|
121
|
+
*
|
|
122
|
+
* `HtmlGenerator` embeds images as base64 `data:` URIs, but EPUB reading systems do not
|
|
123
|
+
* render `data:` URIs - images must be packaged as separate resources referenced by a
|
|
124
|
+
* relative path. So each data-URI image is extracted into `OEBPS/images/`, declared in
|
|
125
|
+
* the manifest, and its `<img src>` rewritten to point at the packaged file.
|
|
126
|
+
*/
|
|
127
|
+
class EpubGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
128
|
+
constructor(ast, config) {
|
|
129
|
+
super('epub', ast, config);
|
|
130
|
+
}
|
|
131
|
+
async generate() {
|
|
132
|
+
const htmlGenerator = new HtmlGenerator_js_1.HtmlGenerator(this.ast, {
|
|
133
|
+
...this.config,
|
|
134
|
+
htmlConfig: { ...this.config.htmlConfig, standalone: false },
|
|
135
|
+
});
|
|
136
|
+
const htmlResult = await htmlGenerator.generate();
|
|
137
|
+
let bodyHtml = typeof htmlResult.value === 'string' ? htmlResult.value : '';
|
|
138
|
+
// Extract base64 data-URI images into packaged files (EPUB readers don't render
|
|
139
|
+
// `data:` URIs). Each distinct image becomes one OEBPS/images/imageN.ext resource,
|
|
140
|
+
// a manifest <item>, and a rewritten relative `src`. Deduped so a repeated image
|
|
141
|
+
// is packaged once.
|
|
142
|
+
const imageResources = {};
|
|
143
|
+
const imageManifestItems = [];
|
|
144
|
+
const dataUriToHref = new Map();
|
|
145
|
+
let imageCounter = 0;
|
|
146
|
+
bodyHtml = bodyHtml.replace(/(<img\b[^>]*\bsrc=")(data:(image\/[a-zA-Z0-9.+-]+);base64,([^"]+))(")/gi, (_full, pre, dataUri, mime, b64, post) => {
|
|
147
|
+
let href = dataUriToHref.get(dataUri);
|
|
148
|
+
if (!href) {
|
|
149
|
+
imageCounter++;
|
|
150
|
+
const ext = MIME_EXT[mime.toLowerCase()] || 'img';
|
|
151
|
+
href = `images/image${imageCounter}.${ext}`;
|
|
152
|
+
dataUriToHref.set(dataUri, href);
|
|
153
|
+
try {
|
|
154
|
+
imageResources[`OEBPS/${href}`] = decodeBase64(b64);
|
|
155
|
+
imageManifestItems.push(`<item id="img${imageCounter}" href="${href}" media-type="${mime}"/>`);
|
|
156
|
+
}
|
|
157
|
+
catch {
|
|
158
|
+
// Undecodable data - leave the original src untouched rather than
|
|
159
|
+
// emit a manifest entry for a resource we couldn't write.
|
|
160
|
+
dataUriToHref.delete(dataUri);
|
|
161
|
+
return `${pre}${dataUri}${post}`;
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return `${pre}${href}${post}`;
|
|
165
|
+
});
|
|
166
|
+
const xhtmlBody = toXhtml(bodyHtml);
|
|
167
|
+
const title = this.ast.metadata?.title || 'Untitled';
|
|
168
|
+
const author = this.ast.metadata?.author;
|
|
169
|
+
const description = this.ast.metadata?.description;
|
|
170
|
+
const nativeProps = (this.ast.metadata?.nativeProperties || {});
|
|
171
|
+
const language = nativeProps.language || 'en';
|
|
172
|
+
const identifier = nativeProps.identifier || `urn:x-officeparser:${this.slugify(title)}-${xhtmlBody.length}`;
|
|
173
|
+
const modified = new Date().toISOString().replace(/\.\d+Z$/, 'Z');
|
|
174
|
+
const chapterXhtml = `<?xml version="1.0" encoding="UTF-8"?>
|
|
175
|
+
<!DOCTYPE html>
|
|
176
|
+
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="${(0, sanitize_js_1.escapeXml)(language)}">
|
|
177
|
+
<head>
|
|
178
|
+
<meta charset="utf-8"/>
|
|
179
|
+
<title>${(0, sanitize_js_1.escapeXml)(title)}</title>
|
|
180
|
+
<style type="text/css">
|
|
181
|
+
${EPUB_STYLESHEET}
|
|
182
|
+
</style>
|
|
183
|
+
</head>
|
|
184
|
+
<body>
|
|
185
|
+
${xhtmlBody}
|
|
186
|
+
</body>
|
|
187
|
+
</html>`;
|
|
188
|
+
const opf = `<?xml version="1.0" encoding="UTF-8"?>
|
|
189
|
+
<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="pub-id">
|
|
190
|
+
<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
|
|
191
|
+
<dc:identifier id="pub-id">${(0, sanitize_js_1.escapeXml)(identifier)}</dc:identifier>
|
|
192
|
+
<dc:title>${(0, sanitize_js_1.escapeXml)(title)}</dc:title>
|
|
193
|
+
${author ? `<dc:creator>${(0, sanitize_js_1.escapeXml)(author)}</dc:creator>` : ''}
|
|
194
|
+
${description ? `<dc:description>${(0, sanitize_js_1.escapeXml)(description)}</dc:description>` : ''}
|
|
195
|
+
<dc:language>${(0, sanitize_js_1.escapeXml)(language)}</dc:language>
|
|
196
|
+
<meta property="dcterms:modified">${modified}</meta>
|
|
197
|
+
</metadata>
|
|
198
|
+
<manifest>
|
|
199
|
+
<item id="chapter1" href="chapter1.xhtml" media-type="application/xhtml+xml"/>
|
|
200
|
+
<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" properties="nav"/>${imageManifestItems.length ? '\n ' + imageManifestItems.join('\n ') : ''}
|
|
201
|
+
</manifest>
|
|
202
|
+
<spine>
|
|
203
|
+
<itemref idref="chapter1"/>
|
|
204
|
+
</spine>
|
|
205
|
+
</package>`;
|
|
206
|
+
const navXhtml = `<?xml version="1.0" encoding="UTF-8"?>
|
|
207
|
+
<!DOCTYPE html>
|
|
208
|
+
<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops">
|
|
209
|
+
<head><meta charset="utf-8"/><title>Navigation</title></head>
|
|
210
|
+
<body>
|
|
211
|
+
<nav epub:type="toc" id="toc">
|
|
212
|
+
<h1>${(0, sanitize_js_1.escapeXml)(title)}</h1>
|
|
213
|
+
<ol>
|
|
214
|
+
<li><a href="chapter1.xhtml">${(0, sanitize_js_1.escapeXml)(title)}</a></li>
|
|
215
|
+
</ol>
|
|
216
|
+
</nav>
|
|
217
|
+
</body>
|
|
218
|
+
</html>`;
|
|
219
|
+
const containerXml = `<?xml version="1.0" encoding="UTF-8"?>
|
|
220
|
+
<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
|
|
221
|
+
<rootfiles>
|
|
222
|
+
<rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/>
|
|
223
|
+
</rootfiles>
|
|
224
|
+
</container>`;
|
|
225
|
+
const encoder = new TextEncoder();
|
|
226
|
+
// EPUB requires the mimetype entry to be the first file in the archive, stored
|
|
227
|
+
// uncompressed (level 0) - readers use it to sniff the format before parsing any XML.
|
|
228
|
+
const zipFiles = {
|
|
229
|
+
mimetype: [encoder.encode('application/epub+zip'), { level: 0 }],
|
|
230
|
+
'META-INF/container.xml': encoder.encode(containerXml),
|
|
231
|
+
'OEBPS/content.opf': encoder.encode(opf),
|
|
232
|
+
'OEBPS/nav.xhtml': encoder.encode(navXhtml),
|
|
233
|
+
'OEBPS/chapter1.xhtml': encoder.encode(chapterXhtml),
|
|
234
|
+
...imageResources,
|
|
235
|
+
};
|
|
236
|
+
return {
|
|
237
|
+
value: (0, fflate_1.zipSync)(zipFiles),
|
|
238
|
+
messages: this.messages,
|
|
239
|
+
};
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
exports.EpubGenerator = EpubGenerator;
|
|
@@ -32,7 +32,19 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
|
|
|
32
32
|
private formatText;
|
|
33
33
|
private getInlineStyles;
|
|
34
34
|
private getPremiumStyles;
|
|
35
|
+
/**
|
|
36
|
+
* Same as `getPremiumStyles()`, but wrapped in a CSS `@scope` block anchored to the
|
|
37
|
+
* `.op-html-scope` wrapper so the rules only apply within the generated fragment - they
|
|
38
|
+
* cannot leak onto a host page's own elements. `:root` and `body` selectors specifically
|
|
39
|
+
* target the real page root/body, so they're remapped to `:scope` (the scope root, i.e. the
|
|
40
|
+
* `.op-html-scope` wrapper) first; every other selector is naturally confined by `@scope`
|
|
41
|
+
* without needing per-selector rewriting. `customCss` is included in this scoping too.
|
|
42
|
+
*/
|
|
43
|
+
private getScopedPremiumStyles;
|
|
35
44
|
protected slugify(text: string): string;
|
|
36
45
|
private getColumnLetter;
|
|
37
46
|
private escape;
|
|
47
|
+
/** Converts a document-supplied date to an ISO string, or '' if it is invalid
|
|
48
|
+
* (a malformed date would otherwise throw a RangeError and abort generation). */
|
|
49
|
+
private toIsoDate;
|
|
38
50
|
}
|