officeparser 7.2.2 → 7.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +161 -17
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +3 -2
- package/dist/defaults.js +3 -3
- package/dist/generators/BaseGenerator.d.ts +11 -0
- package/dist/generators/BaseGenerator.js +29 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +24 -14
- package/dist/generators/EpubGenerator.d.ts +18 -0
- package/dist/generators/EpubGenerator.js +242 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +266 -51
- package/dist/generators/MarkdownGenerator.d.ts +16 -0
- package/dist/generators/MarkdownGenerator.js +173 -24
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +12 -15
- package/dist/generators/TextGenerator.js +11 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +144 -7
- package/dist/officeparser.browser.iife.js +289 -193
- package/dist/officeparser.browser.mjs +289 -193
- package/dist/officeparser.browser.slim.d.ts +2129 -0
- package/dist/officeparser.browser.slim.iife.js +1278 -0
- package/dist/officeparser.browser.slim.mjs +1277 -0
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/HtmlParser.js +284 -20
- package/dist/parsers/MarkdownParser.js +424 -33
- package/dist/parsers/OpenOfficeParser.js +241 -54
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/WordParser.js +2 -2
- package/dist/sbom.cdx.json +111 -223
- package/dist/types.d.ts +146 -7
- package/dist/types.js +2 -0
- package/dist/utils/errorUtils.js +3 -2
- package/dist/utils/sanitize.d.ts +99 -0
- package/dist/utils/sanitize.js +228 -0
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +19 -10
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Parses an EPUB file (a ZIP archive of XHTML content plus an OPF manifest) into the
|
|
4
|
+
* unified OfficeParserAST. Each spine item is parsed via the existing `HtmlParser` and
|
|
5
|
+
* the resulting content/attachments are concatenated in reading order - EPUB is
|
|
6
|
+
* essentially a sequence of XHTML documents, so there's no need for a bespoke content model.
|
|
7
|
+
*/
|
|
8
|
+
export declare const parseEpub: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.parseEpub = void 0;
|
|
4
|
+
const types_js_1 = require("../types.js");
|
|
5
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
6
|
+
const dateUtils_js_1 = require("../utils/dateUtils.js");
|
|
7
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
8
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
9
|
+
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
10
|
+
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
11
|
+
const HtmlParser_js_1 = require("./HtmlParser.js");
|
|
12
|
+
/**
|
|
13
|
+
* Resolves a manifest-relative href against the OPF file's directory, collapsing
|
|
14
|
+
* `./` and `../` segments the way a normal filesystem path resolver would.
|
|
15
|
+
*/
|
|
16
|
+
const resolveOpfPath = (opfDir, href) => {
|
|
17
|
+
const parts = (opfDir + href).split('/');
|
|
18
|
+
const resolved = [];
|
|
19
|
+
for (const part of parts) {
|
|
20
|
+
if (part === '.' || part === '')
|
|
21
|
+
continue;
|
|
22
|
+
if (part === '..')
|
|
23
|
+
resolved.pop();
|
|
24
|
+
else
|
|
25
|
+
resolved.push(part);
|
|
26
|
+
}
|
|
27
|
+
return resolved.join('/');
|
|
28
|
+
};
|
|
29
|
+
/**
|
|
30
|
+
* Parses an EPUB file (a ZIP archive of XHTML content plus an OPF manifest) into the
|
|
31
|
+
* unified OfficeParserAST. Each spine item is parsed via the existing `HtmlParser` and
|
|
32
|
+
* the resulting content/attachments are concatenated in reading order - EPUB is
|
|
33
|
+
* essentially a sequence of XHTML documents, so there's no need for a bespoke content model.
|
|
34
|
+
*/
|
|
35
|
+
const parseEpub = async (buffer, config) => {
|
|
36
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
37
|
+
const files = await (0, zipUtils_js_1.extractFiles)(buffer, (path) => /META-INF\/container\.xml$/i.test(path)
|
|
38
|
+
|| /\.opf$/i.test(path)
|
|
39
|
+
|| /\.(xhtml|html|htm)$/i.test(path)
|
|
40
|
+
|| (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits);
|
|
41
|
+
// The OPF path is authoritative via META-INF/container.xml; fall back to scanning
|
|
42
|
+
// for any .opf file for malformed archives that skip the container manifest.
|
|
43
|
+
let opfPath;
|
|
44
|
+
const containerFile = files.find(f => /META-INF\/container\.xml$/i.test(f.path));
|
|
45
|
+
if (containerFile) {
|
|
46
|
+
const containerXml = (0, xmlUtils_js_1.parseXmlString)(containerFile.content.toString('utf-8'));
|
|
47
|
+
const rootfile = (0, xmlUtils_js_1.getFirstElementByTagName)(containerXml, 'rootfile');
|
|
48
|
+
opfPath = rootfile ? (0, xmlUtils_js_1.getAttribute)(rootfile, 'full-path') : undefined;
|
|
49
|
+
}
|
|
50
|
+
const opfFile = (opfPath && files.find(f => f.path === opfPath)) || files.find(f => /\.opf$/i.test(f.path));
|
|
51
|
+
if (!opfFile) {
|
|
52
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_CORRUPTED, config, 'epub (no OPF manifest found)');
|
|
53
|
+
}
|
|
54
|
+
const opfDir = opfFile.path.includes('/') ? opfFile.path.substring(0, opfFile.path.lastIndexOf('/') + 1) : '';
|
|
55
|
+
const opfXml = (0, xmlUtils_js_1.parseXmlString)(opfFile.content.toString('utf-8'));
|
|
56
|
+
// ─── Metadata (Dublin Core) ─────────────────────────────────────────────
|
|
57
|
+
const metadata = {};
|
|
58
|
+
const metadataEl = (0, xmlUtils_js_1.getFirstElementByTagName)(opfXml, 'metadata');
|
|
59
|
+
if (metadataEl) {
|
|
60
|
+
const nativeProps = {};
|
|
61
|
+
const dcText = (tag) => (0, xmlUtils_js_1.getElementsByTagName)(metadataEl, tag)[0]?.textContent || undefined;
|
|
62
|
+
const title = dcText('dc:title');
|
|
63
|
+
if (title) {
|
|
64
|
+
metadata.title = title;
|
|
65
|
+
nativeProps.title = title;
|
|
66
|
+
}
|
|
67
|
+
const creator = dcText('dc:creator');
|
|
68
|
+
if (creator) {
|
|
69
|
+
metadata.author = creator;
|
|
70
|
+
nativeProps.creator = creator;
|
|
71
|
+
}
|
|
72
|
+
const description = dcText('dc:description');
|
|
73
|
+
if (description) {
|
|
74
|
+
metadata.description = description;
|
|
75
|
+
nativeProps.description = description;
|
|
76
|
+
}
|
|
77
|
+
const subject = dcText('dc:subject');
|
|
78
|
+
if (subject) {
|
|
79
|
+
metadata.subject = subject;
|
|
80
|
+
nativeProps.subject = subject;
|
|
81
|
+
}
|
|
82
|
+
const dateStr = dcText('dc:date');
|
|
83
|
+
if (dateStr) {
|
|
84
|
+
nativeProps.date = dateStr;
|
|
85
|
+
metadata.created = (0, dateUtils_js_1.parseOfficeDate)(dateStr) || (isNaN(Date.parse(dateStr)) ? undefined : new Date(dateStr));
|
|
86
|
+
}
|
|
87
|
+
const publisher = dcText('dc:publisher');
|
|
88
|
+
if (publisher)
|
|
89
|
+
nativeProps.publisher = publisher;
|
|
90
|
+
const language = dcText('dc:language');
|
|
91
|
+
if (language)
|
|
92
|
+
nativeProps.language = language;
|
|
93
|
+
const identifier = dcText('dc:identifier');
|
|
94
|
+
if (identifier)
|
|
95
|
+
nativeProps.identifier = identifier;
|
|
96
|
+
// Calibre/EPUB2-style <meta name="..." content="..."> refinements
|
|
97
|
+
for (const metaTag of (0, xmlUtils_js_1.getElementsByTagName)(metadataEl, 'meta')) {
|
|
98
|
+
const name = (0, xmlUtils_js_1.getAttribute)(metaTag, 'name');
|
|
99
|
+
const content = (0, xmlUtils_js_1.getAttribute)(metaTag, 'content');
|
|
100
|
+
if (name && content)
|
|
101
|
+
nativeProps[name] = content;
|
|
102
|
+
}
|
|
103
|
+
if (Object.keys(nativeProps).length > 0)
|
|
104
|
+
metadata.nativeProperties = nativeProps;
|
|
105
|
+
}
|
|
106
|
+
// ─── Manifest: id -> {href, mediaType} ──────────────────────────────────
|
|
107
|
+
const manifest = new Map();
|
|
108
|
+
let coverImageId;
|
|
109
|
+
for (const item of (0, xmlUtils_js_1.getElementsByTagName)(opfXml, 'item')) {
|
|
110
|
+
const id = (0, xmlUtils_js_1.getAttribute)(item, 'id');
|
|
111
|
+
const href = (0, xmlUtils_js_1.getAttribute)(item, 'href');
|
|
112
|
+
const mediaType = (0, xmlUtils_js_1.getAttribute)(item, 'media-type') || '';
|
|
113
|
+
if (id && href)
|
|
114
|
+
manifest.set(id, { href, mediaType });
|
|
115
|
+
if (((0, xmlUtils_js_1.getAttribute)(item, 'properties') || '').split(/\s+/).includes('cover-image'))
|
|
116
|
+
coverImageId = id;
|
|
117
|
+
}
|
|
118
|
+
if (!coverImageId) {
|
|
119
|
+
// EPUB2-style cover declaration: <meta name="cover" content="{manifest id}">
|
|
120
|
+
const coverMeta = metadataEl && (0, xmlUtils_js_1.getElementsByTagName)(metadataEl, 'meta').find(m => (0, xmlUtils_js_1.getAttribute)(m, 'name') === 'cover');
|
|
121
|
+
coverImageId = coverMeta ? (0, xmlUtils_js_1.getAttribute)(coverMeta, 'content') : undefined;
|
|
122
|
+
}
|
|
123
|
+
// ─── Spine: ordered reading order of XHTML documents ────────────────────
|
|
124
|
+
const spineHrefs = [];
|
|
125
|
+
for (const itemref of (0, xmlUtils_js_1.getElementsByTagName)(opfXml, 'itemref')) {
|
|
126
|
+
const idref = (0, xmlUtils_js_1.getAttribute)(itemref, 'idref');
|
|
127
|
+
const item = idref ? manifest.get(idref) : undefined;
|
|
128
|
+
if (item && /html/i.test(item.mediaType))
|
|
129
|
+
spineHrefs.push(item.href);
|
|
130
|
+
}
|
|
131
|
+
const content = [];
|
|
132
|
+
const attachments = [];
|
|
133
|
+
// Map each in-zip image resource by its resolved path, so inline <img> references can
|
|
134
|
+
// be resolved to real bytes (EPUB images are separate files referenced by relative
|
|
135
|
+
// path, unlike DOCX's embedded parts).
|
|
136
|
+
const imageByPath = new Map();
|
|
137
|
+
if (config.extractAttachments) {
|
|
138
|
+
for (const [, item] of manifest) {
|
|
139
|
+
if (!item.mediaType.startsWith('image/'))
|
|
140
|
+
continue;
|
|
141
|
+
const p = resolveOpfPath(opfDir, item.href);
|
|
142
|
+
const f = files.find(ff => ff.path === p);
|
|
143
|
+
if (f)
|
|
144
|
+
imageByPath.set(p, { content: f.content, mediaType: item.mediaType });
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
const referencedImagePaths = new Set();
|
|
148
|
+
for (const href of spineHrefs) {
|
|
149
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
150
|
+
const xhtmlPath = resolveOpfPath(opfDir, href.split('#')[0]);
|
|
151
|
+
const xhtmlFile = files.find(f => f.path === xhtmlPath);
|
|
152
|
+
if (!xhtmlFile)
|
|
153
|
+
continue;
|
|
154
|
+
let xhtml = xhtmlFile.content.toString('utf-8');
|
|
155
|
+
if (config.extractAttachments && imageByPath.size > 0) {
|
|
156
|
+
// Inline each referenced image as a data URI so HtmlParser extracts it as an
|
|
157
|
+
// attachment (with a real image node linked by name) - the same treatment
|
|
158
|
+
// DOCX images get, and what makes the image survive conversion to any format.
|
|
159
|
+
const xhtmlDir = xhtmlPath.includes('/') ? xhtmlPath.substring(0, xhtmlPath.lastIndexOf('/') + 1) : '';
|
|
160
|
+
xhtml = xhtml.replace(/(<img\b[^>]*\bsrc=")([^"]+)(")/gi, (full, pre, src, post) => {
|
|
161
|
+
if (/^(data:|https?:|\/\/)/i.test(src))
|
|
162
|
+
return full;
|
|
163
|
+
const resolved = resolveOpfPath(xhtmlDir, src.split('#')[0].split('?')[0]);
|
|
164
|
+
const img = imageByPath.get(resolved);
|
|
165
|
+
if (!img)
|
|
166
|
+
return full;
|
|
167
|
+
referencedImagePaths.add(resolved);
|
|
168
|
+
return `${pre}data:${img.mediaType};base64,${img.content.toString('base64')}${post}`;
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
const chapterAst = await (0, HtmlParser_js_1.parseHtml)(Buffer.from(xhtml, 'utf-8'), config);
|
|
172
|
+
content.push(...chapterAst.content);
|
|
173
|
+
attachments.push(...chapterAst.attachments);
|
|
174
|
+
}
|
|
175
|
+
// Keep manifest images that were NOT referenced inline (e.g. cover art, or images used
|
|
176
|
+
// only as CSS list-style bullets) as attachments so the raw assets aren't lost - DOCX
|
|
177
|
+
// likewise exposes such images as attachments even without an inline image node.
|
|
178
|
+
if (config.extractAttachments) {
|
|
179
|
+
const customProperties = {};
|
|
180
|
+
for (const [id, item] of manifest) {
|
|
181
|
+
if (!item.mediaType.startsWith('image/'))
|
|
182
|
+
continue;
|
|
183
|
+
const p = resolveOpfPath(opfDir, item.href);
|
|
184
|
+
const img = imageByPath.get(p);
|
|
185
|
+
if (!img || referencedImagePaths.has(p))
|
|
186
|
+
continue;
|
|
187
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(item.href.split('/').pop() || item.href, img.content);
|
|
188
|
+
attachments.push(attachment);
|
|
189
|
+
if (id === coverImageId)
|
|
190
|
+
customProperties.coverImageName = attachment.name;
|
|
191
|
+
}
|
|
192
|
+
if (Object.keys(customProperties).length > 0) {
|
|
193
|
+
metadata.customProperties = { ...metadata.customProperties, ...customProperties };
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
const toTextSync = () => content.map(n => {
|
|
197
|
+
const getText = (node) => {
|
|
198
|
+
if (node.type === 'text' || node.type === 'code')
|
|
199
|
+
return node.text || '';
|
|
200
|
+
if (node.type === 'break')
|
|
201
|
+
return '\n';
|
|
202
|
+
if (node.type === 'embed')
|
|
203
|
+
return node.metadata?.url || '';
|
|
204
|
+
if (node.type === 'image')
|
|
205
|
+
return node.metadata?.altText || '';
|
|
206
|
+
if (node.children) {
|
|
207
|
+
const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition'].includes(node.type);
|
|
208
|
+
return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
|
|
209
|
+
}
|
|
210
|
+
return '';
|
|
211
|
+
};
|
|
212
|
+
return getText(n);
|
|
213
|
+
}).join(config.newlineDelimiter)
|
|
214
|
+
.replace(/\n{3,}/g, '\n\n');
|
|
215
|
+
return (0, astUtils_js_1.createAST)('epub', metadata, content, attachments, config, undefined, toTextSync);
|
|
216
|
+
};
|
|
217
|
+
exports.parseEpub = parseEpub;
|