officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -0,0 +1,8 @@
1
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
2
+ /**
3
+ * Parses an EPUB file (a ZIP archive of XHTML content plus an OPF manifest) into the
4
+ * unified OfficeParserAST. Each spine item is parsed via the existing `HtmlParser` and
5
+ * the resulting content/attachments are concatenated in reading order - EPUB is
6
+ * essentially a sequence of XHTML documents, so there's no need for a bespoke content model.
7
+ */
8
+ export declare const parseEpub: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
@@ -0,0 +1,217 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.parseEpub = void 0;
4
+ const types_js_1 = require("../types.js");
5
+ const astUtils_js_1 = require("../utils/astUtils.js");
6
+ const dateUtils_js_1 = require("../utils/dateUtils.js");
7
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
8
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
9
+ const xmlUtils_js_1 = require("../utils/xmlUtils.js");
10
+ const zipUtils_js_1 = require("../utils/zipUtils.js");
11
+ const HtmlParser_js_1 = require("./HtmlParser.js");
12
+ /**
13
+ * Resolves a manifest-relative href against the OPF file's directory, collapsing
14
+ * `./` and `../` segments the way a normal filesystem path resolver would.
15
+ */
16
+ const resolveOpfPath = (opfDir, href) => {
17
+ const parts = (opfDir + href).split('/');
18
+ const resolved = [];
19
+ for (const part of parts) {
20
+ if (part === '.' || part === '')
21
+ continue;
22
+ if (part === '..')
23
+ resolved.pop();
24
+ else
25
+ resolved.push(part);
26
+ }
27
+ return resolved.join('/');
28
+ };
29
+ /**
30
+ * Parses an EPUB file (a ZIP archive of XHTML content plus an OPF manifest) into the
31
+ * unified OfficeParserAST. Each spine item is parsed via the existing `HtmlParser` and
32
+ * the resulting content/attachments are concatenated in reading order - EPUB is
33
+ * essentially a sequence of XHTML documents, so there's no need for a bespoke content model.
34
+ */
35
+ const parseEpub = async (buffer, config) => {
36
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
37
+ const files = await (0, zipUtils_js_1.extractFiles)(buffer, (path) => /META-INF\/container\.xml$/i.test(path)
38
+ || /\.opf$/i.test(path)
39
+ || /\.(xhtml|html|htm)$/i.test(path)
40
+ || (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits);
41
+ // The OPF path is authoritative via META-INF/container.xml; fall back to scanning
42
+ // for any .opf file for malformed archives that skip the container manifest.
43
+ let opfPath;
44
+ const containerFile = files.find(f => /META-INF\/container\.xml$/i.test(f.path));
45
+ if (containerFile) {
46
+ const containerXml = (0, xmlUtils_js_1.parseXmlString)(containerFile.content.toString('utf-8'));
47
+ const rootfile = (0, xmlUtils_js_1.getFirstElementByTagName)(containerXml, 'rootfile');
48
+ opfPath = rootfile ? (0, xmlUtils_js_1.getAttribute)(rootfile, 'full-path') : undefined;
49
+ }
50
+ const opfFile = (opfPath && files.find(f => f.path === opfPath)) || files.find(f => /\.opf$/i.test(f.path));
51
+ if (!opfFile) {
52
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_CORRUPTED, config, 'epub (no OPF manifest found)');
53
+ }
54
+ const opfDir = opfFile.path.includes('/') ? opfFile.path.substring(0, opfFile.path.lastIndexOf('/') + 1) : '';
55
+ const opfXml = (0, xmlUtils_js_1.parseXmlString)(opfFile.content.toString('utf-8'));
56
+ // ─── Metadata (Dublin Core) ─────────────────────────────────────────────
57
+ const metadata = {};
58
+ const metadataEl = (0, xmlUtils_js_1.getFirstElementByTagName)(opfXml, 'metadata');
59
+ if (metadataEl) {
60
+ const nativeProps = {};
61
+ const dcText = (tag) => (0, xmlUtils_js_1.getElementsByTagName)(metadataEl, tag)[0]?.textContent || undefined;
62
+ const title = dcText('dc:title');
63
+ if (title) {
64
+ metadata.title = title;
65
+ nativeProps.title = title;
66
+ }
67
+ const creator = dcText('dc:creator');
68
+ if (creator) {
69
+ metadata.author = creator;
70
+ nativeProps.creator = creator;
71
+ }
72
+ const description = dcText('dc:description');
73
+ if (description) {
74
+ metadata.description = description;
75
+ nativeProps.description = description;
76
+ }
77
+ const subject = dcText('dc:subject');
78
+ if (subject) {
79
+ metadata.subject = subject;
80
+ nativeProps.subject = subject;
81
+ }
82
+ const dateStr = dcText('dc:date');
83
+ if (dateStr) {
84
+ nativeProps.date = dateStr;
85
+ metadata.created = (0, dateUtils_js_1.parseOfficeDate)(dateStr) || (isNaN(Date.parse(dateStr)) ? undefined : new Date(dateStr));
86
+ }
87
+ const publisher = dcText('dc:publisher');
88
+ if (publisher)
89
+ nativeProps.publisher = publisher;
90
+ const language = dcText('dc:language');
91
+ if (language)
92
+ nativeProps.language = language;
93
+ const identifier = dcText('dc:identifier');
94
+ if (identifier)
95
+ nativeProps.identifier = identifier;
96
+ // Calibre/EPUB2-style <meta name="..." content="..."> refinements
97
+ for (const metaTag of (0, xmlUtils_js_1.getElementsByTagName)(metadataEl, 'meta')) {
98
+ const name = (0, xmlUtils_js_1.getAttribute)(metaTag, 'name');
99
+ const content = (0, xmlUtils_js_1.getAttribute)(metaTag, 'content');
100
+ if (name && content)
101
+ nativeProps[name] = content;
102
+ }
103
+ if (Object.keys(nativeProps).length > 0)
104
+ metadata.nativeProperties = nativeProps;
105
+ }
106
+ // ─── Manifest: id -> {href, mediaType} ──────────────────────────────────
107
+ const manifest = new Map();
108
+ let coverImageId;
109
+ for (const item of (0, xmlUtils_js_1.getElementsByTagName)(opfXml, 'item')) {
110
+ const id = (0, xmlUtils_js_1.getAttribute)(item, 'id');
111
+ const href = (0, xmlUtils_js_1.getAttribute)(item, 'href');
112
+ const mediaType = (0, xmlUtils_js_1.getAttribute)(item, 'media-type') || '';
113
+ if (id && href)
114
+ manifest.set(id, { href, mediaType });
115
+ if (((0, xmlUtils_js_1.getAttribute)(item, 'properties') || '').split(/\s+/).includes('cover-image'))
116
+ coverImageId = id;
117
+ }
118
+ if (!coverImageId) {
119
+ // EPUB2-style cover declaration: <meta name="cover" content="{manifest id}">
120
+ const coverMeta = metadataEl && (0, xmlUtils_js_1.getElementsByTagName)(metadataEl, 'meta').find(m => (0, xmlUtils_js_1.getAttribute)(m, 'name') === 'cover');
121
+ coverImageId = coverMeta ? (0, xmlUtils_js_1.getAttribute)(coverMeta, 'content') : undefined;
122
+ }
123
+ // ─── Spine: ordered reading order of XHTML documents ────────────────────
124
+ const spineHrefs = [];
125
+ for (const itemref of (0, xmlUtils_js_1.getElementsByTagName)(opfXml, 'itemref')) {
126
+ const idref = (0, xmlUtils_js_1.getAttribute)(itemref, 'idref');
127
+ const item = idref ? manifest.get(idref) : undefined;
128
+ if (item && /html/i.test(item.mediaType))
129
+ spineHrefs.push(item.href);
130
+ }
131
+ const content = [];
132
+ const attachments = [];
133
+ // Map each in-zip image resource by its resolved path, so inline <img> references can
134
+ // be resolved to real bytes (EPUB images are separate files referenced by relative
135
+ // path, unlike DOCX's embedded parts).
136
+ const imageByPath = new Map();
137
+ if (config.extractAttachments) {
138
+ for (const [, item] of manifest) {
139
+ if (!item.mediaType.startsWith('image/'))
140
+ continue;
141
+ const p = resolveOpfPath(opfDir, item.href);
142
+ const f = files.find(ff => ff.path === p);
143
+ if (f)
144
+ imageByPath.set(p, { content: f.content, mediaType: item.mediaType });
145
+ }
146
+ }
147
+ const referencedImagePaths = new Set();
148
+ for (const href of spineHrefs) {
149
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
150
+ const xhtmlPath = resolveOpfPath(opfDir, href.split('#')[0]);
151
+ const xhtmlFile = files.find(f => f.path === xhtmlPath);
152
+ if (!xhtmlFile)
153
+ continue;
154
+ let xhtml = xhtmlFile.content.toString('utf-8');
155
+ if (config.extractAttachments && imageByPath.size > 0) {
156
+ // Inline each referenced image as a data URI so HtmlParser extracts it as an
157
+ // attachment (with a real image node linked by name) - the same treatment
158
+ // DOCX images get, and what makes the image survive conversion to any format.
159
+ const xhtmlDir = xhtmlPath.includes('/') ? xhtmlPath.substring(0, xhtmlPath.lastIndexOf('/') + 1) : '';
160
+ xhtml = xhtml.replace(/(<img\b[^>]*\bsrc=")([^"]+)(")/gi, (full, pre, src, post) => {
161
+ if (/^(data:|https?:|\/\/)/i.test(src))
162
+ return full;
163
+ const resolved = resolveOpfPath(xhtmlDir, src.split('#')[0].split('?')[0]);
164
+ const img = imageByPath.get(resolved);
165
+ if (!img)
166
+ return full;
167
+ referencedImagePaths.add(resolved);
168
+ return `${pre}data:${img.mediaType};base64,${img.content.toString('base64')}${post}`;
169
+ });
170
+ }
171
+ const chapterAst = await (0, HtmlParser_js_1.parseHtml)(Buffer.from(xhtml, 'utf-8'), config);
172
+ content.push(...chapterAst.content);
173
+ attachments.push(...chapterAst.attachments);
174
+ }
175
+ // Keep manifest images that were NOT referenced inline (e.g. cover art, or images used
176
+ // only as CSS list-style bullets) as attachments so the raw assets aren't lost - DOCX
177
+ // likewise exposes such images as attachments even without an inline image node.
178
+ if (config.extractAttachments) {
179
+ const customProperties = {};
180
+ for (const [id, item] of manifest) {
181
+ if (!item.mediaType.startsWith('image/'))
182
+ continue;
183
+ const p = resolveOpfPath(opfDir, item.href);
184
+ const img = imageByPath.get(p);
185
+ if (!img || referencedImagePaths.has(p))
186
+ continue;
187
+ const attachment = (0, imageUtils_js_1.createAttachment)(item.href.split('/').pop() || item.href, img.content);
188
+ attachments.push(attachment);
189
+ if (id === coverImageId)
190
+ customProperties.coverImageName = attachment.name;
191
+ }
192
+ if (Object.keys(customProperties).length > 0) {
193
+ metadata.customProperties = { ...metadata.customProperties, ...customProperties };
194
+ }
195
+ }
196
+ const toTextSync = () => content.map(n => {
197
+ const getText = (node) => {
198
+ if (node.type === 'text' || node.type === 'code')
199
+ return node.text || '';
200
+ if (node.type === 'break')
201
+ return '\n';
202
+ if (node.type === 'embed')
203
+ return node.metadata?.url || '';
204
+ if (node.type === 'image')
205
+ return node.metadata?.altText || '';
206
+ if (node.children) {
207
+ const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition'].includes(node.type);
208
+ return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
209
+ }
210
+ return '';
211
+ };
212
+ return getText(n);
213
+ }).join(config.newlineDelimiter)
214
+ .replace(/\n{3,}/g, '\n\n');
215
+ return (0, astUtils_js_1.createAST)('epub', metadata, content, attachments, config, undefined, toTextSync);
216
+ };
217
+ exports.parseEpub = parseEpub;
@@ -392,6 +392,7 @@ const parseExcel = async (buffer, config) => {
392
392
  const workbookXml = (0, xmlUtils_js_1.parseXmlString)(workbookFile.content.toString());
393
393
  const sheets = (0, xmlUtils_js_1.getElementsByTagName)(workbookXml, "sheet");
394
394
  for (const sheet of sheets) {
395
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
395
396
  const name = sheet.getAttribute("name");
396
397
  const rId = sheet.getAttribute("r:id");
397
398
  if (name && rId && rIdToFile[rId]) {
@@ -486,6 +487,7 @@ const parseExcel = async (buffer, config) => {
486
487
  };
487
488
  let lastRowIndex = -1;
488
489
  for (const rowMatch of rowMatches) {
490
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
489
491
  const rowXml = rowMatch[0];
490
492
  const rowAttrs = rowMatch[1];
491
493
  const isSelfClosing = !!rowMatch[2];