officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -1,4 +1,4 @@
1
- import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType, UniversalGeneratorFormat } from '../types.js';
1
+ import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeMetadata, OfficeParserAST, OfficeWarningType, UniversalGeneratorFormat } from '../types.js';
2
2
  import { StyleMapper } from '../utils/styleMapper.js';
3
3
  /**
4
4
  * Base class for all document generators.
@@ -12,6 +12,28 @@ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat =
12
12
  protected styleMapper: StyleMapper;
13
13
  protected collectedNotes: OfficeContentNode[];
14
14
  constructor(destination: D, ast: OfficeParserAST, config?: GeneratorConfig<D> | FullGeneratorConfig);
15
+ /**
16
+ * The document metadata a generator should write out: `ast.metadata` with
17
+ * `config.metadataOverrides` applied on top, per field.
18
+ *
19
+ * Every generator must read metadata through here rather than touching `this.ast.metadata`
20
+ * directly, so an override reaches all of them uniformly instead of one format at a time.
21
+ *
22
+ * Merged rather than replaced, so overriding one field doesn't blank the rest, and computed
23
+ * fresh rather than cached on the AST: overrides are an output concern, and mutating
24
+ * `ast.metadata` would leak one generation's settings into the next use of the same AST.
25
+ * `custom` merges into `customProperties` so callers see one bucket regardless of origin.
26
+ */
27
+ protected get effectiveMetadata(): OfficeMetadata;
28
+ /**
29
+ * Reports caller-supplied `metadataOverrides.custom` entries that the destination format has
30
+ * no way to represent (EPUB's OPF and RTF's `\info` both have fixed vocabularies).
31
+ *
32
+ * Warning rather than dropping silently: a caller who sets metadata and never sees it in the
33
+ * output otherwise has no way to find out. Only `custom` keys are reported - the named fields
34
+ * map onto something in every format that carries metadata at all.
35
+ */
36
+ protected warnUnrepresentableCustomMetadata(format: string): void;
15
37
  /**
16
38
  * Retrieves the semantic mapping for a node, respecting the includeFormatting flag.
17
39
  * Per design requirements: Style mapping is bypassed if formatting is disabled.
@@ -48,6 +70,17 @@ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat =
48
70
  * Helper to generate a unique ID (slug) from text.
49
71
  */
50
72
  protected slugify(text: string): string;
73
+ private noteFootnoteKeys;
74
+ private usedFootnoteKeys;
75
+ private footnoteKeyCounter;
76
+ /**
77
+ * Assigns a stable, unique reference key to a footnote/endnote node, reused for both
78
+ * its inline reference marker and its collected definition. Source ids aren't reliably
79
+ * unique across a document - DOCX/ODT number footnotes and endnotes in separate
80
+ * sequences, so both can carry noteId "1" - so a source id is only reused when it
81
+ * hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
82
+ */
83
+ protected getFootnoteKey(note: OfficeContentNode): string;
51
84
  /**
52
85
  * Recursively extracts plain text from a node and its children.
53
86
  */
@@ -1,6 +1,7 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.BaseGenerator = void 0;
4
+ const types_js_1 = require("../types.js");
4
5
  const configUtils_js_1 = require("../utils/configUtils.js");
5
6
  const errorUtils_js_1 = require("../utils/errorUtils.js");
6
7
  const styleMapper_js_1 = require("../utils/styleMapper.js");
@@ -21,6 +22,60 @@ class BaseGenerator {
21
22
  this.ast = ast;
22
23
  this.styleMapper = new styleMapper_js_1.StyleMapper(this.config.styleMap, this.config.ignoreDefaultStyleMap);
23
24
  }
25
+ /**
26
+ * The document metadata a generator should write out: `ast.metadata` with
27
+ * `config.metadataOverrides` applied on top, per field.
28
+ *
29
+ * Every generator must read metadata through here rather than touching `this.ast.metadata`
30
+ * directly, so an override reaches all of them uniformly instead of one format at a time.
31
+ *
32
+ * Merged rather than replaced, so overriding one field doesn't blank the rest, and computed
33
+ * fresh rather than cached on the AST: overrides are an output concern, and mutating
34
+ * `ast.metadata` would leak one generation's settings into the next use of the same AST.
35
+ * `custom` merges into `customProperties` so callers see one bucket regardless of origin.
36
+ */
37
+ get effectiveMetadata() {
38
+ const base = this.ast.metadata || {};
39
+ const overrides = this.config.metadataOverrides;
40
+ if (!overrides)
41
+ return base;
42
+ const { custom, language, ...named } = overrides;
43
+ const merged = { ...base };
44
+ // Assign only the fields actually supplied - spreading `named` wholesale would write
45
+ // `undefined` over parsed values for every field the caller left out.
46
+ for (const [key, value] of Object.entries(named)) {
47
+ if (value !== undefined)
48
+ merged[key] = value;
49
+ }
50
+ // `language` has no slot on OfficeMetadata; parsers already surface it through
51
+ // nativeProperties, so an override belongs in the same place rather than widening the
52
+ // parser-side type for a generator concern. Generators reading `nativeProperties.language`
53
+ // then pick it up with no change.
54
+ if (language !== undefined) {
55
+ merged.nativeProperties = { ...(base.nativeProperties || {}), language };
56
+ }
57
+ if (custom && Object.keys(custom).length > 0) {
58
+ merged.customProperties = { ...(base.customProperties || {}), ...custom };
59
+ }
60
+ return merged;
61
+ }
62
+ /**
63
+ * Reports caller-supplied `metadataOverrides.custom` entries that the destination format has
64
+ * no way to represent (EPUB's OPF and RTF's `\info` both have fixed vocabularies).
65
+ *
66
+ * Warning rather than dropping silently: a caller who sets metadata and never sees it in the
67
+ * output otherwise has no way to find out. Only `custom` keys are reported - the named fields
68
+ * map onto something in every format that carries metadata at all.
69
+ */
70
+ warnUnrepresentableCustomMetadata(format) {
71
+ const custom = this.config.metadataOverrides?.custom;
72
+ if (!custom)
73
+ return;
74
+ const keys = Object.keys(custom);
75
+ if (keys.length === 0)
76
+ return;
77
+ this.warn(types_js_1.OfficeWarningType.METADATA_NOT_REPRESENTABLE, { keys, format });
78
+ }
24
79
  /**
25
80
  * Retrieves the semantic mapping for a node, respecting the includeFormatting flag.
26
81
  * Per design requirements: Style mapping is bypassed if formatting is disabled.
@@ -55,6 +110,13 @@ class BaseGenerator {
55
110
  * @returns The generated string for this node and its subtree.
56
111
  */
57
112
  async processNodeRecursive(node, processor) {
113
+ // Every text-based generator funnels its whole traversal through here, so one check
114
+ // makes `abortSignal` effective for all of them. Previously the signal was read once
115
+ // before generation began and never again, which meant it could decline to start work
116
+ // but could not stop work already underway - not much use against a document large
117
+ // enough to be worth aborting. The check is a property read on an optional signal, so
118
+ // the per-node cost is negligible.
119
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
58
120
  const override = await this.handleOnNode(node);
59
121
  if (override === false)
60
122
  return '';
@@ -89,6 +151,42 @@ class BaseGenerator {
89
151
  .replace(/[\s_-]+/g, '-')
90
152
  .replace(/^-+|-+$/g, '');
91
153
  }
154
+ noteFootnoteKeys = new Map();
155
+ usedFootnoteKeys = new Set();
156
+ footnoteKeyCounter = 0;
157
+ /**
158
+ * Assigns a stable, unique reference key to a footnote/endnote node, reused for both
159
+ * its inline reference marker and its collected definition. Source ids aren't reliably
160
+ * unique across a document - DOCX/ODT number footnotes and endnotes in separate
161
+ * sequences, so both can carry noteId "1" - so a source id is only reused when it
162
+ * hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
163
+ */
164
+ getFootnoteKey(note) {
165
+ const cached = this.noteFootnoteKeys.get(note);
166
+ if (cached)
167
+ return cached;
168
+ const preferred = note.metadata?.noteId;
169
+ // The id is document-supplied and lands inside `[^...]` in Markdown, where the parser's
170
+ // own recognizer accepts everything but `]` - so a note id carrying markup round-tripped
171
+ // into the output verbatim. Accept it only when it looks like a label; otherwise fall
172
+ // through to the sequential counter, which is always safe. Real ids are numeric (DOCX),
173
+ // `ftn1`-shaped (ODT) or short slugs (`fn1`), and a Markdown label like `[^my note]`
174
+ // still qualifies, so the fallback fires only for genuinely hostile input.
175
+ const isLabelLike = typeof preferred === 'string' && /^[A-Za-z0-9_.:-][A-Za-z0-9 _.:-]*$/.test(preferred);
176
+ let key;
177
+ if (preferred && isLabelLike && !this.usedFootnoteKeys.has(preferred)) {
178
+ key = preferred;
179
+ }
180
+ else {
181
+ do {
182
+ this.footnoteKeyCounter++;
183
+ key = String(this.footnoteKeyCounter);
184
+ } while (this.usedFootnoteKeys.has(key));
185
+ }
186
+ this.usedFootnoteKeys.add(key);
187
+ this.noteFootnoteKeys.set(note, key);
188
+ return key;
189
+ }
92
190
  /**
93
191
  * Recursively extracts plain text from a node and its children.
94
192
  */
@@ -20,9 +20,17 @@ export declare class CsvGenerator extends BaseGenerator<'csv'> {
20
20
  */
21
21
  private renderNodeToRows;
22
22
  /**
23
- * Escapes a value for CSV formatting.
23
+ * Escapes a value for CSV formatting: RFC 4180 quoting plus a spreadsheet
24
+ * formula-injection guard (see csvSafeCell in ../utils/sanitize.js).
24
25
  */
25
26
  private escapeCsvValue;
27
+ /** Sanitizes a value for a `#` comment line (document metadata or sheet name).
28
+ * Comment lines are free text prefixed with `#`, not RFC-4180 cells, so they
29
+ * can't be quoted; instead we (a) collapse newlines so the value can't break out
30
+ * and inject a new row, and (b) replace the column delimiter with a space so a
31
+ * value like `x,=1+1` can't split into a second cell that a spreadsheet would
32
+ * evaluate as a formula (CSV formula/DDE injection). */
33
+ private sanitizeComment;
26
34
  /**
27
35
  * Renders metadata as comments.
28
36
  */
@@ -4,6 +4,7 @@ exports.CsvGenerator = void 0;
4
4
  const fflate_1 = require("fflate");
5
5
  const types_js_1 = require("../types.js");
6
6
  const sheetUtils_js_1 = require("../utils/sheetUtils.js");
7
+ const sanitize_js_1 = require("../utils/sanitize.js");
7
8
  const BaseGenerator_js_1 = require("./BaseGenerator.js");
8
9
  /**
9
10
  * Generates CSV files from an AST.
@@ -26,7 +27,7 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
26
27
  if (sheetNodes.length === 0) {
27
28
  return { value: '', messages: this.messages };
28
29
  }
29
- const metadataHeader = this.config.renderMetadata ? this.renderMetadata(this.ast) : '';
30
+ const metadataHeader = this.config.renderMetadata ? this.renderMetadata(this.ast, delimiter) : '';
30
31
  // 2. Filter sheets based on range
31
32
  let selectedNodes = sheetNodes;
32
33
  if (csvConfig.sheets) {
@@ -59,7 +60,7 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
59
60
  mergedLines.push(metadataHeader.trim());
60
61
  for (const sheet of sheetData) {
61
62
  if (sheetData.length > 1)
62
- mergedLines.push(`# Sheet: ${sheet.name}`);
63
+ mergedLines.push(`# Sheet: ${this.sanitizeComment(sheet.name, delimiter)}`);
63
64
  for (const row of sheet.rows) {
64
65
  const paddedRow = [...row];
65
66
  // Don't pad comments
@@ -197,34 +198,45 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
197
198
  return rows;
198
199
  }
199
200
  /**
200
- * Escapes a value for CSV formatting.
201
+ * Escapes a value for CSV formatting: RFC 4180 quoting plus a spreadsheet
202
+ * formula-injection guard (see csvSafeCell in ../utils/sanitize.js).
201
203
  */
202
204
  escapeCsvValue(val, delimiter) {
203
- const needsQuotes = val.includes(delimiter) || val.includes('"') || val.includes('\n') || val.includes('\r');
204
- if (!needsQuotes)
205
- return val;
206
- // Double up existing quotes and wrap in quotes
207
- return `"${val.replace(/"/g, '""')}"`;
205
+ return (0, sanitize_js_1.csvSafeCell)(val, delimiter);
206
+ }
207
+ /** Sanitizes a value for a `#` comment line (document metadata or sheet name).
208
+ * Comment lines are free text prefixed with `#`, not RFC-4180 cells, so they
209
+ * can't be quoted; instead we (a) collapse newlines so the value can't break out
210
+ * and inject a new row, and (b) replace the column delimiter with a space so a
211
+ * value like `x,=1+1` can't split into a second cell that a spreadsheet would
212
+ * evaluate as a formula (CSV formula/DDE injection). */
213
+ sanitizeComment(val, delimiter) {
214
+ let s = String(val ?? '').replace(/[\r\n]+/g, ' ');
215
+ if (delimiter)
216
+ s = s.split(delimiter).join(' ');
217
+ return s;
208
218
  }
209
219
  /**
210
220
  * Renders metadata as comments.
211
221
  */
212
- renderMetadata(ast) {
213
- if (!ast.metadata)
222
+ renderMetadata(_ast, delimiter) {
223
+ // Via effectiveMetadata rather than the passed AST so `metadataOverrides` applies here
224
+ // as it does in every other generator.
225
+ const m = this.effectiveMetadata;
226
+ if (!m || Object.keys(m).length === 0)
214
227
  return '';
215
- const m = ast.metadata;
216
228
  let output = '';
217
229
  if (m.title)
218
- output += `# Title: ${m.title}\n`;
230
+ output += `# Title: ${this.sanitizeComment(m.title, delimiter)}\n`;
219
231
  if (m.author)
220
- output += `# Author: ${m.author}\n`;
232
+ output += `# Author: ${this.sanitizeComment(m.author, delimiter)}\n`;
221
233
  if (m.created)
222
- output += `# Created: ${new Date(m.created).toLocaleString()}\n`;
234
+ output += `# Created: ${this.sanitizeComment(new Date(m.created).toLocaleString(), delimiter)}\n`;
223
235
  if (m.modified)
224
- output += `# Modified: ${new Date(m.modified).toLocaleString()}\n`;
236
+ output += `# Modified: ${this.sanitizeComment(new Date(m.modified).toLocaleString(), delimiter)}\n`;
225
237
  if (m.customProperties) {
226
238
  for (const [k, v] of Object.entries(m.customProperties)) {
227
- output += `# ${k}: ${v}\n`;
239
+ output += `# ${this.sanitizeComment(k, delimiter)}: ${this.sanitizeComment(String(v), delimiter)}\n`;
228
240
  }
229
241
  }
230
242
  return output ? output + '\n' : '';
@@ -0,0 +1,43 @@
1
+ import { ConversionResult, GeneratorConfig, OfficeParserAST } from '../types.js';
2
+ import { BaseGenerator } from './BaseGenerator.js';
3
+ /**
4
+ * Generates a minimal, valid EPUB 3 file from an AST.
5
+ *
6
+ * Every AST node is rendered as a single XHTML content document (reusing `HtmlGenerator`
7
+ * for the actual markup, since EPUB content documents are XHTML) and packaged with the
8
+ * required `mimetype`, `META-INF/container.xml`, OPF manifest, and navigation document.
9
+ *
10
+ * `HtmlGenerator` embeds images as base64 `data:` URIs, but EPUB reading systems do not
11
+ * render `data:` URIs - images must be packaged as separate resources referenced by a
12
+ * relative path. So each data-URI image is extracted into `OEBPS/images/`, declared in
13
+ * the manifest, and its `<img src>` rewritten to point at the packaged file.
14
+ */
15
+ export declare class EpubGenerator extends BaseGenerator<'epub'> {
16
+ constructor(ast: OfficeParserAST, config?: GeneratorConfig<'epub'>);
17
+ /**
18
+ * Resolves the modification instant used for both the EPUB 3 `dcterms:modified` property
19
+ * (which the specification requires) and every zip entry's mtime.
20
+ *
21
+ * Takes the value from `effectiveMetadata`, i.e. `metadataOverrides.modified` if the caller
22
+ * set one, otherwise the source document's own `metadata.modified`, and only falls back to
23
+ * the current time when neither exists. That last fallback is the sole non-reproducible
24
+ * option, so it is the last resort rather than the default.
25
+ *
26
+ * **Both outputs matter for reproducibility.** `dcterms:modified` is the visible one, but
27
+ * `zipSync` defaults each entry's mtime to `Date.now()`, so pinning only the OPF still
28
+ * yields archives that differ byte-for-byte on every run. That second source is easy to
29
+ * miss because DOS zip timestamps have two-second granularity - back-to-back generation
30
+ * looks stable and only a gap longer than that reveals it.
31
+ *
32
+ * `iso` is `YYYY-MM-DDThh:mm:ssZ` (UTC, whole seconds) as EPUB requires; `toISOString()`
33
+ * emits milliseconds, so they are stripped.
34
+ */
35
+ private resolveModified;
36
+ /**
37
+ * Zip's DOS timestamp field cannot represent dates outside 1980-2099, and fflate throws
38
+ * rather than clamping. A document legitimately carrying a date outside that window (an
39
+ * unset/epoch-zero mtime is the common case) must not take EPUB generation down with it.
40
+ */
41
+ private clampToZipRange;
42
+ generate(): Promise<ConversionResult<'epub'>>;
43
+ }
@@ -0,0 +1,312 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.EpubGenerator = void 0;
4
+ const fflate_1 = require("fflate");
5
+ const BaseGenerator_js_1 = require("./BaseGenerator.js");
6
+ const HtmlGenerator_js_1 = require("./HtmlGenerator.js");
7
+ const sanitize_js_1 = require("../utils/sanitize.js");
8
+ const VOID_TAGS = ['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr'];
9
+ /** Block-level tags whose HTML5 content model does not permit them inside a <p>. */
10
+ const BLOCK_TAGS_INVALID_IN_P = ['div', 'table', 'ul', 'ol', 'dl', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'blockquote', 'pre', 'section', 'figure'];
11
+ /**
12
+ * HTML5's content model forbids block elements inside <p> (a <p> can only hold "phrasing
13
+ * content"), but browsers silently fix this via the HTML5 parsing algorithm's
14
+ * auto-closing rule: seeing a block start tag implicitly closes the open <p> first.
15
+ * XML parsers have no such rule - they just build the tree exactly as written. Since
16
+ * HtmlGenerator always wraps images in `<div class="image-container">`, a paragraph
17
+ * whose only child is an image becomes `<p><div>...</div></p>`: well-formed XML, but
18
+ * many EPUB rendering engines refuse to lay out a block box found inside a paragraph and
19
+ * simply drop it - silently, with no parse error, which is why the image vanishes.
20
+ *
21
+ * Fixes this by promoting any `<p ...>` that contains a nested block tag to a `<div ...>`
22
+ * instead, matching what a browser's auto-correction effectively produces. Paragraphs
23
+ * don't nest, so the first `</p>` after each `<p>` is always its match.
24
+ */
25
+ const promoteParagraphsWithBlockContent = (html) => {
26
+ const blockTagPattern = new RegExp(`<(?:${BLOCK_TAGS_INVALID_IN_P.join('|')})\\b`, 'i');
27
+ let result = '';
28
+ let cursor = 0;
29
+ const pOpenRegex = /<p(\s[^>]*)?>/gi;
30
+ let match;
31
+ while ((match = pOpenRegex.exec(html)) !== null) {
32
+ if (match.index < cursor)
33
+ continue; // inside content already emitted by a prior promotion
34
+ result += html.slice(cursor, match.index);
35
+ const contentStart = match.index + match[0].length;
36
+ const closeMatch = /<\/p>/i.exec(html.slice(contentStart));
37
+ if (!closeMatch) {
38
+ // No closing tag found (shouldn't happen with well-formed generator output) -
39
+ // leave as-is rather than risk corrupting the rest of the document.
40
+ result += match[0];
41
+ cursor = contentStart;
42
+ pOpenRegex.lastIndex = cursor;
43
+ continue;
44
+ }
45
+ const inner = html.slice(contentStart, contentStart + closeMatch.index);
46
+ const attrs = match[1] || '';
47
+ result += blockTagPattern.test(inner) ? `<div${attrs}>${inner}</div>` : `<p${attrs}>${inner}</p>`;
48
+ cursor = contentStart + closeMatch.index + closeMatch[0].length;
49
+ pOpenRegex.lastIndex = cursor;
50
+ }
51
+ result += html.slice(cursor);
52
+ return result;
53
+ };
54
+ /**
55
+ * Converts HtmlGenerator's HTML output into well-formed XHTML, which EPUB reading
56
+ * systems parse as strict XML (unlike browsers, which tolerate HTML's looseness).
57
+ *
58
+ * This is more than cosmetic: a single raw `&` or unclosed tag makes the whole content
59
+ * document fail to open. The conversion:
60
+ * - strips `<script>` blocks — EpubGenerator renders through HtmlGenerator with
61
+ * `standalone: false`, which already omits the envelope-level stylesheet and Chart.js/
62
+ * spreadsheet scripts entirely, but a chart *node* still emits its own inline
63
+ * `<script>` (chart-init JS) regardless of that flag, since it's content, not envelope.
64
+ * A reading system can't execute it anyway, and its JS operators can contain raw `&`/`<`
65
+ * that are illegal as XML character data, so it's stripped here;
66
+ * - promotes `<p>` tags that contain nested block content (see
67
+ * promoteParagraphsWithBlockContent above) to `<div>`, since XML readers don't apply
68
+ * HTML5's auto-closing correction that hides this in a browser;
69
+ * - normalises HTML named entities (`&nbsp;`) to numeric references, since XML predefines
70
+ * only `&amp;`/`&lt;`/`&gt;`/`&quot;`/`&apos;`;
71
+ * - escapes stray ampersands (e.g. in `href` query strings) not already part of a valid
72
+ * reference;
73
+ * - gives HTML boolean attributes an explicit value (`checked` -> `checked="checked"`);
74
+ * - self-closes void elements (`<br>` -> `<br/>`).
75
+ */
76
+ const toXhtml = (html) => {
77
+ let out = html.replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, '');
78
+ out = promoteParagraphsWithBlockContent(out);
79
+ // Named -> numeric entities (nbsp is the only named entity HtmlGenerator emits).
80
+ out = out.replace(/&nbsp;/g, '&#160;');
81
+ // Escape ampersands that don't already open a valid XML entity reference.
82
+ out = out.replace(/&(?!(?:amp|lt|gt|quot|apos|#\d+|#x[0-9a-fA-F]+);)/g, '&amp;');
83
+ // Give bare boolean attributes an explicit value. Scoped to the specific tags that
84
+ // emit them (checkbox task-list items, media iframes) so body text like "the selected
85
+ // option" is never rewritten.
86
+ out = out.replace(/<input\b([^>]*?)\schecked(\s*\/?>)/gi, '<input$1 checked="checked"$2');
87
+ out = out.replace(/<(iframe|video|audio)\b([^>]*?)\s(allowfullscreen|autoplay|controls|loop|muted)(\s*\/?>|\s)/gi, '<$1$2 $3="$3"$4');
88
+ // Self-close void elements (non-greedy attr capture so an already-present trailing `/`
89
+ // isn't duplicated, e.g. `<meta .../>` must not become `<meta ...//>`).
90
+ const voidTagPattern = new RegExp(`<(${VOID_TAGS.join('|')})((?:\\s[^>]*?)?)\\s*/?>`, 'gi');
91
+ out = out.replace(voidTagPattern, (_m, tag, attrs) => `<${tag}${attrs}/>`);
92
+ return out;
93
+ };
94
+ /**
95
+ * Minimal, book-friendly CSS injected into every EPUB content document. Kept static and
96
+ * free of `&`/`<` so it is XML-safe inline; the reading system supplies typography, so
97
+ * this only covers structural essentials the stripped page-chrome would otherwise lose.
98
+ */
99
+ const EPUB_STYLESHEET = `img { max-width: 100%; height: auto; }
100
+ table { border-collapse: collapse; margin: 1em 0; }
101
+ td, th { border: 1px solid #ccc; padding: 4px 8px; }`;
102
+ /** Maps an image MIME type to a file extension for the packaged resource. */
103
+ const MIME_EXT = {
104
+ 'image/jpeg': 'jpg', 'image/jpg': 'jpg', 'image/png': 'png', 'image/gif': 'gif',
105
+ 'image/svg+xml': 'svg', 'image/webp': 'webp', 'image/bmp': 'bmp', 'image/tiff': 'tiff'
106
+ };
107
+ /** Decodes a base64 string to raw bytes, cross-env (atob exists in Node 16+ and browsers). */
108
+ const decodeBase64 = (b64) => {
109
+ const bin = atob(b64);
110
+ const bytes = new Uint8Array(bin.length);
111
+ for (let i = 0; i < bin.length; i++)
112
+ bytes[i] = bin.charCodeAt(i);
113
+ return bytes;
114
+ };
115
+ /**
116
+ * Generates a minimal, valid EPUB 3 file from an AST.
117
+ *
118
+ * Every AST node is rendered as a single XHTML content document (reusing `HtmlGenerator`
119
+ * for the actual markup, since EPUB content documents are XHTML) and packaged with the
120
+ * required `mimetype`, `META-INF/container.xml`, OPF manifest, and navigation document.
121
+ *
122
+ * `HtmlGenerator` embeds images as base64 `data:` URIs, but EPUB reading systems do not
123
+ * render `data:` URIs - images must be packaged as separate resources referenced by a
124
+ * relative path. So each data-URI image is extracted into `OEBPS/images/`, declared in
125
+ * the manifest, and its `<img src>` rewritten to point at the packaged file.
126
+ */
127
+ class EpubGenerator extends BaseGenerator_js_1.BaseGenerator {
128
+ constructor(ast, config) {
129
+ super('epub', ast, config);
130
+ }
131
+ /**
132
+ * Resolves the modification instant used for both the EPUB 3 `dcterms:modified` property
133
+ * (which the specification requires) and every zip entry's mtime.
134
+ *
135
+ * Takes the value from `effectiveMetadata`, i.e. `metadataOverrides.modified` if the caller
136
+ * set one, otherwise the source document's own `metadata.modified`, and only falls back to
137
+ * the current time when neither exists. That last fallback is the sole non-reproducible
138
+ * option, so it is the last resort rather than the default.
139
+ *
140
+ * **Both outputs matter for reproducibility.** `dcterms:modified` is the visible one, but
141
+ * `zipSync` defaults each entry's mtime to `Date.now()`, so pinning only the OPF still
142
+ * yields archives that differ byte-for-byte on every run. That second source is easy to
143
+ * miss because DOS zip timestamps have two-second granularity - back-to-back generation
144
+ * looks stable and only a gap longer than that reveals it.
145
+ *
146
+ * `iso` is `YYYY-MM-DDThh:mm:ssZ` (UTC, whole seconds) as EPUB requires; `toISOString()`
147
+ * emits milliseconds, so they are stripped.
148
+ */
149
+ resolveModified() {
150
+ const raw = this.effectiveMetadata.modified;
151
+ let resolved = null;
152
+ if (raw instanceof Date && !isNaN(raw.getTime())) {
153
+ resolved = raw;
154
+ }
155
+ else if (typeof raw === 'string' && raw !== '') {
156
+ // A parser may hand back a date-like string rather than a Date.
157
+ const parsed = new Date(raw);
158
+ if (!isNaN(parsed.getTime()))
159
+ resolved = parsed;
160
+ }
161
+ resolved ??= new Date();
162
+ return {
163
+ iso: resolved.toISOString().replace(/\.\d+Z$/, 'Z'),
164
+ mtime: this.clampToZipRange(resolved),
165
+ };
166
+ }
167
+ /**
168
+ * Zip's DOS timestamp field cannot represent dates outside 1980-2099, and fflate throws
169
+ * rather than clamping. A document legitimately carrying a date outside that window (an
170
+ * unset/epoch-zero mtime is the common case) must not take EPUB generation down with it.
171
+ */
172
+ clampToZipRange(date) {
173
+ const MIN = Date.UTC(1980, 0, 1);
174
+ const MAX = Date.UTC(2099, 11, 31, 23, 59, 59);
175
+ const t = date.getTime();
176
+ if (t < MIN)
177
+ return new Date(MIN);
178
+ if (t > MAX)
179
+ return new Date(MAX);
180
+ return date;
181
+ }
182
+ async generate() {
183
+ const htmlGenerator = new HtmlGenerator_js_1.HtmlGenerator(this.ast, {
184
+ ...this.config,
185
+ htmlConfig: { ...this.config.htmlConfig, standalone: false },
186
+ });
187
+ const htmlResult = await htmlGenerator.generate();
188
+ let bodyHtml = typeof htmlResult.value === 'string' ? htmlResult.value : '';
189
+ // Extract base64 data-URI images into packaged files (EPUB readers don't render
190
+ // `data:` URIs). Each distinct image becomes one OEBPS/images/imageN.ext resource,
191
+ // a manifest <item>, and a rewritten relative `src`. Deduped so a repeated image
192
+ // is packaged once.
193
+ const imageResources = {};
194
+ const imageManifestItems = [];
195
+ const dataUriToHref = new Map();
196
+ let imageCounter = 0;
197
+ bodyHtml = bodyHtml.replace(/(<img\b[^>]*\bsrc=")(data:(image\/[a-zA-Z0-9.+-]+);base64,([^"]+))(")/gi, (_full, pre, dataUri, mime, b64, post) => {
198
+ let href = dataUriToHref.get(dataUri);
199
+ if (!href) {
200
+ imageCounter++;
201
+ const ext = MIME_EXT[mime.toLowerCase()] || 'img';
202
+ href = `images/image${imageCounter}.${ext}`;
203
+ dataUriToHref.set(dataUri, href);
204
+ try {
205
+ imageResources[`OEBPS/${href}`] = decodeBase64(b64);
206
+ imageManifestItems.push(`<item id="img${imageCounter}" href="${href}" media-type="${mime}"/>`);
207
+ }
208
+ catch {
209
+ // Undecodable data - leave the original src untouched rather than
210
+ // emit a manifest entry for a resource we couldn't write.
211
+ dataUriToHref.delete(dataUri);
212
+ return `${pre}${dataUri}${post}`;
213
+ }
214
+ }
215
+ return `${pre}${href}${post}`;
216
+ });
217
+ const xhtmlBody = toXhtml(bodyHtml);
218
+ // Via effectiveMetadata so `metadataOverrides` applies here as it does everywhere else.
219
+ const meta = this.effectiveMetadata;
220
+ const title = meta.title || 'Untitled';
221
+ const author = meta.author;
222
+ const description = meta.description;
223
+ // dc:subject is the OPF's only vocabulary slot for either of these; both are repeatable.
224
+ const subject = meta.subject;
225
+ const keywords = meta.keywords;
226
+ const nativeProps = (meta.nativeProperties || {});
227
+ const language = nativeProps.language || 'en';
228
+ // OPF metadata is a closed Dublin Core vocabulary: a caller-defined key has no valid
229
+ // element to live in, and inventing one risks failing EPUB validation outright.
230
+ this.warnUnrepresentableCustomMetadata('EPUB');
231
+ const identifier = nativeProps.identifier || `urn:x-officeparser:${this.slugify(title)}-${xhtmlBody.length}`;
232
+ const { iso: modified, mtime } = this.resolveModified();
233
+ const chapterXhtml = `<?xml version="1.0" encoding="UTF-8"?>
234
+ <!DOCTYPE html>
235
+ <html xmlns="http://www.w3.org/1999/xhtml" xml:lang="${(0, sanitize_js_1.escapeXml)(language)}">
236
+ <head>
237
+ <meta charset="utf-8"/>
238
+ <title>${(0, sanitize_js_1.escapeXml)(title)}</title>
239
+ <style type="text/css">
240
+ ${EPUB_STYLESHEET}
241
+ </style>
242
+ </head>
243
+ <body>
244
+ ${xhtmlBody}
245
+ </body>
246
+ </html>`;
247
+ // Built as a list rather than inline ternaries so an absent field contributes nothing at
248
+ // all. Inline `${x ? ... : ''}` leaves the surrounding indentation and newline behind, so
249
+ // every optional field the document lacks used to emit a stray blank line into the OPF.
250
+ const optionalDcElements = [
251
+ author ? `<dc:creator>${(0, sanitize_js_1.escapeXml)(author)}</dc:creator>` : '',
252
+ description ? `<dc:description>${(0, sanitize_js_1.escapeXml)(description)}</dc:description>` : '',
253
+ // dc:subject is repeatable and is the OPF's only slot for either of these.
254
+ subject ? `<dc:subject>${(0, sanitize_js_1.escapeXml)(subject)}</dc:subject>` : '',
255
+ keywords ? `<dc:subject>${(0, sanitize_js_1.escapeXml)(keywords)}</dc:subject>` : '',
256
+ ].filter(Boolean).map(el => ` ${el}`).join('\n');
257
+ const opf = `<?xml version="1.0" encoding="UTF-8"?>
258
+ <package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="pub-id">
259
+ <metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
260
+ <dc:identifier id="pub-id">${(0, sanitize_js_1.escapeXml)(identifier)}</dc:identifier>
261
+ <dc:title>${(0, sanitize_js_1.escapeXml)(title)}</dc:title>
262
+ ${optionalDcElements}
263
+ <dc:language>${(0, sanitize_js_1.escapeXml)(language)}</dc:language>
264
+ <meta property="dcterms:modified">${modified}</meta>
265
+ </metadata>
266
+ <manifest>
267
+ <item id="chapter1" href="chapter1.xhtml" media-type="application/xhtml+xml"/>
268
+ <item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" properties="nav"/>${imageManifestItems.length ? '\n ' + imageManifestItems.join('\n ') : ''}
269
+ </manifest>
270
+ <spine>
271
+ <itemref idref="chapter1"/>
272
+ </spine>
273
+ </package>`;
274
+ const navXhtml = `<?xml version="1.0" encoding="UTF-8"?>
275
+ <!DOCTYPE html>
276
+ <html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops">
277
+ <head><meta charset="utf-8"/><title>Navigation</title></head>
278
+ <body>
279
+ <nav epub:type="toc" id="toc">
280
+ <h1>${(0, sanitize_js_1.escapeXml)(title)}</h1>
281
+ <ol>
282
+ <li><a href="chapter1.xhtml">${(0, sanitize_js_1.escapeXml)(title)}</a></li>
283
+ </ol>
284
+ </nav>
285
+ </body>
286
+ </html>`;
287
+ const containerXml = `<?xml version="1.0" encoding="UTF-8"?>
288
+ <container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
289
+ <rootfiles>
290
+ <rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/>
291
+ </rootfiles>
292
+ </container>`;
293
+ const encoder = new TextEncoder();
294
+ // EPUB requires the mimetype entry to be the first file in the archive, stored
295
+ // uncompressed (level 0) - readers use it to sniff the format before parsing any XML.
296
+ const zipFiles = {
297
+ mimetype: [encoder.encode('application/epub+zip'), { level: 0 }],
298
+ 'META-INF/container.xml': encoder.encode(containerXml),
299
+ 'OEBPS/content.opf': encoder.encode(opf),
300
+ 'OEBPS/nav.xhtml': encoder.encode(navXhtml),
301
+ 'OEBPS/chapter1.xhtml': encoder.encode(chapterXhtml),
302
+ ...imageResources,
303
+ };
304
+ return {
305
+ // An explicit mtime is what makes the archive reproducible; fflate would otherwise
306
+ // stamp every entry with Date.now(). See resolveModified().
307
+ value: (0, fflate_1.zipSync)(zipFiles, { mtime }),
308
+ messages: this.messages,
309
+ };
310
+ }
311
+ }
312
+ exports.EpubGenerator = EpubGenerator;
@@ -32,7 +32,19 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
32
32
  private formatText;
33
33
  private getInlineStyles;
34
34
  private getPremiumStyles;
35
+ /**
36
+ * Same as `getPremiumStyles()`, but wrapped in a CSS `@scope` block anchored to the
37
+ * `.op-html-scope` wrapper so the rules only apply within the generated fragment - they
38
+ * cannot leak onto a host page's own elements. `:root` and `body` selectors specifically
39
+ * target the real page root/body, so they're remapped to `:scope` (the scope root, i.e. the
40
+ * `.op-html-scope` wrapper) first; every other selector is naturally confined by `@scope`
41
+ * without needing per-selector rewriting. `customCss` is included in this scoping too.
42
+ */
43
+ private getScopedPremiumStyles;
35
44
  protected slugify(text: string): string;
36
45
  private getColumnLetter;
37
46
  private escape;
47
+ /** Converts a document-supplied date to an ISO string, or '' if it is invalid
48
+ * (a malformed date would otherwise throw a RangeError and abort generation). */
49
+ private toIsoDate;
38
50
  }