officeparser 7.4.0 → 7.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +85 -4
- package/dist/OfficeParser.js +54 -6
- package/dist/generators/BaseGenerator.d.ts +15 -0
- package/dist/generators/BaseGenerator.js +31 -0
- package/dist/generators/HtmlGenerator.d.ts +9 -0
- package/dist/generators/HtmlGenerator.js +34 -4
- package/dist/generators/MarkdownGenerator.d.ts +13 -0
- package/dist/generators/MarkdownGenerator.js +115 -41
- package/dist/generators/RtfGenerator.d.ts +13 -0
- package/dist/generators/RtfGenerator.js +23 -2
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +34 -1
- package/dist/officeparser.browser.iife.js +160 -160
- package/dist/officeparser.browser.mjs +198 -198
- package/dist/officeparser.browser.slim.d.ts +34 -1
- package/dist/officeparser.browser.slim.iife.js +186 -186
- package/dist/officeparser.browser.slim.mjs +186 -186
- package/dist/parsers/EpubParser.js +2 -2
- package/dist/parsers/ExcelParser.js +11 -7
- package/dist/parsers/HtmlParser.js +39 -0
- package/dist/parsers/OpenOfficeParser.js +139 -167
- package/dist/parsers/PowerPointParser.js +48 -11
- package/dist/parsers/WordParser.js +33 -7
- package/dist/sbom.cdx.json +92 -92
- package/dist/types.d.ts +34 -1
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.d.ts +15 -2
- package/dist/utils/configUtils.js +58 -13
- package/dist/utils/errorUtils.d.ts +8 -2
- package/dist/utils/errorUtils.js +23 -1
- package/dist/utils/mathUtils.d.ts +42 -0
- package/dist/utils/mathUtils.js +385 -0
- package/dist/utils/zipUtils.d.ts +64 -4
- package/dist/utils/zipUtils.js +188 -4
- package/package.json +9 -5
|
@@ -83,19 +83,33 @@ function isFullParserConfig(config) {
|
|
|
83
83
|
/**
|
|
84
84
|
* Resolves a full parser configuration by merging defaults and user-provided overrides.
|
|
85
85
|
*
|
|
86
|
+
* The returned object always belongs solely to the caller of this function. That matters
|
|
87
|
+
* because a parse installs per-call state on the config it is handed, such as the collector
|
|
88
|
+
* that gathers warnings for one document's `ast.warnings`. Returning the caller's own object
|
|
89
|
+
* would attach that state to an object they may reuse, so a second parse would append its
|
|
90
|
+
* warnings to the first document's already-returned AST, and each parse would retain the
|
|
91
|
+
* previous one's state for as long as the config lived.
|
|
92
|
+
*
|
|
93
|
+
* Only the configuration containers are copied. Callbacks and `abortSignal` keep their
|
|
94
|
+
* identity, since a copy of an `AbortSignal` would no longer be tied to its controller.
|
|
95
|
+
*
|
|
86
96
|
* @param userConfig - Optional configuration provided by the user
|
|
87
|
-
* @returns A fully populated configuration object
|
|
97
|
+
* @returns A fully populated configuration object, owned by the caller
|
|
88
98
|
*/
|
|
89
99
|
function resolveParserConfig(userConfig) {
|
|
90
100
|
if (isFullParserConfig(userConfig)) {
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
101
|
+
const resolved = { ...userConfig };
|
|
102
|
+
resolved.ocrConfig = { ...userConfig.ocrConfig };
|
|
103
|
+
if (userConfig.ocrConfig?.timeout) {
|
|
104
|
+
resolved.ocrConfig.timeout = { ...userConfig.ocrConfig.timeout };
|
|
105
|
+
}
|
|
106
|
+
resolved.decompressionLimits = {
|
|
107
|
+
...(userConfig.decompressionLimits ?? defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.decompressionLimits)
|
|
108
|
+
};
|
|
109
|
+
if (userConfig.htmlParserConfig) {
|
|
110
|
+
resolved.htmlParserConfig = { ...userConfig.htmlParserConfig };
|
|
97
111
|
}
|
|
98
|
-
return
|
|
112
|
+
return resolved;
|
|
99
113
|
}
|
|
100
114
|
// 1. Start with full defaults (deep cloned)
|
|
101
115
|
const config = deepClone(defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG);
|
|
@@ -141,21 +155,52 @@ function resolveParserConfig(userConfig) {
|
|
|
141
155
|
}
|
|
142
156
|
return config;
|
|
143
157
|
}
|
|
158
|
+
/** The per-destination and metadata sub-objects a generator config groups its settings into. */
|
|
159
|
+
const GENERATOR_CONFIG_CONTAINERS = [
|
|
160
|
+
'metadataOverrides', 'htmlConfig', 'mdConfig', 'pdfConfig',
|
|
161
|
+
'csvConfig', 'textConfig', 'rtfConfig', 'chunksConfig',
|
|
162
|
+
];
|
|
163
|
+
/**
|
|
164
|
+
* Copies a generator config's containers so writes during generation cannot reach the caller.
|
|
165
|
+
*
|
|
166
|
+
* One level is enough: the containers are what generation writes to. Everything else is copied
|
|
167
|
+
* by reference on purpose, since callbacks, `styleMap` and `abortSignal` are values whose
|
|
168
|
+
* identity matters, and a duplicated `AbortSignal` would no longer be tied to its controller.
|
|
169
|
+
*
|
|
170
|
+
* @param source - The caller's configuration
|
|
171
|
+
* @returns An equivalent configuration owned by us
|
|
172
|
+
*/
|
|
173
|
+
function copyGeneratorConfigContainers(source) {
|
|
174
|
+
const copy = { ...source };
|
|
175
|
+
for (const key of GENERATOR_CONFIG_CONTAINERS) {
|
|
176
|
+
const container = source[key];
|
|
177
|
+
if (container && typeof container === 'object')
|
|
178
|
+
copy[key] = { ...container };
|
|
179
|
+
}
|
|
180
|
+
return copy;
|
|
181
|
+
}
|
|
144
182
|
/**
|
|
145
183
|
* Resolves a full, destination-specific configuration by merging defaults,
|
|
146
184
|
* AST-level settings, and user-provided overrides.
|
|
147
185
|
*
|
|
186
|
+
* As with {@link resolveParserConfig}, the returned object belongs solely to the caller of this
|
|
187
|
+
* function, so that per-run normalization cannot edit a config the caller still holds.
|
|
188
|
+
*
|
|
148
189
|
* @param destination - The target format
|
|
149
190
|
* @param userConfig - Optional configuration provided by the user
|
|
150
191
|
* @param astConfig - Optional configuration from the source AST (for inheritance)
|
|
151
|
-
* @returns A fully populated configuration object
|
|
192
|
+
* @returns A fully populated configuration object, owned by the caller
|
|
152
193
|
*/
|
|
153
194
|
function resolveGeneratorConfig(destination, astConfig, userConfig) {
|
|
154
|
-
//
|
|
155
|
-
//
|
|
195
|
+
// Already complete, so nothing to merge. Still copied rather than handed straight back, for
|
|
196
|
+
// the same reason as resolveParserConfig: generation writes to the config it is given. The
|
|
197
|
+
// width check below normalizes an invalid `containerWidth` to 'auto', and doing that to the
|
|
198
|
+
// caller's own object both edits a value they still hold and silences the warning on every
|
|
199
|
+
// later run, so the same config would report a problem once and then appear clean.
|
|
156
200
|
if (isFullGeneratorConfig(userConfig) && !astConfig) {
|
|
157
|
-
|
|
158
|
-
|
|
201
|
+
const resolved = copyGeneratorConfigContainers(userConfig);
|
|
202
|
+
validateHtmlConfigWidth(resolved.htmlConfig, resolved);
|
|
203
|
+
return resolved;
|
|
159
204
|
}
|
|
160
205
|
// 1. Start with full defaults (deep cloned to avoid reference sharing)
|
|
161
206
|
const config = deepClone(defaults_js_1.DEFAULT_GENERATOR_CONFIG);
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* It defines standard error types, messages, and handling logic to ensure
|
|
6
6
|
* consistent error reporting across all parsers and the main entry point.
|
|
7
7
|
*/
|
|
8
|
-
import { OfficeErrorType, OfficeParserConfig, OfficeWarningType } from '../types.js';
|
|
8
|
+
import { OfficeError, OfficeErrorType, OfficeParserConfig, OfficeWarningType } from '../types.js';
|
|
9
9
|
/**
|
|
10
10
|
* Creates a formatted warning message for a specific warning type.
|
|
11
11
|
*
|
|
@@ -22,11 +22,17 @@ export declare const getWarningMessage: (type: OfficeWarningType, info?: any) =>
|
|
|
22
22
|
* @param info - Optional additional information
|
|
23
23
|
* @returns The Error object to be thrown
|
|
24
24
|
*/
|
|
25
|
-
export declare const getOfficeError: (type: OfficeErrorType, config?: OfficeParserConfig, info?: any) =>
|
|
25
|
+
export declare const getOfficeError: (type: OfficeErrorType, config?: OfficeParserConfig, info?: any) => OfficeError;
|
|
26
26
|
/**
|
|
27
27
|
* Wraps an existing error with OfficeParser context and performs corruption detection.
|
|
28
28
|
* Optionally logs the error to console.
|
|
29
29
|
*
|
|
30
|
+
* An error already built by {@link getOfficeError} is returned untouched: it carries an
|
|
31
|
+
* `officeIssue`, meaning it has been reported once and already bears the `[OfficeParser]: `
|
|
32
|
+
* header. Re-wrapping it would report the same issue a second time, prepend a second header,
|
|
33
|
+
* and flatten its specific error code to `FILE_CORRUPTED`. This is a marker check on the error
|
|
34
|
+
* object rather than a test against its message text, so it stays independent of wording.
|
|
35
|
+
*
|
|
30
36
|
* **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
|
|
31
37
|
* represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
|
|
32
38
|
* callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
|
package/dist/utils/errorUtils.js
CHANGED
|
@@ -11,6 +11,11 @@ exports.checkAbortSignal = exports.getAbortError = exports.logWarning = exports.
|
|
|
11
11
|
const types_js_1 = require("../types.js");
|
|
12
12
|
/** Error header prefix for all error messages */
|
|
13
13
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
14
|
+
// `OfficeError` (the public shape callers catch) lives in types.ts alongside `OfficeIssue`.
|
|
15
|
+
// Every error built by getOfficeError is branded with its issue, which serves two purposes:
|
|
16
|
+
// consumers branch on `err.officeIssue.code` instead of matching message text, and
|
|
17
|
+
// getWrappedError recognizes an error it has already reported and prefixed, so it neither
|
|
18
|
+
// reports it twice nor prepends a second header.
|
|
14
19
|
/**
|
|
15
20
|
* Lookup table for error messages.
|
|
16
21
|
* Some entries are functions that take parameters to build dynamic messages.
|
|
@@ -34,6 +39,9 @@ const ERROR_MESSAGES = {
|
|
|
34
39
|
[types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
|
|
35
40
|
[types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
|
|
36
41
|
[types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
|
|
42
|
+
[types_js_1.OfficeErrorType.ZIP_NO_ENTRIES_FOUND]: `No readable entries found in ZIP data. The input is corrupt, truncated, or not a ZIP archive: every ZIP-based document format requires at least one entry.`,
|
|
43
|
+
[types_js_1.OfficeErrorType.ZIP_TRUNCATED]: `Malformed ZIP data: no End of Central Directory record was found at the end of the input. Either the file was cut off during download or transfer, or extra data follows the archive; in both cases the entries recovered from it cannot be trusted to be the whole document.`,
|
|
44
|
+
[types_js_1.OfficeErrorType.REQUIRED_PART_MISSING]: (info) => `Your ${info.fileType} file is a readable ZIP archive but is missing its required '${info.part}' part, so it cannot be a valid ${info.fileType} document. The file is corrupt, incomplete, or mislabeled. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce the error.`,
|
|
37
45
|
[types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED]: `Document nesting depth exceeded the safe limit (possible denial-of-service input)`,
|
|
38
46
|
[types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
|
|
39
47
|
};
|
|
@@ -60,6 +68,8 @@ const WARNING_MESSAGES = {
|
|
|
60
68
|
[types_js_1.OfficeWarningType.TABLE_CELL_LIMIT_EXCEEDED]: (limit) => `Table cell limit (${limit}) reached while expanding repeated ODF cells/rows; the remaining cells were not materialized. A few hundred bytes of XML can request an unbounded number of cells via table:number-columns-repeated / table:number-rows-repeated, so this is capped. Raise decompressionLimits.maxTableCells if your documents legitimately exceed it.`,
|
|
61
69
|
[types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH]: (val) => `Invalid HTML containerWidth: ${JSON.stringify(val)}. Falling back to "auto". Width must be a positive number, a valid CSS length string (e.g., "900px", "100%", "50vw"), or "auto".`,
|
|
62
70
|
[types_js_1.OfficeWarningType.METADATA_NOT_REPRESENTABLE]: (info) => `Custom metadata ${info.keys.map(k => `'${k}'`).join(', ')} could not be written to ${info.format} output: the format has a fixed metadata vocabulary with no place for caller-defined keys. The named metadata fields (title, author, etc.) were still applied.`,
|
|
71
|
+
[types_js_1.OfficeWarningType.NO_WORKSHEETS_FOUND]: `Workbook contains no worksheet parts (xl/worksheets/). If the workbook holds only chartsheets this is expected and there is simply no cell text to extract; otherwise the file may be incomplete.`,
|
|
72
|
+
[types_js_1.OfficeWarningType.NO_SLIDES_FOUND]: `Presentation contains no slides (ppt/slides/). A legitimately empty presentation produces this too, but if you expected content the file may be incomplete.`,
|
|
63
73
|
[types_js_1.OfficeWarningType.INVALID_STYLE_MAP_TAG]: (tag) => `styleMap output.tag ${JSON.stringify(tag)} is not an allowed element name and was ignored; the node's default tag was used instead. A tag name is written into both the opening and closing tag, so only a known-safe set of block, heading and inline elements is accepted.`
|
|
64
74
|
};
|
|
65
75
|
/**
|
|
@@ -122,13 +132,23 @@ const getOfficeError = (type, config, info) => {
|
|
|
122
132
|
details: info
|
|
123
133
|
};
|
|
124
134
|
reportIssue(issue, config);
|
|
125
|
-
|
|
135
|
+
const error = new Error(ERRORHEADER + message);
|
|
136
|
+
// Brand the error with the issue that produced it so getWrappedError can tell an
|
|
137
|
+
// already-reported, already-prefixed OfficeParser error from a raw third-party one.
|
|
138
|
+
error.officeIssue = issue;
|
|
139
|
+
return error;
|
|
126
140
|
};
|
|
127
141
|
exports.getOfficeError = getOfficeError;
|
|
128
142
|
/**
|
|
129
143
|
* Wraps an existing error with OfficeParser context and performs corruption detection.
|
|
130
144
|
* Optionally logs the error to console.
|
|
131
145
|
*
|
|
146
|
+
* An error already built by {@link getOfficeError} is returned untouched: it carries an
|
|
147
|
+
* `officeIssue`, meaning it has been reported once and already bears the `[OfficeParser]: `
|
|
148
|
+
* header. Re-wrapping it would report the same issue a second time, prepend a second header,
|
|
149
|
+
* and flatten its specific error code to `FILE_CORRUPTED`. This is a marker check on the error
|
|
150
|
+
* object rather than a test against its message text, so it stays independent of wording.
|
|
151
|
+
*
|
|
132
152
|
* **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
|
|
133
153
|
* represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
|
|
134
154
|
* callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
|
|
@@ -140,6 +160,8 @@ exports.getOfficeError = getOfficeError;
|
|
|
140
160
|
* @returns The wrapped Error object to be thrown
|
|
141
161
|
*/
|
|
142
162
|
const getWrappedError = (error, config, filePath) => {
|
|
163
|
+
if (error?.officeIssue)
|
|
164
|
+
return error;
|
|
143
165
|
let message = error.message || error;
|
|
144
166
|
let code = types_js_1.OfficeErrorType.FILE_CORRUPTED; // Default for wrapped errors
|
|
145
167
|
// Detect file corruption from common library error messages
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Converts an OMML subtree (`<m:oMath>` or any node within one) to LaTeX.
|
|
3
|
+
*
|
|
4
|
+
* Covers the constructs that actually appear in office documents: fractions, sub/superscripts,
|
|
5
|
+
* delimiters, radicals, n-ary operators, functions, accents, bars, boxes and matrices. Anything
|
|
6
|
+
* unrecognized falls through to concatenating its children, which is the old behaviour and the
|
|
7
|
+
* right degradation for a construct that carries no grouping of its own.
|
|
8
|
+
*/
|
|
9
|
+
export declare const ommlToLatex: (node: Node, depth?: number) => string;
|
|
10
|
+
/**
|
|
11
|
+
* The subset of a parsed element every MathML source here can present.
|
|
12
|
+
*
|
|
13
|
+
* MathML reaches this module from two different tree types - the XML DOM (ODF embedded objects)
|
|
14
|
+
* and `HtmlParser`'s own lightweight node (HTML, and EPUB via its spine items) - so the converter
|
|
15
|
+
* is written against this shape and each caller adapts into it. That keeps one implementation of
|
|
16
|
+
* the conversion rather than one per tree type, which is how HTML came to have no MathML support
|
|
17
|
+
* at all while ODF did.
|
|
18
|
+
*/
|
|
19
|
+
export interface MathNode {
|
|
20
|
+
/** Tag name, namespace prefix included or not; `undefined` marks a text node. */
|
|
21
|
+
tagName?: string;
|
|
22
|
+
attributes?: Record<string, string>;
|
|
23
|
+
/** Literal text, for text nodes and for leaf tokens. */
|
|
24
|
+
text?: string;
|
|
25
|
+
children: MathNode[];
|
|
26
|
+
}
|
|
27
|
+
/**
|
|
28
|
+
* Converts a MathML subtree to LaTeX.
|
|
29
|
+
*
|
|
30
|
+
* When the document carries a TeX annotation (`<annotation encoding="application/x-tex">`), that
|
|
31
|
+
* is the author's own source and is used verbatim in preference to anything reconstructed here.
|
|
32
|
+
* ODF's `<annotation encoding="StarMath 5.0">` is deliberately not used - StarMath is not LaTeX,
|
|
33
|
+
* and emitting it would put a second notation back into the output this module exists to unify.
|
|
34
|
+
*/
|
|
35
|
+
export declare const mathmlTreeToLatex: (node: MathNode, depth?: number) => string;
|
|
36
|
+
/** Converts a MathML subtree held in an XML DOM (ODF embedded objects) to LaTeX. */
|
|
37
|
+
export declare const mathmlToLatex: (node: Node, depth?: number) => string;
|
|
38
|
+
/**
|
|
39
|
+
* True when a converted equation carries nothing worth emitting, so callers can drop the node
|
|
40
|
+
* instead of pushing an empty `$$` into the output.
|
|
41
|
+
*/
|
|
42
|
+
export declare const isEmptyMath: (latex: string) => boolean;
|
|
@@ -0,0 +1,385 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.isEmptyMath = exports.mathmlToLatex = exports.mathmlTreeToLatex = exports.ommlToLatex = void 0;
|
|
4
|
+
const xmlUtils_js_1 = require("./xmlUtils.js");
|
|
5
|
+
/**
|
|
6
|
+
* Equation markup normalization.
|
|
7
|
+
*
|
|
8
|
+
* Office documents ship equations in exactly two markups: OOXML's OMML (`<m:oMath>`, used by
|
|
9
|
+
* DOCX and PPTX) and MathML (`<math>`, used by ODF embedded objects, HTML, and EPUB3). Neither
|
|
10
|
+
* is plain text, and neither survives the generic "recurse into unknown elements and concatenate
|
|
11
|
+
* their text" fallback every parser ends with: `<m:num>1</m:num><m:den>2</m:den>` collapses to
|
|
12
|
+
* `12`, which still reads as a number, so nothing downstream can tell the value is wrong.
|
|
13
|
+
*
|
|
14
|
+
* Both markups are converted to LaTeX here rather than to a per-format ad hoc notation, because
|
|
15
|
+
* `MarkdownParser` already emits LaTeX for `$...$` / `$$...$$`. Converging on it means one
|
|
16
|
+
* representation reaches every generator, and a formula survives a docx -> md -> docx round trip
|
|
17
|
+
* instead of degrading at each hop.
|
|
18
|
+
*
|
|
19
|
+
* The output is always emitted as a `code` node carrying `CodeMetadata.math`, matching the
|
|
20
|
+
* contract `MarkdownParser` established - see `src/types.ts`.
|
|
21
|
+
*/
|
|
22
|
+
/**
|
|
23
|
+
* Depth cap for the recursive walks below. Equation markup nests (a fraction inside a
|
|
24
|
+
* superscript inside a fraction), and the nesting is attacker-controlled: a few hundred bytes of
|
|
25
|
+
* hand-written XML can carry thousands of levels. The walks are recursive, so an unbounded
|
|
26
|
+
* document would exhaust the stack rather than merely producing odd output. 64 is far past any
|
|
27
|
+
* real equation - the deepest construct in a typical maths paper is 3 or 4 levels.
|
|
28
|
+
*/
|
|
29
|
+
const MAX_MATH_DEPTH = 64;
|
|
30
|
+
/** Marker substituted for a subtree that exceeded MAX_MATH_DEPTH, so truncation is never silent. */
|
|
31
|
+
const TRUNCATED = '\\ldots';
|
|
32
|
+
/**
|
|
33
|
+
* LaTeX metacharacters, escaped in any literal run of document text.
|
|
34
|
+
*
|
|
35
|
+
* The content of `<m:t>`, `<mi>`, `<mn>` and friends is literal characters, never LaTeX source -
|
|
36
|
+
* an author who types `%` into an equation means a percent sign, not a comment. Escaping is
|
|
37
|
+
* therefore lossless, and it also stops document text from injecting control sequences into the
|
|
38
|
+
* `$...$` span it lands in. Backslash must be replaced first or it would re-escape the
|
|
39
|
+
* backslashes introduced by the later replacements.
|
|
40
|
+
*/
|
|
41
|
+
const escapeLatex = (text) => text.replace(/\\/g, '\\textbackslash{}')
|
|
42
|
+
.replace(/([&%$#_{}])/g, '\\$1')
|
|
43
|
+
.replace(/~/g, '\\textasciitilde{}')
|
|
44
|
+
.replace(/\^/g, '\\textasciicircum{}');
|
|
45
|
+
/**
|
|
46
|
+
* Wraps an expression in braces unless it is already a single token.
|
|
47
|
+
*
|
|
48
|
+
* `x^{2}` and `x^2` render identically, but `x^{2y}` and `x^2y` do not, so the braces cannot be
|
|
49
|
+
* dropped whenever the argument is longer than one character. A lone digit or letter is the only
|
|
50
|
+
* safe case, and it is by far the most common one, so special-casing it keeps ordinary output
|
|
51
|
+
* readable without risking a mis-grouped exponent.
|
|
52
|
+
*/
|
|
53
|
+
const group = (latex) => /^[0-9a-zA-Z]$/.test(latex) ? latex : `{${latex}}`;
|
|
54
|
+
/** Strips any namespace prefix: `m:oMath` -> `omath`, `mml:mfrac` -> `mfrac`. */
|
|
55
|
+
const localName = (element) => element.tagName.toLowerCase().replace(/^.*:/, '');
|
|
56
|
+
/** Element children only, in document order. */
|
|
57
|
+
const elementChildren = (element) => {
|
|
58
|
+
const result = [];
|
|
59
|
+
for (let i = 0; i < (element.childNodes?.length ?? 0); i++) {
|
|
60
|
+
const child = element.childNodes[i];
|
|
61
|
+
if ((0, xmlUtils_js_1.isElement)(child))
|
|
62
|
+
result.push(child);
|
|
63
|
+
}
|
|
64
|
+
return result;
|
|
65
|
+
};
|
|
66
|
+
// ─── OMML (OOXML: DOCX, PPTX) ────────────────────────────────────────────────
|
|
67
|
+
/**
|
|
68
|
+
* Property elements. These carry styling (`m:ctrlPr` even wraps a full `w:rPr`) and never
|
|
69
|
+
* contribute display text, so they must be skipped rather than descended into - the generic
|
|
70
|
+
* fallback descending into them is one of the ways stray formatting text reached the output.
|
|
71
|
+
*/
|
|
72
|
+
const OMML_PROPERTY_TAGS = new Set([
|
|
73
|
+
'fpr', 'rpr', 'ctrlpr', 'ssubpr', 'ssuppr', 'ssubsuppr', 'dpr', 'radpr', 'narypr',
|
|
74
|
+
'funcpr', 'limlowpr', 'limupppr', 'mpr', 'argpr', 'barpr', 'accpr', 'grouppr',
|
|
75
|
+
'phantpr', 'boxpr', 'eqarrpr', 'spre', 'sty', 'scr', 'brk', 'aln', 'nor',
|
|
76
|
+
]);
|
|
77
|
+
/**
|
|
78
|
+
* `m:scr` math alphabets. OOXML encodes `ℝ` as an ASCII `R` plus a script attribute rather than
|
|
79
|
+
* as the Unicode character ODF uses, so these carry meaning and not merely styling.
|
|
80
|
+
*/
|
|
81
|
+
const OMML_MATH_ALPHABETS = {
|
|
82
|
+
'double-struck': '\\mathbb',
|
|
83
|
+
'script': '\\mathcal',
|
|
84
|
+
'fraktur': '\\mathfrak',
|
|
85
|
+
'monospace': '\\mathtt',
|
|
86
|
+
'sans-serif': '\\mathsf',
|
|
87
|
+
'roman': '\\mathrm',
|
|
88
|
+
};
|
|
89
|
+
/** Named OMML functions that map onto a LaTeX command of the same meaning. */
|
|
90
|
+
const OMML_NARY_OPERATORS = {
|
|
91
|
+
'∑': '\\sum', '∏': '\\prod', '∫': '\\int', '∬': '\\iint', '∭': '\\iiint',
|
|
92
|
+
'∮': '\\oint', '⋃': '\\bigcup', '⋂': '\\bigcap', '⋀': '\\bigwedge', '⋁': '\\bigvee',
|
|
93
|
+
};
|
|
94
|
+
/**
|
|
95
|
+
* Converts an OMML subtree (`<m:oMath>` or any node within one) to LaTeX.
|
|
96
|
+
*
|
|
97
|
+
* Covers the constructs that actually appear in office documents: fractions, sub/superscripts,
|
|
98
|
+
* delimiters, radicals, n-ary operators, functions, accents, bars, boxes and matrices. Anything
|
|
99
|
+
* unrecognized falls through to concatenating its children, which is the old behaviour and the
|
|
100
|
+
* right degradation for a construct that carries no grouping of its own.
|
|
101
|
+
*/
|
|
102
|
+
const ommlToLatex = (node, depth = 0) => {
|
|
103
|
+
if (!node)
|
|
104
|
+
return '';
|
|
105
|
+
if (node.nodeType === 3)
|
|
106
|
+
return escapeLatex(node.textContent || '');
|
|
107
|
+
if (!(0, xmlUtils_js_1.isElement)(node))
|
|
108
|
+
return '';
|
|
109
|
+
if (depth > MAX_MATH_DEPTH)
|
|
110
|
+
return TRUNCATED;
|
|
111
|
+
const element = node;
|
|
112
|
+
const tag = localName(element);
|
|
113
|
+
if (OMML_PROPERTY_TAGS.has(tag))
|
|
114
|
+
return '';
|
|
115
|
+
const kids = elementChildren(element);
|
|
116
|
+
/** Concatenates every child, used for containers and for unrecognized constructs. */
|
|
117
|
+
const all = () => kids.map(k => (0, exports.ommlToLatex)(k, depth + 1)).join('');
|
|
118
|
+
/** The LaTeX for the first `<m:xxx>` child with the given local name, or '' when absent. */
|
|
119
|
+
const part = (name) => {
|
|
120
|
+
const found = kids.find(k => localName(k) === name);
|
|
121
|
+
return found ? (0, exports.ommlToLatex)(found, depth + 1) : '';
|
|
122
|
+
};
|
|
123
|
+
switch (tag) {
|
|
124
|
+
case 'r': {
|
|
125
|
+
// A run may carry a math alphabet on its own `m:rPr` - `ℝ` is written as a plain `R`
|
|
126
|
+
// with `<m:scr m:val="double-struck"/>`, so dropping the property loses the meaning
|
|
127
|
+
// of the symbol, not just its look. Both `m:rPr` and `w:rPr` reduce to the same local
|
|
128
|
+
// name, so select on content rather than on the prefix.
|
|
129
|
+
const body = all();
|
|
130
|
+
const rPr = kids.find(k => localName(k) === 'rpr'
|
|
131
|
+
&& elementChildren(k).some(p => localName(p) === 'scr' || localName(p) === 'sty'));
|
|
132
|
+
if (!rPr || !body)
|
|
133
|
+
return body;
|
|
134
|
+
const valueOf = (name) => {
|
|
135
|
+
const el = elementChildren(rPr).find(p => localName(p) === name);
|
|
136
|
+
return el?.getAttribute('m:val') ?? el?.getAttribute('val') ?? '';
|
|
137
|
+
};
|
|
138
|
+
const alphabet = OMML_MATH_ALPHABETS[valueOf('scr')]
|
|
139
|
+
?? (valueOf('sty') === 'b' ? '\\mathbf' : undefined);
|
|
140
|
+
return alphabet ? `${alphabet}${group(body)}` : body;
|
|
141
|
+
}
|
|
142
|
+
// Containers: an equation, an argument, a base.
|
|
143
|
+
case 'omath':
|
|
144
|
+
case 'omathpara':
|
|
145
|
+
case 'e':
|
|
146
|
+
case 'num':
|
|
147
|
+
case 'den':
|
|
148
|
+
case 'sub':
|
|
149
|
+
case 'sup':
|
|
150
|
+
case 'lim':
|
|
151
|
+
case 'fname':
|
|
152
|
+
return all();
|
|
153
|
+
case 't':
|
|
154
|
+
return escapeLatex(element.textContent || '');
|
|
155
|
+
case 'f': {
|
|
156
|
+
// `m:type val="lin"` asks for an inline `a/b` rather than a stacked fraction.
|
|
157
|
+
const fPr = kids.find(k => localName(k) === 'fpr');
|
|
158
|
+
const typeEl = fPr ? elementChildren(fPr).find(k => localName(k) === 'type') : undefined;
|
|
159
|
+
const linear = typeEl?.getAttribute('m:val') === 'lin' || typeEl?.getAttribute('val') === 'lin';
|
|
160
|
+
const num = part('num');
|
|
161
|
+
const den = part('den');
|
|
162
|
+
return linear ? `${group(num)}/${group(den)}` : `\\frac${group(num)}${group(den)}`;
|
|
163
|
+
}
|
|
164
|
+
case 'ssup':
|
|
165
|
+
return `${group(part('e'))}^${group(part('sup'))}`;
|
|
166
|
+
case 'ssub':
|
|
167
|
+
return `${group(part('e'))}_${group(part('sub'))}`;
|
|
168
|
+
case 'ssubsup':
|
|
169
|
+
return `${group(part('e'))}_${group(part('sub'))}^${group(part('sup'))}`;
|
|
170
|
+
case 'spre':
|
|
171
|
+
// Pre-sub/superscript: the scripts precede the base.
|
|
172
|
+
return `{}_${group(part('sub'))}^${group(part('sup'))}${group(part('e'))}`;
|
|
173
|
+
case 'd': {
|
|
174
|
+
// Delimiters. The characters are document-supplied and default to parentheses.
|
|
175
|
+
// Plain delimiters rather than \left...\right: an unbalanced \left would break the
|
|
176
|
+
// whole expression, and a document can legitimately open without closing.
|
|
177
|
+
const dPr = kids.find(k => localName(k) === 'dpr');
|
|
178
|
+
const chr = (name, fallback) => {
|
|
179
|
+
const el = dPr ? elementChildren(dPr).find(k => localName(k) === name) : undefined;
|
|
180
|
+
const raw = el?.getAttribute('m:val') ?? el?.getAttribute('val');
|
|
181
|
+
return raw ? escapeLatex(raw) : fallback;
|
|
182
|
+
};
|
|
183
|
+
const beg = chr('begchr', '(');
|
|
184
|
+
const end = chr('endchr', ')');
|
|
185
|
+
const sep = chr('sepchr', ',');
|
|
186
|
+
const args = kids.filter(k => localName(k) === 'e').map(k => (0, exports.ommlToLatex)(k, depth + 1));
|
|
187
|
+
return `${beg}${args.join(sep)}${end}`;
|
|
188
|
+
}
|
|
189
|
+
case 'rad': {
|
|
190
|
+
const deg = part('deg');
|
|
191
|
+
const base = group(part('e'));
|
|
192
|
+
return deg ? `\\sqrt[${deg}]${base}` : `\\sqrt${base}`;
|
|
193
|
+
}
|
|
194
|
+
case 'nary': {
|
|
195
|
+
// Summation/integral and friends: operator, optional bounds, then the operand.
|
|
196
|
+
const naryPr = kids.find(k => localName(k) === 'narypr');
|
|
197
|
+
const chrEl = naryPr ? elementChildren(naryPr).find(k => localName(k) === 'chr') : undefined;
|
|
198
|
+
const chr = chrEl?.getAttribute('m:val') ?? chrEl?.getAttribute('val') ?? '∫';
|
|
199
|
+
const op = OMML_NARY_OPERATORS[chr] ?? escapeLatex(chr);
|
|
200
|
+
const sub = part('sub');
|
|
201
|
+
const sup = part('sup');
|
|
202
|
+
return `${op}${sub ? `_${group(sub)}` : ''}${sup ? `^${group(sup)}` : ''}${part('e')}`;
|
|
203
|
+
}
|
|
204
|
+
case 'func':
|
|
205
|
+
// `sin`, `log`, ... - the name is document text, so it cannot become a bare command.
|
|
206
|
+
return `\\operatorname${group(part('fname'))}${group(part('e'))}`;
|
|
207
|
+
case 'limlow':
|
|
208
|
+
return `${part('e')}_${group(part('lim'))}`;
|
|
209
|
+
case 'limupp':
|
|
210
|
+
return `${part('e')}^${group(part('lim'))}`;
|
|
211
|
+
case 'bar': {
|
|
212
|
+
const barPr = kids.find(k => localName(k) === 'barpr');
|
|
213
|
+
const posEl = barPr ? elementChildren(barPr).find(k => localName(k) === 'pos') : undefined;
|
|
214
|
+
const pos = posEl?.getAttribute('m:val') ?? posEl?.getAttribute('val');
|
|
215
|
+
return `${pos === 'top' ? '\\overline' : '\\underline'}${group(part('e'))}`;
|
|
216
|
+
}
|
|
217
|
+
case 'acc': {
|
|
218
|
+
const accPr = kids.find(k => localName(k) === 'accpr');
|
|
219
|
+
const chrEl = accPr ? elementChildren(accPr).find(k => localName(k) === 'chr') : undefined;
|
|
220
|
+
const chr = chrEl?.getAttribute('m:val') ?? chrEl?.getAttribute('val') ?? '̂';
|
|
221
|
+
const command = chr === '̄' ? '\\bar' : chr === '⃗' ? '\\vec' : chr === '̇' ? '\\dot' : '\\hat';
|
|
222
|
+
return `${command}${group(part('e'))}`;
|
|
223
|
+
}
|
|
224
|
+
case 'box':
|
|
225
|
+
case 'borderbox':
|
|
226
|
+
case 'phant':
|
|
227
|
+
case 'group':
|
|
228
|
+
case 'groupchr':
|
|
229
|
+
return part('e') || all();
|
|
230
|
+
case 'm': {
|
|
231
|
+
// Matrix: rows of cells. `\begin{matrix}` carries no delimiters of its own, which is
|
|
232
|
+
// correct - a bracketed matrix wraps the `m:m` in an `m:d` that supplies them.
|
|
233
|
+
const rows = kids.filter(k => localName(k) === 'mr').map(row => elementChildren(row)
|
|
234
|
+
.filter(cell => localName(cell) === 'e')
|
|
235
|
+
.map(cell => (0, exports.ommlToLatex)(cell, depth + 1))
|
|
236
|
+
.join(' & '));
|
|
237
|
+
return `\\begin{matrix}${rows.join(' \\\\ ')}\\end{matrix}`;
|
|
238
|
+
}
|
|
239
|
+
default:
|
|
240
|
+
return all();
|
|
241
|
+
}
|
|
242
|
+
};
|
|
243
|
+
exports.ommlToLatex = ommlToLatex;
|
|
244
|
+
// ─── MathML (ODF embedded objects, HTML, EPUB3) ──────────────────────────────
|
|
245
|
+
/** MathML operators that have a dedicated LaTeX command. */
|
|
246
|
+
const MATHML_OPERATORS = {
|
|
247
|
+
'∑': '\\sum', '∏': '\\prod', '∫': '\\int', '∮': '\\oint', '√': '\\sqrt',
|
|
248
|
+
'±': '\\pm', '∓': '\\mp', '×': '\\times', '÷': '\\div', '⋅': '\\cdot',
|
|
249
|
+
'≤': '\\leq', '≥': '\\geq', '≠': '\\neq', '≈': '\\approx', '≡': '\\equiv',
|
|
250
|
+
'∈': '\\in', '∉': '\\notin', '⊂': '\\subset', '⊆': '\\subseteq', '∪': '\\cup',
|
|
251
|
+
'∩': '\\cap', '∞': '\\infty', '→': '\\to', '⇒': '\\Rightarrow', '⇔': '\\Leftrightarrow',
|
|
252
|
+
'∀': '\\forall', '∃': '\\exists', '∂': '\\partial', '∇': '\\nabla', '…': '\\ldots',
|
|
253
|
+
'ℝ': '\\mathbb{R}', 'ℕ': '\\mathbb{N}', 'ℤ': '\\mathbb{Z}', 'ℚ': '\\mathbb{Q}', 'ℂ': '\\mathbb{C}',
|
|
254
|
+
};
|
|
255
|
+
/** Maps a literal run through the operator table, falling back to plain escaped text. */
|
|
256
|
+
const mathmlToken = (raw) => {
|
|
257
|
+
const trimmed = raw.trim();
|
|
258
|
+
return MATHML_OPERATORS[trimmed] ?? escapeLatex(raw);
|
|
259
|
+
};
|
|
260
|
+
/** Local name of a MathNode: `mml:mfrac` -> `mfrac`, `undefined` for a text node. */
|
|
261
|
+
const mathNodeName = (node) => (node.tagName || '').toLowerCase().replace(/^.*:/, '');
|
|
262
|
+
/** Presents an XML DOM node through the MathNode shape. */
|
|
263
|
+
const fromDom = (node) => {
|
|
264
|
+
if (node.nodeType === 3 || !(0, xmlUtils_js_1.isElement)(node)) {
|
|
265
|
+
return { text: node.textContent || '', children: [] };
|
|
266
|
+
}
|
|
267
|
+
const element = node;
|
|
268
|
+
const attributes = {};
|
|
269
|
+
for (let i = 0; i < (element.attributes?.length ?? 0); i++) {
|
|
270
|
+
const attr = element.attributes[i];
|
|
271
|
+
attributes[attr.name] = attr.value;
|
|
272
|
+
}
|
|
273
|
+
return {
|
|
274
|
+
tagName: element.tagName,
|
|
275
|
+
attributes,
|
|
276
|
+
text: element.textContent || '',
|
|
277
|
+
children: elementChildren(element).map(fromDom),
|
|
278
|
+
};
|
|
279
|
+
};
|
|
280
|
+
/**
|
|
281
|
+
* Converts a MathML subtree to LaTeX.
|
|
282
|
+
*
|
|
283
|
+
* When the document carries a TeX annotation (`<annotation encoding="application/x-tex">`), that
|
|
284
|
+
* is the author's own source and is used verbatim in preference to anything reconstructed here.
|
|
285
|
+
* ODF's `<annotation encoding="StarMath 5.0">` is deliberately not used - StarMath is not LaTeX,
|
|
286
|
+
* and emitting it would put a second notation back into the output this module exists to unify.
|
|
287
|
+
*/
|
|
288
|
+
const mathmlTreeToLatex = (node, depth = 0) => {
|
|
289
|
+
if (!node)
|
|
290
|
+
return '';
|
|
291
|
+
if (!node.tagName)
|
|
292
|
+
return mathmlToken(node.text || '');
|
|
293
|
+
if (depth > MAX_MATH_DEPTH)
|
|
294
|
+
return TRUNCATED;
|
|
295
|
+
const tag = mathNodeName(node);
|
|
296
|
+
const kids = node.children ?? [];
|
|
297
|
+
/**
|
|
298
|
+
* The literal content of a leaf token. The DOM adapter fills `text` with `textContent`, but
|
|
299
|
+
* `HtmlParser` keeps an element's text in child text nodes and leaves `text` unset, so fall
|
|
300
|
+
* back to gathering the children rather than emitting an empty token.
|
|
301
|
+
*/
|
|
302
|
+
const tokenText = () => node.text || kids.map(k => k.text || '').join('');
|
|
303
|
+
const all = () => kids.map(k => (0, exports.mathmlTreeToLatex)(k, depth + 1)).join('');
|
|
304
|
+
const arg = (index) => (kids[index] ? (0, exports.mathmlTreeToLatex)(kids[index], depth + 1) : '');
|
|
305
|
+
switch (tag) {
|
|
306
|
+
case 'math':
|
|
307
|
+
case 'semantics': {
|
|
308
|
+
const tex = kids.find(k => mathNodeName(k) === 'annotation'
|
|
309
|
+
&& /tex/i.test(k.attributes?.['encoding'] || ''));
|
|
310
|
+
// Same reason `tokenText` exists below: an element's text is in `text` for the DOM
|
|
311
|
+
// adapter and in child text nodes for `HtmlParser`, so both have to be consulted.
|
|
312
|
+
if (tex)
|
|
313
|
+
return (tex.text || (tex.children ?? []).map(c => c.text || '').join('')).trim();
|
|
314
|
+
return kids.map(k => (0, exports.mathmlTreeToLatex)(k, depth + 1)).join('');
|
|
315
|
+
}
|
|
316
|
+
case 'mrow':
|
|
317
|
+
case 'mstyle':
|
|
318
|
+
case 'mpadded':
|
|
319
|
+
case 'mphantom':
|
|
320
|
+
return all();
|
|
321
|
+
case 'mi':
|
|
322
|
+
case 'mn':
|
|
323
|
+
case 'mo':
|
|
324
|
+
case 'mtext':
|
|
325
|
+
case 'ms':
|
|
326
|
+
return mathmlToken(tokenText());
|
|
327
|
+
case 'mfrac':
|
|
328
|
+
return `\\frac${group(arg(0))}${group(arg(1))}`;
|
|
329
|
+
case 'msup':
|
|
330
|
+
return `${group(arg(0))}^${group(arg(1))}`;
|
|
331
|
+
case 'msub':
|
|
332
|
+
return `${group(arg(0))}_${group(arg(1))}`;
|
|
333
|
+
case 'msubsup':
|
|
334
|
+
return `${group(arg(0))}_${group(arg(1))}^${group(arg(2))}`;
|
|
335
|
+
case 'munder':
|
|
336
|
+
return `\\underset${group(arg(1))}${group(arg(0))}`;
|
|
337
|
+
case 'mover':
|
|
338
|
+
return `\\overset${group(arg(1))}${group(arg(0))}`;
|
|
339
|
+
case 'munderover':
|
|
340
|
+
return `${group(arg(0))}_${group(arg(1))}^${group(arg(2))}`;
|
|
341
|
+
case 'msqrt':
|
|
342
|
+
return `\\sqrt${group(all())}`;
|
|
343
|
+
case 'mroot':
|
|
344
|
+
return `\\sqrt[${arg(1)}]${group(arg(0))}`;
|
|
345
|
+
case 'mfenced': {
|
|
346
|
+
// Deprecated in MathML 3 but still emitted by older producers.
|
|
347
|
+
const open = escapeLatex(node.attributes?.['open'] ?? '(');
|
|
348
|
+
const close = escapeLatex(node.attributes?.['close'] ?? ')');
|
|
349
|
+
const sep = escapeLatex(node.attributes?.['separators'] ?? ',');
|
|
350
|
+
return `${open}${kids.map(k => (0, exports.mathmlTreeToLatex)(k, depth + 1)).join(sep)}${close}`;
|
|
351
|
+
}
|
|
352
|
+
case 'mtable':
|
|
353
|
+
return `\\begin{matrix}${kids
|
|
354
|
+
.filter(k => mathNodeName(k) === 'mtr')
|
|
355
|
+
.map(row => (row.children ?? [])
|
|
356
|
+
.filter(cell => mathNodeName(cell) === 'mtd')
|
|
357
|
+
.map(cell => (0, exports.mathmlTreeToLatex)(cell, depth + 1))
|
|
358
|
+
.join(' & '))
|
|
359
|
+
.join(' \\\\ ')}\\end{matrix}`;
|
|
360
|
+
case 'mtr':
|
|
361
|
+
case 'mtd':
|
|
362
|
+
return all();
|
|
363
|
+
case 'mspace':
|
|
364
|
+
return ' ';
|
|
365
|
+
// Presentation-only wrappers and the annotations themselves carry nothing renderable:
|
|
366
|
+
// `annotation-xml` duplicates the presentation tree in Content MathML, and emitting both
|
|
367
|
+
// would double every formula that has one.
|
|
368
|
+
case 'annotation':
|
|
369
|
+
case 'annotation-xml':
|
|
370
|
+
case 'maction':
|
|
371
|
+
return '';
|
|
372
|
+
default:
|
|
373
|
+
return all();
|
|
374
|
+
}
|
|
375
|
+
};
|
|
376
|
+
exports.mathmlTreeToLatex = mathmlTreeToLatex;
|
|
377
|
+
/** Converts a MathML subtree held in an XML DOM (ODF embedded objects) to LaTeX. */
|
|
378
|
+
const mathmlToLatex = (node, depth = 0) => (0, exports.mathmlTreeToLatex)(fromDom(node), depth);
|
|
379
|
+
exports.mathmlToLatex = mathmlToLatex;
|
|
380
|
+
/**
|
|
381
|
+
* True when a converted equation carries nothing worth emitting, so callers can drop the node
|
|
382
|
+
* instead of pushing an empty `$$` into the output.
|
|
383
|
+
*/
|
|
384
|
+
const isEmptyMath = (latex) => latex.trim().length === 0;
|
|
385
|
+
exports.isEmptyMath = isEmptyMath;
|