officeparser 7.4.0 → 7.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -4
- package/dist/generators/BaseGenerator.d.ts +15 -0
- package/dist/generators/BaseGenerator.js +31 -0
- package/dist/generators/HtmlGenerator.d.ts +9 -0
- package/dist/generators/HtmlGenerator.js +34 -4
- package/dist/generators/MarkdownGenerator.d.ts +13 -0
- package/dist/generators/MarkdownGenerator.js +115 -41
- package/dist/generators/RtfGenerator.d.ts +13 -0
- package/dist/generators/RtfGenerator.js +23 -2
- package/dist/officeparser.browser.iife.js +148 -148
- package/dist/officeparser.browser.mjs +186 -186
- package/dist/officeparser.browser.slim.iife.js +153 -153
- package/dist/officeparser.browser.slim.mjs +153 -153
- package/dist/parsers/HtmlParser.js +39 -0
- package/dist/parsers/OpenOfficeParser.js +122 -164
- package/dist/parsers/PowerPointParser.js +27 -6
- package/dist/parsers/WordParser.js +21 -0
- package/dist/sbom.cdx.json +92 -92
- package/dist/utils/mathUtils.d.ts +42 -0
- package/dist/utils/mathUtils.js +385 -0
- package/package.json +9 -5
|
@@ -0,0 +1,385 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.isEmptyMath = exports.mathmlToLatex = exports.mathmlTreeToLatex = exports.ommlToLatex = void 0;
|
|
4
|
+
const xmlUtils_js_1 = require("./xmlUtils.js");
|
|
5
|
+
/**
|
|
6
|
+
* Equation markup normalization.
|
|
7
|
+
*
|
|
8
|
+
* Office documents ship equations in exactly two markups: OOXML's OMML (`<m:oMath>`, used by
|
|
9
|
+
* DOCX and PPTX) and MathML (`<math>`, used by ODF embedded objects, HTML, and EPUB3). Neither
|
|
10
|
+
* is plain text, and neither survives the generic "recurse into unknown elements and concatenate
|
|
11
|
+
* their text" fallback every parser ends with: `<m:num>1</m:num><m:den>2</m:den>` collapses to
|
|
12
|
+
* `12`, which still reads as a number, so nothing downstream can tell the value is wrong.
|
|
13
|
+
*
|
|
14
|
+
* Both markups are converted to LaTeX here rather than to a per-format ad hoc notation, because
|
|
15
|
+
* `MarkdownParser` already emits LaTeX for `$...$` / `$$...$$`. Converging on it means one
|
|
16
|
+
* representation reaches every generator, and a formula survives a docx -> md -> docx round trip
|
|
17
|
+
* instead of degrading at each hop.
|
|
18
|
+
*
|
|
19
|
+
* The output is always emitted as a `code` node carrying `CodeMetadata.math`, matching the
|
|
20
|
+
* contract `MarkdownParser` established - see `src/types.ts`.
|
|
21
|
+
*/
|
|
22
|
+
/**
|
|
23
|
+
* Depth cap for the recursive walks below. Equation markup nests (a fraction inside a
|
|
24
|
+
* superscript inside a fraction), and the nesting is attacker-controlled: a few hundred bytes of
|
|
25
|
+
* hand-written XML can carry thousands of levels. The walks are recursive, so an unbounded
|
|
26
|
+
* document would exhaust the stack rather than merely producing odd output. 64 is far past any
|
|
27
|
+
* real equation - the deepest construct in a typical maths paper is 3 or 4 levels.
|
|
28
|
+
*/
|
|
29
|
+
const MAX_MATH_DEPTH = 64;
|
|
30
|
+
/** Marker substituted for a subtree that exceeded MAX_MATH_DEPTH, so truncation is never silent. */
|
|
31
|
+
const TRUNCATED = '\\ldots';
|
|
32
|
+
/**
|
|
33
|
+
* LaTeX metacharacters, escaped in any literal run of document text.
|
|
34
|
+
*
|
|
35
|
+
* The content of `<m:t>`, `<mi>`, `<mn>` and friends is literal characters, never LaTeX source -
|
|
36
|
+
* an author who types `%` into an equation means a percent sign, not a comment. Escaping is
|
|
37
|
+
* therefore lossless, and it also stops document text from injecting control sequences into the
|
|
38
|
+
* `$...$` span it lands in. Backslash must be replaced first or it would re-escape the
|
|
39
|
+
* backslashes introduced by the later replacements.
|
|
40
|
+
*/
|
|
41
|
+
const escapeLatex = (text) => text.replace(/\\/g, '\\textbackslash{}')
|
|
42
|
+
.replace(/([&%$#_{}])/g, '\\$1')
|
|
43
|
+
.replace(/~/g, '\\textasciitilde{}')
|
|
44
|
+
.replace(/\^/g, '\\textasciicircum{}');
|
|
45
|
+
/**
|
|
46
|
+
* Wraps an expression in braces unless it is already a single token.
|
|
47
|
+
*
|
|
48
|
+
* `x^{2}` and `x^2` render identically, but `x^{2y}` and `x^2y` do not, so the braces cannot be
|
|
49
|
+
* dropped whenever the argument is longer than one character. A lone digit or letter is the only
|
|
50
|
+
* safe case, and it is by far the most common one, so special-casing it keeps ordinary output
|
|
51
|
+
* readable without risking a mis-grouped exponent.
|
|
52
|
+
*/
|
|
53
|
+
const group = (latex) => /^[0-9a-zA-Z]$/.test(latex) ? latex : `{${latex}}`;
|
|
54
|
+
/** Strips any namespace prefix: `m:oMath` -> `omath`, `mml:mfrac` -> `mfrac`. */
|
|
55
|
+
const localName = (element) => element.tagName.toLowerCase().replace(/^.*:/, '');
|
|
56
|
+
/** Element children only, in document order. */
|
|
57
|
+
const elementChildren = (element) => {
|
|
58
|
+
const result = [];
|
|
59
|
+
for (let i = 0; i < (element.childNodes?.length ?? 0); i++) {
|
|
60
|
+
const child = element.childNodes[i];
|
|
61
|
+
if ((0, xmlUtils_js_1.isElement)(child))
|
|
62
|
+
result.push(child);
|
|
63
|
+
}
|
|
64
|
+
return result;
|
|
65
|
+
};
|
|
66
|
+
// ─── OMML (OOXML: DOCX, PPTX) ────────────────────────────────────────────────
|
|
67
|
+
/**
|
|
68
|
+
* Property elements. These carry styling (`m:ctrlPr` even wraps a full `w:rPr`) and never
|
|
69
|
+
* contribute display text, so they must be skipped rather than descended into - the generic
|
|
70
|
+
* fallback descending into them is one of the ways stray formatting text reached the output.
|
|
71
|
+
*/
|
|
72
|
+
const OMML_PROPERTY_TAGS = new Set([
|
|
73
|
+
'fpr', 'rpr', 'ctrlpr', 'ssubpr', 'ssuppr', 'ssubsuppr', 'dpr', 'radpr', 'narypr',
|
|
74
|
+
'funcpr', 'limlowpr', 'limupppr', 'mpr', 'argpr', 'barpr', 'accpr', 'grouppr',
|
|
75
|
+
'phantpr', 'boxpr', 'eqarrpr', 'spre', 'sty', 'scr', 'brk', 'aln', 'nor',
|
|
76
|
+
]);
|
|
77
|
+
/**
|
|
78
|
+
* `m:scr` math alphabets. OOXML encodes `ℝ` as an ASCII `R` plus a script attribute rather than
|
|
79
|
+
* as the Unicode character ODF uses, so these carry meaning and not merely styling.
|
|
80
|
+
*/
|
|
81
|
+
const OMML_MATH_ALPHABETS = {
|
|
82
|
+
'double-struck': '\\mathbb',
|
|
83
|
+
'script': '\\mathcal',
|
|
84
|
+
'fraktur': '\\mathfrak',
|
|
85
|
+
'monospace': '\\mathtt',
|
|
86
|
+
'sans-serif': '\\mathsf',
|
|
87
|
+
'roman': '\\mathrm',
|
|
88
|
+
};
|
|
89
|
+
/** Named OMML functions that map onto a LaTeX command of the same meaning. */
|
|
90
|
+
const OMML_NARY_OPERATORS = {
|
|
91
|
+
'∑': '\\sum', '∏': '\\prod', '∫': '\\int', '∬': '\\iint', '∭': '\\iiint',
|
|
92
|
+
'∮': '\\oint', '⋃': '\\bigcup', '⋂': '\\bigcap', '⋀': '\\bigwedge', '⋁': '\\bigvee',
|
|
93
|
+
};
|
|
94
|
+
/**
|
|
95
|
+
* Converts an OMML subtree (`<m:oMath>` or any node within one) to LaTeX.
|
|
96
|
+
*
|
|
97
|
+
* Covers the constructs that actually appear in office documents: fractions, sub/superscripts,
|
|
98
|
+
* delimiters, radicals, n-ary operators, functions, accents, bars, boxes and matrices. Anything
|
|
99
|
+
* unrecognized falls through to concatenating its children, which is the old behaviour and the
|
|
100
|
+
* right degradation for a construct that carries no grouping of its own.
|
|
101
|
+
*/
|
|
102
|
+
const ommlToLatex = (node, depth = 0) => {
|
|
103
|
+
if (!node)
|
|
104
|
+
return '';
|
|
105
|
+
if (node.nodeType === 3)
|
|
106
|
+
return escapeLatex(node.textContent || '');
|
|
107
|
+
if (!(0, xmlUtils_js_1.isElement)(node))
|
|
108
|
+
return '';
|
|
109
|
+
if (depth > MAX_MATH_DEPTH)
|
|
110
|
+
return TRUNCATED;
|
|
111
|
+
const element = node;
|
|
112
|
+
const tag = localName(element);
|
|
113
|
+
if (OMML_PROPERTY_TAGS.has(tag))
|
|
114
|
+
return '';
|
|
115
|
+
const kids = elementChildren(element);
|
|
116
|
+
/** Concatenates every child, used for containers and for unrecognized constructs. */
|
|
117
|
+
const all = () => kids.map(k => (0, exports.ommlToLatex)(k, depth + 1)).join('');
|
|
118
|
+
/** The LaTeX for the first `<m:xxx>` child with the given local name, or '' when absent. */
|
|
119
|
+
const part = (name) => {
|
|
120
|
+
const found = kids.find(k => localName(k) === name);
|
|
121
|
+
return found ? (0, exports.ommlToLatex)(found, depth + 1) : '';
|
|
122
|
+
};
|
|
123
|
+
switch (tag) {
|
|
124
|
+
case 'r': {
|
|
125
|
+
// A run may carry a math alphabet on its own `m:rPr` - `ℝ` is written as a plain `R`
|
|
126
|
+
// with `<m:scr m:val="double-struck"/>`, so dropping the property loses the meaning
|
|
127
|
+
// of the symbol, not just its look. Both `m:rPr` and `w:rPr` reduce to the same local
|
|
128
|
+
// name, so select on content rather than on the prefix.
|
|
129
|
+
const body = all();
|
|
130
|
+
const rPr = kids.find(k => localName(k) === 'rpr'
|
|
131
|
+
&& elementChildren(k).some(p => localName(p) === 'scr' || localName(p) === 'sty'));
|
|
132
|
+
if (!rPr || !body)
|
|
133
|
+
return body;
|
|
134
|
+
const valueOf = (name) => {
|
|
135
|
+
const el = elementChildren(rPr).find(p => localName(p) === name);
|
|
136
|
+
return el?.getAttribute('m:val') ?? el?.getAttribute('val') ?? '';
|
|
137
|
+
};
|
|
138
|
+
const alphabet = OMML_MATH_ALPHABETS[valueOf('scr')]
|
|
139
|
+
?? (valueOf('sty') === 'b' ? '\\mathbf' : undefined);
|
|
140
|
+
return alphabet ? `${alphabet}${group(body)}` : body;
|
|
141
|
+
}
|
|
142
|
+
// Containers: an equation, an argument, a base.
|
|
143
|
+
case 'omath':
|
|
144
|
+
case 'omathpara':
|
|
145
|
+
case 'e':
|
|
146
|
+
case 'num':
|
|
147
|
+
case 'den':
|
|
148
|
+
case 'sub':
|
|
149
|
+
case 'sup':
|
|
150
|
+
case 'lim':
|
|
151
|
+
case 'fname':
|
|
152
|
+
return all();
|
|
153
|
+
case 't':
|
|
154
|
+
return escapeLatex(element.textContent || '');
|
|
155
|
+
case 'f': {
|
|
156
|
+
// `m:type val="lin"` asks for an inline `a/b` rather than a stacked fraction.
|
|
157
|
+
const fPr = kids.find(k => localName(k) === 'fpr');
|
|
158
|
+
const typeEl = fPr ? elementChildren(fPr).find(k => localName(k) === 'type') : undefined;
|
|
159
|
+
const linear = typeEl?.getAttribute('m:val') === 'lin' || typeEl?.getAttribute('val') === 'lin';
|
|
160
|
+
const num = part('num');
|
|
161
|
+
const den = part('den');
|
|
162
|
+
return linear ? `${group(num)}/${group(den)}` : `\\frac${group(num)}${group(den)}`;
|
|
163
|
+
}
|
|
164
|
+
case 'ssup':
|
|
165
|
+
return `${group(part('e'))}^${group(part('sup'))}`;
|
|
166
|
+
case 'ssub':
|
|
167
|
+
return `${group(part('e'))}_${group(part('sub'))}`;
|
|
168
|
+
case 'ssubsup':
|
|
169
|
+
return `${group(part('e'))}_${group(part('sub'))}^${group(part('sup'))}`;
|
|
170
|
+
case 'spre':
|
|
171
|
+
// Pre-sub/superscript: the scripts precede the base.
|
|
172
|
+
return `{}_${group(part('sub'))}^${group(part('sup'))}${group(part('e'))}`;
|
|
173
|
+
case 'd': {
|
|
174
|
+
// Delimiters. The characters are document-supplied and default to parentheses.
|
|
175
|
+
// Plain delimiters rather than \left...\right: an unbalanced \left would break the
|
|
176
|
+
// whole expression, and a document can legitimately open without closing.
|
|
177
|
+
const dPr = kids.find(k => localName(k) === 'dpr');
|
|
178
|
+
const chr = (name, fallback) => {
|
|
179
|
+
const el = dPr ? elementChildren(dPr).find(k => localName(k) === name) : undefined;
|
|
180
|
+
const raw = el?.getAttribute('m:val') ?? el?.getAttribute('val');
|
|
181
|
+
return raw ? escapeLatex(raw) : fallback;
|
|
182
|
+
};
|
|
183
|
+
const beg = chr('begchr', '(');
|
|
184
|
+
const end = chr('endchr', ')');
|
|
185
|
+
const sep = chr('sepchr', ',');
|
|
186
|
+
const args = kids.filter(k => localName(k) === 'e').map(k => (0, exports.ommlToLatex)(k, depth + 1));
|
|
187
|
+
return `${beg}${args.join(sep)}${end}`;
|
|
188
|
+
}
|
|
189
|
+
case 'rad': {
|
|
190
|
+
const deg = part('deg');
|
|
191
|
+
const base = group(part('e'));
|
|
192
|
+
return deg ? `\\sqrt[${deg}]${base}` : `\\sqrt${base}`;
|
|
193
|
+
}
|
|
194
|
+
case 'nary': {
|
|
195
|
+
// Summation/integral and friends: operator, optional bounds, then the operand.
|
|
196
|
+
const naryPr = kids.find(k => localName(k) === 'narypr');
|
|
197
|
+
const chrEl = naryPr ? elementChildren(naryPr).find(k => localName(k) === 'chr') : undefined;
|
|
198
|
+
const chr = chrEl?.getAttribute('m:val') ?? chrEl?.getAttribute('val') ?? '∫';
|
|
199
|
+
const op = OMML_NARY_OPERATORS[chr] ?? escapeLatex(chr);
|
|
200
|
+
const sub = part('sub');
|
|
201
|
+
const sup = part('sup');
|
|
202
|
+
return `${op}${sub ? `_${group(sub)}` : ''}${sup ? `^${group(sup)}` : ''}${part('e')}`;
|
|
203
|
+
}
|
|
204
|
+
case 'func':
|
|
205
|
+
// `sin`, `log`, ... - the name is document text, so it cannot become a bare command.
|
|
206
|
+
return `\\operatorname${group(part('fname'))}${group(part('e'))}`;
|
|
207
|
+
case 'limlow':
|
|
208
|
+
return `${part('e')}_${group(part('lim'))}`;
|
|
209
|
+
case 'limupp':
|
|
210
|
+
return `${part('e')}^${group(part('lim'))}`;
|
|
211
|
+
case 'bar': {
|
|
212
|
+
const barPr = kids.find(k => localName(k) === 'barpr');
|
|
213
|
+
const posEl = barPr ? elementChildren(barPr).find(k => localName(k) === 'pos') : undefined;
|
|
214
|
+
const pos = posEl?.getAttribute('m:val') ?? posEl?.getAttribute('val');
|
|
215
|
+
return `${pos === 'top' ? '\\overline' : '\\underline'}${group(part('e'))}`;
|
|
216
|
+
}
|
|
217
|
+
case 'acc': {
|
|
218
|
+
const accPr = kids.find(k => localName(k) === 'accpr');
|
|
219
|
+
const chrEl = accPr ? elementChildren(accPr).find(k => localName(k) === 'chr') : undefined;
|
|
220
|
+
const chr = chrEl?.getAttribute('m:val') ?? chrEl?.getAttribute('val') ?? '̂';
|
|
221
|
+
const command = chr === '̄' ? '\\bar' : chr === '⃗' ? '\\vec' : chr === '̇' ? '\\dot' : '\\hat';
|
|
222
|
+
return `${command}${group(part('e'))}`;
|
|
223
|
+
}
|
|
224
|
+
case 'box':
|
|
225
|
+
case 'borderbox':
|
|
226
|
+
case 'phant':
|
|
227
|
+
case 'group':
|
|
228
|
+
case 'groupchr':
|
|
229
|
+
return part('e') || all();
|
|
230
|
+
case 'm': {
|
|
231
|
+
// Matrix: rows of cells. `\begin{matrix}` carries no delimiters of its own, which is
|
|
232
|
+
// correct - a bracketed matrix wraps the `m:m` in an `m:d` that supplies them.
|
|
233
|
+
const rows = kids.filter(k => localName(k) === 'mr').map(row => elementChildren(row)
|
|
234
|
+
.filter(cell => localName(cell) === 'e')
|
|
235
|
+
.map(cell => (0, exports.ommlToLatex)(cell, depth + 1))
|
|
236
|
+
.join(' & '));
|
|
237
|
+
return `\\begin{matrix}${rows.join(' \\\\ ')}\\end{matrix}`;
|
|
238
|
+
}
|
|
239
|
+
default:
|
|
240
|
+
return all();
|
|
241
|
+
}
|
|
242
|
+
};
|
|
243
|
+
exports.ommlToLatex = ommlToLatex;
|
|
244
|
+
// ─── MathML (ODF embedded objects, HTML, EPUB3) ──────────────────────────────
|
|
245
|
+
/** MathML operators that have a dedicated LaTeX command. */
|
|
246
|
+
const MATHML_OPERATORS = {
|
|
247
|
+
'∑': '\\sum', '∏': '\\prod', '∫': '\\int', '∮': '\\oint', '√': '\\sqrt',
|
|
248
|
+
'±': '\\pm', '∓': '\\mp', '×': '\\times', '÷': '\\div', '⋅': '\\cdot',
|
|
249
|
+
'≤': '\\leq', '≥': '\\geq', '≠': '\\neq', '≈': '\\approx', '≡': '\\equiv',
|
|
250
|
+
'∈': '\\in', '∉': '\\notin', '⊂': '\\subset', '⊆': '\\subseteq', '∪': '\\cup',
|
|
251
|
+
'∩': '\\cap', '∞': '\\infty', '→': '\\to', '⇒': '\\Rightarrow', '⇔': '\\Leftrightarrow',
|
|
252
|
+
'∀': '\\forall', '∃': '\\exists', '∂': '\\partial', '∇': '\\nabla', '…': '\\ldots',
|
|
253
|
+
'ℝ': '\\mathbb{R}', 'ℕ': '\\mathbb{N}', 'ℤ': '\\mathbb{Z}', 'ℚ': '\\mathbb{Q}', 'ℂ': '\\mathbb{C}',
|
|
254
|
+
};
|
|
255
|
+
/** Maps a literal run through the operator table, falling back to plain escaped text. */
|
|
256
|
+
const mathmlToken = (raw) => {
|
|
257
|
+
const trimmed = raw.trim();
|
|
258
|
+
return MATHML_OPERATORS[trimmed] ?? escapeLatex(raw);
|
|
259
|
+
};
|
|
260
|
+
/** Local name of a MathNode: `mml:mfrac` -> `mfrac`, `undefined` for a text node. */
|
|
261
|
+
const mathNodeName = (node) => (node.tagName || '').toLowerCase().replace(/^.*:/, '');
|
|
262
|
+
/** Presents an XML DOM node through the MathNode shape. */
|
|
263
|
+
const fromDom = (node) => {
|
|
264
|
+
if (node.nodeType === 3 || !(0, xmlUtils_js_1.isElement)(node)) {
|
|
265
|
+
return { text: node.textContent || '', children: [] };
|
|
266
|
+
}
|
|
267
|
+
const element = node;
|
|
268
|
+
const attributes = {};
|
|
269
|
+
for (let i = 0; i < (element.attributes?.length ?? 0); i++) {
|
|
270
|
+
const attr = element.attributes[i];
|
|
271
|
+
attributes[attr.name] = attr.value;
|
|
272
|
+
}
|
|
273
|
+
return {
|
|
274
|
+
tagName: element.tagName,
|
|
275
|
+
attributes,
|
|
276
|
+
text: element.textContent || '',
|
|
277
|
+
children: elementChildren(element).map(fromDom),
|
|
278
|
+
};
|
|
279
|
+
};
|
|
280
|
+
/**
|
|
281
|
+
* Converts a MathML subtree to LaTeX.
|
|
282
|
+
*
|
|
283
|
+
* When the document carries a TeX annotation (`<annotation encoding="application/x-tex">`), that
|
|
284
|
+
* is the author's own source and is used verbatim in preference to anything reconstructed here.
|
|
285
|
+
* ODF's `<annotation encoding="StarMath 5.0">` is deliberately not used - StarMath is not LaTeX,
|
|
286
|
+
* and emitting it would put a second notation back into the output this module exists to unify.
|
|
287
|
+
*/
|
|
288
|
+
const mathmlTreeToLatex = (node, depth = 0) => {
|
|
289
|
+
if (!node)
|
|
290
|
+
return '';
|
|
291
|
+
if (!node.tagName)
|
|
292
|
+
return mathmlToken(node.text || '');
|
|
293
|
+
if (depth > MAX_MATH_DEPTH)
|
|
294
|
+
return TRUNCATED;
|
|
295
|
+
const tag = mathNodeName(node);
|
|
296
|
+
const kids = node.children ?? [];
|
|
297
|
+
/**
|
|
298
|
+
* The literal content of a leaf token. The DOM adapter fills `text` with `textContent`, but
|
|
299
|
+
* `HtmlParser` keeps an element's text in child text nodes and leaves `text` unset, so fall
|
|
300
|
+
* back to gathering the children rather than emitting an empty token.
|
|
301
|
+
*/
|
|
302
|
+
const tokenText = () => node.text || kids.map(k => k.text || '').join('');
|
|
303
|
+
const all = () => kids.map(k => (0, exports.mathmlTreeToLatex)(k, depth + 1)).join('');
|
|
304
|
+
const arg = (index) => (kids[index] ? (0, exports.mathmlTreeToLatex)(kids[index], depth + 1) : '');
|
|
305
|
+
switch (tag) {
|
|
306
|
+
case 'math':
|
|
307
|
+
case 'semantics': {
|
|
308
|
+
const tex = kids.find(k => mathNodeName(k) === 'annotation'
|
|
309
|
+
&& /tex/i.test(k.attributes?.['encoding'] || ''));
|
|
310
|
+
// Same reason `tokenText` exists below: an element's text is in `text` for the DOM
|
|
311
|
+
// adapter and in child text nodes for `HtmlParser`, so both have to be consulted.
|
|
312
|
+
if (tex)
|
|
313
|
+
return (tex.text || (tex.children ?? []).map(c => c.text || '').join('')).trim();
|
|
314
|
+
return kids.map(k => (0, exports.mathmlTreeToLatex)(k, depth + 1)).join('');
|
|
315
|
+
}
|
|
316
|
+
case 'mrow':
|
|
317
|
+
case 'mstyle':
|
|
318
|
+
case 'mpadded':
|
|
319
|
+
case 'mphantom':
|
|
320
|
+
return all();
|
|
321
|
+
case 'mi':
|
|
322
|
+
case 'mn':
|
|
323
|
+
case 'mo':
|
|
324
|
+
case 'mtext':
|
|
325
|
+
case 'ms':
|
|
326
|
+
return mathmlToken(tokenText());
|
|
327
|
+
case 'mfrac':
|
|
328
|
+
return `\\frac${group(arg(0))}${group(arg(1))}`;
|
|
329
|
+
case 'msup':
|
|
330
|
+
return `${group(arg(0))}^${group(arg(1))}`;
|
|
331
|
+
case 'msub':
|
|
332
|
+
return `${group(arg(0))}_${group(arg(1))}`;
|
|
333
|
+
case 'msubsup':
|
|
334
|
+
return `${group(arg(0))}_${group(arg(1))}^${group(arg(2))}`;
|
|
335
|
+
case 'munder':
|
|
336
|
+
return `\\underset${group(arg(1))}${group(arg(0))}`;
|
|
337
|
+
case 'mover':
|
|
338
|
+
return `\\overset${group(arg(1))}${group(arg(0))}`;
|
|
339
|
+
case 'munderover':
|
|
340
|
+
return `${group(arg(0))}_${group(arg(1))}^${group(arg(2))}`;
|
|
341
|
+
case 'msqrt':
|
|
342
|
+
return `\\sqrt${group(all())}`;
|
|
343
|
+
case 'mroot':
|
|
344
|
+
return `\\sqrt[${arg(1)}]${group(arg(0))}`;
|
|
345
|
+
case 'mfenced': {
|
|
346
|
+
// Deprecated in MathML 3 but still emitted by older producers.
|
|
347
|
+
const open = escapeLatex(node.attributes?.['open'] ?? '(');
|
|
348
|
+
const close = escapeLatex(node.attributes?.['close'] ?? ')');
|
|
349
|
+
const sep = escapeLatex(node.attributes?.['separators'] ?? ',');
|
|
350
|
+
return `${open}${kids.map(k => (0, exports.mathmlTreeToLatex)(k, depth + 1)).join(sep)}${close}`;
|
|
351
|
+
}
|
|
352
|
+
case 'mtable':
|
|
353
|
+
return `\\begin{matrix}${kids
|
|
354
|
+
.filter(k => mathNodeName(k) === 'mtr')
|
|
355
|
+
.map(row => (row.children ?? [])
|
|
356
|
+
.filter(cell => mathNodeName(cell) === 'mtd')
|
|
357
|
+
.map(cell => (0, exports.mathmlTreeToLatex)(cell, depth + 1))
|
|
358
|
+
.join(' & '))
|
|
359
|
+
.join(' \\\\ ')}\\end{matrix}`;
|
|
360
|
+
case 'mtr':
|
|
361
|
+
case 'mtd':
|
|
362
|
+
return all();
|
|
363
|
+
case 'mspace':
|
|
364
|
+
return ' ';
|
|
365
|
+
// Presentation-only wrappers and the annotations themselves carry nothing renderable:
|
|
366
|
+
// `annotation-xml` duplicates the presentation tree in Content MathML, and emitting both
|
|
367
|
+
// would double every formula that has one.
|
|
368
|
+
case 'annotation':
|
|
369
|
+
case 'annotation-xml':
|
|
370
|
+
case 'maction':
|
|
371
|
+
return '';
|
|
372
|
+
default:
|
|
373
|
+
return all();
|
|
374
|
+
}
|
|
375
|
+
};
|
|
376
|
+
exports.mathmlTreeToLatex = mathmlTreeToLatex;
|
|
377
|
+
/** Converts a MathML subtree held in an XML DOM (ODF embedded objects) to LaTeX. */
|
|
378
|
+
const mathmlToLatex = (node, depth = 0) => (0, exports.mathmlTreeToLatex)(fromDom(node), depth);
|
|
379
|
+
exports.mathmlToLatex = mathmlToLatex;
|
|
380
|
+
/**
|
|
381
|
+
* True when a converted equation carries nothing worth emitting, so callers can drop the node
|
|
382
|
+
* instead of pushing an empty `$$` into the output.
|
|
383
|
+
*/
|
|
384
|
+
const isEmptyMath = (latex) => latex.trim().length === 0;
|
|
385
|
+
exports.isEmptyMath = isEmptyMath;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "7.
|
|
3
|
+
"version": "7.5.0",
|
|
4
4
|
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html, .epub) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, PDF, EPUB, and RAG-focused chunks.",
|
|
5
5
|
"funding": "https://github.com/sponsors/harshankur",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -37,15 +37,19 @@
|
|
|
37
37
|
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/ && cp dist/officeparser.browser.slim.iife.js docs/dist/ && cp dist/officeparser.browser.slim.mjs docs/dist/ && mkdir -p docs/test/files && cp -r test/files/* docs/test/files/",
|
|
38
38
|
"lint": "eslint src",
|
|
39
39
|
"test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:exhaustive && npm run test:generator && npm run test:security && npm run test:cli",
|
|
40
|
+
"test:fast": "npm run lint && npm run test:parser:fast && npm run test:exhaustive && npm run test:generator:fast && npm run test:security && npm run test:cli:fast",
|
|
40
41
|
"test:exhaustive": "npx tsx test/testExhaustive.ts",
|
|
41
42
|
"test:security": "npx tsx test/security/testSanitization.ts",
|
|
42
43
|
"test:baseline": "npm run test:parser:baseline && npm run test:generator:baseline",
|
|
43
44
|
"test:parser": "npx tsx test/parser/testOfficeParser.ts",
|
|
45
|
+
"test:parser:fast": "npx tsx test/parser/testOfficeParser.ts fast",
|
|
44
46
|
"test:parser:baseline": "npx tsx test/parser/testOfficeParser.ts baseline",
|
|
45
47
|
"test:generator": "npx tsx test/generator/testOfficeGenerator.ts",
|
|
48
|
+
"test:generator:fast": "npx tsx test/generator/testOfficeGenerator.ts fast",
|
|
46
49
|
"test:generator:baseline": "npx tsx test/generator/testOfficeGenerator.ts baseline",
|
|
47
50
|
"test:artifacts": "npx tsx test/testShippingArtifacts.ts",
|
|
48
51
|
"test:cli": "npx tsx test/cli/testCli.ts",
|
|
52
|
+
"test:cli:fast": "npx tsx test/cli/testCli.ts fast",
|
|
49
53
|
"test:visualizer": "node test/testVisualizer.js",
|
|
50
54
|
"test:integration": "node test/testIntegration.js",
|
|
51
55
|
"test:license": "npm run sbom && node scripts/validate-licenses.js",
|
|
@@ -126,13 +130,13 @@
|
|
|
126
130
|
},
|
|
127
131
|
"devDependencies": {
|
|
128
132
|
"@types/node": "^26.1.1",
|
|
129
|
-
"@typescript-eslint/eslint-plugin": "^8.
|
|
130
|
-
"@typescript-eslint/parser": "^8.
|
|
133
|
+
"@typescript-eslint/eslint-plugin": "^8.65.0",
|
|
134
|
+
"@typescript-eslint/parser": "^8.65.0",
|
|
131
135
|
"buffer": "^6.0.3",
|
|
132
136
|
"dts-bundle-generator": "^9.5.1",
|
|
133
137
|
"esbuild": "^0.28.1",
|
|
134
138
|
"esbuild-plugins-node-modules-polyfill": "^1.8.2",
|
|
135
|
-
"eslint": "^10.
|
|
139
|
+
"eslint": "^10.8.0",
|
|
136
140
|
"husky": "^9.1.7",
|
|
137
141
|
"postject": "^1.0.0-alpha.6",
|
|
138
142
|
"process": "^0.11.10",
|
|
@@ -140,4 +144,4 @@
|
|
|
140
144
|
"tsx": "^4.23.1",
|
|
141
145
|
"typescript": "^6.0.3"
|
|
142
146
|
}
|
|
143
|
-
}
|
|
147
|
+
}
|