officeparser 7.2.2 → 7.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -31,6 +31,16 @@ const imageUtils_js_1 = require("../utils/imageUtils.js");
31
31
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
32
32
  const xmlUtils_js_1 = require("../utils/xmlUtils.js");
33
33
  const zipUtils_js_1 = require("../utils/zipUtils.js");
34
+ /**
35
+ * Helper to clean and extract attachment name from xlink:href or paths.
36
+ * Handles trailing slashes, leading "./", and subdirectories.
37
+ */
38
+ const cleanAttachmentName = (href) => {
39
+ if (!href)
40
+ return '';
41
+ const cleaned = href.replace(/^\.\//, '').replace(/\/$/, '');
42
+ return cleaned.split('/').pop() || '';
43
+ };
34
44
  /**
35
45
  * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
36
46
  *
@@ -82,6 +92,7 @@ const parseOpenOffice = async (buffer, config) => {
82
92
  let lastListStyle = null;
83
93
  let listIdCounter = 0;
84
94
  let lastWasList = false;
95
+ let traverse;
85
96
  // Helper to parse styles
86
97
  const parseStyles = (xml) => {
87
98
  const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
@@ -170,6 +181,64 @@ const parseOpenOffice = async (buffer, config) => {
170
181
  * Returns the paragraph content without creating a content node.
171
182
  *
172
183
  * @param node - The paragraph element to parse
184
+ */
185
+ const parseMathML = (node) => {
186
+ if (!node)
187
+ return '';
188
+ if (node.nodeType === 3) { // Text node
189
+ return node.textContent || '';
190
+ }
191
+ if (node.nodeType !== 1) { // Not an element
192
+ return '';
193
+ }
194
+ const element = node;
195
+ const tagName = element.tagName.toLowerCase().replace(/^.*:/, ''); // strip namespace prefix
196
+ switch (tagName) {
197
+ case 'math':
198
+ case 'mrow':
199
+ case 'semantics':
200
+ return Array.from(element.childNodes).map(parseMathML).join('');
201
+ case 'mfrac': {
202
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
203
+ if (children.length >= 2) {
204
+ return `(${parseMathML(children[0])})/(${parseMathML(children[1])})`;
205
+ }
206
+ return Array.from(element.childNodes).map(parseMathML).join('');
207
+ }
208
+ case 'msub': {
209
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
210
+ if (children.length >= 2) {
211
+ return `${parseMathML(children[0])}_${parseMathML(children[1])}`;
212
+ }
213
+ return Array.from(element.childNodes).map(parseMathML).join('');
214
+ }
215
+ case 'msup': {
216
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
217
+ if (children.length >= 2) {
218
+ return `${parseMathML(children[0])}^${parseMathML(children[1])}`;
219
+ }
220
+ return Array.from(element.childNodes).map(parseMathML).join('');
221
+ }
222
+ case 'msubsup': {
223
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
224
+ if (children.length >= 3) {
225
+ return `${parseMathML(children[0])}_${parseMathML(children[1])}^${parseMathML(children[2])}`;
226
+ }
227
+ return Array.from(element.childNodes).map(parseMathML).join('');
228
+ }
229
+ case 'mi':
230
+ case 'mn':
231
+ case 'mo':
232
+ case 'mtext':
233
+ case 'ms':
234
+ return element.textContent || '';
235
+ case 'annotation':
236
+ return '';
237
+ default:
238
+ return Array.from(element.childNodes).map(parseMathML).join('');
239
+ }
240
+ };
241
+ /**
173
242
  * Helper to parse inline content (text, spans, links, notes, etc.) recursively.
174
243
  *
175
244
  * @param node - The element to parse (paragraph, span, or link)
@@ -325,41 +394,114 @@ const parseOpenOffice = async (buffer, config) => {
325
394
  }
326
395
  }
327
396
  else if (tagName === 'draw:frame') {
328
- // Inline image
329
397
  const frame = element;
330
- // Extract alt text
331
- let altText = '';
332
- const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
333
- const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
334
- if (svgTitle && svgTitle.textContent) {
335
- altText = svgTitle.textContent;
398
+ const drawTextBox = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "draw:text-box");
399
+ const drawObject = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "draw:object");
400
+ if (drawTextBox) {
401
+ const textBoxChildren = [];
402
+ traverse(drawTextBox, textBoxChildren, false, sourceXml);
403
+ children.push(...textBoxChildren);
404
+ const textBoxText = textBoxChildren.map(c => c.text || '').join('\n');
405
+ fullText += textBoxText;
336
406
  }
337
- else if (svgDesc && svgDesc.textContent) {
338
- altText = svgDesc.textContent;
339
- }
340
- // Extract image href
341
- let imageHref = '';
342
- const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
343
- if (drawImages.length > 0) {
344
- imageHref = drawImages[0].getAttribute("xlink:href") || '';
345
- if (imageHref) {
346
- const parts = imageHref.split('/');
347
- imageHref = parts[parts.length - 1];
407
+ else if (drawObject) {
408
+ const href = drawObject.getAttribute("xlink:href");
409
+ let isFormula = false;
410
+ let formulaText = '';
411
+ let attachmentName = '';
412
+ if (href) {
413
+ attachmentName = cleanAttachmentName(href);
414
+ const objectPath = `${attachmentName}/content.xml`;
415
+ const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
416
+ if (objectFile) {
417
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
418
+ const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
419
+ if (mathNode) {
420
+ isFormula = true;
421
+ formulaText = parseMathML(mathNode).trim();
422
+ }
423
+ }
424
+ }
425
+ if (isFormula) {
426
+ fullText += formulaText;
427
+ const textNode = {
428
+ type: 'text',
429
+ text: formulaText,
430
+ formatting: parentFormatting,
431
+ metadata: linkMetadata ? { ...linkMetadata } : undefined
432
+ };
433
+ if (config.includeRawContent) {
434
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
435
+ }
436
+ children.push(textNode);
437
+ }
438
+ else {
439
+ // Standard inline image extraction fallback if object is not a formula
440
+ let altText = '';
441
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
442
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
443
+ if (svgTitle && svgTitle.textContent) {
444
+ altText = svgTitle.textContent;
445
+ }
446
+ else if (svgDesc && svgDesc.textContent) {
447
+ altText = svgDesc.textContent;
448
+ }
449
+ let imageHref = '';
450
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
451
+ if (drawImages.length > 0) {
452
+ imageHref = drawImages[0].getAttribute("xlink:href") || '';
453
+ if (imageHref) {
454
+ imageHref = cleanAttachmentName(imageHref);
455
+ }
456
+ }
457
+ const imageNode = {
458
+ type: 'image',
459
+ text: '',
460
+ children: [],
461
+ metadata: {
462
+ attachmentName: imageHref || attachmentName,
463
+ ...(altText ? { altText } : {})
464
+ }
465
+ };
466
+ if (config.includeRawContent) {
467
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
468
+ }
469
+ children.push(imageNode);
348
470
  }
349
471
  }
350
- const imageNode = {
351
- type: 'image',
352
- text: '',
353
- children: [],
354
- metadata: {
355
- attachmentName: imageHref,
356
- ...(altText ? { altText } : {})
472
+ else {
473
+ // Standard inline image extraction fallback
474
+ let altText = '';
475
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
476
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
477
+ if (svgTitle && svgTitle.textContent) {
478
+ altText = svgTitle.textContent;
357
479
  }
358
- };
359
- if (config.includeRawContent) {
360
- imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
480
+ else if (svgDesc && svgDesc.textContent) {
481
+ altText = svgDesc.textContent;
482
+ }
483
+ let imageHref = '';
484
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
485
+ if (drawImages.length > 0) {
486
+ imageHref = drawImages[0].getAttribute("xlink:href") || '';
487
+ if (imageHref) {
488
+ imageHref = cleanAttachmentName(imageHref);
489
+ }
490
+ }
491
+ const imageNode = {
492
+ type: 'image',
493
+ text: '',
494
+ children: [],
495
+ metadata: {
496
+ attachmentName: imageHref,
497
+ ...(altText ? { altText } : {})
498
+ }
499
+ };
500
+ if (config.includeRawContent) {
501
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
502
+ }
503
+ children.push(imageNode);
361
504
  }
362
- children.push(imageNode);
363
505
  }
364
506
  }
365
507
  }
@@ -713,7 +855,7 @@ const parseOpenOffice = async (buffer, config) => {
713
855
  * @param sourceXml - The source XML string for raw content extraction
714
856
  * @param asSheet - If true, treats tables as sheets (for ODS)
715
857
  */
716
- function traverse(node, targetArray, forceHeading = false, sourceXml, asSheet = false) {
858
+ traverse = (node, targetArray, forceHeading = false, sourceXml = '', asSheet = false) => {
717
859
  if (node.tagName === "text:p") {
718
860
  const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
719
861
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
@@ -1011,8 +1153,7 @@ const parseOpenOffice = async (buffer, config) => {
1011
1153
  // Extract image href to link to attachment
1012
1154
  let imageHref = image.getAttribute("xlink:href") || '';
1013
1155
  if (imageHref) {
1014
- const parts = imageHref.split('/');
1015
- imageHref = parts[parts.length - 1];
1156
+ imageHref = cleanAttachmentName(imageHref);
1016
1157
  }
1017
1158
  const metadata = {
1018
1159
  attachmentName: imageHref,
@@ -1034,25 +1175,47 @@ const parseOpenOffice = async (buffer, config) => {
1034
1175
  targetArray.push(imageNode);
1035
1176
  }
1036
1177
  else if (object) {
1037
- // Handle embedded objects like charts
1178
+ // Handle embedded objects like charts or math formulas
1038
1179
  const href = object.getAttribute("xlink:href");
1039
1180
  if (href) {
1040
- const attachmentName = href.split('/')[0];
1181
+ const attachmentName = cleanAttachmentName(href);
1041
1182
  const objectPath = `${attachmentName}/content.xml`;
1042
1183
  const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
1043
1184
  if (objectFile) {
1044
- const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
1045
- const chartNode = {
1046
- type: 'chart',
1047
- text: chartData.rawTexts.join(" "),
1048
- metadata: {
1049
- attachmentName: attachmentName,
1050
- chartData
1185
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
1186
+ const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1187
+ if (mathNode) {
1188
+ // Math formula object at block level
1189
+ const formulaText = parseMathML(mathNode).trim();
1190
+ const formulaNode = {
1191
+ type: 'paragraph',
1192
+ text: formulaText,
1193
+ children: [
1194
+ {
1195
+ type: 'text',
1196
+ text: formulaText
1197
+ }
1198
+ ]
1199
+ };
1200
+ if (config.includeRawContent) {
1201
+ formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
1051
1202
  }
1052
- };
1053
- if (config.includeRawContent)
1054
- chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
1055
- targetArray.push(chartNode);
1203
+ targetArray.push(formulaNode);
1204
+ }
1205
+ else {
1206
+ const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
1207
+ const chartNode = {
1208
+ type: 'chart',
1209
+ text: chartData.rawTexts.join(" "),
1210
+ metadata: {
1211
+ attachmentName: attachmentName,
1212
+ chartData
1213
+ }
1214
+ };
1215
+ if (config.includeRawContent)
1216
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
1217
+ targetArray.push(chartNode);
1218
+ }
1056
1219
  }
1057
1220
  else {
1058
1221
  const chartNode = {
@@ -1077,8 +1240,7 @@ const parseOpenOffice = async (buffer, config) => {
1077
1240
  }
1078
1241
  }
1079
1242
  }
1080
- }
1081
- ;
1243
+ };
1082
1244
  // ODS: Spreadsheet
1083
1245
  if (fileType === 'ods') {
1084
1246
  const spreadsheet = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:spreadsheet");
@@ -1156,28 +1318,50 @@ const parseOpenOffice = async (buffer, config) => {
1156
1318
  if (drawImages.length > 0) {
1157
1319
  const rawHref = drawImages[0].getAttribute("xlink:href");
1158
1320
  if (rawHref) {
1159
- const parts = rawHref.split('/');
1160
- imageHref = parts[parts.length - 1];
1321
+ imageHref = cleanAttachmentName(rawHref);
1161
1322
  }
1162
1323
  }
1163
- // Extract chart object href
1324
+ // Extract chart or math object href
1164
1325
  let chartHref = '';
1326
+ let isFormula = false;
1327
+ let formulaText = '';
1165
1328
  const drawObjects = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:object");
1166
1329
  if (drawObjects.length > 0) {
1167
1330
  const href = drawObjects[0].getAttribute("xlink:href");
1168
1331
  if (href) {
1169
- // Object href is usually "./Object 1"
1170
- chartHref = href.split('/')[0];
1332
+ chartHref = cleanAttachmentName(href);
1333
+ const objectPath = `${chartHref}/content.xml`;
1334
+ const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
1335
+ if (objectFile) {
1336
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
1337
+ const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1338
+ if (mathNode) {
1339
+ isFormula = true;
1340
+ formulaText = parseMathML(mathNode).trim();
1341
+ }
1342
+ }
1171
1343
  }
1172
1344
  }
1173
- if (drawImages.length > 0) {
1345
+ if (isFormula) {
1346
+ cellText += formulaText;
1347
+ const textNode = {
1348
+ type: 'text',
1349
+ text: formulaText,
1350
+ formatting: {}
1351
+ };
1352
+ if (config.includeRawContent) {
1353
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1354
+ }
1355
+ children.push(textNode);
1356
+ }
1357
+ else if (drawImages.length > 0) {
1174
1358
  // logic for image node
1175
1359
  const imageNode = {
1176
1360
  type: 'image',
1177
1361
  text: '', // Will be populated by assignAttachmentData
1178
1362
  children: [],
1179
1363
  metadata: {
1180
- attachmentName: imageHref, // Might be empty, will resolve in assignAttachmentData
1364
+ attachmentName: imageHref || chartHref, // Might be empty, will resolve in assignAttachmentData
1181
1365
  ...(altText ? { altText } : {})
1182
1366
  }
1183
1367
  };
@@ -1195,6 +1379,9 @@ const parseOpenOffice = async (buffer, config) => {
1195
1379
  attachmentName: chartHref
1196
1380
  }
1197
1381
  };
1382
+ if (config.includeRawContent) {
1383
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1384
+ }
1198
1385
  children.push(chartNode);
1199
1386
  }
1200
1387
  }
@@ -473,7 +473,7 @@ const parseWord = async (buffer, config) => {
473
473
  const comments = [];
474
474
  // Traverse children of paragraph (runs, hyperlinks, etc.)
475
475
  const processChildNode = (node) => {
476
- if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
476
+ if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'w:r' || node.nodeName === 'm:r')) {
477
477
  const runNode = node;
478
478
  const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:rPr");
479
479
  // Formatting
@@ -512,7 +512,7 @@ const parseWord = async (buffer, config) => {
512
512
  continue;
513
513
  // also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
514
514
  // Text content
515
- if (child.tagName === "w:t" || child.tagName === "t") {
515
+ if (child.tagName === "w:t" || child.tagName === "t" || child.tagName === "m:t") {
516
516
  const tNode = child;
517
517
  const tContent = tNode.textContent || '';
518
518
  text += tContent;
package/dist/types.d.ts CHANGED
@@ -317,7 +317,7 @@ export interface OfficeParserConfig {
317
317
  * The URL/path to the PDF.js worker script.
318
318
  *
319
319
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
320
- * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
320
+ * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs`.
321
321
  * You can override this with your own local path or a different CDN link.
322
322
  */
323
323
  pdfWorkerSrc?: string;
@@ -1872,4 +1872,7 @@ export interface OfficeParserAST {
1872
1872
  */
1873
1873
  to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1874
1874
  }
1875
+ declare global {
1876
+ const __SLIM__: boolean | undefined;
1877
+ }
1875
1878
  export {};
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "7.2.2",
3
+ "version": "7.2.3",
4
4
  "description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, and RAG-focused chunks.",
5
5
  "funding": "https://github.com/sponsors/harshankur",
6
6
  "main": "dist/index.js",
@@ -13,6 +13,11 @@
13
13
  "browser": "./dist/officeparser.browser.mjs",
14
14
  "import": "./dist/index.mjs",
15
15
  "require": "./dist/index.js"
16
+ },
17
+ "./slim": {
18
+ "types": "./dist/officeparser.browser.slim.d.ts",
19
+ "browser": "./dist/officeparser.browser.slim.mjs",
20
+ "import": "./dist/officeparser.browser.slim.mjs"
16
21
  }
17
22
  },
18
23
  "sideEffects": false,
@@ -26,10 +31,10 @@
26
31
  "build": "npm run sync:versions && npm run build:node && npm run build:esm-wrapper && npm run build:browser:types && npm run build:browser",
27
32
  "build:node": "tsc",
28
33
  "build:esm-wrapper": "node scripts/generate-esm-wrapper.js",
29
- "build:browser:types": "dts-bundle-generator --no-check -o dist/officeparser.browser.d.ts src/index.ts",
34
+ "build:browser:types": "dts-bundle-generator --no-check -o dist/officeparser.browser.d.ts src/index.ts && cp dist/officeparser.browser.d.ts dist/officeparser.browser.slim.d.ts",
30
35
  "build:browser": "node build_browser.js && npm run sync:docs",
31
36
  "sync:versions": "node scripts/sync-pdfjs-versions.js",
32
- "sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/ && mkdir -p docs/test/files && cp test/files/* docs/test/files/",
37
+ "sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/ && cp dist/officeparser.browser.slim.iife.js docs/dist/ && cp dist/officeparser.browser.slim.mjs docs/dist/ && mkdir -p docs/test/files && cp test/files/* docs/test/files/",
33
38
  "lint": "eslint src",
34
39
  "test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator && npm run test:cli",
35
40
  "test:baseline": "npm run test:parser:baseline && npm run test:generator:baseline",
@@ -105,9 +110,9 @@
105
110
  "homepage": "https://officeparser.harshankur.com",
106
111
  "dependencies": {
107
112
  "@xmldom/xmldom": "^0.9.10",
108
- "fflate": "^0.8.2",
113
+ "fflate": "^0.8.3",
109
114
  "file-type": "^22.0.1",
110
- "pdfjs-dist": "5.6.205",
115
+ "pdfjs-dist": "6.1.200",
111
116
  "tesseract.js": "^7.0.0"
112
117
  },
113
118
  "peerDependenciesMeta": {
@@ -121,8 +126,8 @@
121
126
  "@typescript-eslint/parser": "^8.59.3",
122
127
  "buffer": "^6.0.3",
123
128
  "dts-bundle-generator": "^9.5.1",
124
- "esbuild": "^0.27.4",
125
- "esbuild-plugins-node-modules-polyfill": "^1.8.1",
129
+ "esbuild": "^0.28.1",
130
+ "esbuild-plugins-node-modules-polyfill": "^1.8.2",
126
131
  "eslint": "^10.3.0",
127
132
  "husky": "^9.1.7",
128
133
  "postject": "^1.0.0-alpha.6",