officeparser 6.0.6 → 6.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +92 -13
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +43 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +116 -0
  6. package/dist/index.d.ts +3 -3
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +756 -0
  10. package/dist/officeparser.browser.iife.js +112 -0
  11. package/dist/officeparser.browser.mjs +111 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +71 -63
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +131 -114
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +85 -88
  20. package/dist/parsers/RtfParser.d.ts +1 -1
  21. package/dist/parsers/RtfParser.js +10 -6
  22. package/dist/parsers/WordParser.d.ts +1 -1
  23. package/dist/parsers/WordParser.js +109 -101
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +69 -1
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +39 -18
  39. package/dist/officeparser.browser.js +0 -153
  40. package/dist/officeparser.browser.js.map +0 -7
@@ -61,12 +61,11 @@
61
61
  */
62
62
  Object.defineProperty(exports, "__esModule", { value: true });
63
63
  exports.parseWord = void 0;
64
- const xmldom_1 = require("@xmldom/xmldom");
65
- const errorUtils_1 = require("../utils/errorUtils");
66
- const imageUtils_1 = require("../utils/imageUtils");
67
- const ocrUtils_1 = require("../utils/ocrUtils");
68
- const xmlUtils_1 = require("../utils/xmlUtils");
69
- const zipUtils_1 = require("../utils/zipUtils");
64
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
65
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
66
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
67
+ const xmlUtils_js_1 = require("../utils/xmlUtils.js");
68
+ const zipUtils_js_1 = require("../utils/zipUtils.js");
70
69
  /**
71
70
  * Parses a Word document (.docx) and extracts content, formatting, and metadata.
72
71
  *
@@ -90,13 +89,13 @@ const parseWord = async (buffer, config) => {
90
89
  const numberingFileRegex = /word\/numbering[\d+]?.xml/;
91
90
  const mediaFileRegex = /(word\/)?media\/.*/;
92
91
  const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
92
+ const customPropsFileRegex = /docProps\/custom\.xml/;
93
93
  const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
94
94
  const stylesFileRegex = /word\/styles[\d+]?.xml/;
95
- const xmlSerializer = new xmldom_1.XMLSerializer();
96
95
  // Helper to extract formatting from run properties XML string
97
96
  const extractFormattingFromXml = (rPr) => {
98
97
  const formatting = {};
99
- const rPrString = xmlSerializer.serializeToString(rPr);
98
+ const rPrString = (0, xmlUtils_js_1.serializeXml)(rPr);
100
99
  // Helper to check boolean properties
101
100
  const getBoolVal = (xmlSnippet, tagName) => {
102
101
  const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
@@ -173,17 +172,24 @@ const parseWord = async (buffer, config) => {
173
172
  }
174
173
  return formatting;
175
174
  };
176
- const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
175
+ const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
177
176
  !!x.match(footnotesFileRegex) ||
178
177
  !!x.match(endnotesFileRegex) ||
179
178
  !!x.match(numberingFileRegex) ||
180
179
  !!x.match(corePropsFileRegex) ||
180
+ !!x.match(customPropsFileRegex) ||
181
181
  !!x.match(relsFileRegex) ||
182
182
  !!x.match(stylesFileRegex) ||
183
183
  (!!config.extractAttachments && !!x.match(mediaFileRegex)));
184
184
  // Extract metadata
185
185
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
186
- const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
186
+ const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
187
+ const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
188
+ if (customPropsFile) {
189
+ const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
190
+ if (Object.keys(customProperties).length > 0)
191
+ metadata.customProperties = customProperties;
192
+ }
187
193
  const footnoteMap = new Map();
188
194
  const endnoteMap = new Map();
189
195
  const collectedNotes = [];
@@ -193,11 +199,11 @@ const parseWord = async (buffer, config) => {
193
199
  const relsFile = files.find(f => f.path.match(relsFileRegex));
194
200
  const relsMap = {};
195
201
  if (relsFile) {
196
- const relsXml = (0, xmlUtils_1.parseXmlString)(relsFile.content.toString());
197
- const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
198
- for (let i = 0; i < relationships.length; i++) {
199
- const id = relationships[i].getAttribute("Id");
200
- const target = relationships[i].getAttribute("Target");
202
+ const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
203
+ const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
204
+ for (const relationship of relationships) {
205
+ const id = relationship.getAttribute("Id");
206
+ const target = relationship.getAttribute("Target");
201
207
  if (id && target) {
202
208
  relsMap[id] = target;
203
209
  }
@@ -206,27 +212,27 @@ const parseWord = async (buffer, config) => {
206
212
  const numberingFile = files.find(f => f.path.match(numberingFileRegex));
207
213
  const numberingMap = {};
208
214
  if (numberingFile) {
209
- const numberingXml = (0, xmlUtils_1.parseXmlString)(numberingFile.content.toString());
210
- const nums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:num");
211
- const abstractNums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:abstractNum");
215
+ const numberingXml = (0, xmlUtils_js_1.parseXmlString)(numberingFile.content.toString());
216
+ const nums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:num");
217
+ const abstractNums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:abstractNum");
212
218
  const abstractNumMap = {};
213
- for (let i = 0; i < abstractNums.length; i++) {
214
- const abstractNumId = abstractNums[i].getAttribute("w:abstractNumId");
219
+ for (const abstractNum of abstractNums) {
220
+ const abstractNumId = abstractNum.getAttribute("w:abstractNumId");
215
221
  if (abstractNumId) {
216
- abstractNumMap[abstractNumId] = abstractNums[i];
222
+ abstractNumMap[abstractNumId] = abstractNum;
217
223
  }
218
224
  }
219
- for (let i = 0; i < nums.length; i++) {
220
- const numId = nums[i].getAttribute("w:numId");
221
- const abstractNumIdNode = (0, xmlUtils_1.getElementsByTagName)(nums[i], "w:abstractNumId")[0];
225
+ for (const num of nums) {
226
+ const numId = num.getAttribute("w:numId");
227
+ const abstractNumIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(num, "w:abstractNumId");
222
228
  const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
223
229
  if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
224
230
  numberingMap[numId] = {};
225
- const lvls = (0, xmlUtils_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
226
- for (let j = 0; j < lvls.length; j++) {
227
- const ilvl = lvls[j].getAttribute("w:ilvl");
228
- const numFmtNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:numFmt")[0];
229
- const lvlTextNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:lvlText")[0];
231
+ const lvls = (0, xmlUtils_js_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
232
+ for (const lvl of lvls) {
233
+ const ilvl = lvl.getAttribute("w:ilvl");
234
+ const numFmtNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:numFmt");
235
+ const lvlTextNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:lvlText");
230
236
  if (ilvl) {
231
237
  numberingMap[numId][ilvl] = {
232
238
  numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
@@ -241,25 +247,25 @@ const parseWord = async (buffer, config) => {
241
247
  const stylesFile = files.find(f => f.path.match(stylesFileRegex));
242
248
  const styleMap = {};
243
249
  if (stylesFile) {
244
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
245
- const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
246
- for (let i = 0; i < styles.length; i++) {
247
- const styleId = styles[i].getAttribute("w:styleId");
250
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
251
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
252
+ for (const style of styles) {
253
+ const styleId = style.getAttribute("w:styleId");
248
254
  if (styleId) {
249
- const rPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:rPr")[0];
250
- const pPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:pPr")[0];
255
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:rPr");
256
+ const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:pPr");
251
257
  const formatting = rPr ? extractFormattingFromXml(rPr) : {};
252
258
  let alignment = undefined;
253
259
  let backgroundColor = undefined;
254
260
  if (pPr) {
255
- const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
261
+ const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
256
262
  if (jc) {
257
263
  const val = jc.getAttribute("w:val");
258
264
  if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
259
265
  alignment = val;
260
266
  }
261
267
  }
262
- const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
268
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
263
269
  if (shd) {
264
270
  const fill = shd.getAttribute("w:fill");
265
271
  if (fill && fill !== 'auto')
@@ -273,12 +279,12 @@ const parseWord = async (buffer, config) => {
273
279
  // Extract document defaults
274
280
  let docDefaults = {};
275
281
  if (stylesFile) {
276
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
277
- const docDefaultsNode = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:docDefaults")[0];
282
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
283
+ const docDefaultsNode = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesXml, "w:docDefaults");
278
284
  if (docDefaultsNode) {
279
- const rPrDefaultNode = (0, xmlUtils_1.getElementsByTagName)(docDefaultsNode, "w:rPrDefault")[0];
285
+ const rPrDefaultNode = (0, xmlUtils_js_1.getFirstElementByTagName)(docDefaultsNode, "w:rPrDefault");
280
286
  if (rPrDefaultNode) {
281
- const rPr = (0, xmlUtils_1.getElementsByTagName)(rPrDefaultNode, "w:rPr")[0];
287
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(rPrDefaultNode, "w:rPr");
282
288
  if (rPr) {
283
289
  docDefaults = extractFormattingFromXml(rPr);
284
290
  }
@@ -288,13 +294,13 @@ const parseWord = async (buffer, config) => {
288
294
  // Detect the default paragraph style (for international compatibility)
289
295
  let defaultParaStyleId = undefined;
290
296
  if (stylesFile) {
291
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
292
- const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
297
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
298
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
293
299
  // Look for a style with w:type="paragraph" and w:default="1"
294
- for (let i = 0; i < styles.length; i++) {
295
- const styleType = styles[i].getAttribute("w:type");
296
- const isDefault = styles[i].getAttribute("w:default");
297
- const styleId = styles[i].getAttribute("w:styleId");
300
+ for (const style of styles) {
301
+ const styleType = style.getAttribute("w:type");
302
+ const isDefault = style.getAttribute("w:default");
303
+ const styleId = style.getAttribute("w:styleId");
298
304
  if (styleType === "paragraph" && isDefault === "1" && styleId) {
299
305
  defaultParaStyleId = styleId;
300
306
  break;
@@ -310,22 +316,22 @@ const parseWord = async (buffer, config) => {
310
316
  const numberingState = {};
311
317
  const listCounters = {}; // Track item index per listId/level
312
318
  // Helper to parse a paragraph node
313
- const parseParagraph = (pNode) => {
314
- const pXml = xmlSerializer.serializeToString(pNode);
319
+ const parseParagraph = (pNode, documentContent) => {
320
+ const pXml = pNode.toString();
315
321
  // Check if it's a list item
316
- const numPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:numPr")[0];
322
+ const numPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:numPr");
317
323
  const isList = !!numPr;
318
324
  // Check if it's a heading
319
- const pPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:pPr")[0];
320
- const pStyle = pPr ? (0, xmlUtils_1.getElementsByTagName)(pPr, "w:pStyle")[0] : null;
321
- const pStyleVal = pStyle ? pStyle.getAttribute("w:val") : null;
325
+ const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:pPr");
326
+ const pStyle = pPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:pStyle") : null;
327
+ const pStyleVal = pStyle?.getAttribute("w:val");
322
328
  const isHeading = pStyleVal ? (pStyleVal.startsWith("Heading") || pStyleVal === "Title") : false;
323
329
  // Extract Paragraph Style Properties
324
330
  const styleProps = pStyleVal && styleMap[pStyleVal] ? styleMap[pStyleVal] : { formatting: {} };
325
331
  // Extract Alignment
326
332
  let alignment = styleProps.alignment;
327
333
  if (pPr) {
328
- const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
334
+ const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
329
335
  if (jc) {
330
336
  const val = jc.getAttribute("w:val");
331
337
  if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
@@ -336,7 +342,7 @@ const parseWord = async (buffer, config) => {
336
342
  // Extract Paragraph Background
337
343
  let paraBackgroundColor = styleProps.backgroundColor;
338
344
  if (pPr) {
339
- const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
345
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
340
346
  if (shd) {
341
347
  const fill = shd.getAttribute("w:fill");
342
348
  if (fill && fill !== 'auto') {
@@ -347,7 +353,7 @@ const parseWord = async (buffer, config) => {
347
353
  // Extract paragraph-level run properties
348
354
  let paragraphRunFormatting = { ...styleProps.formatting };
349
355
  if (pPr) {
350
- const pPrRPr = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:rPr")[0];
356
+ const pPrRPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:rPr");
351
357
  if (pPrRPr) {
352
358
  const pPrFormatting = extractFormattingFromXml(pPrRPr);
353
359
  for (const key in pPrFormatting) {
@@ -366,9 +372,9 @@ const parseWord = async (buffer, config) => {
366
372
  const children = [];
367
373
  // Traverse children of paragraph (runs, hyperlinks, etc.)
368
374
  const processChildNode = (node) => {
369
- if (node.nodeName === 'w:r') {
375
+ if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
370
376
  const runNode = node;
371
- const rPr = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:rPr")[0];
377
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:rPr");
372
378
  // Formatting
373
379
  let formatting = {};
374
380
  // Apply paragraph-level formatting
@@ -376,7 +382,7 @@ const parseWord = async (buffer, config) => {
376
382
  formatting[key] = paragraphRunFormatting[key];
377
383
  }
378
384
  // Check for run style
379
- const rStyle = rPr ? (0, xmlUtils_1.getElementsByTagName)(rPr, "w:rStyle")[0] : null;
385
+ const rStyle = rPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "w:rStyle") : null;
380
386
  const rStyleVal = rStyle ? rStyle.getAttribute("w:val") : pStyleVal;
381
387
  if (rStyleVal && styleMap[rStyleVal]) {
382
388
  for (const key in styleMap[rStyleVal].formatting) {
@@ -401,7 +407,7 @@ const parseWord = async (buffer, config) => {
401
407
  formatting.backgroundColor = paraBackgroundColor;
402
408
  }
403
409
  // Text content
404
- const tNodes = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:t");
410
+ const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:t");
405
411
  for (const tNode of tNodes) {
406
412
  const tContent = tNode.textContent || '';
407
413
  text += tContent;
@@ -411,7 +417,7 @@ const parseWord = async (buffer, config) => {
411
417
  formatting: formatting
412
418
  };
413
419
  if (config.includeRawContent) {
414
- textNode.rawContent = xmlSerializer.serializeToString(tNode);
420
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
415
421
  }
416
422
  // Always set a style: run style > paragraph style > detected default
417
423
  // Use detected default style for international compatibility
@@ -423,25 +429,25 @@ const parseWord = async (buffer, config) => {
423
429
  }
424
430
  // Images/Drawings
425
431
  if (config.extractAttachments) {
426
- const drawings = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:drawing");
427
- const picts = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:pict");
432
+ const drawings = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:drawing");
433
+ const picts = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:pict");
428
434
  const allImages = [...drawings, ...picts];
429
435
  for (const imgNode of allImages) {
430
- const imgXml = xmlSerializer.serializeToString(imgNode);
436
+ const imgXml = (0, xmlUtils_js_1.serializeXml)(imgNode);
431
437
  // Extract Alt Text
432
438
  let altText = '';
433
- const docPr = (0, xmlUtils_1.getElementsByTagName)(imgNode, "wp:docPr")[0];
439
+ const docPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "wp:docPr");
434
440
  if (docPr) {
435
441
  altText = docPr.getAttribute("descr") || docPr.getAttribute("title") || '';
436
442
  }
437
443
  // Extract Relationship ID
438
444
  let rId = '';
439
- const blip = (0, xmlUtils_1.getElementsByTagName)(imgNode, "a:blip")[0];
445
+ const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "a:blip");
440
446
  if (blip) {
441
447
  rId = blip.getAttribute("r:embed") || '';
442
448
  }
443
449
  else {
444
- const imagedata = (0, xmlUtils_1.getElementsByTagName)(imgNode, "v:imagedata")[0];
450
+ const imagedata = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "v:imagedata");
445
451
  if (imagedata) {
446
452
  rId = imagedata.getAttribute("r:id") || '';
447
453
  }
@@ -456,7 +462,7 @@ const parseWord = async (buffer, config) => {
456
462
  metadata: { attachmentName: filename, altText: altText }
457
463
  };
458
464
  if (config.includeRawContent) {
459
- imageNode.rawContent = imgXml;
465
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
460
466
  }
461
467
  children.push(imageNode);
462
468
  }
@@ -467,7 +473,7 @@ const parseWord = async (buffer, config) => {
467
473
  text: '',
468
474
  };
469
475
  if (config.includeRawContent) {
470
- imageNode.rawContent = imgXml;
476
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
471
477
  }
472
478
  children.push(imageNode);
473
479
  }
@@ -475,7 +481,7 @@ const parseWord = async (buffer, config) => {
475
481
  }
476
482
  // Footnotes/Endnotes inside runs
477
483
  if (!config.ignoreNotes) {
478
- const footnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:footnoteReference")[0];
484
+ const footnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:footnoteReference");
479
485
  if (footnoteRef) {
480
486
  const id = footnoteRef.getAttribute("w:id");
481
487
  if (id && footnoteMap.has(id)) {
@@ -494,7 +500,7 @@ const parseWord = async (buffer, config) => {
494
500
  }
495
501
  }
496
502
  }
497
- const endnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:endnoteReference")[0];
503
+ const endnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:endnoteReference");
498
504
  if (endnoteRef) {
499
505
  const id = endnoteRef.getAttribute("w:id");
500
506
  if (id && endnoteMap.has(id)) {
@@ -515,7 +521,7 @@ const parseWord = async (buffer, config) => {
515
521
  }
516
522
  }
517
523
  }
518
- else if (node.nodeName === 'w:hyperlink') {
524
+ else if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:hyperlink') {
519
525
  const hlNode = node;
520
526
  const rId = hlNode.getAttribute("r:id");
521
527
  const anchor = hlNode.getAttribute("w:anchor");
@@ -548,8 +554,8 @@ const parseWord = async (buffer, config) => {
548
554
  processChildNode(child);
549
555
  }
550
556
  if (isList) {
551
- const numIdNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:numId")[0];
552
- const ilvlNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:ilvl")[0];
557
+ const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
558
+ const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
553
559
  const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
554
560
  const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
555
561
  let listType = 'ordered';
@@ -592,7 +598,7 @@ const parseWord = async (buffer, config) => {
592
598
  }
593
599
  };
594
600
  if (config.includeRawContent)
595
- listNode.rawContent = pXml;
601
+ listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
596
602
  return listNode;
597
603
  }
598
604
  else if (isHeading) {
@@ -604,7 +610,7 @@ const parseWord = async (buffer, config) => {
604
610
  metadata: { level, alignment, style: pStyleVal ?? undefined }
605
611
  };
606
612
  if (config.includeRawContent)
607
- headingNode.rawContent = pXml;
613
+ headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
608
614
  return headingNode;
609
615
  }
610
616
  else {
@@ -615,20 +621,20 @@ const parseWord = async (buffer, config) => {
615
621
  metadata: { alignment, style: pStyleVal ?? undefined }
616
622
  };
617
623
  if (config.includeRawContent)
618
- paraNode.rawContent = pXml;
624
+ paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
619
625
  return paraNode;
620
626
  }
621
627
  };
622
628
  // Helper to parse a table node
623
- const parseTable = (tblNode) => {
629
+ const parseTable = (tblNode, documentContent) => {
624
630
  const rows = [];
625
631
  // Only get direct child rows, not nested table rows
626
- const trNodes = (0, xmlUtils_1.getDirectChildren)(tblNode, "w:tr");
632
+ const trNodes = (0, xmlUtils_js_1.getDirectChildren)(tblNode, "w:tr");
627
633
  for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
628
634
  const trNode = trNodes[rIndex];
629
635
  const cells = [];
630
636
  // Only get direct child cells, not nested table cells
631
- const tcNodes = (0, xmlUtils_1.getDirectChildren)(trNode, "w:tc");
637
+ const tcNodes = (0, xmlUtils_js_1.getDirectChildren)(trNode, "w:tc");
632
638
  for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
633
639
  const tcNode = tcNodes[cIndex];
634
640
  const cellChildren = [];
@@ -636,14 +642,14 @@ const parseWord = async (buffer, config) => {
636
642
  // Cells contain paragraphs (and other block-level elements)
637
643
  const cellContentNodes = Array.from(tcNode.childNodes);
638
644
  for (const child of cellContentNodes) {
639
- if (child.nodeName === 'w:p') {
640
- const pNode = parseParagraph(child);
645
+ if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
646
+ const pNode = parseParagraph(child, documentContent);
641
647
  cellChildren.push(pNode);
642
648
  cellText += pNode.text;
643
649
  }
644
- else if (child.nodeName === 'w:tbl') {
650
+ else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
645
651
  // Nested table
646
- const nestedTable = parseTable(child);
652
+ const nestedTable = parseTable(child, documentContent);
647
653
  cellChildren.push(nestedTable);
648
654
  // Don't add nested table text to cell text - it will be handled recursively
649
655
  }
@@ -671,26 +677,28 @@ const parseWord = async (buffer, config) => {
671
677
  if (!config.ignoreNotes) {
672
678
  const footnotesFile = files.find(f => f.path.match(footnotesFileRegex));
673
679
  if (footnotesFile) {
674
- const footnotesDoc = (0, xmlUtils_1.parseXmlString)(footnotesFile.content.toString());
675
- const footnoteNodes = (0, xmlUtils_1.getElementsByTagName)(footnotesDoc, "w:footnote");
680
+ const footnotesDoc = (0, xmlUtils_js_1.parseXmlString)(footnotesFile.content.toString());
681
+ const footnoteXml = footnotesFile.content.toString();
682
+ const footnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(footnotesDoc, "w:footnote");
676
683
  for (const node of footnoteNodes) {
677
684
  const id = node.getAttribute("w:id");
678
685
  if (!id || id === "-1" || id === "0")
679
686
  continue;
680
- const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
681
- footnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
687
+ const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
688
+ footnoteMap.set(id, pNodes.map(p => parseParagraph(p, footnoteXml)));
682
689
  }
683
690
  }
684
691
  const endnotesFile = files.find(f => f.path.match(endnotesFileRegex));
685
692
  if (endnotesFile) {
686
- const endnotesDoc = (0, xmlUtils_1.parseXmlString)(endnotesFile.content.toString());
687
- const endnoteNodes = (0, xmlUtils_1.getElementsByTagName)(endnotesDoc, "w:endnote");
693
+ const endnotesDoc = (0, xmlUtils_js_1.parseXmlString)(endnotesFile.content.toString());
694
+ const endnoteXml = endnotesFile.content.toString();
695
+ const endnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(endnotesDoc, "w:endnote");
688
696
  for (const node of endnoteNodes) {
689
697
  const id = node.getAttribute("w:id");
690
698
  if (!id || id === "-1" || id === "0")
691
699
  continue;
692
- const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
693
- endnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
700
+ const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
701
+ endnoteMap.set(id, pNodes.map(p => parseParagraph(p, endnoteXml)));
694
702
  }
695
703
  }
696
704
  }
@@ -711,16 +719,16 @@ const parseWord = async (buffer, config) => {
711
719
  if (config.includeRawContent) {
712
720
  rawContents.push(documentContent);
713
721
  }
714
- const doc = (0, xmlUtils_1.parseXmlString)(documentContent);
715
- const body = (0, xmlUtils_1.getElementsByTagName)(doc, "w:body")[0];
722
+ const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
723
+ const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
716
724
  if (body) {
717
725
  const bodyChildren = Array.from(body.childNodes);
718
726
  for (const child of bodyChildren) {
719
- if (child.nodeName === 'w:p') {
720
- content.push(parseParagraph(child));
727
+ if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
728
+ content.push(parseParagraph(child, documentContent));
721
729
  }
722
- else if (child.nodeName === 'w:tbl') {
723
- content.push(parseTable(child));
730
+ else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
731
+ content.push(parseTable(child, documentContent));
724
732
  }
725
733
  }
726
734
  }
@@ -728,15 +736,15 @@ const parseWord = async (buffer, config) => {
728
736
  // Extract attachments
729
737
  if (config.extractAttachments) {
730
738
  for (const media of mediaFiles) {
731
- const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
739
+ const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
732
740
  attachments.push(attachment);
733
741
  if (config.ocr) {
734
742
  if (attachment.mimeType.startsWith('image/')) {
735
743
  try {
736
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
744
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
737
745
  }
738
746
  catch (e) {
739
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
747
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
740
748
  }
741
749
  }
742
750
  }