officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
@@ -40,6 +40,7 @@
40
40
  * - `<w:p>` - Paragraph
41
41
  * - `<w:r>` - Run (contiguous text with same formatting)
42
42
  * - `<w:t>` - Text content
43
+ * - `<w:br>` - Line or page break
43
44
  * - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
44
45
  * - `<w:pStyle>` - Paragraph style (for headings)
45
46
  * - `<w:numPr>` - List numbering properties
@@ -61,12 +62,11 @@
61
62
  */
62
63
  Object.defineProperty(exports, "__esModule", { value: true });
63
64
  exports.parseWord = void 0;
64
- const xmldom_1 = require("@xmldom/xmldom");
65
- const errorUtils_1 = require("../utils/errorUtils");
66
- const imageUtils_1 = require("../utils/imageUtils");
67
- const ocrUtils_1 = require("../utils/ocrUtils");
68
- const xmlUtils_1 = require("../utils/xmlUtils");
69
- const zipUtils_1 = require("../utils/zipUtils");
65
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
66
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
67
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
68
+ const xmlUtils_js_1 = require("../utils/xmlUtils.js");
69
+ const zipUtils_js_1 = require("../utils/zipUtils.js");
70
70
  /**
71
71
  * Parses a Word document (.docx) and extracts content, formatting, and metadata.
72
72
  *
@@ -90,13 +90,13 @@ const parseWord = async (buffer, config) => {
90
90
  const numberingFileRegex = /word\/numbering[\d+]?.xml/;
91
91
  const mediaFileRegex = /(word\/)?media\/.*/;
92
92
  const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
93
+ const customPropsFileRegex = /docProps\/custom\.xml/;
93
94
  const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
94
95
  const stylesFileRegex = /word\/styles[\d+]?.xml/;
95
- const xmlSerializer = new xmldom_1.XMLSerializer();
96
96
  // Helper to extract formatting from run properties XML string
97
97
  const extractFormattingFromXml = (rPr) => {
98
98
  const formatting = {};
99
- const rPrString = xmlSerializer.serializeToString(rPr);
99
+ const rPrString = (0, xmlUtils_js_1.serializeXml)(rPr);
100
100
  // Helper to check boolean properties
101
101
  const getBoolVal = (xmlSnippet, tagName) => {
102
102
  const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
@@ -133,7 +133,7 @@ const parseWord = async (buffer, config) => {
133
133
  // Font size
134
134
  const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
135
135
  if (szMatch)
136
- formatting.size = (parseInt(szMatch[1]) / 2).toString() + 'pt';
136
+ formatting.size = (parseInt(szMatch[1], 10) / 2).toString() + 'pt';
137
137
  // Color
138
138
  const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
139
139
  if (colorMatch && colorMatch[1] !== 'auto')
@@ -173,17 +173,45 @@ const parseWord = async (buffer, config) => {
173
173
  }
174
174
  return formatting;
175
175
  };
176
- const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
176
+ // Helper to extract indentation from paragraph properties XML string
177
+ const extractIndentationFromXml = (pPr) => {
178
+ const ind = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:ind");
179
+ if (ind) {
180
+ const indentation = {};
181
+ const left = ind.getAttribute("w:left") || ind.getAttribute("w:start");
182
+ const right = ind.getAttribute("w:right") || ind.getAttribute("w:end");
183
+ const firstLine = ind.getAttribute("w:firstLine");
184
+ const hanging = ind.getAttribute("w:hanging");
185
+ if (left)
186
+ indentation.left = parseInt(left, 10);
187
+ if (right)
188
+ indentation.right = parseInt(right, 10);
189
+ if (firstLine)
190
+ indentation.firstLine = parseInt(firstLine, 10);
191
+ if (hanging)
192
+ indentation.hanging = parseInt(hanging, 10);
193
+ return Object.keys(indentation).length > 0 ? indentation : undefined;
194
+ }
195
+ return undefined;
196
+ };
197
+ const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
177
198
  !!x.match(footnotesFileRegex) ||
178
199
  !!x.match(endnotesFileRegex) ||
179
200
  !!x.match(numberingFileRegex) ||
180
201
  !!x.match(corePropsFileRegex) ||
202
+ !!x.match(customPropsFileRegex) ||
181
203
  !!x.match(relsFileRegex) ||
182
204
  !!x.match(stylesFileRegex) ||
183
205
  (!!config.extractAttachments && !!x.match(mediaFileRegex)));
184
206
  // Extract metadata
185
207
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
186
- const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
208
+ const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
209
+ const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
210
+ if (customPropsFile) {
211
+ const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
212
+ if (Object.keys(customProperties).length > 0)
213
+ metadata.customProperties = customProperties;
214
+ }
187
215
  const footnoteMap = new Map();
188
216
  const endnoteMap = new Map();
189
217
  const collectedNotes = [];
@@ -193,11 +221,11 @@ const parseWord = async (buffer, config) => {
193
221
  const relsFile = files.find(f => f.path.match(relsFileRegex));
194
222
  const relsMap = {};
195
223
  if (relsFile) {
196
- const relsXml = (0, xmlUtils_1.parseXmlString)(relsFile.content.toString());
197
- const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
198
- for (let i = 0; i < relationships.length; i++) {
199
- const id = relationships[i].getAttribute("Id");
200
- const target = relationships[i].getAttribute("Target");
224
+ const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
225
+ const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
226
+ for (const relationship of relationships) {
227
+ const id = relationship.getAttribute("Id");
228
+ const target = relationship.getAttribute("Target");
201
229
  if (id && target) {
202
230
  relsMap[id] = target;
203
231
  }
@@ -206,27 +234,27 @@ const parseWord = async (buffer, config) => {
206
234
  const numberingFile = files.find(f => f.path.match(numberingFileRegex));
207
235
  const numberingMap = {};
208
236
  if (numberingFile) {
209
- const numberingXml = (0, xmlUtils_1.parseXmlString)(numberingFile.content.toString());
210
- const nums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:num");
211
- const abstractNums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:abstractNum");
237
+ const numberingXml = (0, xmlUtils_js_1.parseXmlString)(numberingFile.content.toString());
238
+ const nums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:num");
239
+ const abstractNums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:abstractNum");
212
240
  const abstractNumMap = {};
213
- for (let i = 0; i < abstractNums.length; i++) {
214
- const abstractNumId = abstractNums[i].getAttribute("w:abstractNumId");
241
+ for (const abstractNum of abstractNums) {
242
+ const abstractNumId = abstractNum.getAttribute("w:abstractNumId");
215
243
  if (abstractNumId) {
216
- abstractNumMap[abstractNumId] = abstractNums[i];
244
+ abstractNumMap[abstractNumId] = abstractNum;
217
245
  }
218
246
  }
219
- for (let i = 0; i < nums.length; i++) {
220
- const numId = nums[i].getAttribute("w:numId");
221
- const abstractNumIdNode = (0, xmlUtils_1.getElementsByTagName)(nums[i], "w:abstractNumId")[0];
247
+ for (const num of nums) {
248
+ const numId = num.getAttribute("w:numId");
249
+ const abstractNumIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(num, "w:abstractNumId");
222
250
  const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
223
251
  if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
224
252
  numberingMap[numId] = {};
225
- const lvls = (0, xmlUtils_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
226
- for (let j = 0; j < lvls.length; j++) {
227
- const ilvl = lvls[j].getAttribute("w:ilvl");
228
- const numFmtNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:numFmt")[0];
229
- const lvlTextNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:lvlText")[0];
253
+ const lvls = (0, xmlUtils_js_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
254
+ for (const lvl of lvls) {
255
+ const ilvl = lvl.getAttribute("w:ilvl");
256
+ const numFmtNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:numFmt");
257
+ const lvlTextNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:lvlText");
230
258
  if (ilvl) {
231
259
  numberingMap[numId][ilvl] = {
232
260
  numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
@@ -241,44 +269,48 @@ const parseWord = async (buffer, config) => {
241
269
  const stylesFile = files.find(f => f.path.match(stylesFileRegex));
242
270
  const styleMap = {};
243
271
  if (stylesFile) {
244
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
245
- const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
246
- for (let i = 0; i < styles.length; i++) {
247
- const styleId = styles[i].getAttribute("w:styleId");
272
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
273
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
274
+ for (const style of styles) {
275
+ const styleId = style.getAttribute("w:styleId");
248
276
  if (styleId) {
249
- const rPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:rPr")[0];
250
- const pPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:pPr")[0];
277
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:rPr");
278
+ const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:pPr");
251
279
  const formatting = rPr ? extractFormattingFromXml(rPr) : {};
252
280
  let alignment = undefined;
253
281
  let backgroundColor = undefined;
282
+ let paragraphIndentation = undefined;
254
283
  if (pPr) {
255
- const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
284
+ const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
256
285
  if (jc) {
257
286
  const val = jc.getAttribute("w:val");
258
287
  if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
259
288
  alignment = val;
260
289
  }
261
290
  }
262
- const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
291
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
263
292
  if (shd) {
264
293
  const fill = shd.getAttribute("w:fill");
265
294
  if (fill && fill !== 'auto')
266
295
  backgroundColor = '#' + fill;
267
296
  }
297
+ const ind = extractIndentationFromXml(pPr);
298
+ if (ind)
299
+ paragraphIndentation = ind;
268
300
  }
269
- styleMap[styleId] = { formatting, alignment, backgroundColor };
301
+ styleMap[styleId] = { formatting, alignment, backgroundColor, paragraphIndentation };
270
302
  }
271
303
  }
272
304
  }
273
305
  // Extract document defaults
274
306
  let docDefaults = {};
275
307
  if (stylesFile) {
276
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
277
- const docDefaultsNode = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:docDefaults")[0];
308
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
309
+ const docDefaultsNode = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesXml, "w:docDefaults");
278
310
  if (docDefaultsNode) {
279
- const rPrDefaultNode = (0, xmlUtils_1.getElementsByTagName)(docDefaultsNode, "w:rPrDefault")[0];
311
+ const rPrDefaultNode = (0, xmlUtils_js_1.getFirstElementByTagName)(docDefaultsNode, "w:rPrDefault");
280
312
  if (rPrDefaultNode) {
281
- const rPr = (0, xmlUtils_1.getElementsByTagName)(rPrDefaultNode, "w:rPr")[0];
313
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(rPrDefaultNode, "w:rPr");
282
314
  if (rPr) {
283
315
  docDefaults = extractFormattingFromXml(rPr);
284
316
  }
@@ -288,13 +320,13 @@ const parseWord = async (buffer, config) => {
288
320
  // Detect the default paragraph style (for international compatibility)
289
321
  let defaultParaStyleId = undefined;
290
322
  if (stylesFile) {
291
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
292
- const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
323
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
324
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
293
325
  // Look for a style with w:type="paragraph" and w:default="1"
294
- for (let i = 0; i < styles.length; i++) {
295
- const styleType = styles[i].getAttribute("w:type");
296
- const isDefault = styles[i].getAttribute("w:default");
297
- const styleId = styles[i].getAttribute("w:styleId");
326
+ for (const style of styles) {
327
+ const styleType = style.getAttribute("w:type");
328
+ const isDefault = style.getAttribute("w:default");
329
+ const styleId = style.getAttribute("w:styleId");
298
330
  if (styleType === "paragraph" && isDefault === "1" && styleId) {
299
331
  defaultParaStyleId = styleId;
300
332
  break;
@@ -310,22 +342,22 @@ const parseWord = async (buffer, config) => {
310
342
  const numberingState = {};
311
343
  const listCounters = {}; // Track item index per listId/level
312
344
  // Helper to parse a paragraph node
313
- const parseParagraph = (pNode) => {
314
- const pXml = xmlSerializer.serializeToString(pNode);
345
+ const parseParagraph = (pNode, documentContent) => {
346
+ const pXml = pNode.toString();
315
347
  // Check if it's a list item
316
- const numPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:numPr")[0];
348
+ const numPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:numPr");
317
349
  const isList = !!numPr;
318
350
  // Check if it's a heading
319
- const pPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:pPr")[0];
320
- const pStyle = pPr ? (0, xmlUtils_1.getElementsByTagName)(pPr, "w:pStyle")[0] : null;
321
- const pStyleVal = pStyle ? pStyle.getAttribute("w:val") : null;
351
+ const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:pPr");
352
+ const pStyle = pPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:pStyle") : null;
353
+ const pStyleVal = pStyle?.getAttribute("w:val");
322
354
  const isHeading = pStyleVal ? (pStyleVal.startsWith("Heading") || pStyleVal === "Title") : false;
323
355
  // Extract Paragraph Style Properties
324
356
  const styleProps = pStyleVal && styleMap[pStyleVal] ? styleMap[pStyleVal] : { formatting: {} };
325
357
  // Extract Alignment
326
358
  let alignment = styleProps.alignment;
327
359
  if (pPr) {
328
- const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
360
+ const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
329
361
  if (jc) {
330
362
  const val = jc.getAttribute("w:val");
331
363
  if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
@@ -333,10 +365,18 @@ const parseWord = async (buffer, config) => {
333
365
  }
334
366
  }
335
367
  }
368
+ // Extract Indentation
369
+ let paraIndentation = styleProps.paragraphIndentation;
370
+ if (pPr) {
371
+ const ind = extractIndentationFromXml(pPr);
372
+ if (ind) {
373
+ paraIndentation = { ...paraIndentation, ...ind };
374
+ }
375
+ }
336
376
  // Extract Paragraph Background
337
377
  let paraBackgroundColor = styleProps.backgroundColor;
338
378
  if (pPr) {
339
- const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
379
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
340
380
  if (shd) {
341
381
  const fill = shd.getAttribute("w:fill");
342
382
  if (fill && fill !== 'auto') {
@@ -347,7 +387,7 @@ const parseWord = async (buffer, config) => {
347
387
  // Extract paragraph-level run properties
348
388
  let paragraphRunFormatting = { ...styleProps.formatting };
349
389
  if (pPr) {
350
- const pPrRPr = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:rPr")[0];
390
+ const pPrRPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:rPr");
351
391
  if (pPrRPr) {
352
392
  const pPrFormatting = extractFormattingFromXml(pPrRPr);
353
393
  for (const key in pPrFormatting) {
@@ -366,9 +406,9 @@ const parseWord = async (buffer, config) => {
366
406
  const children = [];
367
407
  // Traverse children of paragraph (runs, hyperlinks, etc.)
368
408
  const processChildNode = (node) => {
369
- if (node.nodeName === 'w:r') {
409
+ if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
370
410
  const runNode = node;
371
- const rPr = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:rPr")[0];
411
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:rPr");
372
412
  // Formatting
373
413
  let formatting = {};
374
414
  // Apply paragraph-level formatting
@@ -376,7 +416,7 @@ const parseWord = async (buffer, config) => {
376
416
  formatting[key] = paragraphRunFormatting[key];
377
417
  }
378
418
  // Check for run style
379
- const rStyle = rPr ? (0, xmlUtils_1.getElementsByTagName)(rPr, "w:rStyle")[0] : null;
419
+ const rStyle = rPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "w:rStyle") : null;
380
420
  const rStyleVal = rStyle ? rStyle.getAttribute("w:val") : pStyleVal;
381
421
  if (rStyleVal && styleMap[rStyleVal]) {
382
422
  for (const key in styleMap[rStyleVal].formatting) {
@@ -400,48 +440,93 @@ const parseWord = async (buffer, config) => {
400
440
  if (!formatting.backgroundColor && paraBackgroundColor) {
401
441
  formatting.backgroundColor = paraBackgroundColor;
402
442
  }
403
- // Text content
404
- const tNodes = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:t");
405
- for (const tNode of tNodes) {
406
- const tContent = tNode.textContent || '';
407
- text += tContent;
408
- const textNode = {
409
- type: 'text',
410
- text: tContent,
411
- formatting: formatting
412
- };
413
- if (config.includeRawContent) {
414
- textNode.rawContent = xmlSerializer.serializeToString(tNode);
443
+ for (const child of runNode.childNodes) {
444
+ if (!(0, xmlUtils_js_1.isElement)(child))
445
+ continue;
446
+ // also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
447
+ // Text content
448
+ if (child.tagName === "w:t" || child.tagName === "t") {
449
+ const tNode = child;
450
+ const tContent = tNode.textContent || '';
451
+ text += tContent;
452
+ const textNode = {
453
+ type: 'text',
454
+ text: tContent,
455
+ formatting: formatting
456
+ };
457
+ if (config.includeRawContent) {
458
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
459
+ }
460
+ // Always set a style: run style > paragraph style > detected default
461
+ // Use detected default style for international compatibility
462
+ const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
463
+ if (nodeStyle) {
464
+ textNode.metadata = { style: nodeStyle };
465
+ }
466
+ children.push(textNode);
467
+ }
468
+ // Break nodes
469
+ else if (config.includeBreakNodes &&
470
+ (child.tagName === "w:br"
471
+ || child.tagName === "br"
472
+ || child.tagName === "w:cr"
473
+ || child.tagName === "cr")) {
474
+ const brNode = child;
475
+ let breakType = 'textWrapping';
476
+ if (child.tagName === "w:cr" || child.tagName === "cr") {
477
+ breakType = 'carriageReturn';
478
+ }
479
+ else {
480
+ const nodeBreakType = brNode.getAttribute("w:type") || brNode.getAttribute("type");
481
+ if (nodeBreakType !== null) {
482
+ breakType = nodeBreakType;
483
+ }
484
+ }
485
+ let breakClear = undefined;
486
+ if (breakType === 'textWrapping' && brNode.getAttribute("w:clear") !== null) {
487
+ breakClear = brNode.getAttribute("w:clear");
488
+ }
489
+ const breakNode = {
490
+ type: 'break',
491
+ metadata: { breakType, clear: breakClear }
492
+ };
493
+ if (config.includeRawContent) {
494
+ breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(brNode, documentContent, config);
495
+ }
496
+ children.push(breakNode);
415
497
  }
416
- // Always set a style: run style > paragraph style > detected default
417
- // Use detected default style for international compatibility
418
- const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
419
- if (nodeStyle) {
420
- textNode.metadata = { style: nodeStyle };
498
+ else if (config.includeBreakNodes && (child.tagName === "w:lastRenderedPageBreak" || child.tagName === "lastRenderedPageBreak")) {
499
+ const breakNode = {
500
+ type: 'break',
501
+ metadata: { breakType: 'lastRenderedPage' }
502
+ };
503
+ if (config.includeRawContent) {
504
+ breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(child, documentContent, config);
505
+ }
506
+ children.push(breakNode);
421
507
  }
422
- children.push(textNode);
423
508
  }
424
509
  // Images/Drawings
425
510
  if (config.extractAttachments) {
426
- const drawings = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:drawing");
427
- const picts = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:pict");
511
+ const drawings = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:drawing");
512
+ const picts = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:pict");
428
513
  const allImages = [...drawings, ...picts];
429
514
  for (const imgNode of allImages) {
430
- const imgXml = xmlSerializer.serializeToString(imgNode);
515
+ const imgXml = (0, xmlUtils_js_1.serializeXml)(imgNode);
431
516
  // Extract Alt Text
432
517
  let altText = '';
433
- const docPr = (0, xmlUtils_1.getElementsByTagName)(imgNode, "wp:docPr")[0];
518
+ const docPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "wp:docPr");
434
519
  if (docPr) {
435
520
  altText = docPr.getAttribute("descr") || docPr.getAttribute("title") || '';
436
521
  }
437
522
  // Extract Relationship ID
438
523
  let rId = '';
439
- const blip = (0, xmlUtils_1.getElementsByTagName)(imgNode, "a:blip")[0];
524
+ const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "a:blip");
440
525
  if (blip) {
441
526
  rId = blip.getAttribute("r:embed") || '';
442
527
  }
443
528
  else {
444
- const imagedata = (0, xmlUtils_1.getElementsByTagName)(imgNode, "v:imagedata")[0];
529
+ const imagedata = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "v:imagedata");
445
530
  if (imagedata) {
446
531
  rId = imagedata.getAttribute("r:id") || '';
447
532
  }
@@ -456,7 +541,7 @@ const parseWord = async (buffer, config) => {
456
541
  metadata: { attachmentName: filename, altText: altText }
457
542
  };
458
543
  if (config.includeRawContent) {
459
- imageNode.rawContent = imgXml;
544
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
460
545
  }
461
546
  children.push(imageNode);
462
547
  }
@@ -467,7 +552,7 @@ const parseWord = async (buffer, config) => {
467
552
  text: '',
468
553
  };
469
554
  if (config.includeRawContent) {
470
- imageNode.rawContent = imgXml;
555
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
471
556
  }
472
557
  children.push(imageNode);
473
558
  }
@@ -475,7 +560,7 @@ const parseWord = async (buffer, config) => {
475
560
  }
476
561
  // Footnotes/Endnotes inside runs
477
562
  if (!config.ignoreNotes) {
478
- const footnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:footnoteReference")[0];
563
+ const footnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:footnoteReference");
479
564
  if (footnoteRef) {
480
565
  const id = footnoteRef.getAttribute("w:id");
481
566
  if (id && footnoteMap.has(id)) {
@@ -494,7 +579,7 @@ const parseWord = async (buffer, config) => {
494
579
  }
495
580
  }
496
581
  }
497
- const endnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:endnoteReference")[0];
582
+ const endnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:endnoteReference");
498
583
  if (endnoteRef) {
499
584
  const id = endnoteRef.getAttribute("w:id");
500
585
  if (id && endnoteMap.has(id)) {
@@ -515,7 +600,7 @@ const parseWord = async (buffer, config) => {
515
600
  }
516
601
  }
517
602
  }
518
- else if (node.nodeName === 'w:hyperlink') {
603
+ else if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:hyperlink') {
519
604
  const hlNode = node;
520
605
  const rId = hlNode.getAttribute("r:id");
521
606
  const anchor = hlNode.getAttribute("w:anchor");
@@ -548,10 +633,10 @@ const parseWord = async (buffer, config) => {
548
633
  processChildNode(child);
549
634
  }
550
635
  if (isList) {
551
- const numIdNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:numId")[0];
552
- const ilvlNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:ilvl")[0];
636
+ const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
637
+ const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
553
638
  const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
554
- const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
639
+ const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0', 10) : 0;
555
640
  let listType = 'ordered';
556
641
  let itemIndex = 0;
557
642
  if (numId && numberingMap[numId]) {
@@ -585,6 +670,7 @@ const parseWord = async (buffer, config) => {
585
670
  metadata: {
586
671
  listType,
587
672
  indentation: ilvl,
673
+ paragraphIndentation: paraIndentation,
588
674
  alignment: (alignment || 'left'),
589
675
  listId: numId,
590
676
  itemIndex: itemIndex,
@@ -592,19 +678,19 @@ const parseWord = async (buffer, config) => {
592
678
  }
593
679
  };
594
680
  if (config.includeRawContent)
595
- listNode.rawContent = pXml;
681
+ listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
596
682
  return listNode;
597
683
  }
598
684
  else if (isHeading) {
599
- const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", "")) || 1 : 1;
685
+ const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", ""), 10) || 1 : 1;
600
686
  const headingNode = {
601
687
  type: 'heading',
602
688
  text: text,
603
689
  children: children,
604
- metadata: { level, alignment, style: pStyleVal ?? undefined }
690
+ metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
605
691
  };
606
692
  if (config.includeRawContent)
607
- headingNode.rawContent = pXml;
693
+ headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
608
694
  return headingNode;
609
695
  }
610
696
  else {
@@ -612,23 +698,23 @@ const parseWord = async (buffer, config) => {
612
698
  type: 'paragraph',
613
699
  text: text,
614
700
  children: children,
615
- metadata: { alignment, style: pStyleVal ?? undefined }
701
+ metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
616
702
  };
617
703
  if (config.includeRawContent)
618
- paraNode.rawContent = pXml;
704
+ paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
619
705
  return paraNode;
620
706
  }
621
707
  };
622
708
  // Helper to parse a table node
623
- const parseTable = (tblNode) => {
709
+ const parseTable = (tblNode, documentContent) => {
624
710
  const rows = [];
625
711
  // Only get direct child rows, not nested table rows
626
- const trNodes = (0, xmlUtils_1.getDirectChildren)(tblNode, "w:tr");
712
+ const trNodes = (0, xmlUtils_js_1.getDirectChildren)(tblNode, "w:tr");
627
713
  for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
628
714
  const trNode = trNodes[rIndex];
629
715
  const cells = [];
630
716
  // Only get direct child cells, not nested table cells
631
- const tcNodes = (0, xmlUtils_1.getDirectChildren)(trNode, "w:tc");
717
+ const tcNodes = (0, xmlUtils_js_1.getDirectChildren)(trNode, "w:tc");
632
718
  for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
633
719
  const tcNode = tcNodes[cIndex];
634
720
  const cellChildren = [];
@@ -636,14 +722,14 @@ const parseWord = async (buffer, config) => {
636
722
  // Cells contain paragraphs (and other block-level elements)
637
723
  const cellContentNodes = Array.from(tcNode.childNodes);
638
724
  for (const child of cellContentNodes) {
639
- if (child.nodeName === 'w:p') {
640
- const pNode = parseParagraph(child);
725
+ if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
726
+ const pNode = parseParagraph(child, documentContent);
641
727
  cellChildren.push(pNode);
642
728
  cellText += pNode.text;
643
729
  }
644
- else if (child.nodeName === 'w:tbl') {
730
+ else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
645
731
  // Nested table
646
- const nestedTable = parseTable(child);
732
+ const nestedTable = parseTable(child, documentContent);
647
733
  cellChildren.push(nestedTable);
648
734
  // Don't add nested table text to cell text - it will be handled recursively
649
735
  }
@@ -671,26 +757,28 @@ const parseWord = async (buffer, config) => {
671
757
  if (!config.ignoreNotes) {
672
758
  const footnotesFile = files.find(f => f.path.match(footnotesFileRegex));
673
759
  if (footnotesFile) {
674
- const footnotesDoc = (0, xmlUtils_1.parseXmlString)(footnotesFile.content.toString());
675
- const footnoteNodes = (0, xmlUtils_1.getElementsByTagName)(footnotesDoc, "w:footnote");
760
+ const footnotesDoc = (0, xmlUtils_js_1.parseXmlString)(footnotesFile.content.toString());
761
+ const footnoteXml = footnotesFile.content.toString();
762
+ const footnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(footnotesDoc, "w:footnote");
676
763
  for (const node of footnoteNodes) {
677
764
  const id = node.getAttribute("w:id");
678
765
  if (!id || id === "-1" || id === "0")
679
766
  continue;
680
- const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
681
- footnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
767
+ const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
768
+ footnoteMap.set(id, pNodes.map(p => parseParagraph(p, footnoteXml)));
682
769
  }
683
770
  }
684
771
  const endnotesFile = files.find(f => f.path.match(endnotesFileRegex));
685
772
  if (endnotesFile) {
686
- const endnotesDoc = (0, xmlUtils_1.parseXmlString)(endnotesFile.content.toString());
687
- const endnoteNodes = (0, xmlUtils_1.getElementsByTagName)(endnotesDoc, "w:endnote");
773
+ const endnotesDoc = (0, xmlUtils_js_1.parseXmlString)(endnotesFile.content.toString());
774
+ const endnoteXml = endnotesFile.content.toString();
775
+ const endnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(endnotesDoc, "w:endnote");
688
776
  for (const node of endnoteNodes) {
689
777
  const id = node.getAttribute("w:id");
690
778
  if (!id || id === "-1" || id === "0")
691
779
  continue;
692
- const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
693
- endnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
780
+ const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
781
+ endnoteMap.set(id, pNodes.map(p => parseParagraph(p, endnoteXml)));
694
782
  }
695
783
  }
696
784
  }
@@ -711,16 +799,16 @@ const parseWord = async (buffer, config) => {
711
799
  if (config.includeRawContent) {
712
800
  rawContents.push(documentContent);
713
801
  }
714
- const doc = (0, xmlUtils_1.parseXmlString)(documentContent);
715
- const body = (0, xmlUtils_1.getElementsByTagName)(doc, "w:body")[0];
802
+ const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
803
+ const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
716
804
  if (body) {
717
805
  const bodyChildren = Array.from(body.childNodes);
718
806
  for (const child of bodyChildren) {
719
- if (child.nodeName === 'w:p') {
720
- content.push(parseParagraph(child));
807
+ if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
808
+ content.push(parseParagraph(child, documentContent));
721
809
  }
722
- else if (child.nodeName === 'w:tbl') {
723
- content.push(parseTable(child));
810
+ else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
811
+ content.push(parseTable(child, documentContent));
724
812
  }
725
813
  }
726
814
  }
@@ -728,15 +816,15 @@ const parseWord = async (buffer, config) => {
728
816
  // Extract attachments
729
817
  if (config.extractAttachments) {
730
818
  for (const media of mediaFiles) {
731
- const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
819
+ const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
732
820
  attachments.push(attachment);
733
821
  if (config.ocr) {
734
822
  if (attachment.mimeType.startsWith('image/')) {
735
823
  try {
736
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
824
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
737
825
  }
738
826
  catch (e) {
739
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
827
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
740
828
  }
741
829
  }
742
830
  }
@@ -776,6 +864,9 @@ const parseWord = async (buffer, config) => {
776
864
  if (node.children) {
777
865
  t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
778
866
  }
867
+ else if (node.type === 'break') {
868
+ t += config.newlineDelimiter ?? '\n';
869
+ }
779
870
  else
780
871
  t += node.text || '';
781
872
  return t;