officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -40,6 +40,7 @@
40
40
  * - `<w:p>` - Paragraph
41
41
  * - `<w:r>` - Run (contiguous text with same formatting)
42
42
  * - `<w:t>` - Text content
43
+ * - `<w:br>` - Line or page break
43
44
  * - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
44
45
  * - `<w:pStyle>` - Paragraph style (for headings)
45
46
  * - `<w:numPr>` - List numbering properties
@@ -61,6 +62,8 @@
61
62
  */
62
63
  Object.defineProperty(exports, "__esModule", { value: true });
63
64
  exports.parseWord = void 0;
65
+ const types_js_1 = require("../types.js");
66
+ const astUtils_js_1 = require("../utils/astUtils.js");
64
67
  const errorUtils_js_1 = require("../utils/errorUtils.js");
65
68
  const imageUtils_js_1 = require("../utils/imageUtils.js");
66
69
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
@@ -95,83 +98,134 @@ const parseWord = async (buffer, config) => {
95
98
  // Helper to extract formatting from run properties XML string
96
99
  const extractFormattingFromXml = (rPr) => {
97
100
  const formatting = {};
98
- const rPrString = (0, xmlUtils_js_1.serializeXml)(rPr);
99
- // Helper to check boolean properties
100
- const getBoolVal = (xmlSnippet, tagName) => {
101
- const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
102
- const match = xmlSnippet.match(regex);
103
- if (match) {
104
- const val = match[1];
105
- if (val === undefined)
101
+ // Helper to check boolean properties (e.g., <w:b />, <w:i w:val="0" />)
102
+ const getBoolVal = (parent, tagName) => {
103
+ const el = (0, xmlUtils_js_1.getFirstElementByTagName)(parent, tagName);
104
+ if (el) {
105
+ const val = el.getAttribute('w:val');
106
+ // In OOXML, if the element is present without w:val, it's true.
107
+ // If w:val is present, it can be '1', 'true', 'on' for true.
108
+ if (val === null)
106
109
  return true;
107
110
  return val === '1' || val === 'true' || val === 'on';
108
111
  }
109
112
  return null;
110
113
  };
111
- const bold = getBoolVal(rPrString, 'w:b');
114
+ const bold = getBoolVal(rPr, 'w:b');
112
115
  if (bold !== null)
113
116
  formatting.bold = bold;
114
- const italic = getBoolVal(rPrString, 'w:i');
117
+ const italic = getBoolVal(rPr, 'w:i');
115
118
  if (italic !== null)
116
119
  formatting.italic = italic;
117
- const underlineMatch = rPrString.match(/<w:u(?: w:val="([^"]+)")?\/?>/);
118
- if (underlineMatch) {
119
- const val = underlineMatch[1];
120
+ const u = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:u');
121
+ if (u) {
122
+ const val = u.getAttribute('w:val');
120
123
  // If val is missing, it's a default underline (true).
121
124
  // If val is present, it's true unless explicit 'none'.
122
125
  if (!val || val !== 'none') {
123
126
  formatting.underline = true;
124
127
  }
125
128
  }
126
- const strike = getBoolVal(rPrString, 'w:strike');
127
- const dstrike = getBoolVal(rPrString, 'w:dstrike');
129
+ const strike = getBoolVal(rPr, 'w:strike');
130
+ const dstrike = getBoolVal(rPr, 'w:dstrike');
128
131
  if (strike !== null)
129
132
  formatting.strikethrough = strike;
130
133
  else if (dstrike !== null)
131
134
  formatting.strikethrough = dstrike;
132
- // Font size
133
- const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
134
- if (szMatch)
135
- formatting.size = (parseInt(szMatch[1]) / 2).toString() + 'pt';
136
- // Color
137
- const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
138
- if (colorMatch && colorMatch[1] !== 'auto')
139
- formatting.color = '#' + colorMatch[1];
140
- // Background color (shading)
141
- const shdMatch = rPrString.match(/<w:shd[^>]*w:fill="([^"]+)"/);
142
- if (shdMatch && shdMatch[1] !== 'auto')
143
- formatting.backgroundColor = '#' + shdMatch[1];
144
- // Highlight (map to backgroundColor)
145
- const highlightMatch = rPrString.match(/<w:highlight w:val="([^"]+)"/);
146
- if (highlightMatch && highlightMatch[1] !== 'none') {
147
- const colorMap = {
148
- 'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
149
- 'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
150
- 'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
151
- 'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
152
- };
153
- formatting.backgroundColor = colorMap[highlightMatch[1]] || highlightMatch[1];
135
+ // Font size (w:sz) - stored in half-points
136
+ const sz = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:sz');
137
+ if (sz) {
138
+ const val = sz.getAttribute('w:val');
139
+ if (val) {
140
+ formatting.size = (parseInt(val, 10) / 2).toString() + 'pt';
141
+ }
154
142
  }
155
- // Font family
156
- const rFontsMatch = rPrString.match(/<w:rFonts[^>]*w:ascii="([^"]+)"/);
157
- if (rFontsMatch) {
158
- formatting.font = rFontsMatch[1];
143
+ // Color (w:color)
144
+ const color = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:color');
145
+ if (color) {
146
+ const val = color.getAttribute('w:val');
147
+ if (val && val !== 'auto') {
148
+ formatting.color = '#' + val;
149
+ }
159
150
  }
160
- else {
161
- const hAnsiMatch = rPrString.match(/<w:rFonts[^>]*w:hAnsi="([^"]+)"/);
162
- if (hAnsiMatch)
163
- formatting.font = hAnsiMatch[1];
151
+ // Background color (w:shd) - shading
152
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:shd');
153
+ if (shd) {
154
+ const val = shd.getAttribute('w:fill');
155
+ if (val && val !== 'auto') {
156
+ formatting.backgroundColor = '#' + val;
157
+ }
164
158
  }
165
- // Subscript/Superscript
166
- const vertAlignMatch = rPrString.match(/<w:vertAlign w:val="([^"]+)"/);
167
- if (vertAlignMatch) {
168
- if (vertAlignMatch[1] === 'subscript')
159
+ // Highlight (w:highlight) - maps to background color in our AST
160
+ const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:highlight');
161
+ if (highlight) {
162
+ const val = highlight.getAttribute('w:val');
163
+ if (val && val !== 'none') {
164
+ const colorMap = {
165
+ 'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
166
+ 'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
167
+ 'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
168
+ 'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
169
+ };
170
+ formatting.backgroundColor = colorMap[val] || val;
171
+ }
172
+ }
173
+ // Font family (w:rFonts)
174
+ const rFonts = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:rFonts');
175
+ if (rFonts) {
176
+ // Priority: ascii (Western) > hAnsi (High ANSI)
177
+ const font = rFonts.getAttribute('w:ascii') || rFonts.getAttribute('w:hAnsi');
178
+ if (font) {
179
+ formatting.font = font;
180
+ }
181
+ }
182
+ // Subscript/Superscript (w:vertAlign)
183
+ const vertAlign = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:vertAlign');
184
+ if (vertAlign) {
185
+ const val = vertAlign.getAttribute('w:val');
186
+ if (val === 'subscript')
169
187
  formatting.subscript = true;
170
- if (vertAlignMatch[1] === 'superscript')
188
+ else if (val === 'superscript')
171
189
  formatting.superscript = true;
172
190
  }
173
191
  return formatting;
174
192
  };
193
+ // Helper to extract indentation from paragraph properties XML string
194
+ const extractIndentationFromXml = (pPr) => {
195
+ const ind = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:ind");
196
+ if (ind) {
197
+ const indentation = {};
198
+ const left = ind.getAttribute("w:left") || ind.getAttribute("w:start");
199
+ const right = ind.getAttribute("w:right") || ind.getAttribute("w:end");
200
+ const firstLine = ind.getAttribute("w:firstLine");
201
+ const hanging = ind.getAttribute("w:hanging");
202
+ if (left)
203
+ indentation.left = parseInt(left, 10);
204
+ if (right)
205
+ indentation.right = parseInt(right, 10);
206
+ if (firstLine)
207
+ indentation.firstLine = parseInt(firstLine, 10);
208
+ if (hanging)
209
+ indentation.hanging = parseInt(hanging, 10);
210
+ return Object.keys(indentation).length > 0 ? indentation : undefined;
211
+ }
212
+ return undefined;
213
+ };
214
+ /**
215
+ * Resolves mc:AlternateContent by preferring mc:Fallback if choice namespace is not recognized,
216
+ * or simply the first available valid child.
217
+ */
218
+ const resolveAlternateContent = (element) => {
219
+ const choice = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "mc:Choice");
220
+ // In most cases, mc:Choice contains the modern version, but mc:Fallback is safer for legacy compatibility
221
+ // Mammoth often skips Choice if it's not handled. We'll try Choice first.
222
+ if (choice)
223
+ return Array.from(choice.childNodes);
224
+ const fallback = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "mc:Fallback");
225
+ if (fallback)
226
+ return Array.from(fallback.childNodes);
227
+ return Array.from(element.childNodes);
228
+ };
175
229
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
176
230
  !!x.match(footnotesFileRegex) ||
177
231
  !!x.match(endnotesFileRegex) ||
@@ -228,18 +282,32 @@ const parseWord = async (buffer, config) => {
228
282
  const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
229
283
  if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
230
284
  numberingMap[numId] = {};
285
+ // Inherit from abstractNum
231
286
  const lvls = (0, xmlUtils_js_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
232
287
  for (const lvl of lvls) {
233
288
  const ilvl = lvl.getAttribute("w:ilvl");
234
289
  const numFmtNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:numFmt");
235
290
  const lvlTextNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:lvlText");
291
+ const startNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:start");
236
292
  if (ilvl) {
237
293
  numberingMap[numId][ilvl] = {
238
294
  numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
239
- lvlText: lvlTextNode?.getAttribute("w:val") || ''
295
+ lvlText: lvlTextNode?.getAttribute("w:val") || '',
296
+ start: parseInt(startNode?.getAttribute("w:val") || '1', 10)
240
297
  };
241
298
  }
242
299
  }
300
+ // Apply instance overrides (w:lvlOverride)
301
+ const overrides = (0, xmlUtils_js_1.getElementsByTagName)(num, "w:lvlOverride");
302
+ for (const override of overrides) {
303
+ const ilvl = override.getAttribute("w:ilvl");
304
+ if (ilvl && numberingMap[numId][ilvl]) {
305
+ const startOverride = (0, xmlUtils_js_1.getFirstElementByTagName)(override, "w:startOverride");
306
+ if (startOverride) {
307
+ numberingMap[numId][ilvl].start = parseInt(startOverride.getAttribute("w:val") || '1', 10);
308
+ }
309
+ }
310
+ }
243
311
  }
244
312
  }
245
313
  }
@@ -257,6 +325,7 @@ const parseWord = async (buffer, config) => {
257
325
  const formatting = rPr ? extractFormattingFromXml(rPr) : {};
258
326
  let alignment = undefined;
259
327
  let backgroundColor = undefined;
328
+ let paragraphIndentation = undefined;
260
329
  if (pPr) {
261
330
  const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
262
331
  if (jc) {
@@ -271,8 +340,11 @@ const parseWord = async (buffer, config) => {
271
340
  if (fill && fill !== 'auto')
272
341
  backgroundColor = '#' + fill;
273
342
  }
343
+ const ind = extractIndentationFromXml(pPr);
344
+ if (ind)
345
+ paragraphIndentation = ind;
274
346
  }
275
- styleMap[styleId] = { formatting, alignment, backgroundColor };
347
+ styleMap[styleId] = { formatting, alignment, backgroundColor, paragraphIndentation };
276
348
  }
277
349
  }
278
350
  }
@@ -316,8 +388,7 @@ const parseWord = async (buffer, config) => {
316
388
  const numberingState = {};
317
389
  const listCounters = {}; // Track item index per listId/level
318
390
  // Helper to parse a paragraph node
319
- const parseParagraph = (pNode, documentContent) => {
320
- const pXml = pNode.toString();
391
+ const parseParagraph = (pNode, documentContent, pendingAnchorIds = []) => {
321
392
  // Check if it's a list item
322
393
  const numPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:numPr");
323
394
  const isList = !!numPr;
@@ -339,6 +410,14 @@ const parseWord = async (buffer, config) => {
339
410
  }
340
411
  }
341
412
  }
413
+ // Extract Indentation
414
+ let paraIndentation = styleProps.paragraphIndentation;
415
+ if (pPr) {
416
+ const ind = extractIndentationFromXml(pPr);
417
+ if (ind) {
418
+ paraIndentation = { ...paraIndentation, ...ind };
419
+ }
420
+ }
342
421
  // Extract Paragraph Background
343
422
  let paraBackgroundColor = styleProps.backgroundColor;
344
423
  if (pPr) {
@@ -406,26 +485,71 @@ const parseWord = async (buffer, config) => {
406
485
  if (!formatting.backgroundColor && paraBackgroundColor) {
407
486
  formatting.backgroundColor = paraBackgroundColor;
408
487
  }
409
- // Text content
410
- const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:t");
411
- for (const tNode of tNodes) {
412
- const tContent = tNode.textContent || '';
413
- text += tContent;
414
- const textNode = {
415
- type: 'text',
416
- text: tContent,
417
- formatting: formatting
418
- };
419
- if (config.includeRawContent) {
420
- textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
488
+ for (const child of runNode.childNodes) {
489
+ if (!(0, xmlUtils_js_1.isElement)(child))
490
+ continue;
491
+ // also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
492
+ // Text content
493
+ if (child.tagName === "w:t" || child.tagName === "t") {
494
+ const tNode = child;
495
+ const tContent = tNode.textContent || '';
496
+ text += tContent;
497
+ const textNode = {
498
+ type: 'text',
499
+ text: tContent,
500
+ formatting: formatting
501
+ };
502
+ if (config.includeRawContent) {
503
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
504
+ }
505
+ // Always set a style: run style > paragraph style > detected default
506
+ // Use detected default style for international compatibility
507
+ const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
508
+ if (nodeStyle) {
509
+ textNode.metadata = { style: nodeStyle };
510
+ }
511
+ children.push(textNode);
512
+ }
513
+ // Break nodes
514
+ else if (config.includeBreakNodes &&
515
+ (child.tagName === "w:br"
516
+ || child.tagName === "br"
517
+ || child.tagName === "w:cr"
518
+ || child.tagName === "cr")) {
519
+ const brNode = child;
520
+ let breakType = 'textWrapping';
521
+ if (child.tagName === "w:cr" || child.tagName === "cr") {
522
+ breakType = 'carriageReturn';
523
+ }
524
+ else {
525
+ const nodeBreakType = brNode.getAttribute("w:type") || brNode.getAttribute("type");
526
+ if (nodeBreakType !== null) {
527
+ breakType = nodeBreakType;
528
+ }
529
+ }
530
+ let breakClear = undefined;
531
+ if (breakType === 'textWrapping' && brNode.getAttribute("w:clear") !== null) {
532
+ breakClear = brNode.getAttribute("w:clear");
533
+ }
534
+ const breakNode = {
535
+ type: 'break',
536
+ metadata: { breakType, clear: breakClear }
537
+ };
538
+ if (config.includeRawContent) {
539
+ breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(brNode, documentContent, config);
540
+ }
541
+ children.push(breakNode);
421
542
  }
422
- // Always set a style: run style > paragraph style > detected default
423
- // Use detected default style for international compatibility
424
- const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
425
- if (nodeStyle) {
426
- textNode.metadata = { style: nodeStyle };
543
+ else if (config.includeBreakNodes && (child.tagName === "w:lastRenderedPageBreak" || child.tagName === "lastRenderedPageBreak")) {
544
+ const breakNode = {
545
+ type: 'break',
546
+ metadata: { breakType: 'lastRenderedPage' }
547
+ };
548
+ if (config.includeRawContent) {
549
+ breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(child, documentContent, config);
550
+ }
551
+ children.push(breakNode);
427
552
  }
428
- children.push(textNode);
429
553
  }
430
554
  // Images/Drawings
431
555
  if (config.extractAttachments) {
@@ -526,7 +650,7 @@ const parseWord = async (buffer, config) => {
526
650
  const rId = hlNode.getAttribute("r:id");
527
651
  const anchor = hlNode.getAttribute("w:anchor");
528
652
  let linkMetadata;
529
- if (anchor) {
653
+ if (anchor && !config.ignoreInternalLinks) {
530
654
  linkMetadata = { link: '#' + anchor, linkType: 'internal' };
531
655
  }
532
656
  else if (rId && relsMap[rId]) {
@@ -548,16 +672,48 @@ const parseWord = async (buffer, config) => {
548
672
  }
549
673
  }
550
674
  }
675
+ else if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:bookmarkStart') {
676
+ const bookmarkName = node.getAttribute("w:name");
677
+ if (bookmarkName && !bookmarkName.startsWith('_GoBack') && !config.ignoreInternalLinks) {
678
+ anchorIds.push(bookmarkName);
679
+ }
680
+ }
681
+ else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'mc:AlternateContent' || node.nodeName === 'AlternateContent')) {
682
+ const resolved = resolveAlternateContent(node);
683
+ for (const rNode of resolved)
684
+ processChildNode(rNode);
685
+ }
686
+ else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'w:pict' || node.nodeName === 'pict' || node.nodeName === 'w:drawing' || node.nodeName === 'drawing')) {
687
+ // Extract text boxes from legacy shapes or modern drawings
688
+ const textBoxes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:txbxContent");
689
+ for (const txbx of textBoxes) {
690
+ const txbxChildren = Array.from(txbx.childNodes);
691
+ for (const txbxChild of txbxChildren) {
692
+ if ((0, xmlUtils_js_1.isElement)(txbxChild) && txbxChild.nodeName === 'w:p') {
693
+ const nestedP = parseParagraph(txbxChild, documentContent);
694
+ children.push(...(nestedP.children || []));
695
+ text += nestedP.text;
696
+ }
697
+ }
698
+ }
699
+ }
700
+ else if (node.childNodes.length > 0) {
701
+ // Generic fallback for unknown elements that might contain content
702
+ for (const child of Array.from(node.childNodes))
703
+ processChildNode(child);
704
+ }
551
705
  };
706
+ const anchorIds = [...pendingAnchorIds];
552
707
  const childNodes = Array.from(pNode.childNodes);
553
708
  for (const child of childNodes) {
554
709
  processChildNode(child);
555
710
  }
711
+ const commonMetadata = anchorIds.length > 0 ? { anchorIds } : {};
556
712
  if (isList) {
557
713
  const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
558
714
  const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
559
715
  const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
560
- const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
716
+ const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0', 10) : 0;
561
717
  let listType = 'ordered';
562
718
  let itemIndex = 0;
563
719
  if (numId && numberingMap[numId]) {
@@ -573,11 +729,11 @@ const parseWord = async (buffer, config) => {
573
729
  }
574
730
  const numFmt = numberingMap[numId][ilvlStr]?.numFmt || 'decimal';
575
731
  listType = numFmt === 'bullet' ? 'unordered' : 'ordered';
576
- // Track itemIndex (starts at 0, continues across interruptions for same listId)
732
+ // Track itemIndex (starts at override or default, continues across interruptions for same listId)
577
733
  if (!listCounters[numId])
578
734
  listCounters[numId] = {};
579
735
  if (listCounters[numId][ilvlStr] === undefined) {
580
- listCounters[numId][ilvlStr] = 0;
736
+ listCounters[numId][ilvlStr] = (numberingMap[numId][ilvlStr]?.start ?? 1) - 1;
581
737
  }
582
738
  else {
583
739
  listCounters[numId][ilvlStr]++;
@@ -591,10 +747,12 @@ const parseWord = async (buffer, config) => {
591
747
  metadata: {
592
748
  listType,
593
749
  indentation: ilvl,
750
+ paragraphIndentation: paraIndentation,
594
751
  alignment: (alignment || 'left'),
595
752
  listId: numId,
596
753
  itemIndex: itemIndex,
597
- style: pStyleVal
754
+ style: pStyleVal,
755
+ ...commonMetadata
598
756
  }
599
757
  };
600
758
  if (config.includeRawContent)
@@ -602,12 +760,12 @@ const parseWord = async (buffer, config) => {
602
760
  return listNode;
603
761
  }
604
762
  else if (isHeading) {
605
- const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", "")) || 1 : 1;
763
+ const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", ""), 10) || 1 : 1;
606
764
  const headingNode = {
607
765
  type: 'heading',
608
766
  text: text,
609
767
  children: children,
610
- metadata: { level, alignment, style: pStyleVal ?? undefined }
768
+ metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
611
769
  };
612
770
  if (config.includeRawContent)
613
771
  headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
@@ -618,7 +776,7 @@ const parseWord = async (buffer, config) => {
618
776
  type: 'paragraph',
619
777
  text: text,
620
778
  children: children,
621
- metadata: { alignment, style: pStyleVal ?? undefined }
779
+ metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
622
780
  };
623
781
  if (config.includeRawContent)
624
782
  paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
@@ -626,17 +784,41 @@ const parseWord = async (buffer, config) => {
626
784
  }
627
785
  };
628
786
  // Helper to parse a table node
629
- const parseTable = (tblNode, documentContent) => {
787
+ const parseTable = (tblNode, documentContent, pendingAnchorIds = []) => {
630
788
  const rows = [];
631
- // Only get direct child rows, not nested table rows
632
789
  const trNodes = (0, xmlUtils_js_1.getDirectChildren)(tblNode, "w:tr");
790
+ // Track vertical merges: colIndex -> { startCellNode, rowSpan }
791
+ const vMergeMap = new Map();
633
792
  for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
634
793
  const trNode = trNodes[rIndex];
635
794
  const cells = [];
636
795
  // Only get direct child cells, not nested table cells
637
796
  const tcNodes = (0, xmlUtils_js_1.getDirectChildren)(trNode, "w:tc");
638
- for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
639
- const tcNode = tcNodes[cIndex];
797
+ let visualCol = 0;
798
+ for (let tcIndex = 0; tcIndex < tcNodes.length; tcIndex++) {
799
+ const tcNode = tcNodes[tcIndex];
800
+ const tcPr = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "w:tcPr");
801
+ // Horizontal merge (colspan)
802
+ let colSpan = 1;
803
+ if (tcPr) {
804
+ const gridSpan = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:gridSpan");
805
+ if (gridSpan) {
806
+ colSpan = parseInt(gridSpan.getAttribute("w:val") || "1", 10);
807
+ }
808
+ }
809
+ let vMergeRestart = false;
810
+ let isVMerge = false;
811
+ if (tcPr) {
812
+ const vMerge = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:vMerge");
813
+ if (vMerge) {
814
+ isVMerge = true;
815
+ const val = vMerge.getAttribute("w:val");
816
+ // If it's explicit restart, or if we don't have an active merge for this column, treat as restart
817
+ if (val === "restart" || !vMergeMap.has(visualCol)) {
818
+ vMergeRestart = true;
819
+ }
820
+ }
821
+ }
640
822
  const cellChildren = [];
641
823
  let cellText = '';
642
824
  // Cells contain paragraphs (and other block-level elements)
@@ -648,23 +830,50 @@ const parseWord = async (buffer, config) => {
648
830
  cellText += pNode.text;
649
831
  }
650
832
  else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
651
- // Nested table
652
833
  const nestedTable = parseTable(child, documentContent);
653
834
  cellChildren.push(nestedTable);
654
- // Don't add nested table text to cell text - it will be handled recursively
655
835
  }
656
836
  }
657
837
  const cellNode = {
658
838
  type: 'cell',
659
839
  text: cellText,
660
840
  children: cellChildren,
661
- metadata: { row: rIndex, col: cIndex }
841
+ metadata: { row: rIndex, col: visualCol }
662
842
  };
663
- cells.push(cellNode);
843
+ if (colSpan > 1)
844
+ cellNode.metadata.colSpan = colSpan;
845
+ if (isVMerge) {
846
+ if (vMergeRestart) {
847
+ vMergeMap.set(visualCol, { node: cellNode, span: 1 });
848
+ cells.push(cellNode);
849
+ }
850
+ else {
851
+ const mergeInfo = vMergeMap.get(visualCol);
852
+ if (mergeInfo) {
853
+ mergeInfo.span++;
854
+ mergeInfo.node.metadata.rowSpan = mergeInfo.span;
855
+ if (cellChildren.length > 0) {
856
+ if (!mergeInfo.node.children)
857
+ mergeInfo.node.children = [];
858
+ mergeInfo.node.children.push(...cellChildren);
859
+ mergeInfo.node.text += " " + cellText;
860
+ }
861
+ }
862
+ else {
863
+ // Fallback: if we found a continue but no restart, treat as normal cell
864
+ cells.push(cellNode);
865
+ }
866
+ }
867
+ }
868
+ else {
869
+ vMergeMap.delete(visualCol);
870
+ cells.push(cellNode);
871
+ }
872
+ visualCol += colSpan;
664
873
  }
665
874
  const rowNode = {
666
875
  type: 'row',
667
- children: cells
876
+ children: cells,
668
877
  };
669
878
  rows.push(rowNode);
670
879
  }
@@ -723,12 +932,23 @@ const parseWord = async (buffer, config) => {
723
932
  const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
724
933
  if (body) {
725
934
  const bodyChildren = Array.from(body.childNodes);
935
+ let pendingAnchorIds = [];
726
936
  for (const child of bodyChildren) {
727
- if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
728
- content.push(parseParagraph(child, documentContent));
729
- }
730
- else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
731
- content.push(parseTable(child, documentContent));
937
+ if ((0, xmlUtils_js_1.isElement)(child)) {
938
+ if (child.nodeName === 'w:p') {
939
+ content.push(parseParagraph(child, documentContent, pendingAnchorIds));
940
+ pendingAnchorIds = [];
941
+ }
942
+ else if (child.nodeName === 'w:tbl') {
943
+ content.push(parseTable(child, documentContent, pendingAnchorIds));
944
+ pendingAnchorIds = [];
945
+ }
946
+ else if (child.nodeName === 'w:bookmarkStart') {
947
+ const bookmarkName = child.getAttribute("w:name");
948
+ if (bookmarkName && !bookmarkName.startsWith('_GoBack') && !config.ignoreInternalLinks) {
949
+ pendingAnchorIds.push(bookmarkName);
950
+ }
951
+ }
732
952
  }
733
953
  }
734
954
  }
@@ -741,10 +961,10 @@ const parseWord = async (buffer, config) => {
741
961
  if (config.ocr) {
742
962
  if (attachment.mimeType.startsWith('image/')) {
743
963
  try {
744
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
964
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
745
965
  }
746
966
  catch (e) {
747
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
967
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
748
968
  }
749
969
  }
750
970
  }
@@ -772,24 +992,22 @@ const parseWord = async (buffer, config) => {
772
992
  if (config.putNotesAtLast && collectedNotes.length > 0) {
773
993
  content.push(...collectedNotes);
774
994
  }
775
- return {
776
- type: 'docx',
777
- metadata: { ...metadata, formatting: docDefaults, styleMap: styleMap },
778
- content: content,
779
- attachments: attachments,
780
- toText: () => content.map(c => {
781
- // Recursive text extraction
782
- const getText = (node) => {
783
- let t = '';
784
- if (node.children) {
785
- t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
786
- }
787
- else
788
- t += node.text || '';
789
- return t;
790
- };
791
- return getText(c);
792
- }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
793
- };
995
+ const toTextSync = () => content.map(c => {
996
+ // Recursive text extraction
997
+ const getText = (node) => {
998
+ let t = '';
999
+ if (node.children) {
1000
+ t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter);
1001
+ }
1002
+ else if (node.type === 'break') {
1003
+ t += config.newlineDelimiter;
1004
+ }
1005
+ else
1006
+ t += node.text || '';
1007
+ return t;
1008
+ };
1009
+ return getText(c);
1010
+ }).filter(t => t != '').join(config.newlineDelimiter);
1011
+ return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, toTextSync);
794
1012
  };
795
1013
  exports.parseWord = parseWord;