officeparser 6.1.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +219 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -29
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +826 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +781 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +27 -8
@@ -59,7 +59,7 @@
59
59
  * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
60
60
  * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
61
61
  */
62
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
62
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
63
63
  /**
64
64
  * Parses a Word document (.docx) and extracts content, formatting, and metadata.
65
65
  *
@@ -76,4 +76,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
76
76
  * @param config - Parser configuration options
77
77
  * @returns A promise resolving to the parsed AST
78
78
  */
79
- export declare const parseWord: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
79
+ export declare const parseWord: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
@@ -62,6 +62,8 @@
62
62
  */
63
63
  Object.defineProperty(exports, "__esModule", { value: true });
64
64
  exports.parseWord = void 0;
65
+ const types_js_1 = require("../types.js");
66
+ const astUtils_js_1 = require("../utils/astUtils.js");
65
67
  const errorUtils_js_1 = require("../utils/errorUtils.js");
66
68
  const imageUtils_js_1 = require("../utils/imageUtils.js");
67
69
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
@@ -96,79 +98,94 @@ const parseWord = async (buffer, config) => {
96
98
  // Helper to extract formatting from run properties XML string
97
99
  const extractFormattingFromXml = (rPr) => {
98
100
  const formatting = {};
99
- const rPrString = (0, xmlUtils_js_1.serializeXml)(rPr);
100
- // Helper to check boolean properties
101
- const getBoolVal = (xmlSnippet, tagName) => {
102
- const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
103
- const match = xmlSnippet.match(regex);
104
- if (match) {
105
- const val = match[1];
106
- if (val === undefined)
101
+ // Helper to check boolean properties (e.g., <w:b />, <w:i w:val="0" />)
102
+ const getBoolVal = (parent, tagName) => {
103
+ const el = (0, xmlUtils_js_1.getFirstElementByTagName)(parent, tagName);
104
+ if (el) {
105
+ const val = el.getAttribute('w:val');
106
+ // In OOXML, if the element is present without w:val, it's true.
107
+ // If w:val is present, it can be '1', 'true', 'on' for true.
108
+ if (val === null)
107
109
  return true;
108
110
  return val === '1' || val === 'true' || val === 'on';
109
111
  }
110
112
  return null;
111
113
  };
112
- const bold = getBoolVal(rPrString, 'w:b');
114
+ const bold = getBoolVal(rPr, 'w:b');
113
115
  if (bold !== null)
114
116
  formatting.bold = bold;
115
- const italic = getBoolVal(rPrString, 'w:i');
117
+ const italic = getBoolVal(rPr, 'w:i');
116
118
  if (italic !== null)
117
119
  formatting.italic = italic;
118
- const underlineMatch = rPrString.match(/<w:u(?: w:val="([^"]+)")?\/?>/);
119
- if (underlineMatch) {
120
- const val = underlineMatch[1];
120
+ const u = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:u');
121
+ if (u) {
122
+ const val = u.getAttribute('w:val');
121
123
  // If val is missing, it's a default underline (true).
122
124
  // If val is present, it's true unless explicit 'none'.
123
125
  if (!val || val !== 'none') {
124
126
  formatting.underline = true;
125
127
  }
126
128
  }
127
- const strike = getBoolVal(rPrString, 'w:strike');
128
- const dstrike = getBoolVal(rPrString, 'w:dstrike');
129
+ const strike = getBoolVal(rPr, 'w:strike');
130
+ const dstrike = getBoolVal(rPr, 'w:dstrike');
129
131
  if (strike !== null)
130
132
  formatting.strikethrough = strike;
131
133
  else if (dstrike !== null)
132
134
  formatting.strikethrough = dstrike;
133
- // Font size
134
- const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
135
- if (szMatch)
136
- formatting.size = (parseInt(szMatch[1], 10) / 2).toString() + 'pt';
137
- // Color
138
- const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
139
- if (colorMatch && colorMatch[1] !== 'auto')
140
- formatting.color = '#' + colorMatch[1];
141
- // Background color (shading)
142
- const shdMatch = rPrString.match(/<w:shd[^>]*w:fill="([^"]+)"/);
143
- if (shdMatch && shdMatch[1] !== 'auto')
144
- formatting.backgroundColor = '#' + shdMatch[1];
145
- // Highlight (map to backgroundColor)
146
- const highlightMatch = rPrString.match(/<w:highlight w:val="([^"]+)"/);
147
- if (highlightMatch && highlightMatch[1] !== 'none') {
148
- const colorMap = {
149
- 'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
150
- 'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
151
- 'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
152
- 'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
153
- };
154
- formatting.backgroundColor = colorMap[highlightMatch[1]] || highlightMatch[1];
135
+ // Font size (w:sz) - stored in half-points
136
+ const sz = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:sz');
137
+ if (sz) {
138
+ const val = sz.getAttribute('w:val');
139
+ if (val) {
140
+ formatting.size = (parseInt(val, 10) / 2).toString() + 'pt';
141
+ }
155
142
  }
156
- // Font family
157
- const rFontsMatch = rPrString.match(/<w:rFonts[^>]*w:ascii="([^"]+)"/);
158
- if (rFontsMatch) {
159
- formatting.font = rFontsMatch[1];
143
+ // Color (w:color)
144
+ const color = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:color');
145
+ if (color) {
146
+ const val = color.getAttribute('w:val');
147
+ if (val && val !== 'auto') {
148
+ formatting.color = '#' + val;
149
+ }
160
150
  }
161
- else {
162
- const hAnsiMatch = rPrString.match(/<w:rFonts[^>]*w:hAnsi="([^"]+)"/);
163
- if (hAnsiMatch)
164
- formatting.font = hAnsiMatch[1];
151
+ // Background color (w:shd) - shading
152
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:shd');
153
+ if (shd) {
154
+ const val = shd.getAttribute('w:fill');
155
+ if (val && val !== 'auto') {
156
+ formatting.backgroundColor = '#' + val;
157
+ }
165
158
  }
166
- // Subscript/Superscript
167
- const vertAlignMatch = rPrString.match(/<w:vertAlign w:val="([^"]+)"/);
168
- if (vertAlignMatch) {
169
- if (vertAlignMatch[1] === 'subscript')
159
+ // Highlight (w:highlight) - maps to background color in our AST
160
+ const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:highlight');
161
+ if (highlight) {
162
+ const val = highlight.getAttribute('w:val');
163
+ if (val && val !== 'none') {
164
+ const colorMap = {
165
+ 'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
166
+ 'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
167
+ 'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
168
+ 'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
169
+ };
170
+ formatting.backgroundColor = colorMap[val] || val;
171
+ }
172
+ }
173
+ // Font family (w:rFonts)
174
+ const rFonts = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:rFonts');
175
+ if (rFonts) {
176
+ // Priority: ascii (Western) > hAnsi (High ANSI)
177
+ const font = rFonts.getAttribute('w:ascii') || rFonts.getAttribute('w:hAnsi');
178
+ if (font) {
179
+ formatting.font = font;
180
+ }
181
+ }
182
+ // Subscript/Superscript (w:vertAlign)
183
+ const vertAlign = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:vertAlign');
184
+ if (vertAlign) {
185
+ const val = vertAlign.getAttribute('w:val');
186
+ if (val === 'subscript')
170
187
  formatting.subscript = true;
171
- if (vertAlignMatch[1] === 'superscript')
188
+ else if (val === 'superscript')
172
189
  formatting.superscript = true;
173
190
  }
174
191
  return formatting;
@@ -194,6 +211,21 @@ const parseWord = async (buffer, config) => {
194
211
  }
195
212
  return undefined;
196
213
  };
214
+ /**
215
+ * Resolves mc:AlternateContent by preferring mc:Fallback if choice namespace is not recognized,
216
+ * or simply the first available valid child.
217
+ */
218
+ const resolveAlternateContent = (element) => {
219
+ const choice = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "mc:Choice");
220
+ // In most cases, mc:Choice contains the modern version, but mc:Fallback is safer for legacy compatibility
221
+ // Mammoth often skips Choice if it's not handled. We'll try Choice first.
222
+ if (choice)
223
+ return Array.from(choice.childNodes);
224
+ const fallback = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "mc:Fallback");
225
+ if (fallback)
226
+ return Array.from(fallback.childNodes);
227
+ return Array.from(element.childNodes);
228
+ };
197
229
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
198
230
  !!x.match(footnotesFileRegex) ||
199
231
  !!x.match(endnotesFileRegex) ||
@@ -250,18 +282,32 @@ const parseWord = async (buffer, config) => {
250
282
  const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
251
283
  if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
252
284
  numberingMap[numId] = {};
285
+ // Inherit from abstractNum
253
286
  const lvls = (0, xmlUtils_js_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
254
287
  for (const lvl of lvls) {
255
288
  const ilvl = lvl.getAttribute("w:ilvl");
256
289
  const numFmtNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:numFmt");
257
290
  const lvlTextNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:lvlText");
291
+ const startNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:start");
258
292
  if (ilvl) {
259
293
  numberingMap[numId][ilvl] = {
260
294
  numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
261
- lvlText: lvlTextNode?.getAttribute("w:val") || ''
295
+ lvlText: lvlTextNode?.getAttribute("w:val") || '',
296
+ start: parseInt(startNode?.getAttribute("w:val") || '1', 10)
262
297
  };
263
298
  }
264
299
  }
300
+ // Apply instance overrides (w:lvlOverride)
301
+ const overrides = (0, xmlUtils_js_1.getElementsByTagName)(num, "w:lvlOverride");
302
+ for (const override of overrides) {
303
+ const ilvl = override.getAttribute("w:ilvl");
304
+ if (ilvl && numberingMap[numId][ilvl]) {
305
+ const startOverride = (0, xmlUtils_js_1.getFirstElementByTagName)(override, "w:startOverride");
306
+ if (startOverride) {
307
+ numberingMap[numId][ilvl].start = parseInt(startOverride.getAttribute("w:val") || '1', 10);
308
+ }
309
+ }
310
+ }
265
311
  }
266
312
  }
267
313
  }
@@ -342,8 +388,7 @@ const parseWord = async (buffer, config) => {
342
388
  const numberingState = {};
343
389
  const listCounters = {}; // Track item index per listId/level
344
390
  // Helper to parse a paragraph node
345
- const parseParagraph = (pNode, documentContent) => {
346
- const pXml = pNode.toString();
391
+ const parseParagraph = (pNode, documentContent, pendingAnchorIds = []) => {
347
392
  // Check if it's a list item
348
393
  const numPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:numPr");
349
394
  const isList = !!numPr;
@@ -605,7 +650,7 @@ const parseWord = async (buffer, config) => {
605
650
  const rId = hlNode.getAttribute("r:id");
606
651
  const anchor = hlNode.getAttribute("w:anchor");
607
652
  let linkMetadata;
608
- if (anchor) {
653
+ if (anchor && !config.ignoreInternalLinks) {
609
654
  linkMetadata = { link: '#' + anchor, linkType: 'internal' };
610
655
  }
611
656
  else if (rId && relsMap[rId]) {
@@ -627,11 +672,43 @@ const parseWord = async (buffer, config) => {
627
672
  }
628
673
  }
629
674
  }
675
+ else if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:bookmarkStart') {
676
+ const bookmarkName = node.getAttribute("w:name");
677
+ if (bookmarkName && !bookmarkName.startsWith('_GoBack') && !config.ignoreInternalLinks) {
678
+ anchorIds.push(bookmarkName);
679
+ }
680
+ }
681
+ else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'mc:AlternateContent' || node.nodeName === 'AlternateContent')) {
682
+ const resolved = resolveAlternateContent(node);
683
+ for (const rNode of resolved)
684
+ processChildNode(rNode);
685
+ }
686
+ else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'w:pict' || node.nodeName === 'pict' || node.nodeName === 'w:drawing' || node.nodeName === 'drawing')) {
687
+ // Extract text boxes from legacy shapes or modern drawings
688
+ const textBoxes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:txbxContent");
689
+ for (const txbx of textBoxes) {
690
+ const txbxChildren = Array.from(txbx.childNodes);
691
+ for (const txbxChild of txbxChildren) {
692
+ if ((0, xmlUtils_js_1.isElement)(txbxChild) && txbxChild.nodeName === 'w:p') {
693
+ const nestedP = parseParagraph(txbxChild, documentContent);
694
+ children.push(...(nestedP.children || []));
695
+ text += nestedP.text;
696
+ }
697
+ }
698
+ }
699
+ }
700
+ else if (node.childNodes.length > 0) {
701
+ // Generic fallback for unknown elements that might contain content
702
+ for (const child of Array.from(node.childNodes))
703
+ processChildNode(child);
704
+ }
630
705
  };
706
+ const anchorIds = [...pendingAnchorIds];
631
707
  const childNodes = Array.from(pNode.childNodes);
632
708
  for (const child of childNodes) {
633
709
  processChildNode(child);
634
710
  }
711
+ const commonMetadata = anchorIds.length > 0 ? { anchorIds } : {};
635
712
  if (isList) {
636
713
  const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
637
714
  const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
@@ -652,11 +729,11 @@ const parseWord = async (buffer, config) => {
652
729
  }
653
730
  const numFmt = numberingMap[numId][ilvlStr]?.numFmt || 'decimal';
654
731
  listType = numFmt === 'bullet' ? 'unordered' : 'ordered';
655
- // Track itemIndex (starts at 0, continues across interruptions for same listId)
732
+ // Track itemIndex (starts at override or default, continues across interruptions for same listId)
656
733
  if (!listCounters[numId])
657
734
  listCounters[numId] = {};
658
735
  if (listCounters[numId][ilvlStr] === undefined) {
659
- listCounters[numId][ilvlStr] = 0;
736
+ listCounters[numId][ilvlStr] = (numberingMap[numId][ilvlStr]?.start ?? 1) - 1;
660
737
  }
661
738
  else {
662
739
  listCounters[numId][ilvlStr]++;
@@ -674,7 +751,8 @@ const parseWord = async (buffer, config) => {
674
751
  alignment: (alignment || 'left'),
675
752
  listId: numId,
676
753
  itemIndex: itemIndex,
677
- style: pStyleVal
754
+ style: pStyleVal,
755
+ ...commonMetadata
678
756
  }
679
757
  };
680
758
  if (config.includeRawContent)
@@ -687,7 +765,7 @@ const parseWord = async (buffer, config) => {
687
765
  type: 'heading',
688
766
  text: text,
689
767
  children: children,
690
- metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
768
+ metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
691
769
  };
692
770
  if (config.includeRawContent)
693
771
  headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
@@ -698,7 +776,7 @@ const parseWord = async (buffer, config) => {
698
776
  type: 'paragraph',
699
777
  text: text,
700
778
  children: children,
701
- metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
779
+ metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
702
780
  };
703
781
  if (config.includeRawContent)
704
782
  paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
@@ -706,17 +784,41 @@ const parseWord = async (buffer, config) => {
706
784
  }
707
785
  };
708
786
  // Helper to parse a table node
709
- const parseTable = (tblNode, documentContent) => {
787
+ const parseTable = (tblNode, documentContent, pendingAnchorIds = []) => {
710
788
  const rows = [];
711
- // Only get direct child rows, not nested table rows
712
789
  const trNodes = (0, xmlUtils_js_1.getDirectChildren)(tblNode, "w:tr");
790
+ // Track vertical merges: colIndex -> { startCellNode, rowSpan }
791
+ const vMergeMap = new Map();
713
792
  for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
714
793
  const trNode = trNodes[rIndex];
715
794
  const cells = [];
716
795
  // Only get direct child cells, not nested table cells
717
796
  const tcNodes = (0, xmlUtils_js_1.getDirectChildren)(trNode, "w:tc");
718
- for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
719
- const tcNode = tcNodes[cIndex];
797
+ let visualCol = 0;
798
+ for (let tcIndex = 0; tcIndex < tcNodes.length; tcIndex++) {
799
+ const tcNode = tcNodes[tcIndex];
800
+ const tcPr = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "w:tcPr");
801
+ // Horizontal merge (colspan)
802
+ let colSpan = 1;
803
+ if (tcPr) {
804
+ const gridSpan = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:gridSpan");
805
+ if (gridSpan) {
806
+ colSpan = parseInt(gridSpan.getAttribute("w:val") || "1", 10);
807
+ }
808
+ }
809
+ let vMergeRestart = false;
810
+ let isVMerge = false;
811
+ if (tcPr) {
812
+ const vMerge = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:vMerge");
813
+ if (vMerge) {
814
+ isVMerge = true;
815
+ const val = vMerge.getAttribute("w:val");
816
+ // If it's explicit restart, or if we don't have an active merge for this column, treat as restart
817
+ if (val === "restart" || !vMergeMap.has(visualCol)) {
818
+ vMergeRestart = true;
819
+ }
820
+ }
821
+ }
720
822
  const cellChildren = [];
721
823
  let cellText = '';
722
824
  // Cells contain paragraphs (and other block-level elements)
@@ -728,23 +830,50 @@ const parseWord = async (buffer, config) => {
728
830
  cellText += pNode.text;
729
831
  }
730
832
  else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
731
- // Nested table
732
833
  const nestedTable = parseTable(child, documentContent);
733
834
  cellChildren.push(nestedTable);
734
- // Don't add nested table text to cell text - it will be handled recursively
735
835
  }
736
836
  }
737
837
  const cellNode = {
738
838
  type: 'cell',
739
839
  text: cellText,
740
840
  children: cellChildren,
741
- metadata: { row: rIndex, col: cIndex }
841
+ metadata: { row: rIndex, col: visualCol }
742
842
  };
743
- cells.push(cellNode);
843
+ if (colSpan > 1)
844
+ cellNode.metadata.colSpan = colSpan;
845
+ if (isVMerge) {
846
+ if (vMergeRestart) {
847
+ vMergeMap.set(visualCol, { node: cellNode, span: 1 });
848
+ cells.push(cellNode);
849
+ }
850
+ else {
851
+ const mergeInfo = vMergeMap.get(visualCol);
852
+ if (mergeInfo) {
853
+ mergeInfo.span++;
854
+ mergeInfo.node.metadata.rowSpan = mergeInfo.span;
855
+ if (cellChildren.length > 0) {
856
+ if (!mergeInfo.node.children)
857
+ mergeInfo.node.children = [];
858
+ mergeInfo.node.children.push(...cellChildren);
859
+ mergeInfo.node.text += " " + cellText;
860
+ }
861
+ }
862
+ else {
863
+ // Fallback: if we found a continue but no restart, treat as normal cell
864
+ cells.push(cellNode);
865
+ }
866
+ }
867
+ }
868
+ else {
869
+ vMergeMap.delete(visualCol);
870
+ cells.push(cellNode);
871
+ }
872
+ visualCol += colSpan;
744
873
  }
745
874
  const rowNode = {
746
875
  type: 'row',
747
- children: cells
876
+ children: cells,
748
877
  };
749
878
  rows.push(rowNode);
750
879
  }
@@ -803,12 +932,23 @@ const parseWord = async (buffer, config) => {
803
932
  const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
804
933
  if (body) {
805
934
  const bodyChildren = Array.from(body.childNodes);
935
+ let pendingAnchorIds = [];
806
936
  for (const child of bodyChildren) {
807
- if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
808
- content.push(parseParagraph(child, documentContent));
809
- }
810
- else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
811
- content.push(parseTable(child, documentContent));
937
+ if ((0, xmlUtils_js_1.isElement)(child)) {
938
+ if (child.nodeName === 'w:p') {
939
+ content.push(parseParagraph(child, documentContent, pendingAnchorIds));
940
+ pendingAnchorIds = [];
941
+ }
942
+ else if (child.nodeName === 'w:tbl') {
943
+ content.push(parseTable(child, documentContent, pendingAnchorIds));
944
+ pendingAnchorIds = [];
945
+ }
946
+ else if (child.nodeName === 'w:bookmarkStart') {
947
+ const bookmarkName = child.getAttribute("w:name");
948
+ if (bookmarkName && !bookmarkName.startsWith('_GoBack') && !config.ignoreInternalLinks) {
949
+ pendingAnchorIds.push(bookmarkName);
950
+ }
951
+ }
812
952
  }
813
953
  }
814
954
  }
@@ -821,10 +961,10 @@ const parseWord = async (buffer, config) => {
821
961
  if (config.ocr) {
822
962
  if (attachment.mimeType.startsWith('image/')) {
823
963
  try {
824
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
964
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
825
965
  }
826
966
  catch (e) {
827
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
967
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
828
968
  }
829
969
  }
830
970
  }
@@ -852,27 +992,22 @@ const parseWord = async (buffer, config) => {
852
992
  if (config.putNotesAtLast && collectedNotes.length > 0) {
853
993
  content.push(...collectedNotes);
854
994
  }
855
- return {
856
- type: 'docx',
857
- metadata: { ...metadata, formatting: docDefaults, styleMap: styleMap },
858
- content: content,
859
- attachments: attachments,
860
- toText: () => content.map(c => {
861
- // Recursive text extraction
862
- const getText = (node) => {
863
- let t = '';
864
- if (node.children) {
865
- t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
866
- }
867
- else if (node.type === 'break') {
868
- t += config.newlineDelimiter ?? '\n';
869
- }
870
- else
871
- t += node.text || '';
872
- return t;
873
- };
874
- return getText(c);
875
- }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
876
- };
995
+ const toTextSync = () => content.map(c => {
996
+ // Recursive text extraction
997
+ const getText = (node) => {
998
+ let t = '';
999
+ if (node.children) {
1000
+ t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter);
1001
+ }
1002
+ else if (node.type === 'break') {
1003
+ t += config.newlineDelimiter;
1004
+ }
1005
+ else
1006
+ t += node.text || '';
1007
+ return t;
1008
+ };
1009
+ return getText(c);
1010
+ }).filter(t => t != '').join(config.newlineDelimiter);
1011
+ return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, toTextSync);
877
1012
  };
878
1013
  exports.parseWord = parseWord;