officeparser 6.1.1 → 7.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +301 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +74 -31
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +828 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +783 -5
- package/dist/types.js +73 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.d.ts +8 -3
- package/dist/utils/envUtils.js +117 -34
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +110 -52
- package/dist/utils/moduleLoader.js +19 -11
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +26 -7
|
@@ -59,7 +59,7 @@
|
|
|
59
59
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
|
|
60
60
|
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
|
|
61
61
|
*/
|
|
62
|
-
import {
|
|
62
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
63
63
|
/**
|
|
64
64
|
* Parses a Word document (.docx) and extracts content, formatting, and metadata.
|
|
65
65
|
*
|
|
@@ -76,4 +76,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
|
76
76
|
* @param config - Parser configuration options
|
|
77
77
|
* @returns A promise resolving to the parsed AST
|
|
78
78
|
*/
|
|
79
|
-
export declare const parseWord: (buffer: Buffer, config:
|
|
79
|
+
export declare const parseWord: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|
|
@@ -62,6 +62,8 @@
|
|
|
62
62
|
*/
|
|
63
63
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
64
64
|
exports.parseWord = void 0;
|
|
65
|
+
const types_js_1 = require("../types.js");
|
|
66
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
65
67
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
66
68
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
67
69
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
@@ -96,79 +98,94 @@ const parseWord = async (buffer, config) => {
|
|
|
96
98
|
// Helper to extract formatting from run properties XML string
|
|
97
99
|
const extractFormattingFromXml = (rPr) => {
|
|
98
100
|
const formatting = {};
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
if (val ===
|
|
101
|
+
// Helper to check boolean properties (e.g., <w:b />, <w:i w:val="0" />)
|
|
102
|
+
const getBoolVal = (parent, tagName) => {
|
|
103
|
+
const el = (0, xmlUtils_js_1.getFirstElementByTagName)(parent, tagName);
|
|
104
|
+
if (el) {
|
|
105
|
+
const val = el.getAttribute('w:val');
|
|
106
|
+
// In OOXML, if the element is present without w:val, it's true.
|
|
107
|
+
// If w:val is present, it can be '1', 'true', 'on' for true.
|
|
108
|
+
if (val === null)
|
|
107
109
|
return true;
|
|
108
110
|
return val === '1' || val === 'true' || val === 'on';
|
|
109
111
|
}
|
|
110
112
|
return null;
|
|
111
113
|
};
|
|
112
|
-
const bold = getBoolVal(
|
|
114
|
+
const bold = getBoolVal(rPr, 'w:b');
|
|
113
115
|
if (bold !== null)
|
|
114
116
|
formatting.bold = bold;
|
|
115
|
-
const italic = getBoolVal(
|
|
117
|
+
const italic = getBoolVal(rPr, 'w:i');
|
|
116
118
|
if (italic !== null)
|
|
117
119
|
formatting.italic = italic;
|
|
118
|
-
const
|
|
119
|
-
if (
|
|
120
|
-
const val =
|
|
120
|
+
const u = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:u');
|
|
121
|
+
if (u) {
|
|
122
|
+
const val = u.getAttribute('w:val');
|
|
121
123
|
// If val is missing, it's a default underline (true).
|
|
122
124
|
// If val is present, it's true unless explicit 'none'.
|
|
123
125
|
if (!val || val !== 'none') {
|
|
124
126
|
formatting.underline = true;
|
|
125
127
|
}
|
|
126
128
|
}
|
|
127
|
-
const strike = getBoolVal(
|
|
128
|
-
const dstrike = getBoolVal(
|
|
129
|
+
const strike = getBoolVal(rPr, 'w:strike');
|
|
130
|
+
const dstrike = getBoolVal(rPr, 'w:dstrike');
|
|
129
131
|
if (strike !== null)
|
|
130
132
|
formatting.strikethrough = strike;
|
|
131
133
|
else if (dstrike !== null)
|
|
132
134
|
formatting.strikethrough = dstrike;
|
|
133
|
-
// Font size
|
|
134
|
-
const
|
|
135
|
-
if (
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
formatting.color = '#' + colorMatch[1];
|
|
141
|
-
// Background color (shading)
|
|
142
|
-
const shdMatch = rPrString.match(/<w:shd[^>]*w:fill="([^"]+)"/);
|
|
143
|
-
if (shdMatch && shdMatch[1] !== 'auto')
|
|
144
|
-
formatting.backgroundColor = '#' + shdMatch[1];
|
|
145
|
-
// Highlight (map to backgroundColor)
|
|
146
|
-
const highlightMatch = rPrString.match(/<w:highlight w:val="([^"]+)"/);
|
|
147
|
-
if (highlightMatch && highlightMatch[1] !== 'none') {
|
|
148
|
-
const colorMap = {
|
|
149
|
-
'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
|
|
150
|
-
'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
|
|
151
|
-
'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
|
|
152
|
-
'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
|
|
153
|
-
};
|
|
154
|
-
formatting.backgroundColor = colorMap[highlightMatch[1]] || highlightMatch[1];
|
|
135
|
+
// Font size (w:sz) - stored in half-points
|
|
136
|
+
const sz = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:sz');
|
|
137
|
+
if (sz) {
|
|
138
|
+
const val = sz.getAttribute('w:val');
|
|
139
|
+
if (val) {
|
|
140
|
+
formatting.size = (parseInt(val, 10) / 2).toString() + 'pt';
|
|
141
|
+
}
|
|
155
142
|
}
|
|
156
|
-
//
|
|
157
|
-
const
|
|
158
|
-
if (
|
|
159
|
-
|
|
143
|
+
// Color (w:color)
|
|
144
|
+
const color = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:color');
|
|
145
|
+
if (color) {
|
|
146
|
+
const val = color.getAttribute('w:val');
|
|
147
|
+
if (val && val !== 'auto') {
|
|
148
|
+
formatting.color = '#' + val;
|
|
149
|
+
}
|
|
160
150
|
}
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
151
|
+
// Background color (w:shd) - shading
|
|
152
|
+
const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:shd');
|
|
153
|
+
if (shd) {
|
|
154
|
+
const val = shd.getAttribute('w:fill');
|
|
155
|
+
if (val && val !== 'auto') {
|
|
156
|
+
formatting.backgroundColor = '#' + val;
|
|
157
|
+
}
|
|
165
158
|
}
|
|
166
|
-
//
|
|
167
|
-
const
|
|
168
|
-
if (
|
|
169
|
-
|
|
159
|
+
// Highlight (w:highlight) - maps to background color in our AST
|
|
160
|
+
const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:highlight');
|
|
161
|
+
if (highlight) {
|
|
162
|
+
const val = highlight.getAttribute('w:val');
|
|
163
|
+
if (val && val !== 'none') {
|
|
164
|
+
const colorMap = {
|
|
165
|
+
'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
|
|
166
|
+
'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
|
|
167
|
+
'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
|
|
168
|
+
'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
|
|
169
|
+
};
|
|
170
|
+
formatting.backgroundColor = colorMap[val] || val;
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
// Font family (w:rFonts)
|
|
174
|
+
const rFonts = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:rFonts');
|
|
175
|
+
if (rFonts) {
|
|
176
|
+
// Priority: ascii (Western) > hAnsi (High ANSI)
|
|
177
|
+
const font = rFonts.getAttribute('w:ascii') || rFonts.getAttribute('w:hAnsi');
|
|
178
|
+
if (font) {
|
|
179
|
+
formatting.font = font;
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
// Subscript/Superscript (w:vertAlign)
|
|
183
|
+
const vertAlign = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, 'w:vertAlign');
|
|
184
|
+
if (vertAlign) {
|
|
185
|
+
const val = vertAlign.getAttribute('w:val');
|
|
186
|
+
if (val === 'subscript')
|
|
170
187
|
formatting.subscript = true;
|
|
171
|
-
if (
|
|
188
|
+
else if (val === 'superscript')
|
|
172
189
|
formatting.superscript = true;
|
|
173
190
|
}
|
|
174
191
|
return formatting;
|
|
@@ -194,6 +211,21 @@ const parseWord = async (buffer, config) => {
|
|
|
194
211
|
}
|
|
195
212
|
return undefined;
|
|
196
213
|
};
|
|
214
|
+
/**
|
|
215
|
+
* Resolves mc:AlternateContent by preferring mc:Fallback if choice namespace is not recognized,
|
|
216
|
+
* or simply the first available valid child.
|
|
217
|
+
*/
|
|
218
|
+
const resolveAlternateContent = (element) => {
|
|
219
|
+
const choice = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "mc:Choice");
|
|
220
|
+
// In most cases, mc:Choice contains the modern version, but mc:Fallback is safer for legacy compatibility
|
|
221
|
+
// Mammoth often skips Choice if it's not handled. We'll try Choice first.
|
|
222
|
+
if (choice)
|
|
223
|
+
return Array.from(choice.childNodes);
|
|
224
|
+
const fallback = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "mc:Fallback");
|
|
225
|
+
if (fallback)
|
|
226
|
+
return Array.from(fallback.childNodes);
|
|
227
|
+
return Array.from(element.childNodes);
|
|
228
|
+
};
|
|
197
229
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
|
|
198
230
|
!!x.match(footnotesFileRegex) ||
|
|
199
231
|
!!x.match(endnotesFileRegex) ||
|
|
@@ -250,18 +282,32 @@ const parseWord = async (buffer, config) => {
|
|
|
250
282
|
const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
|
|
251
283
|
if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
|
|
252
284
|
numberingMap[numId] = {};
|
|
285
|
+
// Inherit from abstractNum
|
|
253
286
|
const lvls = (0, xmlUtils_js_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
|
|
254
287
|
for (const lvl of lvls) {
|
|
255
288
|
const ilvl = lvl.getAttribute("w:ilvl");
|
|
256
289
|
const numFmtNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:numFmt");
|
|
257
290
|
const lvlTextNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:lvlText");
|
|
291
|
+
const startNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:start");
|
|
258
292
|
if (ilvl) {
|
|
259
293
|
numberingMap[numId][ilvl] = {
|
|
260
294
|
numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
|
|
261
|
-
lvlText: lvlTextNode?.getAttribute("w:val") || ''
|
|
295
|
+
lvlText: lvlTextNode?.getAttribute("w:val") || '',
|
|
296
|
+
start: parseInt(startNode?.getAttribute("w:val") || '1', 10)
|
|
262
297
|
};
|
|
263
298
|
}
|
|
264
299
|
}
|
|
300
|
+
// Apply instance overrides (w:lvlOverride)
|
|
301
|
+
const overrides = (0, xmlUtils_js_1.getElementsByTagName)(num, "w:lvlOverride");
|
|
302
|
+
for (const override of overrides) {
|
|
303
|
+
const ilvl = override.getAttribute("w:ilvl");
|
|
304
|
+
if (ilvl && numberingMap[numId][ilvl]) {
|
|
305
|
+
const startOverride = (0, xmlUtils_js_1.getFirstElementByTagName)(override, "w:startOverride");
|
|
306
|
+
if (startOverride) {
|
|
307
|
+
numberingMap[numId][ilvl].start = parseInt(startOverride.getAttribute("w:val") || '1', 10);
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
}
|
|
265
311
|
}
|
|
266
312
|
}
|
|
267
313
|
}
|
|
@@ -342,8 +388,7 @@ const parseWord = async (buffer, config) => {
|
|
|
342
388
|
const numberingState = {};
|
|
343
389
|
const listCounters = {}; // Track item index per listId/level
|
|
344
390
|
// Helper to parse a paragraph node
|
|
345
|
-
const parseParagraph = (pNode, documentContent) => {
|
|
346
|
-
const pXml = pNode.toString();
|
|
391
|
+
const parseParagraph = (pNode, documentContent, pendingAnchorIds = []) => {
|
|
347
392
|
// Check if it's a list item
|
|
348
393
|
const numPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:numPr");
|
|
349
394
|
const isList = !!numPr;
|
|
@@ -605,7 +650,7 @@ const parseWord = async (buffer, config) => {
|
|
|
605
650
|
const rId = hlNode.getAttribute("r:id");
|
|
606
651
|
const anchor = hlNode.getAttribute("w:anchor");
|
|
607
652
|
let linkMetadata;
|
|
608
|
-
if (anchor) {
|
|
653
|
+
if (anchor && !config.ignoreInternalLinks) {
|
|
609
654
|
linkMetadata = { link: '#' + anchor, linkType: 'internal' };
|
|
610
655
|
}
|
|
611
656
|
else if (rId && relsMap[rId]) {
|
|
@@ -627,11 +672,43 @@ const parseWord = async (buffer, config) => {
|
|
|
627
672
|
}
|
|
628
673
|
}
|
|
629
674
|
}
|
|
675
|
+
else if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:bookmarkStart') {
|
|
676
|
+
const bookmarkName = node.getAttribute("w:name");
|
|
677
|
+
if (bookmarkName && !bookmarkName.startsWith('_GoBack') && !config.ignoreInternalLinks) {
|
|
678
|
+
anchorIds.push(bookmarkName);
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'mc:AlternateContent' || node.nodeName === 'AlternateContent')) {
|
|
682
|
+
const resolved = resolveAlternateContent(node);
|
|
683
|
+
for (const rNode of resolved)
|
|
684
|
+
processChildNode(rNode);
|
|
685
|
+
}
|
|
686
|
+
else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'w:pict' || node.nodeName === 'pict' || node.nodeName === 'w:drawing' || node.nodeName === 'drawing')) {
|
|
687
|
+
// Extract text boxes from legacy shapes or modern drawings
|
|
688
|
+
const textBoxes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:txbxContent");
|
|
689
|
+
for (const txbx of textBoxes) {
|
|
690
|
+
const txbxChildren = Array.from(txbx.childNodes);
|
|
691
|
+
for (const txbxChild of txbxChildren) {
|
|
692
|
+
if ((0, xmlUtils_js_1.isElement)(txbxChild) && txbxChild.nodeName === 'w:p') {
|
|
693
|
+
const nestedP = parseParagraph(txbxChild, documentContent);
|
|
694
|
+
children.push(...(nestedP.children || []));
|
|
695
|
+
text += nestedP.text;
|
|
696
|
+
}
|
|
697
|
+
}
|
|
698
|
+
}
|
|
699
|
+
}
|
|
700
|
+
else if (node.childNodes.length > 0) {
|
|
701
|
+
// Generic fallback for unknown elements that might contain content
|
|
702
|
+
for (const child of Array.from(node.childNodes))
|
|
703
|
+
processChildNode(child);
|
|
704
|
+
}
|
|
630
705
|
};
|
|
706
|
+
const anchorIds = [...pendingAnchorIds];
|
|
631
707
|
const childNodes = Array.from(pNode.childNodes);
|
|
632
708
|
for (const child of childNodes) {
|
|
633
709
|
processChildNode(child);
|
|
634
710
|
}
|
|
711
|
+
const commonMetadata = anchorIds.length > 0 ? { anchorIds } : {};
|
|
635
712
|
if (isList) {
|
|
636
713
|
const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
|
|
637
714
|
const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
|
|
@@ -652,11 +729,11 @@ const parseWord = async (buffer, config) => {
|
|
|
652
729
|
}
|
|
653
730
|
const numFmt = numberingMap[numId][ilvlStr]?.numFmt || 'decimal';
|
|
654
731
|
listType = numFmt === 'bullet' ? 'unordered' : 'ordered';
|
|
655
|
-
// Track itemIndex (starts at
|
|
732
|
+
// Track itemIndex (starts at override or default, continues across interruptions for same listId)
|
|
656
733
|
if (!listCounters[numId])
|
|
657
734
|
listCounters[numId] = {};
|
|
658
735
|
if (listCounters[numId][ilvlStr] === undefined) {
|
|
659
|
-
listCounters[numId][ilvlStr] =
|
|
736
|
+
listCounters[numId][ilvlStr] = (numberingMap[numId][ilvlStr]?.start ?? 1) - 1;
|
|
660
737
|
}
|
|
661
738
|
else {
|
|
662
739
|
listCounters[numId][ilvlStr]++;
|
|
@@ -674,7 +751,8 @@ const parseWord = async (buffer, config) => {
|
|
|
674
751
|
alignment: (alignment || 'left'),
|
|
675
752
|
listId: numId,
|
|
676
753
|
itemIndex: itemIndex,
|
|
677
|
-
style: pStyleVal
|
|
754
|
+
style: pStyleVal,
|
|
755
|
+
...commonMetadata
|
|
678
756
|
}
|
|
679
757
|
};
|
|
680
758
|
if (config.includeRawContent)
|
|
@@ -687,7 +765,7 @@ const parseWord = async (buffer, config) => {
|
|
|
687
765
|
type: 'heading',
|
|
688
766
|
text: text,
|
|
689
767
|
children: children,
|
|
690
|
-
metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
|
|
768
|
+
metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
|
|
691
769
|
};
|
|
692
770
|
if (config.includeRawContent)
|
|
693
771
|
headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
@@ -698,7 +776,7 @@ const parseWord = async (buffer, config) => {
|
|
|
698
776
|
type: 'paragraph',
|
|
699
777
|
text: text,
|
|
700
778
|
children: children,
|
|
701
|
-
metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
|
|
779
|
+
metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
|
|
702
780
|
};
|
|
703
781
|
if (config.includeRawContent)
|
|
704
782
|
paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
@@ -706,17 +784,41 @@ const parseWord = async (buffer, config) => {
|
|
|
706
784
|
}
|
|
707
785
|
};
|
|
708
786
|
// Helper to parse a table node
|
|
709
|
-
const parseTable = (tblNode, documentContent) => {
|
|
787
|
+
const parseTable = (tblNode, documentContent, pendingAnchorIds = []) => {
|
|
710
788
|
const rows = [];
|
|
711
|
-
// Only get direct child rows, not nested table rows
|
|
712
789
|
const trNodes = (0, xmlUtils_js_1.getDirectChildren)(tblNode, "w:tr");
|
|
790
|
+
// Track vertical merges: colIndex -> { startCellNode, rowSpan }
|
|
791
|
+
const vMergeMap = new Map();
|
|
713
792
|
for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
|
|
714
793
|
const trNode = trNodes[rIndex];
|
|
715
794
|
const cells = [];
|
|
716
795
|
// Only get direct child cells, not nested table cells
|
|
717
796
|
const tcNodes = (0, xmlUtils_js_1.getDirectChildren)(trNode, "w:tc");
|
|
718
|
-
|
|
719
|
-
|
|
797
|
+
let visualCol = 0;
|
|
798
|
+
for (let tcIndex = 0; tcIndex < tcNodes.length; tcIndex++) {
|
|
799
|
+
const tcNode = tcNodes[tcIndex];
|
|
800
|
+
const tcPr = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "w:tcPr");
|
|
801
|
+
// Horizontal merge (colspan)
|
|
802
|
+
let colSpan = 1;
|
|
803
|
+
if (tcPr) {
|
|
804
|
+
const gridSpan = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:gridSpan");
|
|
805
|
+
if (gridSpan) {
|
|
806
|
+
colSpan = parseInt(gridSpan.getAttribute("w:val") || "1", 10);
|
|
807
|
+
}
|
|
808
|
+
}
|
|
809
|
+
let vMergeRestart = false;
|
|
810
|
+
let isVMerge = false;
|
|
811
|
+
if (tcPr) {
|
|
812
|
+
const vMerge = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:vMerge");
|
|
813
|
+
if (vMerge) {
|
|
814
|
+
isVMerge = true;
|
|
815
|
+
const val = vMerge.getAttribute("w:val");
|
|
816
|
+
// If it's explicit restart, or if we don't have an active merge for this column, treat as restart
|
|
817
|
+
if (val === "restart" || !vMergeMap.has(visualCol)) {
|
|
818
|
+
vMergeRestart = true;
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
}
|
|
720
822
|
const cellChildren = [];
|
|
721
823
|
let cellText = '';
|
|
722
824
|
// Cells contain paragraphs (and other block-level elements)
|
|
@@ -728,23 +830,50 @@ const parseWord = async (buffer, config) => {
|
|
|
728
830
|
cellText += pNode.text;
|
|
729
831
|
}
|
|
730
832
|
else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
|
|
731
|
-
// Nested table
|
|
732
833
|
const nestedTable = parseTable(child, documentContent);
|
|
733
834
|
cellChildren.push(nestedTable);
|
|
734
|
-
// Don't add nested table text to cell text - it will be handled recursively
|
|
735
835
|
}
|
|
736
836
|
}
|
|
737
837
|
const cellNode = {
|
|
738
838
|
type: 'cell',
|
|
739
839
|
text: cellText,
|
|
740
840
|
children: cellChildren,
|
|
741
|
-
metadata: { row: rIndex, col:
|
|
841
|
+
metadata: { row: rIndex, col: visualCol }
|
|
742
842
|
};
|
|
743
|
-
|
|
843
|
+
if (colSpan > 1)
|
|
844
|
+
cellNode.metadata.colSpan = colSpan;
|
|
845
|
+
if (isVMerge) {
|
|
846
|
+
if (vMergeRestart) {
|
|
847
|
+
vMergeMap.set(visualCol, { node: cellNode, span: 1 });
|
|
848
|
+
cells.push(cellNode);
|
|
849
|
+
}
|
|
850
|
+
else {
|
|
851
|
+
const mergeInfo = vMergeMap.get(visualCol);
|
|
852
|
+
if (mergeInfo) {
|
|
853
|
+
mergeInfo.span++;
|
|
854
|
+
mergeInfo.node.metadata.rowSpan = mergeInfo.span;
|
|
855
|
+
if (cellChildren.length > 0) {
|
|
856
|
+
if (!mergeInfo.node.children)
|
|
857
|
+
mergeInfo.node.children = [];
|
|
858
|
+
mergeInfo.node.children.push(...cellChildren);
|
|
859
|
+
mergeInfo.node.text += " " + cellText;
|
|
860
|
+
}
|
|
861
|
+
}
|
|
862
|
+
else {
|
|
863
|
+
// Fallback: if we found a continue but no restart, treat as normal cell
|
|
864
|
+
cells.push(cellNode);
|
|
865
|
+
}
|
|
866
|
+
}
|
|
867
|
+
}
|
|
868
|
+
else {
|
|
869
|
+
vMergeMap.delete(visualCol);
|
|
870
|
+
cells.push(cellNode);
|
|
871
|
+
}
|
|
872
|
+
visualCol += colSpan;
|
|
744
873
|
}
|
|
745
874
|
const rowNode = {
|
|
746
875
|
type: 'row',
|
|
747
|
-
children: cells
|
|
876
|
+
children: cells,
|
|
748
877
|
};
|
|
749
878
|
rows.push(rowNode);
|
|
750
879
|
}
|
|
@@ -803,12 +932,23 @@ const parseWord = async (buffer, config) => {
|
|
|
803
932
|
const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
|
|
804
933
|
if (body) {
|
|
805
934
|
const bodyChildren = Array.from(body.childNodes);
|
|
935
|
+
let pendingAnchorIds = [];
|
|
806
936
|
for (const child of bodyChildren) {
|
|
807
|
-
if ((0, xmlUtils_js_1.isElement)(child)
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
937
|
+
if ((0, xmlUtils_js_1.isElement)(child)) {
|
|
938
|
+
if (child.nodeName === 'w:p') {
|
|
939
|
+
content.push(parseParagraph(child, documentContent, pendingAnchorIds));
|
|
940
|
+
pendingAnchorIds = [];
|
|
941
|
+
}
|
|
942
|
+
else if (child.nodeName === 'w:tbl') {
|
|
943
|
+
content.push(parseTable(child, documentContent, pendingAnchorIds));
|
|
944
|
+
pendingAnchorIds = [];
|
|
945
|
+
}
|
|
946
|
+
else if (child.nodeName === 'w:bookmarkStart') {
|
|
947
|
+
const bookmarkName = child.getAttribute("w:name");
|
|
948
|
+
if (bookmarkName && !bookmarkName.startsWith('_GoBack') && !config.ignoreInternalLinks) {
|
|
949
|
+
pendingAnchorIds.push(bookmarkName);
|
|
950
|
+
}
|
|
951
|
+
}
|
|
812
952
|
}
|
|
813
953
|
}
|
|
814
954
|
}
|
|
@@ -821,10 +961,10 @@ const parseWord = async (buffer, config) => {
|
|
|
821
961
|
if (config.ocr) {
|
|
822
962
|
if (attachment.mimeType.startsWith('image/')) {
|
|
823
963
|
try {
|
|
824
|
-
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, {
|
|
964
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
|
|
825
965
|
}
|
|
826
966
|
catch (e) {
|
|
827
|
-
(0, errorUtils_js_1.logWarning)(
|
|
967
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
|
|
828
968
|
}
|
|
829
969
|
}
|
|
830
970
|
}
|
|
@@ -852,27 +992,22 @@ const parseWord = async (buffer, config) => {
|
|
|
852
992
|
if (config.putNotesAtLast && collectedNotes.length > 0) {
|
|
853
993
|
content.push(...collectedNotes);
|
|
854
994
|
}
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
return t;
|
|
873
|
-
};
|
|
874
|
-
return getText(c);
|
|
875
|
-
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
876
|
-
};
|
|
995
|
+
const toTextSync = () => content.map(c => {
|
|
996
|
+
// Recursive text extraction
|
|
997
|
+
const getText = (node) => {
|
|
998
|
+
let t = '';
|
|
999
|
+
if (node.children) {
|
|
1000
|
+
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter);
|
|
1001
|
+
}
|
|
1002
|
+
else if (node.type === 'break') {
|
|
1003
|
+
t += config.newlineDelimiter;
|
|
1004
|
+
}
|
|
1005
|
+
else
|
|
1006
|
+
t += node.text || '';
|
|
1007
|
+
return t;
|
|
1008
|
+
};
|
|
1009
|
+
return getText(c);
|
|
1010
|
+
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
1011
|
+
return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, toTextSync);
|
|
877
1012
|
};
|
|
878
1013
|
exports.parseWord = parseWord;
|