officeparser 5.2.2 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,787 @@
1
+ "use strict";
2
+ /**
3
+ * Word Document (DOCX) Parser
4
+ *
5
+ * **DOCX Format Overview:**
6
+ * DOCX is the default format for Microsoft Word documents since Office 2007.
7
+ * It's based on the Office Open XML (OOXML) standard (ECMA-376, ISO/IEC 29500).
8
+ *
9
+ * **File Structure:**
10
+ * DOCX files are ZIP archives containing:
11
+ * - `word/document.xml` - Main document content
12
+ * - `word/styles.xml` - Style definitions
13
+ * - `word/numbering.xml` - List numbering definitions
14
+ * - `word/footnotes.xml` - Footnotes content
15
+ * - `word/media/*` - Embedded images and media
16
+ * - `docProps/core.xml` - Document metadata
17
+ * - `[Content_Types].xml` - MIME type mappings
18
+ *
19
+ * **XML Structure (word/document.xml):**
20
+ * ```xml
21
+ * <w:document>
22
+ * <w:body>
23
+ * <w:p> <!-- Paragraph -->
24
+ * <w:pPr> <!-- Paragraph properties -->
25
+ * <w:pStyle w:val="Heading1"/>
26
+ * </w:pPr>
27
+ * <w:r> <!-- Run (text with same formatting) -->
28
+ * <w:rPr> <!-- Run properties -->
29
+ * <w:b/> <!-- Bold -->
30
+ * <w:sz w:val="24"/> <!-- Font size (half-points) -->
31
+ * </w:rPr>
32
+ * <w:t>Hello</w:t> <!-- Text -->
33
+ * </w:r>
34
+ * </w:p>
35
+ * </w:body>
36
+ * </w:document>
37
+ * ```
38
+ *
39
+ * **Key OOXML Elements:**
40
+ * - `<w:p>` - Paragraph
41
+ * - `<w:r>` - Run (contiguous text with same formatting)
42
+ * - `<w:t>` - Text content
43
+ * - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
44
+ * - `<w:pStyle>` - Paragraph style (for headings)
45
+ * - `<w:numPr>` - List numbering properties
46
+ * - `<w:tbl>` - Table
47
+ * - `<w:drawing>` - Drawing/image
48
+ *
49
+ * **Parsing Approach:**
50
+ * 1. Extract ZIP contents
51
+ * 2. Parse word/document.xml for structure and text
52
+ * 3. Extract formatting from run properties (rPr)
53
+ * 4. Identify headings via paragraph styles
54
+ * 5. Extract footnotes from word/footnotes.xml
55
+ * 6. Process embedded images from word/media/*
56
+ * 7. Parse metadata from docProps/core.xml
57
+ *
58
+ * @module WordParser
59
+ * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
60
+ * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
61
+ */
62
+ Object.defineProperty(exports, "__esModule", { value: true });
63
+ exports.parseWord = void 0;
64
+ const xmldom_1 = require("@xmldom/xmldom");
65
+ const errorUtils_1 = require("../utils/errorUtils");
66
+ const imageUtils_1 = require("../utils/imageUtils");
67
+ const ocrUtils_1 = require("../utils/ocrUtils");
68
+ const xmlUtils_1 = require("../utils/xmlUtils");
69
+ const zipUtils_1 = require("../utils/zipUtils");
70
+ /**
71
+ * Parses a Word document (.docx) and extracts content, formatting, and metadata.
72
+ *
73
+ * The parsing process:
74
+ * 1. Unzip the DOCX file
75
+ * 2. Parse word/document.xml to extract paragraphs and runs
76
+ * 3. Extract text formatting from run properties
77
+ * 4. Identify headings from paragraph styles
78
+ * 5. Process lists from numbering properties
79
+ * 6. Extract images and optionally perform OCR
80
+ * 7. Parse document metadata
81
+ *
82
+ * @param buffer - The DOCX file as a Buffer
83
+ * @param config - Parser configuration options
84
+ * @returns A promise resolving to the parsed AST
85
+ */
86
+ const parseWord = async (buffer, config) => {
87
+ const documentFileRegex = /word\/document[\d+]?.xml/;
88
+ const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
89
+ const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
90
+ const numberingFileRegex = /word\/numbering[\d+]?.xml/;
91
+ const mediaFileRegex = /(word\/)?media\/.*/;
92
+ const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
93
+ const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
94
+ const stylesFileRegex = /word\/styles[\d+]?.xml/;
95
+ const xmlSerializer = new xmldom_1.XMLSerializer();
96
+ // Helper to extract formatting from run properties XML string
97
+ const extractFormattingFromXml = (rPr) => {
98
+ const formatting = {};
99
+ const rPrString = xmlSerializer.serializeToString(rPr);
100
+ // Helper to check boolean properties
101
+ const getBoolVal = (xmlSnippet, tagName) => {
102
+ const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
103
+ const match = xmlSnippet.match(regex);
104
+ if (match) {
105
+ const val = match[1];
106
+ if (val === undefined)
107
+ return true;
108
+ return val === '1' || val === 'true' || val === 'on';
109
+ }
110
+ return null;
111
+ };
112
+ const bold = getBoolVal(rPrString, 'w:b');
113
+ if (bold !== null)
114
+ formatting.bold = bold;
115
+ const italic = getBoolVal(rPrString, 'w:i');
116
+ if (italic !== null)
117
+ formatting.italic = italic;
118
+ const underlineMatch = rPrString.match(/<w:u(?: w:val="([^"]+)")?\/?>/);
119
+ if (underlineMatch) {
120
+ const val = underlineMatch[1];
121
+ // If val is missing, it's a default underline (true).
122
+ // If val is present, it's true unless explicit 'none'.
123
+ if (!val || val !== 'none') {
124
+ formatting.underline = true;
125
+ }
126
+ }
127
+ const strike = getBoolVal(rPrString, 'w:strike');
128
+ const dstrike = getBoolVal(rPrString, 'w:dstrike');
129
+ if (strike !== null)
130
+ formatting.strikethrough = strike;
131
+ else if (dstrike !== null)
132
+ formatting.strikethrough = dstrike;
133
+ // Font size
134
+ const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
135
+ if (szMatch)
136
+ formatting.size = (parseInt(szMatch[1]) / 2).toString() + 'pt';
137
+ // Color
138
+ const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
139
+ if (colorMatch && colorMatch[1] !== 'auto')
140
+ formatting.color = '#' + colorMatch[1];
141
+ // Background color (shading)
142
+ const shdMatch = rPrString.match(/<w:shd[^>]*w:fill="([^"]+)"/);
143
+ if (shdMatch && shdMatch[1] !== 'auto')
144
+ formatting.backgroundColor = '#' + shdMatch[1];
145
+ // Highlight (map to backgroundColor)
146
+ const highlightMatch = rPrString.match(/<w:highlight w:val="([^"]+)"/);
147
+ if (highlightMatch && highlightMatch[1] !== 'none') {
148
+ const colorMap = {
149
+ 'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
150
+ 'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
151
+ 'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
152
+ 'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
153
+ };
154
+ formatting.backgroundColor = colorMap[highlightMatch[1]] || highlightMatch[1];
155
+ }
156
+ // Font family
157
+ const rFontsMatch = rPrString.match(/<w:rFonts[^>]*w:ascii="([^"]+)"/);
158
+ if (rFontsMatch) {
159
+ formatting.font = rFontsMatch[1];
160
+ }
161
+ else {
162
+ const hAnsiMatch = rPrString.match(/<w:rFonts[^>]*w:hAnsi="([^"]+)"/);
163
+ if (hAnsiMatch)
164
+ formatting.font = hAnsiMatch[1];
165
+ }
166
+ // Subscript/Superscript
167
+ const vertAlignMatch = rPrString.match(/<w:vertAlign w:val="([^"]+)"/);
168
+ if (vertAlignMatch) {
169
+ if (vertAlignMatch[1] === 'subscript')
170
+ formatting.subscript = true;
171
+ if (vertAlignMatch[1] === 'superscript')
172
+ formatting.superscript = true;
173
+ }
174
+ return formatting;
175
+ };
176
+ const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
177
+ !!x.match(footnotesFileRegex) ||
178
+ !!x.match(endnotesFileRegex) ||
179
+ !!x.match(numberingFileRegex) ||
180
+ !!x.match(corePropsFileRegex) ||
181
+ !!x.match(relsFileRegex) ||
182
+ !!x.match(stylesFileRegex) ||
183
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)));
184
+ // Extract metadata
185
+ const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
186
+ const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
187
+ const footnoteMap = new Map();
188
+ const endnoteMap = new Map();
189
+ const collectedNotes = [];
190
+ const attachments = [];
191
+ const mediaFiles = files.filter(f => f.path.match(mediaFileRegex));
192
+ // Extract relationships
193
+ const relsFile = files.find(f => f.path.match(relsFileRegex));
194
+ const relsMap = {};
195
+ if (relsFile) {
196
+ const relsXml = (0, xmlUtils_1.parseXmlString)(relsFile.content.toString());
197
+ const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
198
+ for (let i = 0; i < relationships.length; i++) {
199
+ const id = relationships[i].getAttribute("Id");
200
+ const target = relationships[i].getAttribute("Target");
201
+ if (id && target) {
202
+ relsMap[id] = target;
203
+ }
204
+ }
205
+ }
206
+ const numberingFile = files.find(f => f.path.match(numberingFileRegex));
207
+ const numberingMap = {};
208
+ if (numberingFile) {
209
+ const numberingXml = (0, xmlUtils_1.parseXmlString)(numberingFile.content.toString());
210
+ const nums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:num");
211
+ const abstractNums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:abstractNum");
212
+ const abstractNumMap = {};
213
+ for (let i = 0; i < abstractNums.length; i++) {
214
+ const abstractNumId = abstractNums[i].getAttribute("w:abstractNumId");
215
+ if (abstractNumId) {
216
+ abstractNumMap[abstractNumId] = abstractNums[i];
217
+ }
218
+ }
219
+ for (let i = 0; i < nums.length; i++) {
220
+ const numId = nums[i].getAttribute("w:numId");
221
+ const abstractNumIdNode = (0, xmlUtils_1.getElementsByTagName)(nums[i], "w:abstractNumId")[0];
222
+ const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
223
+ if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
224
+ numberingMap[numId] = {};
225
+ const lvls = (0, xmlUtils_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
226
+ for (let j = 0; j < lvls.length; j++) {
227
+ const ilvl = lvls[j].getAttribute("w:ilvl");
228
+ const numFmtNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:numFmt")[0];
229
+ const lvlTextNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:lvlText")[0];
230
+ if (ilvl) {
231
+ numberingMap[numId][ilvl] = {
232
+ numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
233
+ lvlText: lvlTextNode?.getAttribute("w:val") || ''
234
+ };
235
+ }
236
+ }
237
+ }
238
+ }
239
+ }
240
+ // Parse Styles
241
+ const stylesFile = files.find(f => f.path.match(stylesFileRegex));
242
+ const styleMap = {};
243
+ if (stylesFile) {
244
+ const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
245
+ const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
246
+ for (let i = 0; i < styles.length; i++) {
247
+ const styleId = styles[i].getAttribute("w:styleId");
248
+ if (styleId) {
249
+ const rPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:rPr")[0];
250
+ const pPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:pPr")[0];
251
+ const formatting = rPr ? extractFormattingFromXml(rPr) : {};
252
+ let alignment = undefined;
253
+ let backgroundColor = undefined;
254
+ if (pPr) {
255
+ const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
256
+ if (jc) {
257
+ const val = jc.getAttribute("w:val");
258
+ if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
259
+ alignment = val;
260
+ }
261
+ }
262
+ const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
263
+ if (shd) {
264
+ const fill = shd.getAttribute("w:fill");
265
+ if (fill && fill !== 'auto')
266
+ backgroundColor = '#' + fill;
267
+ }
268
+ }
269
+ styleMap[styleId] = { formatting, alignment, backgroundColor };
270
+ }
271
+ }
272
+ }
273
+ // Extract document defaults
274
+ let docDefaults = {};
275
+ if (stylesFile) {
276
+ const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
277
+ const docDefaultsNode = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:docDefaults")[0];
278
+ if (docDefaultsNode) {
279
+ const rPrDefaultNode = (0, xmlUtils_1.getElementsByTagName)(docDefaultsNode, "w:rPrDefault")[0];
280
+ if (rPrDefaultNode) {
281
+ const rPr = (0, xmlUtils_1.getElementsByTagName)(rPrDefaultNode, "w:rPr")[0];
282
+ if (rPr) {
283
+ docDefaults = extractFormattingFromXml(rPr);
284
+ }
285
+ }
286
+ }
287
+ }
288
+ // Detect the default paragraph style (for international compatibility)
289
+ let defaultParaStyleId = undefined;
290
+ if (stylesFile) {
291
+ const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
292
+ const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
293
+ // Look for a style with w:type="paragraph" and w:default="1"
294
+ for (let i = 0; i < styles.length; i++) {
295
+ const styleType = styles[i].getAttribute("w:type");
296
+ const isDefault = styles[i].getAttribute("w:default");
297
+ const styleId = styles[i].getAttribute("w:styleId");
298
+ if (styleType === "paragraph" && isDefault === "1" && styleId) {
299
+ defaultParaStyleId = styleId;
300
+ break;
301
+ }
302
+ }
303
+ // Fallback: if no default found, try "Normal"
304
+ if (!defaultParaStyleId && styleMap["Normal"]) {
305
+ defaultParaStyleId = "Normal";
306
+ }
307
+ }
308
+ const content = [];
309
+ const rawContents = [];
310
+ const numberingState = {};
311
+ const listCounters = {}; // Track item index per listId/level
312
+ // Helper to parse a paragraph node
313
+ const parseParagraph = (pNode) => {
314
+ const pXml = xmlSerializer.serializeToString(pNode);
315
+ // Check if it's a list item
316
+ const numPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:numPr")[0];
317
+ const isList = !!numPr;
318
+ // Check if it's a heading
319
+ const pPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:pPr")[0];
320
+ const pStyle = pPr ? (0, xmlUtils_1.getElementsByTagName)(pPr, "w:pStyle")[0] : null;
321
+ const pStyleVal = pStyle ? pStyle.getAttribute("w:val") : null;
322
+ const isHeading = pStyleVal ? (pStyleVal.startsWith("Heading") || pStyleVal === "Title") : false;
323
+ // Extract Paragraph Style Properties
324
+ const styleProps = pStyleVal && styleMap[pStyleVal] ? styleMap[pStyleVal] : { formatting: {} };
325
+ // Extract Alignment
326
+ let alignment = styleProps.alignment;
327
+ if (pPr) {
328
+ const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
329
+ if (jc) {
330
+ const val = jc.getAttribute("w:val");
331
+ if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
332
+ alignment = val;
333
+ }
334
+ }
335
+ }
336
+ // Extract Paragraph Background
337
+ let paraBackgroundColor = styleProps.backgroundColor;
338
+ if (pPr) {
339
+ const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
340
+ if (shd) {
341
+ const fill = shd.getAttribute("w:fill");
342
+ if (fill && fill !== 'auto') {
343
+ paraBackgroundColor = '#' + fill;
344
+ }
345
+ }
346
+ }
347
+ // Extract paragraph-level run properties
348
+ let paragraphRunFormatting = { ...styleProps.formatting };
349
+ if (pPr) {
350
+ const pPrRPr = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:rPr")[0];
351
+ if (pPrRPr) {
352
+ const pPrFormatting = extractFormattingFromXml(pPrRPr);
353
+ for (const key in pPrFormatting) {
354
+ const value = pPrFormatting[key];
355
+ if (value === false) {
356
+ delete paragraphRunFormatting[key];
357
+ }
358
+ else if (value !== undefined) {
359
+ paragraphRunFormatting[key] = value;
360
+ }
361
+ }
362
+ }
363
+ }
364
+ // Extract text and children
365
+ let text = '';
366
+ const children = [];
367
+ // Traverse children of paragraph (runs, hyperlinks, etc.)
368
+ const processChildNode = (node) => {
369
+ if (node.nodeName === 'w:r') {
370
+ const runNode = node;
371
+ const rPr = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:rPr")[0];
372
+ // Formatting
373
+ let formatting = {};
374
+ // Apply paragraph-level formatting
375
+ for (const key in paragraphRunFormatting) {
376
+ formatting[key] = paragraphRunFormatting[key];
377
+ }
378
+ // Check for run style
379
+ const rStyle = rPr ? (0, xmlUtils_1.getElementsByTagName)(rPr, "w:rStyle")[0] : null;
380
+ const rStyleVal = rStyle ? rStyle.getAttribute("w:val") : pStyleVal;
381
+ if (rStyleVal && styleMap[rStyleVal]) {
382
+ for (const key in styleMap[rStyleVal].formatting) {
383
+ formatting[key] = styleMap[rStyleVal].formatting[key];
384
+ }
385
+ }
386
+ // Apply direct run properties
387
+ if (rPr) {
388
+ const directFormatting = extractFormattingFromXml(rPr);
389
+ for (const key in directFormatting) {
390
+ const value = directFormatting[key];
391
+ if (value === false) {
392
+ delete formatting[key];
393
+ }
394
+ else if (value !== undefined) {
395
+ formatting[key] = value;
396
+ }
397
+ }
398
+ }
399
+ // Inherit paragraph background
400
+ if (!formatting.backgroundColor && paraBackgroundColor) {
401
+ formatting.backgroundColor = paraBackgroundColor;
402
+ }
403
+ // Text content
404
+ const tNodes = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:t");
405
+ for (const tNode of tNodes) {
406
+ const tContent = tNode.textContent || '';
407
+ text += tContent;
408
+ const textNode = {
409
+ type: 'text',
410
+ text: tContent,
411
+ formatting: formatting
412
+ };
413
+ if (config.includeRawContent) {
414
+ textNode.rawContent = xmlSerializer.serializeToString(tNode);
415
+ }
416
+ // Always set a style: run style > paragraph style > detected default
417
+ // Use detected default style for international compatibility
418
+ const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
419
+ if (nodeStyle) {
420
+ textNode.metadata = { style: nodeStyle };
421
+ }
422
+ children.push(textNode);
423
+ }
424
+ // Images/Drawings
425
+ if (config.extractAttachments) {
426
+ const drawings = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:drawing");
427
+ const picts = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:pict");
428
+ const allImages = [...drawings, ...picts];
429
+ for (const imgNode of allImages) {
430
+ const imgXml = xmlSerializer.serializeToString(imgNode);
431
+ // Extract Alt Text
432
+ let altText = '';
433
+ const docPr = (0, xmlUtils_1.getElementsByTagName)(imgNode, "wp:docPr")[0];
434
+ if (docPr) {
435
+ altText = docPr.getAttribute("descr") || docPr.getAttribute("title") || '';
436
+ }
437
+ // Extract Relationship ID
438
+ let rId = '';
439
+ const blip = (0, xmlUtils_1.getElementsByTagName)(imgNode, "a:blip")[0];
440
+ if (blip) {
441
+ rId = blip.getAttribute("r:embed") || '';
442
+ }
443
+ else {
444
+ const imagedata = (0, xmlUtils_1.getElementsByTagName)(imgNode, "v:imagedata")[0];
445
+ if (imagedata) {
446
+ rId = imagedata.getAttribute("r:id") || '';
447
+ }
448
+ }
449
+ if (rId && relsMap[rId]) {
450
+ const target = relsMap[rId];
451
+ const filename = target.split('/').pop();
452
+ if (filename) {
453
+ const imageNode = {
454
+ type: 'image',
455
+ text: '',
456
+ metadata: { attachmentName: filename, altText: altText }
457
+ };
458
+ if (config.includeRawContent) {
459
+ imageNode.rawContent = imgXml;
460
+ }
461
+ children.push(imageNode);
462
+ }
463
+ }
464
+ else {
465
+ const imageNode = {
466
+ type: 'image',
467
+ text: '',
468
+ };
469
+ if (config.includeRawContent) {
470
+ imageNode.rawContent = imgXml;
471
+ }
472
+ children.push(imageNode);
473
+ }
474
+ }
475
+ }
476
+ // Footnotes/Endnotes inside runs
477
+ if (!config.ignoreNotes) {
478
+ const footnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:footnoteReference")[0];
479
+ if (footnoteRef) {
480
+ const id = footnoteRef.getAttribute("w:id");
481
+ if (id && footnoteMap.has(id)) {
482
+ const noteNodes = footnoteMap.get(id);
483
+ const noteNode = {
484
+ type: 'note',
485
+ text: noteNodes.map((n) => n.text).join(' '),
486
+ children: noteNodes,
487
+ metadata: { noteType: 'footnote', noteId: id }
488
+ };
489
+ if (config.putNotesAtLast) {
490
+ collectedNotes.push(noteNode);
491
+ }
492
+ else {
493
+ children.push(noteNode);
494
+ }
495
+ }
496
+ }
497
+ const endnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:endnoteReference")[0];
498
+ if (endnoteRef) {
499
+ const id = endnoteRef.getAttribute("w:id");
500
+ if (id && endnoteMap.has(id)) {
501
+ const noteNodes = endnoteMap.get(id);
502
+ const noteNode = {
503
+ type: 'note',
504
+ text: noteNodes.map((n) => n.text).join(' '),
505
+ children: noteNodes,
506
+ metadata: { noteType: 'endnote', noteId: id }
507
+ };
508
+ if (config.putNotesAtLast) {
509
+ collectedNotes.push(noteNode);
510
+ }
511
+ else {
512
+ children.push(noteNode);
513
+ }
514
+ }
515
+ }
516
+ }
517
+ }
518
+ else if (node.nodeName === 'w:hyperlink') {
519
+ const hlNode = node;
520
+ const rId = hlNode.getAttribute("r:id");
521
+ const anchor = hlNode.getAttribute("w:anchor");
522
+ let linkMetadata;
523
+ if (anchor) {
524
+ linkMetadata = { link: '#' + anchor, linkType: 'internal' };
525
+ }
526
+ else if (rId && relsMap[rId]) {
527
+ linkMetadata = { link: relsMap[rId], linkType: 'external' };
528
+ }
529
+ // Process children of hyperlink (usually runs)
530
+ const hlChildren = Array.from(hlNode.childNodes);
531
+ for (const child of hlChildren) {
532
+ // Capture the current length of children to apply metadata to new nodes
533
+ const startIndex = children.length;
534
+ processChildNode(child);
535
+ // Apply link metadata to the newly added text nodes
536
+ if (linkMetadata) {
537
+ for (let i = startIndex; i < children.length; i++) {
538
+ if (children[i].type === 'text') {
539
+ children[i].metadata = { ...(children[i].metadata ?? {}), ...linkMetadata };
540
+ }
541
+ }
542
+ }
543
+ }
544
+ }
545
+ };
546
+ const childNodes = Array.from(pNode.childNodes);
547
+ for (const child of childNodes) {
548
+ processChildNode(child);
549
+ }
550
+ if (isList) {
551
+ const numIdNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:numId")[0];
552
+ const ilvlNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:ilvl")[0];
553
+ const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
554
+ const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
555
+ let listType = 'ordered';
556
+ let itemIndex = 0;
557
+ if (numId && numberingMap[numId]) {
558
+ const ilvlStr = ilvl.toString();
559
+ if (!numberingState[numId])
560
+ numberingState[numId] = {};
561
+ if (!numberingState[numId][ilvlStr])
562
+ numberingState[numId][ilvlStr] = 0;
563
+ numberingState[numId][ilvlStr]++;
564
+ for (let k = ilvl + 1; k < 10; k++) {
565
+ if (numberingState[numId][k.toString()])
566
+ numberingState[numId][k.toString()] = 0;
567
+ }
568
+ const numFmt = numberingMap[numId][ilvlStr]?.numFmt || 'decimal';
569
+ listType = numFmt === 'bullet' ? 'unordered' : 'ordered';
570
+ // Track itemIndex (starts at 0, continues across interruptions for same listId)
571
+ if (!listCounters[numId])
572
+ listCounters[numId] = {};
573
+ if (listCounters[numId][ilvlStr] === undefined) {
574
+ listCounters[numId][ilvlStr] = 0;
575
+ }
576
+ else {
577
+ listCounters[numId][ilvlStr]++;
578
+ }
579
+ itemIndex = listCounters[numId][ilvlStr];
580
+ }
581
+ const listNode = {
582
+ type: 'list',
583
+ text: text,
584
+ children: children,
585
+ metadata: {
586
+ listType,
587
+ indentation: ilvl,
588
+ alignment: (alignment || 'left'),
589
+ listId: numId,
590
+ itemIndex: itemIndex,
591
+ style: pStyleVal
592
+ }
593
+ };
594
+ if (config.includeRawContent)
595
+ listNode.rawContent = pXml;
596
+ return listNode;
597
+ }
598
+ else if (isHeading) {
599
+ const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", "")) || 1 : 1;
600
+ const headingNode = {
601
+ type: 'heading',
602
+ text: text,
603
+ children: children,
604
+ metadata: { level, alignment, style: pStyleVal ?? undefined }
605
+ };
606
+ if (config.includeRawContent)
607
+ headingNode.rawContent = pXml;
608
+ return headingNode;
609
+ }
610
+ else {
611
+ const paraNode = {
612
+ type: 'paragraph',
613
+ text: text,
614
+ children: children,
615
+ metadata: { alignment, style: pStyleVal ?? undefined }
616
+ };
617
+ if (config.includeRawContent)
618
+ paraNode.rawContent = pXml;
619
+ return paraNode;
620
+ }
621
+ };
622
+ // Helper to parse a table node
623
+ const parseTable = (tblNode) => {
624
+ const rows = [];
625
+ // Only get direct child rows, not nested table rows
626
+ const trNodes = (0, xmlUtils_1.getDirectChildren)(tblNode, "w:tr");
627
+ for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
628
+ const trNode = trNodes[rIndex];
629
+ const cells = [];
630
+ // Only get direct child cells, not nested table cells
631
+ const tcNodes = (0, xmlUtils_1.getDirectChildren)(trNode, "w:tc");
632
+ for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
633
+ const tcNode = tcNodes[cIndex];
634
+ const cellChildren = [];
635
+ let cellText = '';
636
+ // Cells contain paragraphs (and other block-level elements)
637
+ const cellContentNodes = Array.from(tcNode.childNodes);
638
+ for (const child of cellContentNodes) {
639
+ if (child.nodeName === 'w:p') {
640
+ const pNode = parseParagraph(child);
641
+ cellChildren.push(pNode);
642
+ cellText += pNode.text;
643
+ }
644
+ else if (child.nodeName === 'w:tbl') {
645
+ // Nested table
646
+ const nestedTable = parseTable(child);
647
+ cellChildren.push(nestedTable);
648
+ // Don't add nested table text to cell text - it will be handled recursively
649
+ }
650
+ }
651
+ const cellNode = {
652
+ type: 'cell',
653
+ text: cellText,
654
+ children: cellChildren,
655
+ metadata: { row: rIndex, col: cIndex }
656
+ };
657
+ cells.push(cellNode);
658
+ }
659
+ const rowNode = {
660
+ type: 'row',
661
+ children: cells
662
+ };
663
+ rows.push(rowNode);
664
+ }
665
+ return {
666
+ type: 'table',
667
+ children: rows
668
+ };
669
+ };
670
+ // Pre-process footnotes and endnotes to be inserted inline later
671
+ if (!config.ignoreNotes) {
672
+ const footnotesFile = files.find(f => f.path.match(footnotesFileRegex));
673
+ if (footnotesFile) {
674
+ const footnotesDoc = (0, xmlUtils_1.parseXmlString)(footnotesFile.content.toString());
675
+ const footnoteNodes = (0, xmlUtils_1.getElementsByTagName)(footnotesDoc, "w:footnote");
676
+ for (const node of footnoteNodes) {
677
+ const id = node.getAttribute("w:id");
678
+ if (!id || id === "-1" || id === "0")
679
+ continue;
680
+ const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
681
+ footnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
682
+ }
683
+ }
684
+ const endnotesFile = files.find(f => f.path.match(endnotesFileRegex));
685
+ if (endnotesFile) {
686
+ const endnotesDoc = (0, xmlUtils_1.parseXmlString)(endnotesFile.content.toString());
687
+ const endnoteNodes = (0, xmlUtils_1.getElementsByTagName)(endnotesDoc, "w:endnote");
688
+ for (const node of endnoteNodes) {
689
+ const id = node.getAttribute("w:id");
690
+ if (!id || id === "-1" || id === "0")
691
+ continue;
692
+ const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
693
+ endnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
694
+ }
695
+ }
696
+ }
697
+ for (const file of files) {
698
+ if (file.path.match(mediaFileRegex))
699
+ continue;
700
+ if (file.path.match(numberingFileRegex))
701
+ continue;
702
+ if (file.path.match(relsFileRegex))
703
+ continue;
704
+ if (file.path.match(stylesFileRegex))
705
+ continue;
706
+ if (file.path.match(footnotesFileRegex))
707
+ continue;
708
+ if (file.path.match(endnotesFileRegex))
709
+ continue;
710
+ const documentContent = file.content.toString();
711
+ if (config.includeRawContent) {
712
+ rawContents.push(documentContent);
713
+ }
714
+ const doc = (0, xmlUtils_1.parseXmlString)(documentContent);
715
+ const body = (0, xmlUtils_1.getElementsByTagName)(doc, "w:body")[0];
716
+ if (body) {
717
+ const bodyChildren = Array.from(body.childNodes);
718
+ for (const child of bodyChildren) {
719
+ if (child.nodeName === 'w:p') {
720
+ content.push(parseParagraph(child));
721
+ }
722
+ else if (child.nodeName === 'w:tbl') {
723
+ content.push(parseTable(child));
724
+ }
725
+ }
726
+ }
727
+ }
728
+ // Extract attachments
729
+ if (config.extractAttachments) {
730
+ for (const media of mediaFiles) {
731
+ const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
732
+ attachments.push(attachment);
733
+ if (config.ocr) {
734
+ if (attachment.mimeType.startsWith('image/')) {
735
+ try {
736
+ attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
737
+ }
738
+ catch (e) {
739
+ (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
740
+ }
741
+ }
742
+ }
743
+ }
744
+ // Assign OCR text to image nodes
745
+ if (config.ocr) {
746
+ const assignOcr = (nodes) => {
747
+ for (const node of nodes) {
748
+ if (node.type === 'image' && 'attachmentName' in (node.metadata || {})) {
749
+ const meta = node.metadata;
750
+ const attachment = attachments.find(a => a.name === meta.attachmentName);
751
+ if (attachment && attachment.ocrText) {
752
+ node.text = attachment.ocrText;
753
+ attachment.altText = meta.altText;
754
+ }
755
+ }
756
+ if (node.children) {
757
+ assignOcr(node.children);
758
+ }
759
+ }
760
+ };
761
+ assignOcr(content);
762
+ }
763
+ }
764
+ if (config.putNotesAtLast && collectedNotes.length > 0) {
765
+ content.push(...collectedNotes);
766
+ }
767
+ return {
768
+ type: 'docx',
769
+ metadata: { ...metadata, formatting: docDefaults, styleMap: styleMap },
770
+ content: content,
771
+ attachments: attachments,
772
+ toText: () => content.map(c => {
773
+ // Recursive text extraction
774
+ const getText = (node) => {
775
+ let t = '';
776
+ if (node.children) {
777
+ t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
778
+ }
779
+ else
780
+ t += node.text || '';
781
+ return t;
782
+ };
783
+ return getText(c);
784
+ }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
785
+ };
786
+ };
787
+ exports.parseWord = parseWord;