officeparser 5.2.2 → 6.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +165 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +847 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -790
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,787 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Word Document (DOCX) Parser
|
|
4
|
+
*
|
|
5
|
+
* **DOCX Format Overview:**
|
|
6
|
+
* DOCX is the default format for Microsoft Word documents since Office 2007.
|
|
7
|
+
* It's based on the Office Open XML (OOXML) standard (ECMA-376, ISO/IEC 29500).
|
|
8
|
+
*
|
|
9
|
+
* **File Structure:**
|
|
10
|
+
* DOCX files are ZIP archives containing:
|
|
11
|
+
* - `word/document.xml` - Main document content
|
|
12
|
+
* - `word/styles.xml` - Style definitions
|
|
13
|
+
* - `word/numbering.xml` - List numbering definitions
|
|
14
|
+
* - `word/footnotes.xml` - Footnotes content
|
|
15
|
+
* - `word/media/*` - Embedded images and media
|
|
16
|
+
* - `docProps/core.xml` - Document metadata
|
|
17
|
+
* - `[Content_Types].xml` - MIME type mappings
|
|
18
|
+
*
|
|
19
|
+
* **XML Structure (word/document.xml):**
|
|
20
|
+
* ```xml
|
|
21
|
+
* <w:document>
|
|
22
|
+
* <w:body>
|
|
23
|
+
* <w:p> <!-- Paragraph -->
|
|
24
|
+
* <w:pPr> <!-- Paragraph properties -->
|
|
25
|
+
* <w:pStyle w:val="Heading1"/>
|
|
26
|
+
* </w:pPr>
|
|
27
|
+
* <w:r> <!-- Run (text with same formatting) -->
|
|
28
|
+
* <w:rPr> <!-- Run properties -->
|
|
29
|
+
* <w:b/> <!-- Bold -->
|
|
30
|
+
* <w:sz w:val="24"/> <!-- Font size (half-points) -->
|
|
31
|
+
* </w:rPr>
|
|
32
|
+
* <w:t>Hello</w:t> <!-- Text -->
|
|
33
|
+
* </w:r>
|
|
34
|
+
* </w:p>
|
|
35
|
+
* </w:body>
|
|
36
|
+
* </w:document>
|
|
37
|
+
* ```
|
|
38
|
+
*
|
|
39
|
+
* **Key OOXML Elements:**
|
|
40
|
+
* - `<w:p>` - Paragraph
|
|
41
|
+
* - `<w:r>` - Run (contiguous text with same formatting)
|
|
42
|
+
* - `<w:t>` - Text content
|
|
43
|
+
* - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
|
|
44
|
+
* - `<w:pStyle>` - Paragraph style (for headings)
|
|
45
|
+
* - `<w:numPr>` - List numbering properties
|
|
46
|
+
* - `<w:tbl>` - Table
|
|
47
|
+
* - `<w:drawing>` - Drawing/image
|
|
48
|
+
*
|
|
49
|
+
* **Parsing Approach:**
|
|
50
|
+
* 1. Extract ZIP contents
|
|
51
|
+
* 2. Parse word/document.xml for structure and text
|
|
52
|
+
* 3. Extract formatting from run properties (rPr)
|
|
53
|
+
* 4. Identify headings via paragraph styles
|
|
54
|
+
* 5. Extract footnotes from word/footnotes.xml
|
|
55
|
+
* 6. Process embedded images from word/media/*
|
|
56
|
+
* 7. Parse metadata from docProps/core.xml
|
|
57
|
+
*
|
|
58
|
+
* @module WordParser
|
|
59
|
+
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
|
|
60
|
+
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
|
|
61
|
+
*/
|
|
62
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
63
|
+
exports.parseWord = void 0;
|
|
64
|
+
const xmldom_1 = require("@xmldom/xmldom");
|
|
65
|
+
const errorUtils_1 = require("../utils/errorUtils");
|
|
66
|
+
const imageUtils_1 = require("../utils/imageUtils");
|
|
67
|
+
const ocrUtils_1 = require("../utils/ocrUtils");
|
|
68
|
+
const xmlUtils_1 = require("../utils/xmlUtils");
|
|
69
|
+
const zipUtils_1 = require("../utils/zipUtils");
|
|
70
|
+
/**
|
|
71
|
+
* Parses a Word document (.docx) and extracts content, formatting, and metadata.
|
|
72
|
+
*
|
|
73
|
+
* The parsing process:
|
|
74
|
+
* 1. Unzip the DOCX file
|
|
75
|
+
* 2. Parse word/document.xml to extract paragraphs and runs
|
|
76
|
+
* 3. Extract text formatting from run properties
|
|
77
|
+
* 4. Identify headings from paragraph styles
|
|
78
|
+
* 5. Process lists from numbering properties
|
|
79
|
+
* 6. Extract images and optionally perform OCR
|
|
80
|
+
* 7. Parse document metadata
|
|
81
|
+
*
|
|
82
|
+
* @param buffer - The DOCX file as a Buffer
|
|
83
|
+
* @param config - Parser configuration options
|
|
84
|
+
* @returns A promise resolving to the parsed AST
|
|
85
|
+
*/
|
|
86
|
+
const parseWord = async (buffer, config) => {
|
|
87
|
+
const documentFileRegex = /word\/document[\d+]?.xml/;
|
|
88
|
+
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
|
|
89
|
+
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
|
|
90
|
+
const numberingFileRegex = /word\/numbering[\d+]?.xml/;
|
|
91
|
+
const mediaFileRegex = /(word\/)?media\/.*/;
|
|
92
|
+
const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
|
|
93
|
+
const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
|
|
94
|
+
const stylesFileRegex = /word\/styles[\d+]?.xml/;
|
|
95
|
+
const xmlSerializer = new xmldom_1.XMLSerializer();
|
|
96
|
+
// Helper to extract formatting from run properties XML string
|
|
97
|
+
const extractFormattingFromXml = (rPr) => {
|
|
98
|
+
const formatting = {};
|
|
99
|
+
const rPrString = xmlSerializer.serializeToString(rPr);
|
|
100
|
+
// Helper to check boolean properties
|
|
101
|
+
const getBoolVal = (xmlSnippet, tagName) => {
|
|
102
|
+
const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
|
|
103
|
+
const match = xmlSnippet.match(regex);
|
|
104
|
+
if (match) {
|
|
105
|
+
const val = match[1];
|
|
106
|
+
if (val === undefined)
|
|
107
|
+
return true;
|
|
108
|
+
return val === '1' || val === 'true' || val === 'on';
|
|
109
|
+
}
|
|
110
|
+
return null;
|
|
111
|
+
};
|
|
112
|
+
const bold = getBoolVal(rPrString, 'w:b');
|
|
113
|
+
if (bold !== null)
|
|
114
|
+
formatting.bold = bold;
|
|
115
|
+
const italic = getBoolVal(rPrString, 'w:i');
|
|
116
|
+
if (italic !== null)
|
|
117
|
+
formatting.italic = italic;
|
|
118
|
+
const underlineMatch = rPrString.match(/<w:u(?: w:val="([^"]+)")?\/?>/);
|
|
119
|
+
if (underlineMatch) {
|
|
120
|
+
const val = underlineMatch[1];
|
|
121
|
+
// If val is missing, it's a default underline (true).
|
|
122
|
+
// If val is present, it's true unless explicit 'none'.
|
|
123
|
+
if (!val || val !== 'none') {
|
|
124
|
+
formatting.underline = true;
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
const strike = getBoolVal(rPrString, 'w:strike');
|
|
128
|
+
const dstrike = getBoolVal(rPrString, 'w:dstrike');
|
|
129
|
+
if (strike !== null)
|
|
130
|
+
formatting.strikethrough = strike;
|
|
131
|
+
else if (dstrike !== null)
|
|
132
|
+
formatting.strikethrough = dstrike;
|
|
133
|
+
// Font size
|
|
134
|
+
const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
|
|
135
|
+
if (szMatch)
|
|
136
|
+
formatting.size = (parseInt(szMatch[1]) / 2).toString() + 'pt';
|
|
137
|
+
// Color
|
|
138
|
+
const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
|
|
139
|
+
if (colorMatch && colorMatch[1] !== 'auto')
|
|
140
|
+
formatting.color = '#' + colorMatch[1];
|
|
141
|
+
// Background color (shading)
|
|
142
|
+
const shdMatch = rPrString.match(/<w:shd[^>]*w:fill="([^"]+)"/);
|
|
143
|
+
if (shdMatch && shdMatch[1] !== 'auto')
|
|
144
|
+
formatting.backgroundColor = '#' + shdMatch[1];
|
|
145
|
+
// Highlight (map to backgroundColor)
|
|
146
|
+
const highlightMatch = rPrString.match(/<w:highlight w:val="([^"]+)"/);
|
|
147
|
+
if (highlightMatch && highlightMatch[1] !== 'none') {
|
|
148
|
+
const colorMap = {
|
|
149
|
+
'yellow': '#FFFF00', 'green': '#00FF00', 'cyan': '#00FFFF', 'magenta': '#FF00FF',
|
|
150
|
+
'blue': '#0000FF', 'red': '#FF0000', 'darkBlue': '#00008B', 'darkCyan': '#008B8B',
|
|
151
|
+
'darkGreen': '#006400', 'darkMagenta': '#8B008B', 'darkRed': '#8B0000',
|
|
152
|
+
'darkYellow': '#808000', 'darkGray': '#A9A9A9', 'lightGray': '#D3D3D3', 'black': '#000000'
|
|
153
|
+
};
|
|
154
|
+
formatting.backgroundColor = colorMap[highlightMatch[1]] || highlightMatch[1];
|
|
155
|
+
}
|
|
156
|
+
// Font family
|
|
157
|
+
const rFontsMatch = rPrString.match(/<w:rFonts[^>]*w:ascii="([^"]+)"/);
|
|
158
|
+
if (rFontsMatch) {
|
|
159
|
+
formatting.font = rFontsMatch[1];
|
|
160
|
+
}
|
|
161
|
+
else {
|
|
162
|
+
const hAnsiMatch = rPrString.match(/<w:rFonts[^>]*w:hAnsi="([^"]+)"/);
|
|
163
|
+
if (hAnsiMatch)
|
|
164
|
+
formatting.font = hAnsiMatch[1];
|
|
165
|
+
}
|
|
166
|
+
// Subscript/Superscript
|
|
167
|
+
const vertAlignMatch = rPrString.match(/<w:vertAlign w:val="([^"]+)"/);
|
|
168
|
+
if (vertAlignMatch) {
|
|
169
|
+
if (vertAlignMatch[1] === 'subscript')
|
|
170
|
+
formatting.subscript = true;
|
|
171
|
+
if (vertAlignMatch[1] === 'superscript')
|
|
172
|
+
formatting.superscript = true;
|
|
173
|
+
}
|
|
174
|
+
return formatting;
|
|
175
|
+
};
|
|
176
|
+
const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
|
|
177
|
+
!!x.match(footnotesFileRegex) ||
|
|
178
|
+
!!x.match(endnotesFileRegex) ||
|
|
179
|
+
!!x.match(numberingFileRegex) ||
|
|
180
|
+
!!x.match(corePropsFileRegex) ||
|
|
181
|
+
!!x.match(relsFileRegex) ||
|
|
182
|
+
!!x.match(stylesFileRegex) ||
|
|
183
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
184
|
+
// Extract metadata
|
|
185
|
+
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
186
|
+
const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
187
|
+
const footnoteMap = new Map();
|
|
188
|
+
const endnoteMap = new Map();
|
|
189
|
+
const collectedNotes = [];
|
|
190
|
+
const attachments = [];
|
|
191
|
+
const mediaFiles = files.filter(f => f.path.match(mediaFileRegex));
|
|
192
|
+
// Extract relationships
|
|
193
|
+
const relsFile = files.find(f => f.path.match(relsFileRegex));
|
|
194
|
+
const relsMap = {};
|
|
195
|
+
if (relsFile) {
|
|
196
|
+
const relsXml = (0, xmlUtils_1.parseXmlString)(relsFile.content.toString());
|
|
197
|
+
const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
|
|
198
|
+
for (let i = 0; i < relationships.length; i++) {
|
|
199
|
+
const id = relationships[i].getAttribute("Id");
|
|
200
|
+
const target = relationships[i].getAttribute("Target");
|
|
201
|
+
if (id && target) {
|
|
202
|
+
relsMap[id] = target;
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
const numberingFile = files.find(f => f.path.match(numberingFileRegex));
|
|
207
|
+
const numberingMap = {};
|
|
208
|
+
if (numberingFile) {
|
|
209
|
+
const numberingXml = (0, xmlUtils_1.parseXmlString)(numberingFile.content.toString());
|
|
210
|
+
const nums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:num");
|
|
211
|
+
const abstractNums = (0, xmlUtils_1.getElementsByTagName)(numberingXml, "w:abstractNum");
|
|
212
|
+
const abstractNumMap = {};
|
|
213
|
+
for (let i = 0; i < abstractNums.length; i++) {
|
|
214
|
+
const abstractNumId = abstractNums[i].getAttribute("w:abstractNumId");
|
|
215
|
+
if (abstractNumId) {
|
|
216
|
+
abstractNumMap[abstractNumId] = abstractNums[i];
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
for (let i = 0; i < nums.length; i++) {
|
|
220
|
+
const numId = nums[i].getAttribute("w:numId");
|
|
221
|
+
const abstractNumIdNode = (0, xmlUtils_1.getElementsByTagName)(nums[i], "w:abstractNumId")[0];
|
|
222
|
+
const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
|
|
223
|
+
if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
|
|
224
|
+
numberingMap[numId] = {};
|
|
225
|
+
const lvls = (0, xmlUtils_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
|
|
226
|
+
for (let j = 0; j < lvls.length; j++) {
|
|
227
|
+
const ilvl = lvls[j].getAttribute("w:ilvl");
|
|
228
|
+
const numFmtNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:numFmt")[0];
|
|
229
|
+
const lvlTextNode = (0, xmlUtils_1.getElementsByTagName)(lvls[j], "w:lvlText")[0];
|
|
230
|
+
if (ilvl) {
|
|
231
|
+
numberingMap[numId][ilvl] = {
|
|
232
|
+
numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
|
|
233
|
+
lvlText: lvlTextNode?.getAttribute("w:val") || ''
|
|
234
|
+
};
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
// Parse Styles
|
|
241
|
+
const stylesFile = files.find(f => f.path.match(stylesFileRegex));
|
|
242
|
+
const styleMap = {};
|
|
243
|
+
if (stylesFile) {
|
|
244
|
+
const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
|
|
245
|
+
const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
|
|
246
|
+
for (let i = 0; i < styles.length; i++) {
|
|
247
|
+
const styleId = styles[i].getAttribute("w:styleId");
|
|
248
|
+
if (styleId) {
|
|
249
|
+
const rPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:rPr")[0];
|
|
250
|
+
const pPr = (0, xmlUtils_1.getElementsByTagName)(styles[i], "w:pPr")[0];
|
|
251
|
+
const formatting = rPr ? extractFormattingFromXml(rPr) : {};
|
|
252
|
+
let alignment = undefined;
|
|
253
|
+
let backgroundColor = undefined;
|
|
254
|
+
if (pPr) {
|
|
255
|
+
const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
|
|
256
|
+
if (jc) {
|
|
257
|
+
const val = jc.getAttribute("w:val");
|
|
258
|
+
if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
|
|
259
|
+
alignment = val;
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
|
|
263
|
+
if (shd) {
|
|
264
|
+
const fill = shd.getAttribute("w:fill");
|
|
265
|
+
if (fill && fill !== 'auto')
|
|
266
|
+
backgroundColor = '#' + fill;
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
styleMap[styleId] = { formatting, alignment, backgroundColor };
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
// Extract document defaults
|
|
274
|
+
let docDefaults = {};
|
|
275
|
+
if (stylesFile) {
|
|
276
|
+
const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
|
|
277
|
+
const docDefaultsNode = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:docDefaults")[0];
|
|
278
|
+
if (docDefaultsNode) {
|
|
279
|
+
const rPrDefaultNode = (0, xmlUtils_1.getElementsByTagName)(docDefaultsNode, "w:rPrDefault")[0];
|
|
280
|
+
if (rPrDefaultNode) {
|
|
281
|
+
const rPr = (0, xmlUtils_1.getElementsByTagName)(rPrDefaultNode, "w:rPr")[0];
|
|
282
|
+
if (rPr) {
|
|
283
|
+
docDefaults = extractFormattingFromXml(rPr);
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
// Detect the default paragraph style (for international compatibility)
|
|
289
|
+
let defaultParaStyleId = undefined;
|
|
290
|
+
if (stylesFile) {
|
|
291
|
+
const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
|
|
292
|
+
const styles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "w:style");
|
|
293
|
+
// Look for a style with w:type="paragraph" and w:default="1"
|
|
294
|
+
for (let i = 0; i < styles.length; i++) {
|
|
295
|
+
const styleType = styles[i].getAttribute("w:type");
|
|
296
|
+
const isDefault = styles[i].getAttribute("w:default");
|
|
297
|
+
const styleId = styles[i].getAttribute("w:styleId");
|
|
298
|
+
if (styleType === "paragraph" && isDefault === "1" && styleId) {
|
|
299
|
+
defaultParaStyleId = styleId;
|
|
300
|
+
break;
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
// Fallback: if no default found, try "Normal"
|
|
304
|
+
if (!defaultParaStyleId && styleMap["Normal"]) {
|
|
305
|
+
defaultParaStyleId = "Normal";
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
const content = [];
|
|
309
|
+
const rawContents = [];
|
|
310
|
+
const numberingState = {};
|
|
311
|
+
const listCounters = {}; // Track item index per listId/level
|
|
312
|
+
// Helper to parse a paragraph node
|
|
313
|
+
const parseParagraph = (pNode) => {
|
|
314
|
+
const pXml = xmlSerializer.serializeToString(pNode);
|
|
315
|
+
// Check if it's a list item
|
|
316
|
+
const numPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:numPr")[0];
|
|
317
|
+
const isList = !!numPr;
|
|
318
|
+
// Check if it's a heading
|
|
319
|
+
const pPr = (0, xmlUtils_1.getElementsByTagName)(pNode, "w:pPr")[0];
|
|
320
|
+
const pStyle = pPr ? (0, xmlUtils_1.getElementsByTagName)(pPr, "w:pStyle")[0] : null;
|
|
321
|
+
const pStyleVal = pStyle ? pStyle.getAttribute("w:val") : null;
|
|
322
|
+
const isHeading = pStyleVal ? (pStyleVal.startsWith("Heading") || pStyleVal === "Title") : false;
|
|
323
|
+
// Extract Paragraph Style Properties
|
|
324
|
+
const styleProps = pStyleVal && styleMap[pStyleVal] ? styleMap[pStyleVal] : { formatting: {} };
|
|
325
|
+
// Extract Alignment
|
|
326
|
+
let alignment = styleProps.alignment;
|
|
327
|
+
if (pPr) {
|
|
328
|
+
const jc = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:jc")[0];
|
|
329
|
+
if (jc) {
|
|
330
|
+
const val = jc.getAttribute("w:val");
|
|
331
|
+
if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
|
|
332
|
+
alignment = val;
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
// Extract Paragraph Background
|
|
337
|
+
let paraBackgroundColor = styleProps.backgroundColor;
|
|
338
|
+
if (pPr) {
|
|
339
|
+
const shd = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:shd")[0];
|
|
340
|
+
if (shd) {
|
|
341
|
+
const fill = shd.getAttribute("w:fill");
|
|
342
|
+
if (fill && fill !== 'auto') {
|
|
343
|
+
paraBackgroundColor = '#' + fill;
|
|
344
|
+
}
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
// Extract paragraph-level run properties
|
|
348
|
+
let paragraphRunFormatting = { ...styleProps.formatting };
|
|
349
|
+
if (pPr) {
|
|
350
|
+
const pPrRPr = (0, xmlUtils_1.getElementsByTagName)(pPr, "w:rPr")[0];
|
|
351
|
+
if (pPrRPr) {
|
|
352
|
+
const pPrFormatting = extractFormattingFromXml(pPrRPr);
|
|
353
|
+
for (const key in pPrFormatting) {
|
|
354
|
+
const value = pPrFormatting[key];
|
|
355
|
+
if (value === false) {
|
|
356
|
+
delete paragraphRunFormatting[key];
|
|
357
|
+
}
|
|
358
|
+
else if (value !== undefined) {
|
|
359
|
+
paragraphRunFormatting[key] = value;
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
// Extract text and children
|
|
365
|
+
let text = '';
|
|
366
|
+
const children = [];
|
|
367
|
+
// Traverse children of paragraph (runs, hyperlinks, etc.)
|
|
368
|
+
const processChildNode = (node) => {
|
|
369
|
+
if (node.nodeName === 'w:r') {
|
|
370
|
+
const runNode = node;
|
|
371
|
+
const rPr = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:rPr")[0];
|
|
372
|
+
// Formatting
|
|
373
|
+
let formatting = {};
|
|
374
|
+
// Apply paragraph-level formatting
|
|
375
|
+
for (const key in paragraphRunFormatting) {
|
|
376
|
+
formatting[key] = paragraphRunFormatting[key];
|
|
377
|
+
}
|
|
378
|
+
// Check for run style
|
|
379
|
+
const rStyle = rPr ? (0, xmlUtils_1.getElementsByTagName)(rPr, "w:rStyle")[0] : null;
|
|
380
|
+
const rStyleVal = rStyle ? rStyle.getAttribute("w:val") : pStyleVal;
|
|
381
|
+
if (rStyleVal && styleMap[rStyleVal]) {
|
|
382
|
+
for (const key in styleMap[rStyleVal].formatting) {
|
|
383
|
+
formatting[key] = styleMap[rStyleVal].formatting[key];
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
// Apply direct run properties
|
|
387
|
+
if (rPr) {
|
|
388
|
+
const directFormatting = extractFormattingFromXml(rPr);
|
|
389
|
+
for (const key in directFormatting) {
|
|
390
|
+
const value = directFormatting[key];
|
|
391
|
+
if (value === false) {
|
|
392
|
+
delete formatting[key];
|
|
393
|
+
}
|
|
394
|
+
else if (value !== undefined) {
|
|
395
|
+
formatting[key] = value;
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
// Inherit paragraph background
|
|
400
|
+
if (!formatting.backgroundColor && paraBackgroundColor) {
|
|
401
|
+
formatting.backgroundColor = paraBackgroundColor;
|
|
402
|
+
}
|
|
403
|
+
// Text content
|
|
404
|
+
const tNodes = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:t");
|
|
405
|
+
for (const tNode of tNodes) {
|
|
406
|
+
const tContent = tNode.textContent || '';
|
|
407
|
+
text += tContent;
|
|
408
|
+
const textNode = {
|
|
409
|
+
type: 'text',
|
|
410
|
+
text: tContent,
|
|
411
|
+
formatting: formatting
|
|
412
|
+
};
|
|
413
|
+
if (config.includeRawContent) {
|
|
414
|
+
textNode.rawContent = xmlSerializer.serializeToString(tNode);
|
|
415
|
+
}
|
|
416
|
+
// Always set a style: run style > paragraph style > detected default
|
|
417
|
+
// Use detected default style for international compatibility
|
|
418
|
+
const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
|
|
419
|
+
if (nodeStyle) {
|
|
420
|
+
textNode.metadata = { style: nodeStyle };
|
|
421
|
+
}
|
|
422
|
+
children.push(textNode);
|
|
423
|
+
}
|
|
424
|
+
// Images/Drawings
|
|
425
|
+
if (config.extractAttachments) {
|
|
426
|
+
const drawings = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:drawing");
|
|
427
|
+
const picts = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:pict");
|
|
428
|
+
const allImages = [...drawings, ...picts];
|
|
429
|
+
for (const imgNode of allImages) {
|
|
430
|
+
const imgXml = xmlSerializer.serializeToString(imgNode);
|
|
431
|
+
// Extract Alt Text
|
|
432
|
+
let altText = '';
|
|
433
|
+
const docPr = (0, xmlUtils_1.getElementsByTagName)(imgNode, "wp:docPr")[0];
|
|
434
|
+
if (docPr) {
|
|
435
|
+
altText = docPr.getAttribute("descr") || docPr.getAttribute("title") || '';
|
|
436
|
+
}
|
|
437
|
+
// Extract Relationship ID
|
|
438
|
+
let rId = '';
|
|
439
|
+
const blip = (0, xmlUtils_1.getElementsByTagName)(imgNode, "a:blip")[0];
|
|
440
|
+
if (blip) {
|
|
441
|
+
rId = blip.getAttribute("r:embed") || '';
|
|
442
|
+
}
|
|
443
|
+
else {
|
|
444
|
+
const imagedata = (0, xmlUtils_1.getElementsByTagName)(imgNode, "v:imagedata")[0];
|
|
445
|
+
if (imagedata) {
|
|
446
|
+
rId = imagedata.getAttribute("r:id") || '';
|
|
447
|
+
}
|
|
448
|
+
}
|
|
449
|
+
if (rId && relsMap[rId]) {
|
|
450
|
+
const target = relsMap[rId];
|
|
451
|
+
const filename = target.split('/').pop();
|
|
452
|
+
if (filename) {
|
|
453
|
+
const imageNode = {
|
|
454
|
+
type: 'image',
|
|
455
|
+
text: '',
|
|
456
|
+
metadata: { attachmentName: filename, altText: altText }
|
|
457
|
+
};
|
|
458
|
+
if (config.includeRawContent) {
|
|
459
|
+
imageNode.rawContent = imgXml;
|
|
460
|
+
}
|
|
461
|
+
children.push(imageNode);
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
else {
|
|
465
|
+
const imageNode = {
|
|
466
|
+
type: 'image',
|
|
467
|
+
text: '',
|
|
468
|
+
};
|
|
469
|
+
if (config.includeRawContent) {
|
|
470
|
+
imageNode.rawContent = imgXml;
|
|
471
|
+
}
|
|
472
|
+
children.push(imageNode);
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
}
|
|
476
|
+
// Footnotes/Endnotes inside runs
|
|
477
|
+
if (!config.ignoreNotes) {
|
|
478
|
+
const footnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:footnoteReference")[0];
|
|
479
|
+
if (footnoteRef) {
|
|
480
|
+
const id = footnoteRef.getAttribute("w:id");
|
|
481
|
+
if (id && footnoteMap.has(id)) {
|
|
482
|
+
const noteNodes = footnoteMap.get(id);
|
|
483
|
+
const noteNode = {
|
|
484
|
+
type: 'note',
|
|
485
|
+
text: noteNodes.map((n) => n.text).join(' '),
|
|
486
|
+
children: noteNodes,
|
|
487
|
+
metadata: { noteType: 'footnote', noteId: id }
|
|
488
|
+
};
|
|
489
|
+
if (config.putNotesAtLast) {
|
|
490
|
+
collectedNotes.push(noteNode);
|
|
491
|
+
}
|
|
492
|
+
else {
|
|
493
|
+
children.push(noteNode);
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
const endnoteRef = (0, xmlUtils_1.getElementsByTagName)(runNode, "w:endnoteReference")[0];
|
|
498
|
+
if (endnoteRef) {
|
|
499
|
+
const id = endnoteRef.getAttribute("w:id");
|
|
500
|
+
if (id && endnoteMap.has(id)) {
|
|
501
|
+
const noteNodes = endnoteMap.get(id);
|
|
502
|
+
const noteNode = {
|
|
503
|
+
type: 'note',
|
|
504
|
+
text: noteNodes.map((n) => n.text).join(' '),
|
|
505
|
+
children: noteNodes,
|
|
506
|
+
metadata: { noteType: 'endnote', noteId: id }
|
|
507
|
+
};
|
|
508
|
+
if (config.putNotesAtLast) {
|
|
509
|
+
collectedNotes.push(noteNode);
|
|
510
|
+
}
|
|
511
|
+
else {
|
|
512
|
+
children.push(noteNode);
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
else if (node.nodeName === 'w:hyperlink') {
|
|
519
|
+
const hlNode = node;
|
|
520
|
+
const rId = hlNode.getAttribute("r:id");
|
|
521
|
+
const anchor = hlNode.getAttribute("w:anchor");
|
|
522
|
+
let linkMetadata;
|
|
523
|
+
if (anchor) {
|
|
524
|
+
linkMetadata = { link: '#' + anchor, linkType: 'internal' };
|
|
525
|
+
}
|
|
526
|
+
else if (rId && relsMap[rId]) {
|
|
527
|
+
linkMetadata = { link: relsMap[rId], linkType: 'external' };
|
|
528
|
+
}
|
|
529
|
+
// Process children of hyperlink (usually runs)
|
|
530
|
+
const hlChildren = Array.from(hlNode.childNodes);
|
|
531
|
+
for (const child of hlChildren) {
|
|
532
|
+
// Capture the current length of children to apply metadata to new nodes
|
|
533
|
+
const startIndex = children.length;
|
|
534
|
+
processChildNode(child);
|
|
535
|
+
// Apply link metadata to the newly added text nodes
|
|
536
|
+
if (linkMetadata) {
|
|
537
|
+
for (let i = startIndex; i < children.length; i++) {
|
|
538
|
+
if (children[i].type === 'text') {
|
|
539
|
+
children[i].metadata = { ...(children[i].metadata ?? {}), ...linkMetadata };
|
|
540
|
+
}
|
|
541
|
+
}
|
|
542
|
+
}
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
};
|
|
546
|
+
const childNodes = Array.from(pNode.childNodes);
|
|
547
|
+
for (const child of childNodes) {
|
|
548
|
+
processChildNode(child);
|
|
549
|
+
}
|
|
550
|
+
if (isList) {
|
|
551
|
+
const numIdNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:numId")[0];
|
|
552
|
+
const ilvlNode = (0, xmlUtils_1.getElementsByTagName)(numPr, "w:ilvl")[0];
|
|
553
|
+
const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
|
|
554
|
+
const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
|
|
555
|
+
let listType = 'ordered';
|
|
556
|
+
let itemIndex = 0;
|
|
557
|
+
if (numId && numberingMap[numId]) {
|
|
558
|
+
const ilvlStr = ilvl.toString();
|
|
559
|
+
if (!numberingState[numId])
|
|
560
|
+
numberingState[numId] = {};
|
|
561
|
+
if (!numberingState[numId][ilvlStr])
|
|
562
|
+
numberingState[numId][ilvlStr] = 0;
|
|
563
|
+
numberingState[numId][ilvlStr]++;
|
|
564
|
+
for (let k = ilvl + 1; k < 10; k++) {
|
|
565
|
+
if (numberingState[numId][k.toString()])
|
|
566
|
+
numberingState[numId][k.toString()] = 0;
|
|
567
|
+
}
|
|
568
|
+
const numFmt = numberingMap[numId][ilvlStr]?.numFmt || 'decimal';
|
|
569
|
+
listType = numFmt === 'bullet' ? 'unordered' : 'ordered';
|
|
570
|
+
// Track itemIndex (starts at 0, continues across interruptions for same listId)
|
|
571
|
+
if (!listCounters[numId])
|
|
572
|
+
listCounters[numId] = {};
|
|
573
|
+
if (listCounters[numId][ilvlStr] === undefined) {
|
|
574
|
+
listCounters[numId][ilvlStr] = 0;
|
|
575
|
+
}
|
|
576
|
+
else {
|
|
577
|
+
listCounters[numId][ilvlStr]++;
|
|
578
|
+
}
|
|
579
|
+
itemIndex = listCounters[numId][ilvlStr];
|
|
580
|
+
}
|
|
581
|
+
const listNode = {
|
|
582
|
+
type: 'list',
|
|
583
|
+
text: text,
|
|
584
|
+
children: children,
|
|
585
|
+
metadata: {
|
|
586
|
+
listType,
|
|
587
|
+
indentation: ilvl,
|
|
588
|
+
alignment: (alignment || 'left'),
|
|
589
|
+
listId: numId,
|
|
590
|
+
itemIndex: itemIndex,
|
|
591
|
+
style: pStyleVal
|
|
592
|
+
}
|
|
593
|
+
};
|
|
594
|
+
if (config.includeRawContent)
|
|
595
|
+
listNode.rawContent = pXml;
|
|
596
|
+
return listNode;
|
|
597
|
+
}
|
|
598
|
+
else if (isHeading) {
|
|
599
|
+
const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", "")) || 1 : 1;
|
|
600
|
+
const headingNode = {
|
|
601
|
+
type: 'heading',
|
|
602
|
+
text: text,
|
|
603
|
+
children: children,
|
|
604
|
+
metadata: { level, alignment, style: pStyleVal ?? undefined }
|
|
605
|
+
};
|
|
606
|
+
if (config.includeRawContent)
|
|
607
|
+
headingNode.rawContent = pXml;
|
|
608
|
+
return headingNode;
|
|
609
|
+
}
|
|
610
|
+
else {
|
|
611
|
+
const paraNode = {
|
|
612
|
+
type: 'paragraph',
|
|
613
|
+
text: text,
|
|
614
|
+
children: children,
|
|
615
|
+
metadata: { alignment, style: pStyleVal ?? undefined }
|
|
616
|
+
};
|
|
617
|
+
if (config.includeRawContent)
|
|
618
|
+
paraNode.rawContent = pXml;
|
|
619
|
+
return paraNode;
|
|
620
|
+
}
|
|
621
|
+
};
|
|
622
|
+
// Helper to parse a table node
|
|
623
|
+
const parseTable = (tblNode) => {
|
|
624
|
+
const rows = [];
|
|
625
|
+
// Only get direct child rows, not nested table rows
|
|
626
|
+
const trNodes = (0, xmlUtils_1.getDirectChildren)(tblNode, "w:tr");
|
|
627
|
+
for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
|
|
628
|
+
const trNode = trNodes[rIndex];
|
|
629
|
+
const cells = [];
|
|
630
|
+
// Only get direct child cells, not nested table cells
|
|
631
|
+
const tcNodes = (0, xmlUtils_1.getDirectChildren)(trNode, "w:tc");
|
|
632
|
+
for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
|
|
633
|
+
const tcNode = tcNodes[cIndex];
|
|
634
|
+
const cellChildren = [];
|
|
635
|
+
let cellText = '';
|
|
636
|
+
// Cells contain paragraphs (and other block-level elements)
|
|
637
|
+
const cellContentNodes = Array.from(tcNode.childNodes);
|
|
638
|
+
for (const child of cellContentNodes) {
|
|
639
|
+
if (child.nodeName === 'w:p') {
|
|
640
|
+
const pNode = parseParagraph(child);
|
|
641
|
+
cellChildren.push(pNode);
|
|
642
|
+
cellText += pNode.text;
|
|
643
|
+
}
|
|
644
|
+
else if (child.nodeName === 'w:tbl') {
|
|
645
|
+
// Nested table
|
|
646
|
+
const nestedTable = parseTable(child);
|
|
647
|
+
cellChildren.push(nestedTable);
|
|
648
|
+
// Don't add nested table text to cell text - it will be handled recursively
|
|
649
|
+
}
|
|
650
|
+
}
|
|
651
|
+
const cellNode = {
|
|
652
|
+
type: 'cell',
|
|
653
|
+
text: cellText,
|
|
654
|
+
children: cellChildren,
|
|
655
|
+
metadata: { row: rIndex, col: cIndex }
|
|
656
|
+
};
|
|
657
|
+
cells.push(cellNode);
|
|
658
|
+
}
|
|
659
|
+
const rowNode = {
|
|
660
|
+
type: 'row',
|
|
661
|
+
children: cells
|
|
662
|
+
};
|
|
663
|
+
rows.push(rowNode);
|
|
664
|
+
}
|
|
665
|
+
return {
|
|
666
|
+
type: 'table',
|
|
667
|
+
children: rows
|
|
668
|
+
};
|
|
669
|
+
};
|
|
670
|
+
// Pre-process footnotes and endnotes to be inserted inline later
|
|
671
|
+
if (!config.ignoreNotes) {
|
|
672
|
+
const footnotesFile = files.find(f => f.path.match(footnotesFileRegex));
|
|
673
|
+
if (footnotesFile) {
|
|
674
|
+
const footnotesDoc = (0, xmlUtils_1.parseXmlString)(footnotesFile.content.toString());
|
|
675
|
+
const footnoteNodes = (0, xmlUtils_1.getElementsByTagName)(footnotesDoc, "w:footnote");
|
|
676
|
+
for (const node of footnoteNodes) {
|
|
677
|
+
const id = node.getAttribute("w:id");
|
|
678
|
+
if (!id || id === "-1" || id === "0")
|
|
679
|
+
continue;
|
|
680
|
+
const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
|
|
681
|
+
footnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
|
|
682
|
+
}
|
|
683
|
+
}
|
|
684
|
+
const endnotesFile = files.find(f => f.path.match(endnotesFileRegex));
|
|
685
|
+
if (endnotesFile) {
|
|
686
|
+
const endnotesDoc = (0, xmlUtils_1.parseXmlString)(endnotesFile.content.toString());
|
|
687
|
+
const endnoteNodes = (0, xmlUtils_1.getElementsByTagName)(endnotesDoc, "w:endnote");
|
|
688
|
+
for (const node of endnoteNodes) {
|
|
689
|
+
const id = node.getAttribute("w:id");
|
|
690
|
+
if (!id || id === "-1" || id === "0")
|
|
691
|
+
continue;
|
|
692
|
+
const pNodes = (0, xmlUtils_1.getElementsByTagName)(node, "w:p");
|
|
693
|
+
endnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
}
|
|
697
|
+
for (const file of files) {
|
|
698
|
+
if (file.path.match(mediaFileRegex))
|
|
699
|
+
continue;
|
|
700
|
+
if (file.path.match(numberingFileRegex))
|
|
701
|
+
continue;
|
|
702
|
+
if (file.path.match(relsFileRegex))
|
|
703
|
+
continue;
|
|
704
|
+
if (file.path.match(stylesFileRegex))
|
|
705
|
+
continue;
|
|
706
|
+
if (file.path.match(footnotesFileRegex))
|
|
707
|
+
continue;
|
|
708
|
+
if (file.path.match(endnotesFileRegex))
|
|
709
|
+
continue;
|
|
710
|
+
const documentContent = file.content.toString();
|
|
711
|
+
if (config.includeRawContent) {
|
|
712
|
+
rawContents.push(documentContent);
|
|
713
|
+
}
|
|
714
|
+
const doc = (0, xmlUtils_1.parseXmlString)(documentContent);
|
|
715
|
+
const body = (0, xmlUtils_1.getElementsByTagName)(doc, "w:body")[0];
|
|
716
|
+
if (body) {
|
|
717
|
+
const bodyChildren = Array.from(body.childNodes);
|
|
718
|
+
for (const child of bodyChildren) {
|
|
719
|
+
if (child.nodeName === 'w:p') {
|
|
720
|
+
content.push(parseParagraph(child));
|
|
721
|
+
}
|
|
722
|
+
else if (child.nodeName === 'w:tbl') {
|
|
723
|
+
content.push(parseTable(child));
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
}
|
|
727
|
+
}
|
|
728
|
+
// Extract attachments
|
|
729
|
+
if (config.extractAttachments) {
|
|
730
|
+
for (const media of mediaFiles) {
|
|
731
|
+
const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
732
|
+
attachments.push(attachment);
|
|
733
|
+
if (config.ocr) {
|
|
734
|
+
if (attachment.mimeType.startsWith('image/')) {
|
|
735
|
+
try {
|
|
736
|
+
attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
|
|
737
|
+
}
|
|
738
|
+
catch (e) {
|
|
739
|
+
(0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
740
|
+
}
|
|
741
|
+
}
|
|
742
|
+
}
|
|
743
|
+
}
|
|
744
|
+
// Assign OCR text to image nodes
|
|
745
|
+
if (config.ocr) {
|
|
746
|
+
const assignOcr = (nodes) => {
|
|
747
|
+
for (const node of nodes) {
|
|
748
|
+
if (node.type === 'image' && 'attachmentName' in (node.metadata || {})) {
|
|
749
|
+
const meta = node.metadata;
|
|
750
|
+
const attachment = attachments.find(a => a.name === meta.attachmentName);
|
|
751
|
+
if (attachment && attachment.ocrText) {
|
|
752
|
+
node.text = attachment.ocrText;
|
|
753
|
+
attachment.altText = meta.altText;
|
|
754
|
+
}
|
|
755
|
+
}
|
|
756
|
+
if (node.children) {
|
|
757
|
+
assignOcr(node.children);
|
|
758
|
+
}
|
|
759
|
+
}
|
|
760
|
+
};
|
|
761
|
+
assignOcr(content);
|
|
762
|
+
}
|
|
763
|
+
}
|
|
764
|
+
if (config.putNotesAtLast && collectedNotes.length > 0) {
|
|
765
|
+
content.push(...collectedNotes);
|
|
766
|
+
}
|
|
767
|
+
return {
|
|
768
|
+
type: 'docx',
|
|
769
|
+
metadata: { ...metadata, formatting: docDefaults, styleMap: styleMap },
|
|
770
|
+
content: content,
|
|
771
|
+
attachments: attachments,
|
|
772
|
+
toText: () => content.map(c => {
|
|
773
|
+
// Recursive text extraction
|
|
774
|
+
const getText = (node) => {
|
|
775
|
+
let t = '';
|
|
776
|
+
if (node.children) {
|
|
777
|
+
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
|
|
778
|
+
}
|
|
779
|
+
else
|
|
780
|
+
t += node.text || '';
|
|
781
|
+
return t;
|
|
782
|
+
};
|
|
783
|
+
return getText(c);
|
|
784
|
+
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
785
|
+
};
|
|
786
|
+
};
|
|
787
|
+
exports.parseWord = parseWord;
|