officeparser 6.0.7 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -13
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +43 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +116 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +79 -1
- package/dist/officeparser.browser.iife.js +112 -0
- package/dist/officeparser.browser.mjs +111 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +71 -63
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +131 -114
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +85 -88
- package/dist/parsers/RtfParser.d.ts +1 -1
- package/dist/parsers/RtfParser.js +10 -6
- package/dist/parsers/WordParser.d.ts +1 -1
- package/dist/parsers/WordParser.js +109 -101
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +69 -1
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
|
@@ -61,12 +61,11 @@
|
|
|
61
61
|
*/
|
|
62
62
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
63
63
|
exports.parseWord = void 0;
|
|
64
|
-
const
|
|
65
|
-
const
|
|
66
|
-
const
|
|
67
|
-
const
|
|
68
|
-
const
|
|
69
|
-
const zipUtils_1 = require("../utils/zipUtils");
|
|
64
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
65
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
66
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
67
|
+
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
68
|
+
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
70
69
|
/**
|
|
71
70
|
* Parses a Word document (.docx) and extracts content, formatting, and metadata.
|
|
72
71
|
*
|
|
@@ -90,13 +89,13 @@ const parseWord = async (buffer, config) => {
|
|
|
90
89
|
const numberingFileRegex = /word\/numbering[\d+]?.xml/;
|
|
91
90
|
const mediaFileRegex = /(word\/)?media\/.*/;
|
|
92
91
|
const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
|
|
92
|
+
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
93
93
|
const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
|
|
94
94
|
const stylesFileRegex = /word\/styles[\d+]?.xml/;
|
|
95
|
-
const xmlSerializer = new xmldom_1.XMLSerializer();
|
|
96
95
|
// Helper to extract formatting from run properties XML string
|
|
97
96
|
const extractFormattingFromXml = (rPr) => {
|
|
98
97
|
const formatting = {};
|
|
99
|
-
const rPrString =
|
|
98
|
+
const rPrString = (0, xmlUtils_js_1.serializeXml)(rPr);
|
|
100
99
|
// Helper to check boolean properties
|
|
101
100
|
const getBoolVal = (xmlSnippet, tagName) => {
|
|
102
101
|
const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
|
|
@@ -173,17 +172,24 @@ const parseWord = async (buffer, config) => {
|
|
|
173
172
|
}
|
|
174
173
|
return formatting;
|
|
175
174
|
};
|
|
176
|
-
const files = await (0,
|
|
175
|
+
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
|
|
177
176
|
!!x.match(footnotesFileRegex) ||
|
|
178
177
|
!!x.match(endnotesFileRegex) ||
|
|
179
178
|
!!x.match(numberingFileRegex) ||
|
|
180
179
|
!!x.match(corePropsFileRegex) ||
|
|
180
|
+
!!x.match(customPropsFileRegex) ||
|
|
181
181
|
!!x.match(relsFileRegex) ||
|
|
182
182
|
!!x.match(stylesFileRegex) ||
|
|
183
183
|
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
184
184
|
// Extract metadata
|
|
185
185
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
186
|
-
const metadata = corePropsFile ? (0,
|
|
186
|
+
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
187
|
+
const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
|
|
188
|
+
if (customPropsFile) {
|
|
189
|
+
const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
|
|
190
|
+
if (Object.keys(customProperties).length > 0)
|
|
191
|
+
metadata.customProperties = customProperties;
|
|
192
|
+
}
|
|
187
193
|
const footnoteMap = new Map();
|
|
188
194
|
const endnoteMap = new Map();
|
|
189
195
|
const collectedNotes = [];
|
|
@@ -193,11 +199,11 @@ const parseWord = async (buffer, config) => {
|
|
|
193
199
|
const relsFile = files.find(f => f.path.match(relsFileRegex));
|
|
194
200
|
const relsMap = {};
|
|
195
201
|
if (relsFile) {
|
|
196
|
-
const relsXml = (0,
|
|
197
|
-
const relationships = (0,
|
|
198
|
-
for (
|
|
199
|
-
const id =
|
|
200
|
-
const target =
|
|
202
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
|
|
203
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
204
|
+
for (const relationship of relationships) {
|
|
205
|
+
const id = relationship.getAttribute("Id");
|
|
206
|
+
const target = relationship.getAttribute("Target");
|
|
201
207
|
if (id && target) {
|
|
202
208
|
relsMap[id] = target;
|
|
203
209
|
}
|
|
@@ -206,27 +212,27 @@ const parseWord = async (buffer, config) => {
|
|
|
206
212
|
const numberingFile = files.find(f => f.path.match(numberingFileRegex));
|
|
207
213
|
const numberingMap = {};
|
|
208
214
|
if (numberingFile) {
|
|
209
|
-
const numberingXml = (0,
|
|
210
|
-
const nums = (0,
|
|
211
|
-
const abstractNums = (0,
|
|
215
|
+
const numberingXml = (0, xmlUtils_js_1.parseXmlString)(numberingFile.content.toString());
|
|
216
|
+
const nums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:num");
|
|
217
|
+
const abstractNums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:abstractNum");
|
|
212
218
|
const abstractNumMap = {};
|
|
213
|
-
for (
|
|
214
|
-
const abstractNumId =
|
|
219
|
+
for (const abstractNum of abstractNums) {
|
|
220
|
+
const abstractNumId = abstractNum.getAttribute("w:abstractNumId");
|
|
215
221
|
if (abstractNumId) {
|
|
216
|
-
abstractNumMap[abstractNumId] =
|
|
222
|
+
abstractNumMap[abstractNumId] = abstractNum;
|
|
217
223
|
}
|
|
218
224
|
}
|
|
219
|
-
for (
|
|
220
|
-
const numId =
|
|
221
|
-
const abstractNumIdNode = (0,
|
|
225
|
+
for (const num of nums) {
|
|
226
|
+
const numId = num.getAttribute("w:numId");
|
|
227
|
+
const abstractNumIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(num, "w:abstractNumId");
|
|
222
228
|
const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
|
|
223
229
|
if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
|
|
224
230
|
numberingMap[numId] = {};
|
|
225
|
-
const lvls = (0,
|
|
226
|
-
for (
|
|
227
|
-
const ilvl =
|
|
228
|
-
const numFmtNode = (0,
|
|
229
|
-
const lvlTextNode = (0,
|
|
231
|
+
const lvls = (0, xmlUtils_js_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
|
|
232
|
+
for (const lvl of lvls) {
|
|
233
|
+
const ilvl = lvl.getAttribute("w:ilvl");
|
|
234
|
+
const numFmtNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:numFmt");
|
|
235
|
+
const lvlTextNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:lvlText");
|
|
230
236
|
if (ilvl) {
|
|
231
237
|
numberingMap[numId][ilvl] = {
|
|
232
238
|
numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
|
|
@@ -241,25 +247,25 @@ const parseWord = async (buffer, config) => {
|
|
|
241
247
|
const stylesFile = files.find(f => f.path.match(stylesFileRegex));
|
|
242
248
|
const styleMap = {};
|
|
243
249
|
if (stylesFile) {
|
|
244
|
-
const stylesXml = (0,
|
|
245
|
-
const styles = (0,
|
|
246
|
-
for (
|
|
247
|
-
const styleId =
|
|
250
|
+
const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
251
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
|
|
252
|
+
for (const style of styles) {
|
|
253
|
+
const styleId = style.getAttribute("w:styleId");
|
|
248
254
|
if (styleId) {
|
|
249
|
-
const rPr = (0,
|
|
250
|
-
const pPr = (0,
|
|
255
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:rPr");
|
|
256
|
+
const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:pPr");
|
|
251
257
|
const formatting = rPr ? extractFormattingFromXml(rPr) : {};
|
|
252
258
|
let alignment = undefined;
|
|
253
259
|
let backgroundColor = undefined;
|
|
254
260
|
if (pPr) {
|
|
255
|
-
const jc = (0,
|
|
261
|
+
const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
|
|
256
262
|
if (jc) {
|
|
257
263
|
const val = jc.getAttribute("w:val");
|
|
258
264
|
if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
|
|
259
265
|
alignment = val;
|
|
260
266
|
}
|
|
261
267
|
}
|
|
262
|
-
const shd = (0,
|
|
268
|
+
const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
|
|
263
269
|
if (shd) {
|
|
264
270
|
const fill = shd.getAttribute("w:fill");
|
|
265
271
|
if (fill && fill !== 'auto')
|
|
@@ -273,12 +279,12 @@ const parseWord = async (buffer, config) => {
|
|
|
273
279
|
// Extract document defaults
|
|
274
280
|
let docDefaults = {};
|
|
275
281
|
if (stylesFile) {
|
|
276
|
-
const stylesXml = (0,
|
|
277
|
-
const docDefaultsNode = (0,
|
|
282
|
+
const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
283
|
+
const docDefaultsNode = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesXml, "w:docDefaults");
|
|
278
284
|
if (docDefaultsNode) {
|
|
279
|
-
const rPrDefaultNode = (0,
|
|
285
|
+
const rPrDefaultNode = (0, xmlUtils_js_1.getFirstElementByTagName)(docDefaultsNode, "w:rPrDefault");
|
|
280
286
|
if (rPrDefaultNode) {
|
|
281
|
-
const rPr = (0,
|
|
287
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(rPrDefaultNode, "w:rPr");
|
|
282
288
|
if (rPr) {
|
|
283
289
|
docDefaults = extractFormattingFromXml(rPr);
|
|
284
290
|
}
|
|
@@ -288,13 +294,13 @@ const parseWord = async (buffer, config) => {
|
|
|
288
294
|
// Detect the default paragraph style (for international compatibility)
|
|
289
295
|
let defaultParaStyleId = undefined;
|
|
290
296
|
if (stylesFile) {
|
|
291
|
-
const stylesXml = (0,
|
|
292
|
-
const styles = (0,
|
|
297
|
+
const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
298
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
|
|
293
299
|
// Look for a style with w:type="paragraph" and w:default="1"
|
|
294
|
-
for (
|
|
295
|
-
const styleType =
|
|
296
|
-
const isDefault =
|
|
297
|
-
const styleId =
|
|
300
|
+
for (const style of styles) {
|
|
301
|
+
const styleType = style.getAttribute("w:type");
|
|
302
|
+
const isDefault = style.getAttribute("w:default");
|
|
303
|
+
const styleId = style.getAttribute("w:styleId");
|
|
298
304
|
if (styleType === "paragraph" && isDefault === "1" && styleId) {
|
|
299
305
|
defaultParaStyleId = styleId;
|
|
300
306
|
break;
|
|
@@ -310,22 +316,22 @@ const parseWord = async (buffer, config) => {
|
|
|
310
316
|
const numberingState = {};
|
|
311
317
|
const listCounters = {}; // Track item index per listId/level
|
|
312
318
|
// Helper to parse a paragraph node
|
|
313
|
-
const parseParagraph = (pNode) => {
|
|
314
|
-
const pXml =
|
|
319
|
+
const parseParagraph = (pNode, documentContent) => {
|
|
320
|
+
const pXml = pNode.toString();
|
|
315
321
|
// Check if it's a list item
|
|
316
|
-
const numPr = (0,
|
|
322
|
+
const numPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:numPr");
|
|
317
323
|
const isList = !!numPr;
|
|
318
324
|
// Check if it's a heading
|
|
319
|
-
const pPr = (0,
|
|
320
|
-
const pStyle = pPr ? (0,
|
|
321
|
-
const pStyleVal = pStyle
|
|
325
|
+
const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:pPr");
|
|
326
|
+
const pStyle = pPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:pStyle") : null;
|
|
327
|
+
const pStyleVal = pStyle?.getAttribute("w:val");
|
|
322
328
|
const isHeading = pStyleVal ? (pStyleVal.startsWith("Heading") || pStyleVal === "Title") : false;
|
|
323
329
|
// Extract Paragraph Style Properties
|
|
324
330
|
const styleProps = pStyleVal && styleMap[pStyleVal] ? styleMap[pStyleVal] : { formatting: {} };
|
|
325
331
|
// Extract Alignment
|
|
326
332
|
let alignment = styleProps.alignment;
|
|
327
333
|
if (pPr) {
|
|
328
|
-
const jc = (0,
|
|
334
|
+
const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
|
|
329
335
|
if (jc) {
|
|
330
336
|
const val = jc.getAttribute("w:val");
|
|
331
337
|
if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
|
|
@@ -336,7 +342,7 @@ const parseWord = async (buffer, config) => {
|
|
|
336
342
|
// Extract Paragraph Background
|
|
337
343
|
let paraBackgroundColor = styleProps.backgroundColor;
|
|
338
344
|
if (pPr) {
|
|
339
|
-
const shd = (0,
|
|
345
|
+
const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
|
|
340
346
|
if (shd) {
|
|
341
347
|
const fill = shd.getAttribute("w:fill");
|
|
342
348
|
if (fill && fill !== 'auto') {
|
|
@@ -347,7 +353,7 @@ const parseWord = async (buffer, config) => {
|
|
|
347
353
|
// Extract paragraph-level run properties
|
|
348
354
|
let paragraphRunFormatting = { ...styleProps.formatting };
|
|
349
355
|
if (pPr) {
|
|
350
|
-
const pPrRPr = (0,
|
|
356
|
+
const pPrRPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:rPr");
|
|
351
357
|
if (pPrRPr) {
|
|
352
358
|
const pPrFormatting = extractFormattingFromXml(pPrRPr);
|
|
353
359
|
for (const key in pPrFormatting) {
|
|
@@ -366,9 +372,9 @@ const parseWord = async (buffer, config) => {
|
|
|
366
372
|
const children = [];
|
|
367
373
|
// Traverse children of paragraph (runs, hyperlinks, etc.)
|
|
368
374
|
const processChildNode = (node) => {
|
|
369
|
-
if (node.nodeName === 'w:r') {
|
|
375
|
+
if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
|
|
370
376
|
const runNode = node;
|
|
371
|
-
const rPr = (0,
|
|
377
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:rPr");
|
|
372
378
|
// Formatting
|
|
373
379
|
let formatting = {};
|
|
374
380
|
// Apply paragraph-level formatting
|
|
@@ -376,7 +382,7 @@ const parseWord = async (buffer, config) => {
|
|
|
376
382
|
formatting[key] = paragraphRunFormatting[key];
|
|
377
383
|
}
|
|
378
384
|
// Check for run style
|
|
379
|
-
const rStyle = rPr ? (0,
|
|
385
|
+
const rStyle = rPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "w:rStyle") : null;
|
|
380
386
|
const rStyleVal = rStyle ? rStyle.getAttribute("w:val") : pStyleVal;
|
|
381
387
|
if (rStyleVal && styleMap[rStyleVal]) {
|
|
382
388
|
for (const key in styleMap[rStyleVal].formatting) {
|
|
@@ -401,7 +407,7 @@ const parseWord = async (buffer, config) => {
|
|
|
401
407
|
formatting.backgroundColor = paraBackgroundColor;
|
|
402
408
|
}
|
|
403
409
|
// Text content
|
|
404
|
-
const tNodes = (0,
|
|
410
|
+
const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:t");
|
|
405
411
|
for (const tNode of tNodes) {
|
|
406
412
|
const tContent = tNode.textContent || '';
|
|
407
413
|
text += tContent;
|
|
@@ -411,7 +417,7 @@ const parseWord = async (buffer, config) => {
|
|
|
411
417
|
formatting: formatting
|
|
412
418
|
};
|
|
413
419
|
if (config.includeRawContent) {
|
|
414
|
-
textNode.rawContent =
|
|
420
|
+
textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
|
|
415
421
|
}
|
|
416
422
|
// Always set a style: run style > paragraph style > detected default
|
|
417
423
|
// Use detected default style for international compatibility
|
|
@@ -423,25 +429,25 @@ const parseWord = async (buffer, config) => {
|
|
|
423
429
|
}
|
|
424
430
|
// Images/Drawings
|
|
425
431
|
if (config.extractAttachments) {
|
|
426
|
-
const drawings = (0,
|
|
427
|
-
const picts = (0,
|
|
432
|
+
const drawings = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:drawing");
|
|
433
|
+
const picts = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:pict");
|
|
428
434
|
const allImages = [...drawings, ...picts];
|
|
429
435
|
for (const imgNode of allImages) {
|
|
430
|
-
const imgXml =
|
|
436
|
+
const imgXml = (0, xmlUtils_js_1.serializeXml)(imgNode);
|
|
431
437
|
// Extract Alt Text
|
|
432
438
|
let altText = '';
|
|
433
|
-
const docPr = (0,
|
|
439
|
+
const docPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "wp:docPr");
|
|
434
440
|
if (docPr) {
|
|
435
441
|
altText = docPr.getAttribute("descr") || docPr.getAttribute("title") || '';
|
|
436
442
|
}
|
|
437
443
|
// Extract Relationship ID
|
|
438
444
|
let rId = '';
|
|
439
|
-
const blip = (0,
|
|
445
|
+
const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "a:blip");
|
|
440
446
|
if (blip) {
|
|
441
447
|
rId = blip.getAttribute("r:embed") || '';
|
|
442
448
|
}
|
|
443
449
|
else {
|
|
444
|
-
const imagedata = (0,
|
|
450
|
+
const imagedata = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "v:imagedata");
|
|
445
451
|
if (imagedata) {
|
|
446
452
|
rId = imagedata.getAttribute("r:id") || '';
|
|
447
453
|
}
|
|
@@ -456,7 +462,7 @@ const parseWord = async (buffer, config) => {
|
|
|
456
462
|
metadata: { attachmentName: filename, altText: altText }
|
|
457
463
|
};
|
|
458
464
|
if (config.includeRawContent) {
|
|
459
|
-
imageNode.rawContent =
|
|
465
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
|
|
460
466
|
}
|
|
461
467
|
children.push(imageNode);
|
|
462
468
|
}
|
|
@@ -467,7 +473,7 @@ const parseWord = async (buffer, config) => {
|
|
|
467
473
|
text: '',
|
|
468
474
|
};
|
|
469
475
|
if (config.includeRawContent) {
|
|
470
|
-
imageNode.rawContent =
|
|
476
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
|
|
471
477
|
}
|
|
472
478
|
children.push(imageNode);
|
|
473
479
|
}
|
|
@@ -475,7 +481,7 @@ const parseWord = async (buffer, config) => {
|
|
|
475
481
|
}
|
|
476
482
|
// Footnotes/Endnotes inside runs
|
|
477
483
|
if (!config.ignoreNotes) {
|
|
478
|
-
const footnoteRef = (0,
|
|
484
|
+
const footnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:footnoteReference");
|
|
479
485
|
if (footnoteRef) {
|
|
480
486
|
const id = footnoteRef.getAttribute("w:id");
|
|
481
487
|
if (id && footnoteMap.has(id)) {
|
|
@@ -494,7 +500,7 @@ const parseWord = async (buffer, config) => {
|
|
|
494
500
|
}
|
|
495
501
|
}
|
|
496
502
|
}
|
|
497
|
-
const endnoteRef = (0,
|
|
503
|
+
const endnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:endnoteReference");
|
|
498
504
|
if (endnoteRef) {
|
|
499
505
|
const id = endnoteRef.getAttribute("w:id");
|
|
500
506
|
if (id && endnoteMap.has(id)) {
|
|
@@ -515,7 +521,7 @@ const parseWord = async (buffer, config) => {
|
|
|
515
521
|
}
|
|
516
522
|
}
|
|
517
523
|
}
|
|
518
|
-
else if (node.nodeName === 'w:hyperlink') {
|
|
524
|
+
else if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:hyperlink') {
|
|
519
525
|
const hlNode = node;
|
|
520
526
|
const rId = hlNode.getAttribute("r:id");
|
|
521
527
|
const anchor = hlNode.getAttribute("w:anchor");
|
|
@@ -548,8 +554,8 @@ const parseWord = async (buffer, config) => {
|
|
|
548
554
|
processChildNode(child);
|
|
549
555
|
}
|
|
550
556
|
if (isList) {
|
|
551
|
-
const numIdNode = (0,
|
|
552
|
-
const ilvlNode = (0,
|
|
557
|
+
const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
|
|
558
|
+
const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
|
|
553
559
|
const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
|
|
554
560
|
const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
|
|
555
561
|
let listType = 'ordered';
|
|
@@ -592,7 +598,7 @@ const parseWord = async (buffer, config) => {
|
|
|
592
598
|
}
|
|
593
599
|
};
|
|
594
600
|
if (config.includeRawContent)
|
|
595
|
-
listNode.rawContent =
|
|
601
|
+
listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
596
602
|
return listNode;
|
|
597
603
|
}
|
|
598
604
|
else if (isHeading) {
|
|
@@ -604,7 +610,7 @@ const parseWord = async (buffer, config) => {
|
|
|
604
610
|
metadata: { level, alignment, style: pStyleVal ?? undefined }
|
|
605
611
|
};
|
|
606
612
|
if (config.includeRawContent)
|
|
607
|
-
headingNode.rawContent =
|
|
613
|
+
headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
608
614
|
return headingNode;
|
|
609
615
|
}
|
|
610
616
|
else {
|
|
@@ -615,20 +621,20 @@ const parseWord = async (buffer, config) => {
|
|
|
615
621
|
metadata: { alignment, style: pStyleVal ?? undefined }
|
|
616
622
|
};
|
|
617
623
|
if (config.includeRawContent)
|
|
618
|
-
paraNode.rawContent =
|
|
624
|
+
paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
619
625
|
return paraNode;
|
|
620
626
|
}
|
|
621
627
|
};
|
|
622
628
|
// Helper to parse a table node
|
|
623
|
-
const parseTable = (tblNode) => {
|
|
629
|
+
const parseTable = (tblNode, documentContent) => {
|
|
624
630
|
const rows = [];
|
|
625
631
|
// Only get direct child rows, not nested table rows
|
|
626
|
-
const trNodes = (0,
|
|
632
|
+
const trNodes = (0, xmlUtils_js_1.getDirectChildren)(tblNode, "w:tr");
|
|
627
633
|
for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
|
|
628
634
|
const trNode = trNodes[rIndex];
|
|
629
635
|
const cells = [];
|
|
630
636
|
// Only get direct child cells, not nested table cells
|
|
631
|
-
const tcNodes = (0,
|
|
637
|
+
const tcNodes = (0, xmlUtils_js_1.getDirectChildren)(trNode, "w:tc");
|
|
632
638
|
for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
|
|
633
639
|
const tcNode = tcNodes[cIndex];
|
|
634
640
|
const cellChildren = [];
|
|
@@ -636,14 +642,14 @@ const parseWord = async (buffer, config) => {
|
|
|
636
642
|
// Cells contain paragraphs (and other block-level elements)
|
|
637
643
|
const cellContentNodes = Array.from(tcNode.childNodes);
|
|
638
644
|
for (const child of cellContentNodes) {
|
|
639
|
-
if (child.nodeName === 'w:p') {
|
|
640
|
-
const pNode = parseParagraph(child);
|
|
645
|
+
if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
|
|
646
|
+
const pNode = parseParagraph(child, documentContent);
|
|
641
647
|
cellChildren.push(pNode);
|
|
642
648
|
cellText += pNode.text;
|
|
643
649
|
}
|
|
644
|
-
else if (child.nodeName === 'w:tbl') {
|
|
650
|
+
else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
|
|
645
651
|
// Nested table
|
|
646
|
-
const nestedTable = parseTable(child);
|
|
652
|
+
const nestedTable = parseTable(child, documentContent);
|
|
647
653
|
cellChildren.push(nestedTable);
|
|
648
654
|
// Don't add nested table text to cell text - it will be handled recursively
|
|
649
655
|
}
|
|
@@ -671,26 +677,28 @@ const parseWord = async (buffer, config) => {
|
|
|
671
677
|
if (!config.ignoreNotes) {
|
|
672
678
|
const footnotesFile = files.find(f => f.path.match(footnotesFileRegex));
|
|
673
679
|
if (footnotesFile) {
|
|
674
|
-
const footnotesDoc = (0,
|
|
675
|
-
const
|
|
680
|
+
const footnotesDoc = (0, xmlUtils_js_1.parseXmlString)(footnotesFile.content.toString());
|
|
681
|
+
const footnoteXml = footnotesFile.content.toString();
|
|
682
|
+
const footnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(footnotesDoc, "w:footnote");
|
|
676
683
|
for (const node of footnoteNodes) {
|
|
677
684
|
const id = node.getAttribute("w:id");
|
|
678
685
|
if (!id || id === "-1" || id === "0")
|
|
679
686
|
continue;
|
|
680
|
-
const pNodes = (0,
|
|
681
|
-
footnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
|
|
687
|
+
const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
|
|
688
|
+
footnoteMap.set(id, pNodes.map(p => parseParagraph(p, footnoteXml)));
|
|
682
689
|
}
|
|
683
690
|
}
|
|
684
691
|
const endnotesFile = files.find(f => f.path.match(endnotesFileRegex));
|
|
685
692
|
if (endnotesFile) {
|
|
686
|
-
const endnotesDoc = (0,
|
|
687
|
-
const
|
|
693
|
+
const endnotesDoc = (0, xmlUtils_js_1.parseXmlString)(endnotesFile.content.toString());
|
|
694
|
+
const endnoteXml = endnotesFile.content.toString();
|
|
695
|
+
const endnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(endnotesDoc, "w:endnote");
|
|
688
696
|
for (const node of endnoteNodes) {
|
|
689
697
|
const id = node.getAttribute("w:id");
|
|
690
698
|
if (!id || id === "-1" || id === "0")
|
|
691
699
|
continue;
|
|
692
|
-
const pNodes = (0,
|
|
693
|
-
endnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
|
|
700
|
+
const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
|
|
701
|
+
endnoteMap.set(id, pNodes.map(p => parseParagraph(p, endnoteXml)));
|
|
694
702
|
}
|
|
695
703
|
}
|
|
696
704
|
}
|
|
@@ -711,16 +719,16 @@ const parseWord = async (buffer, config) => {
|
|
|
711
719
|
if (config.includeRawContent) {
|
|
712
720
|
rawContents.push(documentContent);
|
|
713
721
|
}
|
|
714
|
-
const doc = (0,
|
|
715
|
-
const body = (0,
|
|
722
|
+
const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
|
|
723
|
+
const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
|
|
716
724
|
if (body) {
|
|
717
725
|
const bodyChildren = Array.from(body.childNodes);
|
|
718
726
|
for (const child of bodyChildren) {
|
|
719
|
-
if (child.nodeName === 'w:p') {
|
|
720
|
-
content.push(parseParagraph(child));
|
|
727
|
+
if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
|
|
728
|
+
content.push(parseParagraph(child, documentContent));
|
|
721
729
|
}
|
|
722
|
-
else if (child.nodeName === 'w:tbl') {
|
|
723
|
-
content.push(parseTable(child));
|
|
730
|
+
else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
|
|
731
|
+
content.push(parseTable(child, documentContent));
|
|
724
732
|
}
|
|
725
733
|
}
|
|
726
734
|
}
|
|
@@ -728,15 +736,15 @@ const parseWord = async (buffer, config) => {
|
|
|
728
736
|
// Extract attachments
|
|
729
737
|
if (config.extractAttachments) {
|
|
730
738
|
for (const media of mediaFiles) {
|
|
731
|
-
const attachment = (0,
|
|
739
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
732
740
|
attachments.push(attachment);
|
|
733
741
|
if (config.ocr) {
|
|
734
742
|
if (attachment.mimeType.startsWith('image/')) {
|
|
735
743
|
try {
|
|
736
|
-
attachment.ocrText = (await (0,
|
|
744
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
737
745
|
}
|
|
738
746
|
catch (e) {
|
|
739
|
-
(0,
|
|
747
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
740
748
|
}
|
|
741
749
|
}
|
|
742
750
|
}
|