officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
|
@@ -40,6 +40,7 @@
|
|
|
40
40
|
* - `<w:p>` - Paragraph
|
|
41
41
|
* - `<w:r>` - Run (contiguous text with same formatting)
|
|
42
42
|
* - `<w:t>` - Text content
|
|
43
|
+
* - `<w:br>` - Line or page break
|
|
43
44
|
* - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
|
|
44
45
|
* - `<w:pStyle>` - Paragraph style (for headings)
|
|
45
46
|
* - `<w:numPr>` - List numbering properties
|
|
@@ -61,12 +62,11 @@
|
|
|
61
62
|
*/
|
|
62
63
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
63
64
|
exports.parseWord = void 0;
|
|
64
|
-
const
|
|
65
|
-
const
|
|
66
|
-
const
|
|
67
|
-
const
|
|
68
|
-
const
|
|
69
|
-
const zipUtils_1 = require("../utils/zipUtils");
|
|
65
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
66
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
67
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
68
|
+
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
69
|
+
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
70
70
|
/**
|
|
71
71
|
* Parses a Word document (.docx) and extracts content, formatting, and metadata.
|
|
72
72
|
*
|
|
@@ -90,13 +90,13 @@ const parseWord = async (buffer, config) => {
|
|
|
90
90
|
const numberingFileRegex = /word\/numbering[\d+]?.xml/;
|
|
91
91
|
const mediaFileRegex = /(word\/)?media\/.*/;
|
|
92
92
|
const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
|
|
93
|
+
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
93
94
|
const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
|
|
94
95
|
const stylesFileRegex = /word\/styles[\d+]?.xml/;
|
|
95
|
-
const xmlSerializer = new xmldom_1.XMLSerializer();
|
|
96
96
|
// Helper to extract formatting from run properties XML string
|
|
97
97
|
const extractFormattingFromXml = (rPr) => {
|
|
98
98
|
const formatting = {};
|
|
99
|
-
const rPrString =
|
|
99
|
+
const rPrString = (0, xmlUtils_js_1.serializeXml)(rPr);
|
|
100
100
|
// Helper to check boolean properties
|
|
101
101
|
const getBoolVal = (xmlSnippet, tagName) => {
|
|
102
102
|
const regex = new RegExp(`<${tagName}(?:\\s+w:val="([^"]+)")?\\s*\\/?>`);
|
|
@@ -133,7 +133,7 @@ const parseWord = async (buffer, config) => {
|
|
|
133
133
|
// Font size
|
|
134
134
|
const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
|
|
135
135
|
if (szMatch)
|
|
136
|
-
formatting.size = (parseInt(szMatch[1]) / 2).toString() + 'pt';
|
|
136
|
+
formatting.size = (parseInt(szMatch[1], 10) / 2).toString() + 'pt';
|
|
137
137
|
// Color
|
|
138
138
|
const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
|
|
139
139
|
if (colorMatch && colorMatch[1] !== 'auto')
|
|
@@ -173,17 +173,45 @@ const parseWord = async (buffer, config) => {
|
|
|
173
173
|
}
|
|
174
174
|
return formatting;
|
|
175
175
|
};
|
|
176
|
-
|
|
176
|
+
// Helper to extract indentation from paragraph properties XML string
|
|
177
|
+
const extractIndentationFromXml = (pPr) => {
|
|
178
|
+
const ind = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:ind");
|
|
179
|
+
if (ind) {
|
|
180
|
+
const indentation = {};
|
|
181
|
+
const left = ind.getAttribute("w:left") || ind.getAttribute("w:start");
|
|
182
|
+
const right = ind.getAttribute("w:right") || ind.getAttribute("w:end");
|
|
183
|
+
const firstLine = ind.getAttribute("w:firstLine");
|
|
184
|
+
const hanging = ind.getAttribute("w:hanging");
|
|
185
|
+
if (left)
|
|
186
|
+
indentation.left = parseInt(left, 10);
|
|
187
|
+
if (right)
|
|
188
|
+
indentation.right = parseInt(right, 10);
|
|
189
|
+
if (firstLine)
|
|
190
|
+
indentation.firstLine = parseInt(firstLine, 10);
|
|
191
|
+
if (hanging)
|
|
192
|
+
indentation.hanging = parseInt(hanging, 10);
|
|
193
|
+
return Object.keys(indentation).length > 0 ? indentation : undefined;
|
|
194
|
+
}
|
|
195
|
+
return undefined;
|
|
196
|
+
};
|
|
197
|
+
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
|
|
177
198
|
!!x.match(footnotesFileRegex) ||
|
|
178
199
|
!!x.match(endnotesFileRegex) ||
|
|
179
200
|
!!x.match(numberingFileRegex) ||
|
|
180
201
|
!!x.match(corePropsFileRegex) ||
|
|
202
|
+
!!x.match(customPropsFileRegex) ||
|
|
181
203
|
!!x.match(relsFileRegex) ||
|
|
182
204
|
!!x.match(stylesFileRegex) ||
|
|
183
205
|
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
184
206
|
// Extract metadata
|
|
185
207
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
186
|
-
const metadata = corePropsFile ? (0,
|
|
208
|
+
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
209
|
+
const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
|
|
210
|
+
if (customPropsFile) {
|
|
211
|
+
const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
|
|
212
|
+
if (Object.keys(customProperties).length > 0)
|
|
213
|
+
metadata.customProperties = customProperties;
|
|
214
|
+
}
|
|
187
215
|
const footnoteMap = new Map();
|
|
188
216
|
const endnoteMap = new Map();
|
|
189
217
|
const collectedNotes = [];
|
|
@@ -193,11 +221,11 @@ const parseWord = async (buffer, config) => {
|
|
|
193
221
|
const relsFile = files.find(f => f.path.match(relsFileRegex));
|
|
194
222
|
const relsMap = {};
|
|
195
223
|
if (relsFile) {
|
|
196
|
-
const relsXml = (0,
|
|
197
|
-
const relationships = (0,
|
|
198
|
-
for (
|
|
199
|
-
const id =
|
|
200
|
-
const target =
|
|
224
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
|
|
225
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
226
|
+
for (const relationship of relationships) {
|
|
227
|
+
const id = relationship.getAttribute("Id");
|
|
228
|
+
const target = relationship.getAttribute("Target");
|
|
201
229
|
if (id && target) {
|
|
202
230
|
relsMap[id] = target;
|
|
203
231
|
}
|
|
@@ -206,27 +234,27 @@ const parseWord = async (buffer, config) => {
|
|
|
206
234
|
const numberingFile = files.find(f => f.path.match(numberingFileRegex));
|
|
207
235
|
const numberingMap = {};
|
|
208
236
|
if (numberingFile) {
|
|
209
|
-
const numberingXml = (0,
|
|
210
|
-
const nums = (0,
|
|
211
|
-
const abstractNums = (0,
|
|
237
|
+
const numberingXml = (0, xmlUtils_js_1.parseXmlString)(numberingFile.content.toString());
|
|
238
|
+
const nums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:num");
|
|
239
|
+
const abstractNums = (0, xmlUtils_js_1.getElementsByTagName)(numberingXml, "w:abstractNum");
|
|
212
240
|
const abstractNumMap = {};
|
|
213
|
-
for (
|
|
214
|
-
const abstractNumId =
|
|
241
|
+
for (const abstractNum of abstractNums) {
|
|
242
|
+
const abstractNumId = abstractNum.getAttribute("w:abstractNumId");
|
|
215
243
|
if (abstractNumId) {
|
|
216
|
-
abstractNumMap[abstractNumId] =
|
|
244
|
+
abstractNumMap[abstractNumId] = abstractNum;
|
|
217
245
|
}
|
|
218
246
|
}
|
|
219
|
-
for (
|
|
220
|
-
const numId =
|
|
221
|
-
const abstractNumIdNode = (0,
|
|
247
|
+
for (const num of nums) {
|
|
248
|
+
const numId = num.getAttribute("w:numId");
|
|
249
|
+
const abstractNumIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(num, "w:abstractNumId");
|
|
222
250
|
const abstractNumId = abstractNumIdNode?.getAttribute("w:val");
|
|
223
251
|
if (numId && abstractNumId && abstractNumMap[abstractNumId]) {
|
|
224
252
|
numberingMap[numId] = {};
|
|
225
|
-
const lvls = (0,
|
|
226
|
-
for (
|
|
227
|
-
const ilvl =
|
|
228
|
-
const numFmtNode = (0,
|
|
229
|
-
const lvlTextNode = (0,
|
|
253
|
+
const lvls = (0, xmlUtils_js_1.getElementsByTagName)(abstractNumMap[abstractNumId], "w:lvl");
|
|
254
|
+
for (const lvl of lvls) {
|
|
255
|
+
const ilvl = lvl.getAttribute("w:ilvl");
|
|
256
|
+
const numFmtNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:numFmt");
|
|
257
|
+
const lvlTextNode = (0, xmlUtils_js_1.getFirstElementByTagName)(lvl, "w:lvlText");
|
|
230
258
|
if (ilvl) {
|
|
231
259
|
numberingMap[numId][ilvl] = {
|
|
232
260
|
numFmt: numFmtNode?.getAttribute("w:val") || 'decimal',
|
|
@@ -241,44 +269,48 @@ const parseWord = async (buffer, config) => {
|
|
|
241
269
|
const stylesFile = files.find(f => f.path.match(stylesFileRegex));
|
|
242
270
|
const styleMap = {};
|
|
243
271
|
if (stylesFile) {
|
|
244
|
-
const stylesXml = (0,
|
|
245
|
-
const styles = (0,
|
|
246
|
-
for (
|
|
247
|
-
const styleId =
|
|
272
|
+
const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
273
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
|
|
274
|
+
for (const style of styles) {
|
|
275
|
+
const styleId = style.getAttribute("w:styleId");
|
|
248
276
|
if (styleId) {
|
|
249
|
-
const rPr = (0,
|
|
250
|
-
const pPr = (0,
|
|
277
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:rPr");
|
|
278
|
+
const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "w:pPr");
|
|
251
279
|
const formatting = rPr ? extractFormattingFromXml(rPr) : {};
|
|
252
280
|
let alignment = undefined;
|
|
253
281
|
let backgroundColor = undefined;
|
|
282
|
+
let paragraphIndentation = undefined;
|
|
254
283
|
if (pPr) {
|
|
255
|
-
const jc = (0,
|
|
284
|
+
const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
|
|
256
285
|
if (jc) {
|
|
257
286
|
const val = jc.getAttribute("w:val");
|
|
258
287
|
if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
|
|
259
288
|
alignment = val;
|
|
260
289
|
}
|
|
261
290
|
}
|
|
262
|
-
const shd = (0,
|
|
291
|
+
const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
|
|
263
292
|
if (shd) {
|
|
264
293
|
const fill = shd.getAttribute("w:fill");
|
|
265
294
|
if (fill && fill !== 'auto')
|
|
266
295
|
backgroundColor = '#' + fill;
|
|
267
296
|
}
|
|
297
|
+
const ind = extractIndentationFromXml(pPr);
|
|
298
|
+
if (ind)
|
|
299
|
+
paragraphIndentation = ind;
|
|
268
300
|
}
|
|
269
|
-
styleMap[styleId] = { formatting, alignment, backgroundColor };
|
|
301
|
+
styleMap[styleId] = { formatting, alignment, backgroundColor, paragraphIndentation };
|
|
270
302
|
}
|
|
271
303
|
}
|
|
272
304
|
}
|
|
273
305
|
// Extract document defaults
|
|
274
306
|
let docDefaults = {};
|
|
275
307
|
if (stylesFile) {
|
|
276
|
-
const stylesXml = (0,
|
|
277
|
-
const docDefaultsNode = (0,
|
|
308
|
+
const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
309
|
+
const docDefaultsNode = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesXml, "w:docDefaults");
|
|
278
310
|
if (docDefaultsNode) {
|
|
279
|
-
const rPrDefaultNode = (0,
|
|
311
|
+
const rPrDefaultNode = (0, xmlUtils_js_1.getFirstElementByTagName)(docDefaultsNode, "w:rPrDefault");
|
|
280
312
|
if (rPrDefaultNode) {
|
|
281
|
-
const rPr = (0,
|
|
313
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(rPrDefaultNode, "w:rPr");
|
|
282
314
|
if (rPr) {
|
|
283
315
|
docDefaults = extractFormattingFromXml(rPr);
|
|
284
316
|
}
|
|
@@ -288,13 +320,13 @@ const parseWord = async (buffer, config) => {
|
|
|
288
320
|
// Detect the default paragraph style (for international compatibility)
|
|
289
321
|
let defaultParaStyleId = undefined;
|
|
290
322
|
if (stylesFile) {
|
|
291
|
-
const stylesXml = (0,
|
|
292
|
-
const styles = (0,
|
|
323
|
+
const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
324
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "w:style");
|
|
293
325
|
// Look for a style with w:type="paragraph" and w:default="1"
|
|
294
|
-
for (
|
|
295
|
-
const styleType =
|
|
296
|
-
const isDefault =
|
|
297
|
-
const styleId =
|
|
326
|
+
for (const style of styles) {
|
|
327
|
+
const styleType = style.getAttribute("w:type");
|
|
328
|
+
const isDefault = style.getAttribute("w:default");
|
|
329
|
+
const styleId = style.getAttribute("w:styleId");
|
|
298
330
|
if (styleType === "paragraph" && isDefault === "1" && styleId) {
|
|
299
331
|
defaultParaStyleId = styleId;
|
|
300
332
|
break;
|
|
@@ -310,22 +342,22 @@ const parseWord = async (buffer, config) => {
|
|
|
310
342
|
const numberingState = {};
|
|
311
343
|
const listCounters = {}; // Track item index per listId/level
|
|
312
344
|
// Helper to parse a paragraph node
|
|
313
|
-
const parseParagraph = (pNode) => {
|
|
314
|
-
const pXml =
|
|
345
|
+
const parseParagraph = (pNode, documentContent) => {
|
|
346
|
+
const pXml = pNode.toString();
|
|
315
347
|
// Check if it's a list item
|
|
316
|
-
const numPr = (0,
|
|
348
|
+
const numPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:numPr");
|
|
317
349
|
const isList = !!numPr;
|
|
318
350
|
// Check if it's a heading
|
|
319
|
-
const pPr = (0,
|
|
320
|
-
const pStyle = pPr ? (0,
|
|
321
|
-
const pStyleVal = pStyle
|
|
351
|
+
const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pNode, "w:pPr");
|
|
352
|
+
const pStyle = pPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:pStyle") : null;
|
|
353
|
+
const pStyleVal = pStyle?.getAttribute("w:val");
|
|
322
354
|
const isHeading = pStyleVal ? (pStyleVal.startsWith("Heading") || pStyleVal === "Title") : false;
|
|
323
355
|
// Extract Paragraph Style Properties
|
|
324
356
|
const styleProps = pStyleVal && styleMap[pStyleVal] ? styleMap[pStyleVal] : { formatting: {} };
|
|
325
357
|
// Extract Alignment
|
|
326
358
|
let alignment = styleProps.alignment;
|
|
327
359
|
if (pPr) {
|
|
328
|
-
const jc = (0,
|
|
360
|
+
const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
|
|
329
361
|
if (jc) {
|
|
330
362
|
const val = jc.getAttribute("w:val");
|
|
331
363
|
if (val === 'left' || val === 'center' || val === 'right' || val === 'justify') {
|
|
@@ -333,10 +365,18 @@ const parseWord = async (buffer, config) => {
|
|
|
333
365
|
}
|
|
334
366
|
}
|
|
335
367
|
}
|
|
368
|
+
// Extract Indentation
|
|
369
|
+
let paraIndentation = styleProps.paragraphIndentation;
|
|
370
|
+
if (pPr) {
|
|
371
|
+
const ind = extractIndentationFromXml(pPr);
|
|
372
|
+
if (ind) {
|
|
373
|
+
paraIndentation = { ...paraIndentation, ...ind };
|
|
374
|
+
}
|
|
375
|
+
}
|
|
336
376
|
// Extract Paragraph Background
|
|
337
377
|
let paraBackgroundColor = styleProps.backgroundColor;
|
|
338
378
|
if (pPr) {
|
|
339
|
-
const shd = (0,
|
|
379
|
+
const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:shd");
|
|
340
380
|
if (shd) {
|
|
341
381
|
const fill = shd.getAttribute("w:fill");
|
|
342
382
|
if (fill && fill !== 'auto') {
|
|
@@ -347,7 +387,7 @@ const parseWord = async (buffer, config) => {
|
|
|
347
387
|
// Extract paragraph-level run properties
|
|
348
388
|
let paragraphRunFormatting = { ...styleProps.formatting };
|
|
349
389
|
if (pPr) {
|
|
350
|
-
const pPrRPr = (0,
|
|
390
|
+
const pPrRPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:rPr");
|
|
351
391
|
if (pPrRPr) {
|
|
352
392
|
const pPrFormatting = extractFormattingFromXml(pPrRPr);
|
|
353
393
|
for (const key in pPrFormatting) {
|
|
@@ -366,9 +406,9 @@ const parseWord = async (buffer, config) => {
|
|
|
366
406
|
const children = [];
|
|
367
407
|
// Traverse children of paragraph (runs, hyperlinks, etc.)
|
|
368
408
|
const processChildNode = (node) => {
|
|
369
|
-
if (node.nodeName === 'w:r') {
|
|
409
|
+
if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
|
|
370
410
|
const runNode = node;
|
|
371
|
-
const rPr = (0,
|
|
411
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:rPr");
|
|
372
412
|
// Formatting
|
|
373
413
|
let formatting = {};
|
|
374
414
|
// Apply paragraph-level formatting
|
|
@@ -376,7 +416,7 @@ const parseWord = async (buffer, config) => {
|
|
|
376
416
|
formatting[key] = paragraphRunFormatting[key];
|
|
377
417
|
}
|
|
378
418
|
// Check for run style
|
|
379
|
-
const rStyle = rPr ? (0,
|
|
419
|
+
const rStyle = rPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "w:rStyle") : null;
|
|
380
420
|
const rStyleVal = rStyle ? rStyle.getAttribute("w:val") : pStyleVal;
|
|
381
421
|
if (rStyleVal && styleMap[rStyleVal]) {
|
|
382
422
|
for (const key in styleMap[rStyleVal].formatting) {
|
|
@@ -400,48 +440,93 @@ const parseWord = async (buffer, config) => {
|
|
|
400
440
|
if (!formatting.backgroundColor && paraBackgroundColor) {
|
|
401
441
|
formatting.backgroundColor = paraBackgroundColor;
|
|
402
442
|
}
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
443
|
+
for (const child of runNode.childNodes) {
|
|
444
|
+
if (!(0, xmlUtils_js_1.isElement)(child))
|
|
445
|
+
continue;
|
|
446
|
+
// also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
|
|
447
|
+
// Text content
|
|
448
|
+
if (child.tagName === "w:t" || child.tagName === "t") {
|
|
449
|
+
const tNode = child;
|
|
450
|
+
const tContent = tNode.textContent || '';
|
|
451
|
+
text += tContent;
|
|
452
|
+
const textNode = {
|
|
453
|
+
type: 'text',
|
|
454
|
+
text: tContent,
|
|
455
|
+
formatting: formatting
|
|
456
|
+
};
|
|
457
|
+
if (config.includeRawContent) {
|
|
458
|
+
textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
|
|
459
|
+
}
|
|
460
|
+
// Always set a style: run style > paragraph style > detected default
|
|
461
|
+
// Use detected default style for international compatibility
|
|
462
|
+
const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
|
|
463
|
+
if (nodeStyle) {
|
|
464
|
+
textNode.metadata = { style: nodeStyle };
|
|
465
|
+
}
|
|
466
|
+
children.push(textNode);
|
|
467
|
+
}
|
|
468
|
+
// Break nodes
|
|
469
|
+
else if (config.includeBreakNodes &&
|
|
470
|
+
(child.tagName === "w:br"
|
|
471
|
+
|| child.tagName === "br"
|
|
472
|
+
|| child.tagName === "w:cr"
|
|
473
|
+
|| child.tagName === "cr")) {
|
|
474
|
+
const brNode = child;
|
|
475
|
+
let breakType = 'textWrapping';
|
|
476
|
+
if (child.tagName === "w:cr" || child.tagName === "cr") {
|
|
477
|
+
breakType = 'carriageReturn';
|
|
478
|
+
}
|
|
479
|
+
else {
|
|
480
|
+
const nodeBreakType = brNode.getAttribute("w:type") || brNode.getAttribute("type");
|
|
481
|
+
if (nodeBreakType !== null) {
|
|
482
|
+
breakType = nodeBreakType;
|
|
483
|
+
}
|
|
484
|
+
}
|
|
485
|
+
let breakClear = undefined;
|
|
486
|
+
if (breakType === 'textWrapping' && brNode.getAttribute("w:clear") !== null) {
|
|
487
|
+
breakClear = brNode.getAttribute("w:clear");
|
|
488
|
+
}
|
|
489
|
+
const breakNode = {
|
|
490
|
+
type: 'break',
|
|
491
|
+
metadata: { breakType, clear: breakClear }
|
|
492
|
+
};
|
|
493
|
+
if (config.includeRawContent) {
|
|
494
|
+
breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(brNode, documentContent, config);
|
|
495
|
+
}
|
|
496
|
+
children.push(breakNode);
|
|
415
497
|
}
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
498
|
+
else if (config.includeBreakNodes && (child.tagName === "w:lastRenderedPageBreak" || child.tagName === "lastRenderedPageBreak")) {
|
|
499
|
+
const breakNode = {
|
|
500
|
+
type: 'break',
|
|
501
|
+
metadata: { breakType: 'lastRenderedPage' }
|
|
502
|
+
};
|
|
503
|
+
if (config.includeRawContent) {
|
|
504
|
+
breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(child, documentContent, config);
|
|
505
|
+
}
|
|
506
|
+
children.push(breakNode);
|
|
421
507
|
}
|
|
422
|
-
children.push(textNode);
|
|
423
508
|
}
|
|
424
509
|
// Images/Drawings
|
|
425
510
|
if (config.extractAttachments) {
|
|
426
|
-
const drawings = (0,
|
|
427
|
-
const picts = (0,
|
|
511
|
+
const drawings = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:drawing");
|
|
512
|
+
const picts = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:pict");
|
|
428
513
|
const allImages = [...drawings, ...picts];
|
|
429
514
|
for (const imgNode of allImages) {
|
|
430
|
-
const imgXml =
|
|
515
|
+
const imgXml = (0, xmlUtils_js_1.serializeXml)(imgNode);
|
|
431
516
|
// Extract Alt Text
|
|
432
517
|
let altText = '';
|
|
433
|
-
const docPr = (0,
|
|
518
|
+
const docPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "wp:docPr");
|
|
434
519
|
if (docPr) {
|
|
435
520
|
altText = docPr.getAttribute("descr") || docPr.getAttribute("title") || '';
|
|
436
521
|
}
|
|
437
522
|
// Extract Relationship ID
|
|
438
523
|
let rId = '';
|
|
439
|
-
const blip = (0,
|
|
524
|
+
const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "a:blip");
|
|
440
525
|
if (blip) {
|
|
441
526
|
rId = blip.getAttribute("r:embed") || '';
|
|
442
527
|
}
|
|
443
528
|
else {
|
|
444
|
-
const imagedata = (0,
|
|
529
|
+
const imagedata = (0, xmlUtils_js_1.getFirstElementByTagName)(imgNode, "v:imagedata");
|
|
445
530
|
if (imagedata) {
|
|
446
531
|
rId = imagedata.getAttribute("r:id") || '';
|
|
447
532
|
}
|
|
@@ -456,7 +541,7 @@ const parseWord = async (buffer, config) => {
|
|
|
456
541
|
metadata: { attachmentName: filename, altText: altText }
|
|
457
542
|
};
|
|
458
543
|
if (config.includeRawContent) {
|
|
459
|
-
imageNode.rawContent =
|
|
544
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
|
|
460
545
|
}
|
|
461
546
|
children.push(imageNode);
|
|
462
547
|
}
|
|
@@ -467,7 +552,7 @@ const parseWord = async (buffer, config) => {
|
|
|
467
552
|
text: '',
|
|
468
553
|
};
|
|
469
554
|
if (config.includeRawContent) {
|
|
470
|
-
imageNode.rawContent =
|
|
555
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(imgNode, documentContent, config);
|
|
471
556
|
}
|
|
472
557
|
children.push(imageNode);
|
|
473
558
|
}
|
|
@@ -475,7 +560,7 @@ const parseWord = async (buffer, config) => {
|
|
|
475
560
|
}
|
|
476
561
|
// Footnotes/Endnotes inside runs
|
|
477
562
|
if (!config.ignoreNotes) {
|
|
478
|
-
const footnoteRef = (0,
|
|
563
|
+
const footnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:footnoteReference");
|
|
479
564
|
if (footnoteRef) {
|
|
480
565
|
const id = footnoteRef.getAttribute("w:id");
|
|
481
566
|
if (id && footnoteMap.has(id)) {
|
|
@@ -494,7 +579,7 @@ const parseWord = async (buffer, config) => {
|
|
|
494
579
|
}
|
|
495
580
|
}
|
|
496
581
|
}
|
|
497
|
-
const endnoteRef = (0,
|
|
582
|
+
const endnoteRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:endnoteReference");
|
|
498
583
|
if (endnoteRef) {
|
|
499
584
|
const id = endnoteRef.getAttribute("w:id");
|
|
500
585
|
if (id && endnoteMap.has(id)) {
|
|
@@ -515,7 +600,7 @@ const parseWord = async (buffer, config) => {
|
|
|
515
600
|
}
|
|
516
601
|
}
|
|
517
602
|
}
|
|
518
|
-
else if (node.nodeName === 'w:hyperlink') {
|
|
603
|
+
else if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:hyperlink') {
|
|
519
604
|
const hlNode = node;
|
|
520
605
|
const rId = hlNode.getAttribute("r:id");
|
|
521
606
|
const anchor = hlNode.getAttribute("w:anchor");
|
|
@@ -548,10 +633,10 @@ const parseWord = async (buffer, config) => {
|
|
|
548
633
|
processChildNode(child);
|
|
549
634
|
}
|
|
550
635
|
if (isList) {
|
|
551
|
-
const numIdNode = (0,
|
|
552
|
-
const ilvlNode = (0,
|
|
636
|
+
const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
|
|
637
|
+
const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
|
|
553
638
|
const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
|
|
554
|
-
const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
|
|
639
|
+
const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0', 10) : 0;
|
|
555
640
|
let listType = 'ordered';
|
|
556
641
|
let itemIndex = 0;
|
|
557
642
|
if (numId && numberingMap[numId]) {
|
|
@@ -585,6 +670,7 @@ const parseWord = async (buffer, config) => {
|
|
|
585
670
|
metadata: {
|
|
586
671
|
listType,
|
|
587
672
|
indentation: ilvl,
|
|
673
|
+
paragraphIndentation: paraIndentation,
|
|
588
674
|
alignment: (alignment || 'left'),
|
|
589
675
|
listId: numId,
|
|
590
676
|
itemIndex: itemIndex,
|
|
@@ -592,19 +678,19 @@ const parseWord = async (buffer, config) => {
|
|
|
592
678
|
}
|
|
593
679
|
};
|
|
594
680
|
if (config.includeRawContent)
|
|
595
|
-
listNode.rawContent =
|
|
681
|
+
listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
596
682
|
return listNode;
|
|
597
683
|
}
|
|
598
684
|
else if (isHeading) {
|
|
599
|
-
const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", "")) || 1 : 1;
|
|
685
|
+
const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", ""), 10) || 1 : 1;
|
|
600
686
|
const headingNode = {
|
|
601
687
|
type: 'heading',
|
|
602
688
|
text: text,
|
|
603
689
|
children: children,
|
|
604
|
-
metadata: { level, alignment, style: pStyleVal ?? undefined }
|
|
690
|
+
metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
|
|
605
691
|
};
|
|
606
692
|
if (config.includeRawContent)
|
|
607
|
-
headingNode.rawContent =
|
|
693
|
+
headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
608
694
|
return headingNode;
|
|
609
695
|
}
|
|
610
696
|
else {
|
|
@@ -612,23 +698,23 @@ const parseWord = async (buffer, config) => {
|
|
|
612
698
|
type: 'paragraph',
|
|
613
699
|
text: text,
|
|
614
700
|
children: children,
|
|
615
|
-
metadata: { alignment, style: pStyleVal ?? undefined }
|
|
701
|
+
metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
|
|
616
702
|
};
|
|
617
703
|
if (config.includeRawContent)
|
|
618
|
-
paraNode.rawContent =
|
|
704
|
+
paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
619
705
|
return paraNode;
|
|
620
706
|
}
|
|
621
707
|
};
|
|
622
708
|
// Helper to parse a table node
|
|
623
|
-
const parseTable = (tblNode) => {
|
|
709
|
+
const parseTable = (tblNode, documentContent) => {
|
|
624
710
|
const rows = [];
|
|
625
711
|
// Only get direct child rows, not nested table rows
|
|
626
|
-
const trNodes = (0,
|
|
712
|
+
const trNodes = (0, xmlUtils_js_1.getDirectChildren)(tblNode, "w:tr");
|
|
627
713
|
for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
|
|
628
714
|
const trNode = trNodes[rIndex];
|
|
629
715
|
const cells = [];
|
|
630
716
|
// Only get direct child cells, not nested table cells
|
|
631
|
-
const tcNodes = (0,
|
|
717
|
+
const tcNodes = (0, xmlUtils_js_1.getDirectChildren)(trNode, "w:tc");
|
|
632
718
|
for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
|
|
633
719
|
const tcNode = tcNodes[cIndex];
|
|
634
720
|
const cellChildren = [];
|
|
@@ -636,14 +722,14 @@ const parseWord = async (buffer, config) => {
|
|
|
636
722
|
// Cells contain paragraphs (and other block-level elements)
|
|
637
723
|
const cellContentNodes = Array.from(tcNode.childNodes);
|
|
638
724
|
for (const child of cellContentNodes) {
|
|
639
|
-
if (child.nodeName === 'w:p') {
|
|
640
|
-
const pNode = parseParagraph(child);
|
|
725
|
+
if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
|
|
726
|
+
const pNode = parseParagraph(child, documentContent);
|
|
641
727
|
cellChildren.push(pNode);
|
|
642
728
|
cellText += pNode.text;
|
|
643
729
|
}
|
|
644
|
-
else if (child.nodeName === 'w:tbl') {
|
|
730
|
+
else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
|
|
645
731
|
// Nested table
|
|
646
|
-
const nestedTable = parseTable(child);
|
|
732
|
+
const nestedTable = parseTable(child, documentContent);
|
|
647
733
|
cellChildren.push(nestedTable);
|
|
648
734
|
// Don't add nested table text to cell text - it will be handled recursively
|
|
649
735
|
}
|
|
@@ -671,26 +757,28 @@ const parseWord = async (buffer, config) => {
|
|
|
671
757
|
if (!config.ignoreNotes) {
|
|
672
758
|
const footnotesFile = files.find(f => f.path.match(footnotesFileRegex));
|
|
673
759
|
if (footnotesFile) {
|
|
674
|
-
const footnotesDoc = (0,
|
|
675
|
-
const
|
|
760
|
+
const footnotesDoc = (0, xmlUtils_js_1.parseXmlString)(footnotesFile.content.toString());
|
|
761
|
+
const footnoteXml = footnotesFile.content.toString();
|
|
762
|
+
const footnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(footnotesDoc, "w:footnote");
|
|
676
763
|
for (const node of footnoteNodes) {
|
|
677
764
|
const id = node.getAttribute("w:id");
|
|
678
765
|
if (!id || id === "-1" || id === "0")
|
|
679
766
|
continue;
|
|
680
|
-
const pNodes = (0,
|
|
681
|
-
footnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
|
|
767
|
+
const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
|
|
768
|
+
footnoteMap.set(id, pNodes.map(p => parseParagraph(p, footnoteXml)));
|
|
682
769
|
}
|
|
683
770
|
}
|
|
684
771
|
const endnotesFile = files.find(f => f.path.match(endnotesFileRegex));
|
|
685
772
|
if (endnotesFile) {
|
|
686
|
-
const endnotesDoc = (0,
|
|
687
|
-
const
|
|
773
|
+
const endnotesDoc = (0, xmlUtils_js_1.parseXmlString)(endnotesFile.content.toString());
|
|
774
|
+
const endnoteXml = endnotesFile.content.toString();
|
|
775
|
+
const endnoteNodes = (0, xmlUtils_js_1.getElementsByTagName)(endnotesDoc, "w:endnote");
|
|
688
776
|
for (const node of endnoteNodes) {
|
|
689
777
|
const id = node.getAttribute("w:id");
|
|
690
778
|
if (!id || id === "-1" || id === "0")
|
|
691
779
|
continue;
|
|
692
|
-
const pNodes = (0,
|
|
693
|
-
endnoteMap.set(id, pNodes.map(p => parseParagraph(p)));
|
|
780
|
+
const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
|
|
781
|
+
endnoteMap.set(id, pNodes.map(p => parseParagraph(p, endnoteXml)));
|
|
694
782
|
}
|
|
695
783
|
}
|
|
696
784
|
}
|
|
@@ -711,16 +799,16 @@ const parseWord = async (buffer, config) => {
|
|
|
711
799
|
if (config.includeRawContent) {
|
|
712
800
|
rawContents.push(documentContent);
|
|
713
801
|
}
|
|
714
|
-
const doc = (0,
|
|
715
|
-
const body = (0,
|
|
802
|
+
const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
|
|
803
|
+
const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
|
|
716
804
|
if (body) {
|
|
717
805
|
const bodyChildren = Array.from(body.childNodes);
|
|
718
806
|
for (const child of bodyChildren) {
|
|
719
|
-
if (child.nodeName === 'w:p') {
|
|
720
|
-
content.push(parseParagraph(child));
|
|
807
|
+
if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:p') {
|
|
808
|
+
content.push(parseParagraph(child, documentContent));
|
|
721
809
|
}
|
|
722
|
-
else if (child.nodeName === 'w:tbl') {
|
|
723
|
-
content.push(parseTable(child));
|
|
810
|
+
else if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === 'w:tbl') {
|
|
811
|
+
content.push(parseTable(child, documentContent));
|
|
724
812
|
}
|
|
725
813
|
}
|
|
726
814
|
}
|
|
@@ -728,15 +816,15 @@ const parseWord = async (buffer, config) => {
|
|
|
728
816
|
// Extract attachments
|
|
729
817
|
if (config.extractAttachments) {
|
|
730
818
|
for (const media of mediaFiles) {
|
|
731
|
-
const attachment = (0,
|
|
819
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
732
820
|
attachments.push(attachment);
|
|
733
821
|
if (config.ocr) {
|
|
734
822
|
if (attachment.mimeType.startsWith('image/')) {
|
|
735
823
|
try {
|
|
736
|
-
attachment.ocrText = (await (0,
|
|
824
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
737
825
|
}
|
|
738
826
|
catch (e) {
|
|
739
|
-
(0,
|
|
827
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
740
828
|
}
|
|
741
829
|
}
|
|
742
830
|
}
|
|
@@ -776,6 +864,9 @@ const parseWord = async (buffer, config) => {
|
|
|
776
864
|
if (node.children) {
|
|
777
865
|
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
|
|
778
866
|
}
|
|
867
|
+
else if (node.type === 'break') {
|
|
868
|
+
t += config.newlineDelimiter ?? '\n';
|
|
869
|
+
}
|
|
779
870
|
else
|
|
780
871
|
t += node.text || '';
|
|
781
872
|
return t;
|