officeparser 6.1.0 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -63
- package/dist/OfficeParser.js +1 -0
- package/dist/cli.js +1 -0
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +54 -2
- package/dist/officeparser.browser.iife.js +49 -46
- package/dist/officeparser.browser.mjs +49 -46
- package/dist/parsers/ExcelParser.js +5 -5
- package/dist/parsers/OpenOfficeParser.js +97 -49
- package/dist/parsers/PowerPointParser.js +112 -100
- package/dist/parsers/RtfParser.d.ts +20 -0
- package/dist/parsers/RtfParser.js +107 -42
- package/dist/parsers/WordParser.d.ts +1 -0
- package/dist/parsers/WordParser.js +107 -24
- package/dist/sbom.cdx.json +102 -102
- package/dist/types.d.ts +54 -2
- package/package.json +2 -2
|
@@ -92,6 +92,12 @@ class SimpleRtfParser {
|
|
|
92
92
|
index = 0;
|
|
93
93
|
/** The RTF content as a Buffer */
|
|
94
94
|
buffer;
|
|
95
|
+
/** Current code page for character decoding (default is Windows-1252) */
|
|
96
|
+
codePage = 1252;
|
|
97
|
+
/** Cached TextDecoders for different code pages */
|
|
98
|
+
decoders = {};
|
|
99
|
+
/** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
|
|
100
|
+
pendingBytes = [];
|
|
95
101
|
/** Total length of the buffer */
|
|
96
102
|
length;
|
|
97
103
|
/**
|
|
@@ -110,12 +116,14 @@ class SimpleRtfParser {
|
|
|
110
116
|
const currentGroup = stack[stack.length - 1];
|
|
111
117
|
if (char === 0x7B) { // '{'
|
|
112
118
|
this.index++;
|
|
119
|
+
this.flushPendingText(currentGroup);
|
|
113
120
|
const newGroup = { type: 'group', content: [] };
|
|
114
121
|
currentGroup.content.push(newGroup);
|
|
115
122
|
stack.push(newGroup);
|
|
116
123
|
}
|
|
117
124
|
else if (char === 0x7D) { // '}'
|
|
118
125
|
this.index++;
|
|
126
|
+
this.flushPendingText(currentGroup);
|
|
119
127
|
if (stack.length > 1) {
|
|
120
128
|
stack.pop();
|
|
121
129
|
}
|
|
@@ -133,6 +141,7 @@ class SimpleRtfParser {
|
|
|
133
141
|
this.parseText(currentGroup);
|
|
134
142
|
}
|
|
135
143
|
}
|
|
144
|
+
this.flushPendingText(root);
|
|
136
145
|
return root;
|
|
137
146
|
}
|
|
138
147
|
parseControl(group) {
|
|
@@ -141,55 +150,23 @@ class SimpleRtfParser {
|
|
|
141
150
|
const char = this.buffer[this.index];
|
|
142
151
|
// Special control symbols
|
|
143
152
|
if (char === 0x7B || char === 0x7D || char === 0x5C) { // \{ \} \\
|
|
144
|
-
|
|
153
|
+
this.pendingBytes.push(char);
|
|
145
154
|
this.index++;
|
|
146
155
|
return;
|
|
147
156
|
}
|
|
148
157
|
if (char === 0x27) { // \'xx (hex)
|
|
149
158
|
this.index++;
|
|
150
159
|
if (this.index + 1 < this.length) {
|
|
151
|
-
const hex = this.buffer
|
|
160
|
+
const hex = String.fromCharCode(this.buffer[this.index], this.buffer[this.index + 1]);
|
|
152
161
|
const code = parseInt(hex, 16);
|
|
153
162
|
if (!isNaN(code)) {
|
|
154
|
-
|
|
155
|
-
// Characters 0x80-0x9F in Windows-1252 don't map directly to Unicode
|
|
156
|
-
// We need to convert them properly
|
|
157
|
-
const windows1252ToUnicode = {
|
|
158
|
-
0x80: 0x20AC, // €
|
|
159
|
-
0x82: 0x201A, // ‚
|
|
160
|
-
0x83: 0x0192, // ƒ
|
|
161
|
-
0x84: 0x201E, // „
|
|
162
|
-
0x85: 0x2026, // …
|
|
163
|
-
0x86: 0x2020, // †
|
|
164
|
-
0x87: 0x2021, // ‡
|
|
165
|
-
0x88: 0x02C6, // ˆ
|
|
166
|
-
0x89: 0x2030, // ‰
|
|
167
|
-
0x8A: 0x0160, // Š
|
|
168
|
-
0x8B: 0x2039, // ‹
|
|
169
|
-
0x8C: 0x0152, // Œ
|
|
170
|
-
0x8E: 0x017D, // Ž
|
|
171
|
-
0x91: 0x2018, // '
|
|
172
|
-
0x92: 0x2019, // '
|
|
173
|
-
0x93: 0x201C, // "
|
|
174
|
-
0x94: 0x201D, // "
|
|
175
|
-
0x95: 0x2022, // •
|
|
176
|
-
0x96: 0x2013, // –
|
|
177
|
-
0x97: 0x2014, // —
|
|
178
|
-
0x98: 0x02DC, // ˜
|
|
179
|
-
0x99: 0x2122, // ™
|
|
180
|
-
0x9A: 0x0161, // š
|
|
181
|
-
0x9B: 0x203A, // ›
|
|
182
|
-
0x9C: 0x0153, // œ
|
|
183
|
-
0x9E: 0x017E, // ž
|
|
184
|
-
0x9F: 0x0178 // Ÿ
|
|
185
|
-
};
|
|
186
|
-
const unicodeCode = windows1252ToUnicode[code] || code;
|
|
187
|
-
group.content.push({ type: 'text', value: String.fromCharCode(unicodeCode) });
|
|
163
|
+
this.pendingBytes.push(code);
|
|
188
164
|
}
|
|
189
165
|
this.index += 2;
|
|
190
166
|
}
|
|
191
167
|
return;
|
|
192
168
|
}
|
|
169
|
+
this.flushPendingText(group);
|
|
193
170
|
if (char === 0x2A) { // \* (ignorable destination)
|
|
194
171
|
// We treat this as a control word named '*'
|
|
195
172
|
group.content.push({ type: 'control', value: '*' });
|
|
@@ -231,7 +208,7 @@ class SimpleRtfParser {
|
|
|
231
208
|
param = parseInt(paramStr, 10);
|
|
232
209
|
}
|
|
233
210
|
// Space after control word is consumed
|
|
234
|
-
if (this.index < this.length && this.buffer[this.index] === 0x20) {
|
|
211
|
+
if (name !== '' && this.index < this.length && this.buffer[this.index] === 0x20) {
|
|
235
212
|
this.index++;
|
|
236
213
|
}
|
|
237
214
|
// Handle \binN
|
|
@@ -241,6 +218,22 @@ class SimpleRtfParser {
|
|
|
241
218
|
// \binN is not added to content as we want to ignore it
|
|
242
219
|
return;
|
|
243
220
|
}
|
|
221
|
+
// Handle encoding control words
|
|
222
|
+
if (name === 'ansicpg' && param !== undefined) {
|
|
223
|
+
this.codePage = param;
|
|
224
|
+
}
|
|
225
|
+
else if (name === 'ansi') {
|
|
226
|
+
this.codePage = 1252;
|
|
227
|
+
}
|
|
228
|
+
else if (name === 'mac') {
|
|
229
|
+
this.codePage = 10000;
|
|
230
|
+
}
|
|
231
|
+
else if (name === 'pc') {
|
|
232
|
+
this.codePage = 437;
|
|
233
|
+
}
|
|
234
|
+
else if (name === 'pca') {
|
|
235
|
+
this.codePage = 850;
|
|
236
|
+
}
|
|
244
237
|
group.content.push({ type: 'control', value: name, param });
|
|
245
238
|
// If this is the first control word in the group, it might be the destination
|
|
246
239
|
if (group.content.length === 1 && group.type === 'group') {
|
|
@@ -252,19 +245,91 @@ class SimpleRtfParser {
|
|
|
252
245
|
}
|
|
253
246
|
}
|
|
254
247
|
parseText(group) {
|
|
255
|
-
let text = '';
|
|
256
248
|
while (this.index < this.length) {
|
|
257
249
|
const char = this.buffer[this.index];
|
|
250
|
+
if (char === undefined)
|
|
251
|
+
break;
|
|
258
252
|
if (char === 0x7B || char === 0x7D || char === 0x5C || char === 0x0D || char === 0x0A) {
|
|
259
253
|
break;
|
|
260
254
|
}
|
|
261
|
-
|
|
262
|
-
text += String.fromCharCode(char);
|
|
255
|
+
this.pendingBytes.push(char);
|
|
263
256
|
this.index++;
|
|
264
257
|
}
|
|
265
|
-
|
|
266
|
-
|
|
258
|
+
}
|
|
259
|
+
/**
|
|
260
|
+
* Flushes the pending bytes buffer as a text node to the current group.
|
|
261
|
+
* @param group The group to append the text node to
|
|
262
|
+
*/
|
|
263
|
+
flushPendingText(group) {
|
|
264
|
+
if (this.pendingBytes.length > 0) {
|
|
265
|
+
group.content.push({ type: 'text', value: this.decodeBytes(this.pendingBytes, this.codePage) });
|
|
266
|
+
this.pendingBytes = [];
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Decodes a byte array using a "UTF-8 first" strategy.
|
|
271
|
+
* If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
|
|
272
|
+
* Otherwise, falls back to the specified code page.
|
|
273
|
+
* @param bytes The bytes to decode
|
|
274
|
+
* @param codePage The RTF code page ID
|
|
275
|
+
* @returns The decoded string
|
|
276
|
+
*/
|
|
277
|
+
decodeBytes(bytes, codePage) {
|
|
278
|
+
const uint8 = new Uint8Array(bytes);
|
|
279
|
+
// Try UTF-8 first if there are any non-ASCII bytes.
|
|
280
|
+
// Many modern RTF generators (like calibre or web-based tools) dump UTF-8 bytes
|
|
281
|
+
// into the RTF even if the header claims a different code page.
|
|
282
|
+
if (bytes.some(b => b > 127)) {
|
|
283
|
+
try {
|
|
284
|
+
// Use fatal: true to ensure we fall back on invalid UTF-8 sequences
|
|
285
|
+
const utf8Decoder = new TextDecoder('utf-8', { fatal: true });
|
|
286
|
+
return utf8Decoder.decode(uint8);
|
|
287
|
+
}
|
|
288
|
+
catch (e) {
|
|
289
|
+
// Not valid UTF-8, continue to code page fallback
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
// Fallback to specified code page
|
|
293
|
+
if (!this.decoders[codePage]) {
|
|
294
|
+
let encoding = `windows-${codePage}`;
|
|
295
|
+
if (codePage === 10000)
|
|
296
|
+
encoding = 'macintosh';
|
|
297
|
+
else if (codePage === 437)
|
|
298
|
+
encoding = 'ibm437';
|
|
299
|
+
else if (codePage === 850)
|
|
300
|
+
encoding = 'ibm850';
|
|
301
|
+
try {
|
|
302
|
+
this.decoders[codePage] = new TextDecoder(encoding);
|
|
303
|
+
}
|
|
304
|
+
catch (e) {
|
|
305
|
+
if (codePage !== 1252) {
|
|
306
|
+
try {
|
|
307
|
+
this.decoders[codePage] = new TextDecoder('windows-1252');
|
|
308
|
+
}
|
|
309
|
+
catch (e2) {
|
|
310
|
+
return String.fromCharCode(...bytes);
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
else {
|
|
314
|
+
return String.fromCharCode(...bytes);
|
|
315
|
+
}
|
|
316
|
+
}
|
|
267
317
|
}
|
|
318
|
+
let result = this.decoders[codePage].decode(uint8);
|
|
319
|
+
// Safety override for Windows-1252 0x80-0x9F range if TextDecoder behaves like Latin-1.
|
|
320
|
+
// We replace control characters in the decoded string with their proper 1252 equivalents.
|
|
321
|
+
if (codePage === 1252 && /[\u0080-\u009F]/.test(result)) {
|
|
322
|
+
const map = {
|
|
323
|
+
'\u0080': '€', '\u0082': '‚', '\u0083': 'ƒ', '\u0084': '„', '\u0085': '…',
|
|
324
|
+
'\u0086': '†', '\u0087': '‡', '\u0088': 'ˆ', '\u0089': '‰', '\u008A': 'Š',
|
|
325
|
+
'\u008B': '‹', '\u008C': 'Œ', '\u008E': 'Ž', '\u0091': '‘', '\u0092': '’',
|
|
326
|
+
'\u0093': '“', '\u0094': '”', '\u0095': '•', '\u0096': '–', '\u0097': '—',
|
|
327
|
+
'\u0098': '˜', '\u0099': '™', '\u009A': 'š', '\u009B': '›', '\u009C': 'œ',
|
|
328
|
+
'\u009E': 'ž', '\u009F': 'Ÿ'
|
|
329
|
+
};
|
|
330
|
+
return result.replace(/[\u0080-\u009F]/g, m => map[m] || m);
|
|
331
|
+
}
|
|
332
|
+
return result;
|
|
268
333
|
}
|
|
269
334
|
}
|
|
270
335
|
exports.SimpleRtfParser = SimpleRtfParser;
|
|
@@ -39,6 +39,7 @@
|
|
|
39
39
|
* - `<w:p>` - Paragraph
|
|
40
40
|
* - `<w:r>` - Run (contiguous text with same formatting)
|
|
41
41
|
* - `<w:t>` - Text content
|
|
42
|
+
* - `<w:br>` - Line or page break
|
|
42
43
|
* - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
|
|
43
44
|
* - `<w:pStyle>` - Paragraph style (for headings)
|
|
44
45
|
* - `<w:numPr>` - List numbering properties
|
|
@@ -40,6 +40,7 @@
|
|
|
40
40
|
* - `<w:p>` - Paragraph
|
|
41
41
|
* - `<w:r>` - Run (contiguous text with same formatting)
|
|
42
42
|
* - `<w:t>` - Text content
|
|
43
|
+
* - `<w:br>` - Line or page break
|
|
43
44
|
* - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
|
|
44
45
|
* - `<w:pStyle>` - Paragraph style (for headings)
|
|
45
46
|
* - `<w:numPr>` - List numbering properties
|
|
@@ -132,7 +133,7 @@ const parseWord = async (buffer, config) => {
|
|
|
132
133
|
// Font size
|
|
133
134
|
const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
|
|
134
135
|
if (szMatch)
|
|
135
|
-
formatting.size = (parseInt(szMatch[1]) / 2).toString() + 'pt';
|
|
136
|
+
formatting.size = (parseInt(szMatch[1], 10) / 2).toString() + 'pt';
|
|
136
137
|
// Color
|
|
137
138
|
const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
|
|
138
139
|
if (colorMatch && colorMatch[1] !== 'auto')
|
|
@@ -172,6 +173,27 @@ const parseWord = async (buffer, config) => {
|
|
|
172
173
|
}
|
|
173
174
|
return formatting;
|
|
174
175
|
};
|
|
176
|
+
// Helper to extract indentation from paragraph properties XML string
|
|
177
|
+
const extractIndentationFromXml = (pPr) => {
|
|
178
|
+
const ind = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:ind");
|
|
179
|
+
if (ind) {
|
|
180
|
+
const indentation = {};
|
|
181
|
+
const left = ind.getAttribute("w:left") || ind.getAttribute("w:start");
|
|
182
|
+
const right = ind.getAttribute("w:right") || ind.getAttribute("w:end");
|
|
183
|
+
const firstLine = ind.getAttribute("w:firstLine");
|
|
184
|
+
const hanging = ind.getAttribute("w:hanging");
|
|
185
|
+
if (left)
|
|
186
|
+
indentation.left = parseInt(left, 10);
|
|
187
|
+
if (right)
|
|
188
|
+
indentation.right = parseInt(right, 10);
|
|
189
|
+
if (firstLine)
|
|
190
|
+
indentation.firstLine = parseInt(firstLine, 10);
|
|
191
|
+
if (hanging)
|
|
192
|
+
indentation.hanging = parseInt(hanging, 10);
|
|
193
|
+
return Object.keys(indentation).length > 0 ? indentation : undefined;
|
|
194
|
+
}
|
|
195
|
+
return undefined;
|
|
196
|
+
};
|
|
175
197
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
|
|
176
198
|
!!x.match(footnotesFileRegex) ||
|
|
177
199
|
!!x.match(endnotesFileRegex) ||
|
|
@@ -257,6 +279,7 @@ const parseWord = async (buffer, config) => {
|
|
|
257
279
|
const formatting = rPr ? extractFormattingFromXml(rPr) : {};
|
|
258
280
|
let alignment = undefined;
|
|
259
281
|
let backgroundColor = undefined;
|
|
282
|
+
let paragraphIndentation = undefined;
|
|
260
283
|
if (pPr) {
|
|
261
284
|
const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
|
|
262
285
|
if (jc) {
|
|
@@ -271,8 +294,11 @@ const parseWord = async (buffer, config) => {
|
|
|
271
294
|
if (fill && fill !== 'auto')
|
|
272
295
|
backgroundColor = '#' + fill;
|
|
273
296
|
}
|
|
297
|
+
const ind = extractIndentationFromXml(pPr);
|
|
298
|
+
if (ind)
|
|
299
|
+
paragraphIndentation = ind;
|
|
274
300
|
}
|
|
275
|
-
styleMap[styleId] = { formatting, alignment, backgroundColor };
|
|
301
|
+
styleMap[styleId] = { formatting, alignment, backgroundColor, paragraphIndentation };
|
|
276
302
|
}
|
|
277
303
|
}
|
|
278
304
|
}
|
|
@@ -339,6 +365,14 @@ const parseWord = async (buffer, config) => {
|
|
|
339
365
|
}
|
|
340
366
|
}
|
|
341
367
|
}
|
|
368
|
+
// Extract Indentation
|
|
369
|
+
let paraIndentation = styleProps.paragraphIndentation;
|
|
370
|
+
if (pPr) {
|
|
371
|
+
const ind = extractIndentationFromXml(pPr);
|
|
372
|
+
if (ind) {
|
|
373
|
+
paraIndentation = { ...paraIndentation, ...ind };
|
|
374
|
+
}
|
|
375
|
+
}
|
|
342
376
|
// Extract Paragraph Background
|
|
343
377
|
let paraBackgroundColor = styleProps.backgroundColor;
|
|
344
378
|
if (pPr) {
|
|
@@ -406,26 +440,71 @@ const parseWord = async (buffer, config) => {
|
|
|
406
440
|
if (!formatting.backgroundColor && paraBackgroundColor) {
|
|
407
441
|
formatting.backgroundColor = paraBackgroundColor;
|
|
408
442
|
}
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
443
|
+
for (const child of runNode.childNodes) {
|
|
444
|
+
if (!(0, xmlUtils_js_1.isElement)(child))
|
|
445
|
+
continue;
|
|
446
|
+
// also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
|
|
447
|
+
// Text content
|
|
448
|
+
if (child.tagName === "w:t" || child.tagName === "t") {
|
|
449
|
+
const tNode = child;
|
|
450
|
+
const tContent = tNode.textContent || '';
|
|
451
|
+
text += tContent;
|
|
452
|
+
const textNode = {
|
|
453
|
+
type: 'text',
|
|
454
|
+
text: tContent,
|
|
455
|
+
formatting: formatting
|
|
456
|
+
};
|
|
457
|
+
if (config.includeRawContent) {
|
|
458
|
+
textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
|
|
459
|
+
}
|
|
460
|
+
// Always set a style: run style > paragraph style > detected default
|
|
461
|
+
// Use detected default style for international compatibility
|
|
462
|
+
const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
|
|
463
|
+
if (nodeStyle) {
|
|
464
|
+
textNode.metadata = { style: nodeStyle };
|
|
465
|
+
}
|
|
466
|
+
children.push(textNode);
|
|
421
467
|
}
|
|
422
|
-
//
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
468
|
+
// Break nodes
|
|
469
|
+
else if (config.includeBreakNodes &&
|
|
470
|
+
(child.tagName === "w:br"
|
|
471
|
+
|| child.tagName === "br"
|
|
472
|
+
|| child.tagName === "w:cr"
|
|
473
|
+
|| child.tagName === "cr")) {
|
|
474
|
+
const brNode = child;
|
|
475
|
+
let breakType = 'textWrapping';
|
|
476
|
+
if (child.tagName === "w:cr" || child.tagName === "cr") {
|
|
477
|
+
breakType = 'carriageReturn';
|
|
478
|
+
}
|
|
479
|
+
else {
|
|
480
|
+
const nodeBreakType = brNode.getAttribute("w:type") || brNode.getAttribute("type");
|
|
481
|
+
if (nodeBreakType !== null) {
|
|
482
|
+
breakType = nodeBreakType;
|
|
483
|
+
}
|
|
484
|
+
}
|
|
485
|
+
let breakClear = undefined;
|
|
486
|
+
if (breakType === 'textWrapping' && brNode.getAttribute("w:clear") !== null) {
|
|
487
|
+
breakClear = brNode.getAttribute("w:clear");
|
|
488
|
+
}
|
|
489
|
+
const breakNode = {
|
|
490
|
+
type: 'break',
|
|
491
|
+
metadata: { breakType, clear: breakClear }
|
|
492
|
+
};
|
|
493
|
+
if (config.includeRawContent) {
|
|
494
|
+
breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(brNode, documentContent, config);
|
|
495
|
+
}
|
|
496
|
+
children.push(breakNode);
|
|
497
|
+
}
|
|
498
|
+
else if (config.includeBreakNodes && (child.tagName === "w:lastRenderedPageBreak" || child.tagName === "lastRenderedPageBreak")) {
|
|
499
|
+
const breakNode = {
|
|
500
|
+
type: 'break',
|
|
501
|
+
metadata: { breakType: 'lastRenderedPage' }
|
|
502
|
+
};
|
|
503
|
+
if (config.includeRawContent) {
|
|
504
|
+
breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(child, documentContent, config);
|
|
505
|
+
}
|
|
506
|
+
children.push(breakNode);
|
|
427
507
|
}
|
|
428
|
-
children.push(textNode);
|
|
429
508
|
}
|
|
430
509
|
// Images/Drawings
|
|
431
510
|
if (config.extractAttachments) {
|
|
@@ -557,7 +636,7 @@ const parseWord = async (buffer, config) => {
|
|
|
557
636
|
const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
|
|
558
637
|
const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
|
|
559
638
|
const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
|
|
560
|
-
const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
|
|
639
|
+
const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0', 10) : 0;
|
|
561
640
|
let listType = 'ordered';
|
|
562
641
|
let itemIndex = 0;
|
|
563
642
|
if (numId && numberingMap[numId]) {
|
|
@@ -591,6 +670,7 @@ const parseWord = async (buffer, config) => {
|
|
|
591
670
|
metadata: {
|
|
592
671
|
listType,
|
|
593
672
|
indentation: ilvl,
|
|
673
|
+
paragraphIndentation: paraIndentation,
|
|
594
674
|
alignment: (alignment || 'left'),
|
|
595
675
|
listId: numId,
|
|
596
676
|
itemIndex: itemIndex,
|
|
@@ -602,12 +682,12 @@ const parseWord = async (buffer, config) => {
|
|
|
602
682
|
return listNode;
|
|
603
683
|
}
|
|
604
684
|
else if (isHeading) {
|
|
605
|
-
const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", "")) || 1 : 1;
|
|
685
|
+
const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", ""), 10) || 1 : 1;
|
|
606
686
|
const headingNode = {
|
|
607
687
|
type: 'heading',
|
|
608
688
|
text: text,
|
|
609
689
|
children: children,
|
|
610
|
-
metadata: { level, alignment, style: pStyleVal ?? undefined }
|
|
690
|
+
metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
|
|
611
691
|
};
|
|
612
692
|
if (config.includeRawContent)
|
|
613
693
|
headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
@@ -618,7 +698,7 @@ const parseWord = async (buffer, config) => {
|
|
|
618
698
|
type: 'paragraph',
|
|
619
699
|
text: text,
|
|
620
700
|
children: children,
|
|
621
|
-
metadata: { alignment, style: pStyleVal ?? undefined }
|
|
701
|
+
metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
|
|
622
702
|
};
|
|
623
703
|
if (config.includeRawContent)
|
|
624
704
|
paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
|
|
@@ -784,6 +864,9 @@ const parseWord = async (buffer, config) => {
|
|
|
784
864
|
if (node.children) {
|
|
785
865
|
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
|
|
786
866
|
}
|
|
867
|
+
else if (node.type === 'break') {
|
|
868
|
+
t += config.newlineDelimiter ?? '\n';
|
|
869
|
+
}
|
|
787
870
|
else
|
|
788
871
|
t += node.text || '';
|
|
789
872
|
return t;
|