officeparser 6.1.0 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -92,6 +92,12 @@ class SimpleRtfParser {
92
92
  index = 0;
93
93
  /** The RTF content as a Buffer */
94
94
  buffer;
95
+ /** Current code page for character decoding (default is Windows-1252) */
96
+ codePage = 1252;
97
+ /** Cached TextDecoders for different code pages */
98
+ decoders = {};
99
+ /** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
100
+ pendingBytes = [];
95
101
  /** Total length of the buffer */
96
102
  length;
97
103
  /**
@@ -110,12 +116,14 @@ class SimpleRtfParser {
110
116
  const currentGroup = stack[stack.length - 1];
111
117
  if (char === 0x7B) { // '{'
112
118
  this.index++;
119
+ this.flushPendingText(currentGroup);
113
120
  const newGroup = { type: 'group', content: [] };
114
121
  currentGroup.content.push(newGroup);
115
122
  stack.push(newGroup);
116
123
  }
117
124
  else if (char === 0x7D) { // '}'
118
125
  this.index++;
126
+ this.flushPendingText(currentGroup);
119
127
  if (stack.length > 1) {
120
128
  stack.pop();
121
129
  }
@@ -133,6 +141,7 @@ class SimpleRtfParser {
133
141
  this.parseText(currentGroup);
134
142
  }
135
143
  }
144
+ this.flushPendingText(root);
136
145
  return root;
137
146
  }
138
147
  parseControl(group) {
@@ -141,55 +150,23 @@ class SimpleRtfParser {
141
150
  const char = this.buffer[this.index];
142
151
  // Special control symbols
143
152
  if (char === 0x7B || char === 0x7D || char === 0x5C) { // \{ \} \\
144
- group.content.push({ type: 'text', value: String.fromCharCode(char) });
153
+ this.pendingBytes.push(char);
145
154
  this.index++;
146
155
  return;
147
156
  }
148
157
  if (char === 0x27) { // \'xx (hex)
149
158
  this.index++;
150
159
  if (this.index + 1 < this.length) {
151
- const hex = this.buffer.toString('utf8', this.index, this.index + 2);
160
+ const hex = String.fromCharCode(this.buffer[this.index], this.buffer[this.index + 1]);
152
161
  const code = parseInt(hex, 16);
153
162
  if (!isNaN(code)) {
154
- // RTF hex escapes represent bytes in the document's code page (usually Windows-1252)
155
- // Characters 0x80-0x9F in Windows-1252 don't map directly to Unicode
156
- // We need to convert them properly
157
- const windows1252ToUnicode = {
158
- 0x80: 0x20AC, // €
159
- 0x82: 0x201A, // ‚
160
- 0x83: 0x0192, // ƒ
161
- 0x84: 0x201E, // „
162
- 0x85: 0x2026, // …
163
- 0x86: 0x2020, // †
164
- 0x87: 0x2021, // ‡
165
- 0x88: 0x02C6, // ˆ
166
- 0x89: 0x2030, // ‰
167
- 0x8A: 0x0160, // Š
168
- 0x8B: 0x2039, // ‹
169
- 0x8C: 0x0152, // Œ
170
- 0x8E: 0x017D, // Ž
171
- 0x91: 0x2018, // '
172
- 0x92: 0x2019, // '
173
- 0x93: 0x201C, // "
174
- 0x94: 0x201D, // "
175
- 0x95: 0x2022, // •
176
- 0x96: 0x2013, // –
177
- 0x97: 0x2014, // —
178
- 0x98: 0x02DC, // ˜
179
- 0x99: 0x2122, // ™
180
- 0x9A: 0x0161, // š
181
- 0x9B: 0x203A, // ›
182
- 0x9C: 0x0153, // œ
183
- 0x9E: 0x017E, // ž
184
- 0x9F: 0x0178 // Ÿ
185
- };
186
- const unicodeCode = windows1252ToUnicode[code] || code;
187
- group.content.push({ type: 'text', value: String.fromCharCode(unicodeCode) });
163
+ this.pendingBytes.push(code);
188
164
  }
189
165
  this.index += 2;
190
166
  }
191
167
  return;
192
168
  }
169
+ this.flushPendingText(group);
193
170
  if (char === 0x2A) { // \* (ignorable destination)
194
171
  // We treat this as a control word named '*'
195
172
  group.content.push({ type: 'control', value: '*' });
@@ -231,7 +208,7 @@ class SimpleRtfParser {
231
208
  param = parseInt(paramStr, 10);
232
209
  }
233
210
  // Space after control word is consumed
234
- if (this.index < this.length && this.buffer[this.index] === 0x20) {
211
+ if (name !== '' && this.index < this.length && this.buffer[this.index] === 0x20) {
235
212
  this.index++;
236
213
  }
237
214
  // Handle \binN
@@ -241,6 +218,22 @@ class SimpleRtfParser {
241
218
  // \binN is not added to content as we want to ignore it
242
219
  return;
243
220
  }
221
+ // Handle encoding control words
222
+ if (name === 'ansicpg' && param !== undefined) {
223
+ this.codePage = param;
224
+ }
225
+ else if (name === 'ansi') {
226
+ this.codePage = 1252;
227
+ }
228
+ else if (name === 'mac') {
229
+ this.codePage = 10000;
230
+ }
231
+ else if (name === 'pc') {
232
+ this.codePage = 437;
233
+ }
234
+ else if (name === 'pca') {
235
+ this.codePage = 850;
236
+ }
244
237
  group.content.push({ type: 'control', value: name, param });
245
238
  // If this is the first control word in the group, it might be the destination
246
239
  if (group.content.length === 1 && group.type === 'group') {
@@ -252,19 +245,91 @@ class SimpleRtfParser {
252
245
  }
253
246
  }
254
247
  parseText(group) {
255
- let text = '';
256
248
  while (this.index < this.length) {
257
249
  const char = this.buffer[this.index];
250
+ if (char === undefined)
251
+ break;
258
252
  if (char === 0x7B || char === 0x7D || char === 0x5C || char === 0x0D || char === 0x0A) {
259
253
  break;
260
254
  }
261
- // Basic ASCII text.
262
- text += String.fromCharCode(char);
255
+ this.pendingBytes.push(char);
263
256
  this.index++;
264
257
  }
265
- if (text.length > 0) {
266
- group.content.push({ type: 'text', value: text });
258
+ }
259
+ /**
260
+ * Flushes the pending bytes buffer as a text node to the current group.
261
+ * @param group The group to append the text node to
262
+ */
263
+ flushPendingText(group) {
264
+ if (this.pendingBytes.length > 0) {
265
+ group.content.push({ type: 'text', value: this.decodeBytes(this.pendingBytes, this.codePage) });
266
+ this.pendingBytes = [];
267
+ }
268
+ }
269
+ /**
270
+ * Decodes a byte array using a "UTF-8 first" strategy.
271
+ * If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
272
+ * Otherwise, falls back to the specified code page.
273
+ * @param bytes The bytes to decode
274
+ * @param codePage The RTF code page ID
275
+ * @returns The decoded string
276
+ */
277
+ decodeBytes(bytes, codePage) {
278
+ const uint8 = new Uint8Array(bytes);
279
+ // Try UTF-8 first if there are any non-ASCII bytes.
280
+ // Many modern RTF generators (like calibre or web-based tools) dump UTF-8 bytes
281
+ // into the RTF even if the header claims a different code page.
282
+ if (bytes.some(b => b > 127)) {
283
+ try {
284
+ // Use fatal: true to ensure we fall back on invalid UTF-8 sequences
285
+ const utf8Decoder = new TextDecoder('utf-8', { fatal: true });
286
+ return utf8Decoder.decode(uint8);
287
+ }
288
+ catch (e) {
289
+ // Not valid UTF-8, continue to code page fallback
290
+ }
291
+ }
292
+ // Fallback to specified code page
293
+ if (!this.decoders[codePage]) {
294
+ let encoding = `windows-${codePage}`;
295
+ if (codePage === 10000)
296
+ encoding = 'macintosh';
297
+ else if (codePage === 437)
298
+ encoding = 'ibm437';
299
+ else if (codePage === 850)
300
+ encoding = 'ibm850';
301
+ try {
302
+ this.decoders[codePage] = new TextDecoder(encoding);
303
+ }
304
+ catch (e) {
305
+ if (codePage !== 1252) {
306
+ try {
307
+ this.decoders[codePage] = new TextDecoder('windows-1252');
308
+ }
309
+ catch (e2) {
310
+ return String.fromCharCode(...bytes);
311
+ }
312
+ }
313
+ else {
314
+ return String.fromCharCode(...bytes);
315
+ }
316
+ }
267
317
  }
318
+ let result = this.decoders[codePage].decode(uint8);
319
+ // Safety override for Windows-1252 0x80-0x9F range if TextDecoder behaves like Latin-1.
320
+ // We replace control characters in the decoded string with their proper 1252 equivalents.
321
+ if (codePage === 1252 && /[\u0080-\u009F]/.test(result)) {
322
+ const map = {
323
+ '\u0080': '€', '\u0082': '‚', '\u0083': 'ƒ', '\u0084': '„', '\u0085': '…',
324
+ '\u0086': '†', '\u0087': '‡', '\u0088': 'ˆ', '\u0089': '‰', '\u008A': 'Š',
325
+ '\u008B': '‹', '\u008C': 'Œ', '\u008E': 'Ž', '\u0091': '‘', '\u0092': '’',
326
+ '\u0093': '“', '\u0094': '”', '\u0095': '•', '\u0096': '–', '\u0097': '—',
327
+ '\u0098': '˜', '\u0099': '™', '\u009A': 'š', '\u009B': '›', '\u009C': 'œ',
328
+ '\u009E': 'ž', '\u009F': 'Ÿ'
329
+ };
330
+ return result.replace(/[\u0080-\u009F]/g, m => map[m] || m);
331
+ }
332
+ return result;
268
333
  }
269
334
  }
270
335
  exports.SimpleRtfParser = SimpleRtfParser;
@@ -39,6 +39,7 @@
39
39
  * - `<w:p>` - Paragraph
40
40
  * - `<w:r>` - Run (contiguous text with same formatting)
41
41
  * - `<w:t>` - Text content
42
+ * - `<w:br>` - Line or page break
42
43
  * - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
43
44
  * - `<w:pStyle>` - Paragraph style (for headings)
44
45
  * - `<w:numPr>` - List numbering properties
@@ -40,6 +40,7 @@
40
40
  * - `<w:p>` - Paragraph
41
41
  * - `<w:r>` - Run (contiguous text with same formatting)
42
42
  * - `<w:t>` - Text content
43
+ * - `<w:br>` - Line or page break
43
44
  * - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
44
45
  * - `<w:pStyle>` - Paragraph style (for headings)
45
46
  * - `<w:numPr>` - List numbering properties
@@ -132,7 +133,7 @@ const parseWord = async (buffer, config) => {
132
133
  // Font size
133
134
  const szMatch = rPrString.match(/<w:sz w:val="(\d+)"/);
134
135
  if (szMatch)
135
- formatting.size = (parseInt(szMatch[1]) / 2).toString() + 'pt';
136
+ formatting.size = (parseInt(szMatch[1], 10) / 2).toString() + 'pt';
136
137
  // Color
137
138
  const colorMatch = rPrString.match(/<w:color w:val="([^"]+)"/);
138
139
  if (colorMatch && colorMatch[1] !== 'auto')
@@ -172,6 +173,27 @@ const parseWord = async (buffer, config) => {
172
173
  }
173
174
  return formatting;
174
175
  };
176
+ // Helper to extract indentation from paragraph properties XML string
177
+ const extractIndentationFromXml = (pPr) => {
178
+ const ind = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:ind");
179
+ if (ind) {
180
+ const indentation = {};
181
+ const left = ind.getAttribute("w:left") || ind.getAttribute("w:start");
182
+ const right = ind.getAttribute("w:right") || ind.getAttribute("w:end");
183
+ const firstLine = ind.getAttribute("w:firstLine");
184
+ const hanging = ind.getAttribute("w:hanging");
185
+ if (left)
186
+ indentation.left = parseInt(left, 10);
187
+ if (right)
188
+ indentation.right = parseInt(right, 10);
189
+ if (firstLine)
190
+ indentation.firstLine = parseInt(firstLine, 10);
191
+ if (hanging)
192
+ indentation.hanging = parseInt(hanging, 10);
193
+ return Object.keys(indentation).length > 0 ? indentation : undefined;
194
+ }
195
+ return undefined;
196
+ };
175
197
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(documentFileRegex) ||
176
198
  !!x.match(footnotesFileRegex) ||
177
199
  !!x.match(endnotesFileRegex) ||
@@ -257,6 +279,7 @@ const parseWord = async (buffer, config) => {
257
279
  const formatting = rPr ? extractFormattingFromXml(rPr) : {};
258
280
  let alignment = undefined;
259
281
  let backgroundColor = undefined;
282
+ let paragraphIndentation = undefined;
260
283
  if (pPr) {
261
284
  const jc = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:jc");
262
285
  if (jc) {
@@ -271,8 +294,11 @@ const parseWord = async (buffer, config) => {
271
294
  if (fill && fill !== 'auto')
272
295
  backgroundColor = '#' + fill;
273
296
  }
297
+ const ind = extractIndentationFromXml(pPr);
298
+ if (ind)
299
+ paragraphIndentation = ind;
274
300
  }
275
- styleMap[styleId] = { formatting, alignment, backgroundColor };
301
+ styleMap[styleId] = { formatting, alignment, backgroundColor, paragraphIndentation };
276
302
  }
277
303
  }
278
304
  }
@@ -339,6 +365,14 @@ const parseWord = async (buffer, config) => {
339
365
  }
340
366
  }
341
367
  }
368
+ // Extract Indentation
369
+ let paraIndentation = styleProps.paragraphIndentation;
370
+ if (pPr) {
371
+ const ind = extractIndentationFromXml(pPr);
372
+ if (ind) {
373
+ paraIndentation = { ...paraIndentation, ...ind };
374
+ }
375
+ }
342
376
  // Extract Paragraph Background
343
377
  let paraBackgroundColor = styleProps.backgroundColor;
344
378
  if (pPr) {
@@ -406,26 +440,71 @@ const parseWord = async (buffer, config) => {
406
440
  if (!formatting.backgroundColor && paraBackgroundColor) {
407
441
  formatting.backgroundColor = paraBackgroundColor;
408
442
  }
409
- // Text content
410
- const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(runNode, "w:t");
411
- for (const tNode of tNodes) {
412
- const tContent = tNode.textContent || '';
413
- text += tContent;
414
- const textNode = {
415
- type: 'text',
416
- text: tContent,
417
- formatting: formatting
418
- };
419
- if (config.includeRawContent) {
420
- textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
443
+ for (const child of runNode.childNodes) {
444
+ if (!(0, xmlUtils_js_1.isElement)(child))
445
+ continue;
446
+ // also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
447
+ // Text content
448
+ if (child.tagName === "w:t" || child.tagName === "t") {
449
+ const tNode = child;
450
+ const tContent = tNode.textContent || '';
451
+ text += tContent;
452
+ const textNode = {
453
+ type: 'text',
454
+ text: tContent,
455
+ formatting: formatting
456
+ };
457
+ if (config.includeRawContent) {
458
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(tNode, documentContent, config);
459
+ }
460
+ // Always set a style: run style > paragraph style > detected default
461
+ // Use detected default style for international compatibility
462
+ const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
463
+ if (nodeStyle) {
464
+ textNode.metadata = { style: nodeStyle };
465
+ }
466
+ children.push(textNode);
421
467
  }
422
- // Always set a style: run style > paragraph style > detected default
423
- // Use detected default style for international compatibility
424
- const nodeStyle = rStyleVal || pStyleVal || defaultParaStyleId;
425
- if (nodeStyle) {
426
- textNode.metadata = { style: nodeStyle };
468
+ // Break nodes
469
+ else if (config.includeBreakNodes &&
470
+ (child.tagName === "w:br"
471
+ || child.tagName === "br"
472
+ || child.tagName === "w:cr"
473
+ || child.tagName === "cr")) {
474
+ const brNode = child;
475
+ let breakType = 'textWrapping';
476
+ if (child.tagName === "w:cr" || child.tagName === "cr") {
477
+ breakType = 'carriageReturn';
478
+ }
479
+ else {
480
+ const nodeBreakType = brNode.getAttribute("w:type") || brNode.getAttribute("type");
481
+ if (nodeBreakType !== null) {
482
+ breakType = nodeBreakType;
483
+ }
484
+ }
485
+ let breakClear = undefined;
486
+ if (breakType === 'textWrapping' && brNode.getAttribute("w:clear") !== null) {
487
+ breakClear = brNode.getAttribute("w:clear");
488
+ }
489
+ const breakNode = {
490
+ type: 'break',
491
+ metadata: { breakType, clear: breakClear }
492
+ };
493
+ if (config.includeRawContent) {
494
+ breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(brNode, documentContent, config);
495
+ }
496
+ children.push(breakNode);
497
+ }
498
+ else if (config.includeBreakNodes && (child.tagName === "w:lastRenderedPageBreak" || child.tagName === "lastRenderedPageBreak")) {
499
+ const breakNode = {
500
+ type: 'break',
501
+ metadata: { breakType: 'lastRenderedPage' }
502
+ };
503
+ if (config.includeRawContent) {
504
+ breakNode.rawContent = (0, xmlUtils_js_1.getRawContent)(child, documentContent, config);
505
+ }
506
+ children.push(breakNode);
427
507
  }
428
- children.push(textNode);
429
508
  }
430
509
  // Images/Drawings
431
510
  if (config.extractAttachments) {
@@ -557,7 +636,7 @@ const parseWord = async (buffer, config) => {
557
636
  const numIdNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:numId");
558
637
  const ilvlNode = (0, xmlUtils_js_1.getFirstElementByTagName)(numPr, "w:ilvl");
559
638
  const numId = numIdNode ? numIdNode.getAttribute("w:val") || '0' : '0';
560
- const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0') : 0;
639
+ const ilvl = ilvlNode ? parseInt(ilvlNode.getAttribute("w:val") || '0', 10) : 0;
561
640
  let listType = 'ordered';
562
641
  let itemIndex = 0;
563
642
  if (numId && numberingMap[numId]) {
@@ -591,6 +670,7 @@ const parseWord = async (buffer, config) => {
591
670
  metadata: {
592
671
  listType,
593
672
  indentation: ilvl,
673
+ paragraphIndentation: paraIndentation,
594
674
  alignment: (alignment || 'left'),
595
675
  listId: numId,
596
676
  itemIndex: itemIndex,
@@ -602,12 +682,12 @@ const parseWord = async (buffer, config) => {
602
682
  return listNode;
603
683
  }
604
684
  else if (isHeading) {
605
- const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", "")) || 1 : 1;
685
+ const level = pStyleVal ? parseInt(pStyleVal.replace("Heading", ""), 10) || 1 : 1;
606
686
  const headingNode = {
607
687
  type: 'heading',
608
688
  text: text,
609
689
  children: children,
610
- metadata: { level, alignment, style: pStyleVal ?? undefined }
690
+ metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
611
691
  };
612
692
  if (config.includeRawContent)
613
693
  headingNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
@@ -618,7 +698,7 @@ const parseWord = async (buffer, config) => {
618
698
  type: 'paragraph',
619
699
  text: text,
620
700
  children: children,
621
- metadata: { alignment, style: pStyleVal ?? undefined }
701
+ metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined }
622
702
  };
623
703
  if (config.includeRawContent)
624
704
  paraNode.rawContent = (0, xmlUtils_js_1.getRawContent)(pNode, documentContent, config);
@@ -784,6 +864,9 @@ const parseWord = async (buffer, config) => {
784
864
  if (node.children) {
785
865
  t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
786
866
  }
867
+ else if (node.type === 'break') {
868
+ t += config.newlineDelimiter ?? '\n';
869
+ }
787
870
  else
788
871
  t += node.text || '';
789
872
  return t;