officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
@@ -39,7 +39,7 @@
39
39
  * @see https://www.biblioscape.com/rtf15_spec.htm RTF 1.5 Specification
40
40
  * @see https://latex2rtf.sourceforge.net/RTF-Spec-1.2.pdf RTF 1.2 Specification
41
41
  */
42
- import { OfficeParserAST, OfficeParserConfig } from '../types';
42
+ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
43
43
  /**
44
44
  * Represents an RTF group (content enclosed in braces).
45
45
  * Groups create formatting scopes and can contain other groups, control words, or text.
@@ -130,6 +130,12 @@ export declare class SimpleRtfParser {
130
130
  private index;
131
131
  /** The RTF content as a Buffer */
132
132
  private buffer;
133
+ /** Current code page for character decoding (default is Windows-1252) */
134
+ private codePage;
135
+ /** Cached TextDecoders for different code pages */
136
+ private decoders;
137
+ /** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
138
+ private pendingBytes;
133
139
  /** Total length of the buffer */
134
140
  private length;
135
141
  /**
@@ -140,6 +146,20 @@ export declare class SimpleRtfParser {
140
146
  parse(): RtfGroup;
141
147
  private parseControl;
142
148
  private parseText;
149
+ /**
150
+ * Flushes the pending bytes buffer as a text node to the current group.
151
+ * @param group The group to append the text node to
152
+ */
153
+ private flushPendingText;
154
+ /**
155
+ * Decodes a byte array using a "UTF-8 first" strategy.
156
+ * If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
157
+ * Otherwise, falls back to the specified code page.
158
+ * @param bytes The bytes to decode
159
+ * @param codePage The RTF code page ID
160
+ * @returns The decoded string
161
+ */
162
+ private decodeBytes;
143
163
  }
144
164
  /**
145
165
  * Parses an RTF file and returns the AST.
@@ -42,8 +42,8 @@
42
42
  */
43
43
  Object.defineProperty(exports, "__esModule", { value: true });
44
44
  exports.parseRtf = exports.SimpleRtfParser = void 0;
45
- const errorUtils_1 = require("../utils/errorUtils");
46
- const ocrUtils_1 = require("../utils/ocrUtils");
45
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
46
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
47
47
  /**
48
48
  * Lookup table mapping RTF control words to internal formats.
49
49
  * Fully typed: if a key maps to an unsupported format, TypeScript throws an error.
@@ -88,13 +88,23 @@ const IMAGE_MIME_MAP = {
88
88
  * ```
89
89
  */
90
90
  class SimpleRtfParser {
91
+ /** Current position in the buffer */
92
+ index = 0;
93
+ /** The RTF content as a Buffer */
94
+ buffer;
95
+ /** Current code page for character decoding (default is Windows-1252) */
96
+ codePage = 1252;
97
+ /** Cached TextDecoders for different code pages */
98
+ decoders = {};
99
+ /** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
100
+ pendingBytes = [];
101
+ /** Total length of the buffer */
102
+ length;
91
103
  /**
92
104
  * Creates a new RTF parser.
93
105
  * @param buffer - The RTF file content as a Buffer
94
106
  */
95
107
  constructor(buffer) {
96
- /** Current position in the buffer */
97
- this.index = 0;
98
108
  this.buffer = buffer;
99
109
  this.length = buffer.length;
100
110
  }
@@ -106,12 +116,14 @@ class SimpleRtfParser {
106
116
  const currentGroup = stack[stack.length - 1];
107
117
  if (char === 0x7B) { // '{'
108
118
  this.index++;
119
+ this.flushPendingText(currentGroup);
109
120
  const newGroup = { type: 'group', content: [] };
110
121
  currentGroup.content.push(newGroup);
111
122
  stack.push(newGroup);
112
123
  }
113
124
  else if (char === 0x7D) { // '}'
114
125
  this.index++;
126
+ this.flushPendingText(currentGroup);
115
127
  if (stack.length > 1) {
116
128
  stack.pop();
117
129
  }
@@ -129,6 +141,7 @@ class SimpleRtfParser {
129
141
  this.parseText(currentGroup);
130
142
  }
131
143
  }
144
+ this.flushPendingText(root);
132
145
  return root;
133
146
  }
134
147
  parseControl(group) {
@@ -137,55 +150,23 @@ class SimpleRtfParser {
137
150
  const char = this.buffer[this.index];
138
151
  // Special control symbols
139
152
  if (char === 0x7B || char === 0x7D || char === 0x5C) { // \{ \} \\
140
- group.content.push({ type: 'text', value: String.fromCharCode(char) });
153
+ this.pendingBytes.push(char);
141
154
  this.index++;
142
155
  return;
143
156
  }
144
157
  if (char === 0x27) { // \'xx (hex)
145
158
  this.index++;
146
159
  if (this.index + 1 < this.length) {
147
- const hex = this.buffer.toString('utf8', this.index, this.index + 2);
160
+ const hex = String.fromCharCode(this.buffer[this.index], this.buffer[this.index + 1]);
148
161
  const code = parseInt(hex, 16);
149
162
  if (!isNaN(code)) {
150
- // RTF hex escapes represent bytes in the document's code page (usually Windows-1252)
151
- // Characters 0x80-0x9F in Windows-1252 don't map directly to Unicode
152
- // We need to convert them properly
153
- const windows1252ToUnicode = {
154
- 0x80: 0x20AC, // €
155
- 0x82: 0x201A, // ‚
156
- 0x83: 0x0192, // ƒ
157
- 0x84: 0x201E, // „
158
- 0x85: 0x2026, // …
159
- 0x86: 0x2020, // †
160
- 0x87: 0x2021, // ‡
161
- 0x88: 0x02C6, // ˆ
162
- 0x89: 0x2030, // ‰
163
- 0x8A: 0x0160, // Š
164
- 0x8B: 0x2039, // ‹
165
- 0x8C: 0x0152, // Œ
166
- 0x8E: 0x017D, // Ž
167
- 0x91: 0x2018, // '
168
- 0x92: 0x2019, // '
169
- 0x93: 0x201C, // "
170
- 0x94: 0x201D, // "
171
- 0x95: 0x2022, // •
172
- 0x96: 0x2013, // –
173
- 0x97: 0x2014, // —
174
- 0x98: 0x02DC, // ˜
175
- 0x99: 0x2122, // ™
176
- 0x9A: 0x0161, // š
177
- 0x9B: 0x203A, // ›
178
- 0x9C: 0x0153, // œ
179
- 0x9E: 0x017E, // ž
180
- 0x9F: 0x0178 // Ÿ
181
- };
182
- const unicodeCode = windows1252ToUnicode[code] || code;
183
- group.content.push({ type: 'text', value: String.fromCharCode(unicodeCode) });
163
+ this.pendingBytes.push(code);
184
164
  }
185
165
  this.index += 2;
186
166
  }
187
167
  return;
188
168
  }
169
+ this.flushPendingText(group);
189
170
  if (char === 0x2A) { // \* (ignorable destination)
190
171
  // We treat this as a control word named '*'
191
172
  group.content.push({ type: 'control', value: '*' });
@@ -227,7 +208,7 @@ class SimpleRtfParser {
227
208
  param = parseInt(paramStr, 10);
228
209
  }
229
210
  // Space after control word is consumed
230
- if (this.index < this.length && this.buffer[this.index] === 0x20) {
211
+ if (name !== '' && this.index < this.length && this.buffer[this.index] === 0x20) {
231
212
  this.index++;
232
213
  }
233
214
  // Handle \binN
@@ -237,6 +218,22 @@ class SimpleRtfParser {
237
218
  // \binN is not added to content as we want to ignore it
238
219
  return;
239
220
  }
221
+ // Handle encoding control words
222
+ if (name === 'ansicpg' && param !== undefined) {
223
+ this.codePage = param;
224
+ }
225
+ else if (name === 'ansi') {
226
+ this.codePage = 1252;
227
+ }
228
+ else if (name === 'mac') {
229
+ this.codePage = 10000;
230
+ }
231
+ else if (name === 'pc') {
232
+ this.codePage = 437;
233
+ }
234
+ else if (name === 'pca') {
235
+ this.codePage = 850;
236
+ }
240
237
  group.content.push({ type: 'control', value: name, param });
241
238
  // If this is the first control word in the group, it might be the destination
242
239
  if (group.content.length === 1 && group.type === 'group') {
@@ -248,19 +245,91 @@ class SimpleRtfParser {
248
245
  }
249
246
  }
250
247
  parseText(group) {
251
- let text = '';
252
248
  while (this.index < this.length) {
253
249
  const char = this.buffer[this.index];
250
+ if (char === undefined)
251
+ break;
254
252
  if (char === 0x7B || char === 0x7D || char === 0x5C || char === 0x0D || char === 0x0A) {
255
253
  break;
256
254
  }
257
- // Basic ASCII text.
258
- text += String.fromCharCode(char);
255
+ this.pendingBytes.push(char);
259
256
  this.index++;
260
257
  }
261
- if (text.length > 0) {
262
- group.content.push({ type: 'text', value: text });
258
+ }
259
+ /**
260
+ * Flushes the pending bytes buffer as a text node to the current group.
261
+ * @param group The group to append the text node to
262
+ */
263
+ flushPendingText(group) {
264
+ if (this.pendingBytes.length > 0) {
265
+ group.content.push({ type: 'text', value: this.decodeBytes(this.pendingBytes, this.codePage) });
266
+ this.pendingBytes = [];
267
+ }
268
+ }
269
+ /**
270
+ * Decodes a byte array using a "UTF-8 first" strategy.
271
+ * If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
272
+ * Otherwise, falls back to the specified code page.
273
+ * @param bytes The bytes to decode
274
+ * @param codePage The RTF code page ID
275
+ * @returns The decoded string
276
+ */
277
+ decodeBytes(bytes, codePage) {
278
+ const uint8 = new Uint8Array(bytes);
279
+ // Try UTF-8 first if there are any non-ASCII bytes.
280
+ // Many modern RTF generators (like calibre or web-based tools) dump UTF-8 bytes
281
+ // into the RTF even if the header claims a different code page.
282
+ if (bytes.some(b => b > 127)) {
283
+ try {
284
+ // Use fatal: true to ensure we fall back on invalid UTF-8 sequences
285
+ const utf8Decoder = new TextDecoder('utf-8', { fatal: true });
286
+ return utf8Decoder.decode(uint8);
287
+ }
288
+ catch (e) {
289
+ // Not valid UTF-8, continue to code page fallback
290
+ }
291
+ }
292
+ // Fallback to specified code page
293
+ if (!this.decoders[codePage]) {
294
+ let encoding = `windows-${codePage}`;
295
+ if (codePage === 10000)
296
+ encoding = 'macintosh';
297
+ else if (codePage === 437)
298
+ encoding = 'ibm437';
299
+ else if (codePage === 850)
300
+ encoding = 'ibm850';
301
+ try {
302
+ this.decoders[codePage] = new TextDecoder(encoding);
303
+ }
304
+ catch (e) {
305
+ if (codePage !== 1252) {
306
+ try {
307
+ this.decoders[codePage] = new TextDecoder('windows-1252');
308
+ }
309
+ catch (e2) {
310
+ return String.fromCharCode(...bytes);
311
+ }
312
+ }
313
+ else {
314
+ return String.fromCharCode(...bytes);
315
+ }
316
+ }
263
317
  }
318
+ let result = this.decoders[codePage].decode(uint8);
319
+ // Safety override for Windows-1252 0x80-0x9F range if TextDecoder behaves like Latin-1.
320
+ // We replace control characters in the decoded string with their proper 1252 equivalents.
321
+ if (codePage === 1252 && /[\u0080-\u009F]/.test(result)) {
322
+ const map = {
323
+ '\u0080': '€', '\u0082': '‚', '\u0083': 'ƒ', '\u0084': '„', '\u0085': '…',
324
+ '\u0086': '†', '\u0087': '‡', '\u0088': 'ˆ', '\u0089': '‰', '\u008A': 'Š',
325
+ '\u008B': '‹', '\u008C': 'Œ', '\u008E': 'Ž', '\u0091': '‘', '\u0092': '’',
326
+ '\u0093': '“', '\u0094': '”', '\u0095': '•', '\u0096': '–', '\u0097': '—',
327
+ '\u0098': '˜', '\u0099': '™', '\u009A': 'š', '\u009B': '›', '\u009C': 'œ',
328
+ '\u009E': 'ž', '\u009F': 'Ÿ'
329
+ };
330
+ return result.replace(/[\u0080-\u009F]/g, m => map[m] || m);
331
+ }
332
+ return result;
264
333
  }
265
334
  }
266
335
  exports.SimpleRtfParser = SimpleRtfParser;
@@ -1510,10 +1579,10 @@ const parseRtf = async (buffer, config) => {
1510
1579
  // Passing base64 string directly would be interpreted as a file path,
1511
1580
  // causing ENAMETOOLONG error for large images.
1512
1581
  const imageBuffer = Buffer.from(attachment.data, 'base64');
1513
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(imageBuffer, config.ocrLanguage)).trim();
1582
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1514
1583
  }
1515
1584
  catch (e) {
1516
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1585
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1517
1586
  }
1518
1587
  }
1519
1588
  }
@@ -39,6 +39,7 @@
39
39
  * - `<w:p>` - Paragraph
40
40
  * - `<w:r>` - Run (contiguous text with same formatting)
41
41
  * - `<w:t>` - Text content
42
+ * - `<w:br>` - Line or page break
42
43
  * - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
43
44
  * - `<w:pStyle>` - Paragraph style (for headings)
44
45
  * - `<w:numPr>` - List numbering properties
@@ -58,7 +59,7 @@
58
59
  * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
59
60
  * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
60
61
  */
61
- import { OfficeParserAST, OfficeParserConfig } from '../types';
62
+ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
62
63
  /**
63
64
  * Parses a Word document (.docx) and extracts content, formatting, and metadata.
64
65
  *