officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
|
@@ -39,7 +39,7 @@
|
|
|
39
39
|
* @see https://www.biblioscape.com/rtf15_spec.htm RTF 1.5 Specification
|
|
40
40
|
* @see https://latex2rtf.sourceforge.net/RTF-Spec-1.2.pdf RTF 1.2 Specification
|
|
41
41
|
*/
|
|
42
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
42
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
43
43
|
/**
|
|
44
44
|
* Represents an RTF group (content enclosed in braces).
|
|
45
45
|
* Groups create formatting scopes and can contain other groups, control words, or text.
|
|
@@ -130,6 +130,12 @@ export declare class SimpleRtfParser {
|
|
|
130
130
|
private index;
|
|
131
131
|
/** The RTF content as a Buffer */
|
|
132
132
|
private buffer;
|
|
133
|
+
/** Current code page for character decoding (default is Windows-1252) */
|
|
134
|
+
private codePage;
|
|
135
|
+
/** Cached TextDecoders for different code pages */
|
|
136
|
+
private decoders;
|
|
137
|
+
/** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
|
|
138
|
+
private pendingBytes;
|
|
133
139
|
/** Total length of the buffer */
|
|
134
140
|
private length;
|
|
135
141
|
/**
|
|
@@ -140,6 +146,20 @@ export declare class SimpleRtfParser {
|
|
|
140
146
|
parse(): RtfGroup;
|
|
141
147
|
private parseControl;
|
|
142
148
|
private parseText;
|
|
149
|
+
/**
|
|
150
|
+
* Flushes the pending bytes buffer as a text node to the current group.
|
|
151
|
+
* @param group The group to append the text node to
|
|
152
|
+
*/
|
|
153
|
+
private flushPendingText;
|
|
154
|
+
/**
|
|
155
|
+
* Decodes a byte array using a "UTF-8 first" strategy.
|
|
156
|
+
* If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
|
|
157
|
+
* Otherwise, falls back to the specified code page.
|
|
158
|
+
* @param bytes The bytes to decode
|
|
159
|
+
* @param codePage The RTF code page ID
|
|
160
|
+
* @returns The decoded string
|
|
161
|
+
*/
|
|
162
|
+
private decodeBytes;
|
|
143
163
|
}
|
|
144
164
|
/**
|
|
145
165
|
* Parses an RTF file and returns the AST.
|
|
@@ -42,8 +42,8 @@
|
|
|
42
42
|
*/
|
|
43
43
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
44
44
|
exports.parseRtf = exports.SimpleRtfParser = void 0;
|
|
45
|
-
const
|
|
46
|
-
const
|
|
45
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
46
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
47
47
|
/**
|
|
48
48
|
* Lookup table mapping RTF control words to internal formats.
|
|
49
49
|
* Fully typed: if a key maps to an unsupported format, TypeScript throws an error.
|
|
@@ -88,13 +88,23 @@ const IMAGE_MIME_MAP = {
|
|
|
88
88
|
* ```
|
|
89
89
|
*/
|
|
90
90
|
class SimpleRtfParser {
|
|
91
|
+
/** Current position in the buffer */
|
|
92
|
+
index = 0;
|
|
93
|
+
/** The RTF content as a Buffer */
|
|
94
|
+
buffer;
|
|
95
|
+
/** Current code page for character decoding (default is Windows-1252) */
|
|
96
|
+
codePage = 1252;
|
|
97
|
+
/** Cached TextDecoders for different code pages */
|
|
98
|
+
decoders = {};
|
|
99
|
+
/** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
|
|
100
|
+
pendingBytes = [];
|
|
101
|
+
/** Total length of the buffer */
|
|
102
|
+
length;
|
|
91
103
|
/**
|
|
92
104
|
* Creates a new RTF parser.
|
|
93
105
|
* @param buffer - The RTF file content as a Buffer
|
|
94
106
|
*/
|
|
95
107
|
constructor(buffer) {
|
|
96
|
-
/** Current position in the buffer */
|
|
97
|
-
this.index = 0;
|
|
98
108
|
this.buffer = buffer;
|
|
99
109
|
this.length = buffer.length;
|
|
100
110
|
}
|
|
@@ -106,12 +116,14 @@ class SimpleRtfParser {
|
|
|
106
116
|
const currentGroup = stack[stack.length - 1];
|
|
107
117
|
if (char === 0x7B) { // '{'
|
|
108
118
|
this.index++;
|
|
119
|
+
this.flushPendingText(currentGroup);
|
|
109
120
|
const newGroup = { type: 'group', content: [] };
|
|
110
121
|
currentGroup.content.push(newGroup);
|
|
111
122
|
stack.push(newGroup);
|
|
112
123
|
}
|
|
113
124
|
else if (char === 0x7D) { // '}'
|
|
114
125
|
this.index++;
|
|
126
|
+
this.flushPendingText(currentGroup);
|
|
115
127
|
if (stack.length > 1) {
|
|
116
128
|
stack.pop();
|
|
117
129
|
}
|
|
@@ -129,6 +141,7 @@ class SimpleRtfParser {
|
|
|
129
141
|
this.parseText(currentGroup);
|
|
130
142
|
}
|
|
131
143
|
}
|
|
144
|
+
this.flushPendingText(root);
|
|
132
145
|
return root;
|
|
133
146
|
}
|
|
134
147
|
parseControl(group) {
|
|
@@ -137,55 +150,23 @@ class SimpleRtfParser {
|
|
|
137
150
|
const char = this.buffer[this.index];
|
|
138
151
|
// Special control symbols
|
|
139
152
|
if (char === 0x7B || char === 0x7D || char === 0x5C) { // \{ \} \\
|
|
140
|
-
|
|
153
|
+
this.pendingBytes.push(char);
|
|
141
154
|
this.index++;
|
|
142
155
|
return;
|
|
143
156
|
}
|
|
144
157
|
if (char === 0x27) { // \'xx (hex)
|
|
145
158
|
this.index++;
|
|
146
159
|
if (this.index + 1 < this.length) {
|
|
147
|
-
const hex = this.buffer
|
|
160
|
+
const hex = String.fromCharCode(this.buffer[this.index], this.buffer[this.index + 1]);
|
|
148
161
|
const code = parseInt(hex, 16);
|
|
149
162
|
if (!isNaN(code)) {
|
|
150
|
-
|
|
151
|
-
// Characters 0x80-0x9F in Windows-1252 don't map directly to Unicode
|
|
152
|
-
// We need to convert them properly
|
|
153
|
-
const windows1252ToUnicode = {
|
|
154
|
-
0x80: 0x20AC, // €
|
|
155
|
-
0x82: 0x201A, // ‚
|
|
156
|
-
0x83: 0x0192, // ƒ
|
|
157
|
-
0x84: 0x201E, // „
|
|
158
|
-
0x85: 0x2026, // …
|
|
159
|
-
0x86: 0x2020, // †
|
|
160
|
-
0x87: 0x2021, // ‡
|
|
161
|
-
0x88: 0x02C6, // ˆ
|
|
162
|
-
0x89: 0x2030, // ‰
|
|
163
|
-
0x8A: 0x0160, // Š
|
|
164
|
-
0x8B: 0x2039, // ‹
|
|
165
|
-
0x8C: 0x0152, // Œ
|
|
166
|
-
0x8E: 0x017D, // Ž
|
|
167
|
-
0x91: 0x2018, // '
|
|
168
|
-
0x92: 0x2019, // '
|
|
169
|
-
0x93: 0x201C, // "
|
|
170
|
-
0x94: 0x201D, // "
|
|
171
|
-
0x95: 0x2022, // •
|
|
172
|
-
0x96: 0x2013, // –
|
|
173
|
-
0x97: 0x2014, // —
|
|
174
|
-
0x98: 0x02DC, // ˜
|
|
175
|
-
0x99: 0x2122, // ™
|
|
176
|
-
0x9A: 0x0161, // š
|
|
177
|
-
0x9B: 0x203A, // ›
|
|
178
|
-
0x9C: 0x0153, // œ
|
|
179
|
-
0x9E: 0x017E, // ž
|
|
180
|
-
0x9F: 0x0178 // Ÿ
|
|
181
|
-
};
|
|
182
|
-
const unicodeCode = windows1252ToUnicode[code] || code;
|
|
183
|
-
group.content.push({ type: 'text', value: String.fromCharCode(unicodeCode) });
|
|
163
|
+
this.pendingBytes.push(code);
|
|
184
164
|
}
|
|
185
165
|
this.index += 2;
|
|
186
166
|
}
|
|
187
167
|
return;
|
|
188
168
|
}
|
|
169
|
+
this.flushPendingText(group);
|
|
189
170
|
if (char === 0x2A) { // \* (ignorable destination)
|
|
190
171
|
// We treat this as a control word named '*'
|
|
191
172
|
group.content.push({ type: 'control', value: '*' });
|
|
@@ -227,7 +208,7 @@ class SimpleRtfParser {
|
|
|
227
208
|
param = parseInt(paramStr, 10);
|
|
228
209
|
}
|
|
229
210
|
// Space after control word is consumed
|
|
230
|
-
if (this.index < this.length && this.buffer[this.index] === 0x20) {
|
|
211
|
+
if (name !== '' && this.index < this.length && this.buffer[this.index] === 0x20) {
|
|
231
212
|
this.index++;
|
|
232
213
|
}
|
|
233
214
|
// Handle \binN
|
|
@@ -237,6 +218,22 @@ class SimpleRtfParser {
|
|
|
237
218
|
// \binN is not added to content as we want to ignore it
|
|
238
219
|
return;
|
|
239
220
|
}
|
|
221
|
+
// Handle encoding control words
|
|
222
|
+
if (name === 'ansicpg' && param !== undefined) {
|
|
223
|
+
this.codePage = param;
|
|
224
|
+
}
|
|
225
|
+
else if (name === 'ansi') {
|
|
226
|
+
this.codePage = 1252;
|
|
227
|
+
}
|
|
228
|
+
else if (name === 'mac') {
|
|
229
|
+
this.codePage = 10000;
|
|
230
|
+
}
|
|
231
|
+
else if (name === 'pc') {
|
|
232
|
+
this.codePage = 437;
|
|
233
|
+
}
|
|
234
|
+
else if (name === 'pca') {
|
|
235
|
+
this.codePage = 850;
|
|
236
|
+
}
|
|
240
237
|
group.content.push({ type: 'control', value: name, param });
|
|
241
238
|
// If this is the first control word in the group, it might be the destination
|
|
242
239
|
if (group.content.length === 1 && group.type === 'group') {
|
|
@@ -248,19 +245,91 @@ class SimpleRtfParser {
|
|
|
248
245
|
}
|
|
249
246
|
}
|
|
250
247
|
parseText(group) {
|
|
251
|
-
let text = '';
|
|
252
248
|
while (this.index < this.length) {
|
|
253
249
|
const char = this.buffer[this.index];
|
|
250
|
+
if (char === undefined)
|
|
251
|
+
break;
|
|
254
252
|
if (char === 0x7B || char === 0x7D || char === 0x5C || char === 0x0D || char === 0x0A) {
|
|
255
253
|
break;
|
|
256
254
|
}
|
|
257
|
-
|
|
258
|
-
text += String.fromCharCode(char);
|
|
255
|
+
this.pendingBytes.push(char);
|
|
259
256
|
this.index++;
|
|
260
257
|
}
|
|
261
|
-
|
|
262
|
-
|
|
258
|
+
}
|
|
259
|
+
/**
|
|
260
|
+
* Flushes the pending bytes buffer as a text node to the current group.
|
|
261
|
+
* @param group The group to append the text node to
|
|
262
|
+
*/
|
|
263
|
+
flushPendingText(group) {
|
|
264
|
+
if (this.pendingBytes.length > 0) {
|
|
265
|
+
group.content.push({ type: 'text', value: this.decodeBytes(this.pendingBytes, this.codePage) });
|
|
266
|
+
this.pendingBytes = [];
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Decodes a byte array using a "UTF-8 first" strategy.
|
|
271
|
+
* If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
|
|
272
|
+
* Otherwise, falls back to the specified code page.
|
|
273
|
+
* @param bytes The bytes to decode
|
|
274
|
+
* @param codePage The RTF code page ID
|
|
275
|
+
* @returns The decoded string
|
|
276
|
+
*/
|
|
277
|
+
decodeBytes(bytes, codePage) {
|
|
278
|
+
const uint8 = new Uint8Array(bytes);
|
|
279
|
+
// Try UTF-8 first if there are any non-ASCII bytes.
|
|
280
|
+
// Many modern RTF generators (like calibre or web-based tools) dump UTF-8 bytes
|
|
281
|
+
// into the RTF even if the header claims a different code page.
|
|
282
|
+
if (bytes.some(b => b > 127)) {
|
|
283
|
+
try {
|
|
284
|
+
// Use fatal: true to ensure we fall back on invalid UTF-8 sequences
|
|
285
|
+
const utf8Decoder = new TextDecoder('utf-8', { fatal: true });
|
|
286
|
+
return utf8Decoder.decode(uint8);
|
|
287
|
+
}
|
|
288
|
+
catch (e) {
|
|
289
|
+
// Not valid UTF-8, continue to code page fallback
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
// Fallback to specified code page
|
|
293
|
+
if (!this.decoders[codePage]) {
|
|
294
|
+
let encoding = `windows-${codePage}`;
|
|
295
|
+
if (codePage === 10000)
|
|
296
|
+
encoding = 'macintosh';
|
|
297
|
+
else if (codePage === 437)
|
|
298
|
+
encoding = 'ibm437';
|
|
299
|
+
else if (codePage === 850)
|
|
300
|
+
encoding = 'ibm850';
|
|
301
|
+
try {
|
|
302
|
+
this.decoders[codePage] = new TextDecoder(encoding);
|
|
303
|
+
}
|
|
304
|
+
catch (e) {
|
|
305
|
+
if (codePage !== 1252) {
|
|
306
|
+
try {
|
|
307
|
+
this.decoders[codePage] = new TextDecoder('windows-1252');
|
|
308
|
+
}
|
|
309
|
+
catch (e2) {
|
|
310
|
+
return String.fromCharCode(...bytes);
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
else {
|
|
314
|
+
return String.fromCharCode(...bytes);
|
|
315
|
+
}
|
|
316
|
+
}
|
|
263
317
|
}
|
|
318
|
+
let result = this.decoders[codePage].decode(uint8);
|
|
319
|
+
// Safety override for Windows-1252 0x80-0x9F range if TextDecoder behaves like Latin-1.
|
|
320
|
+
// We replace control characters in the decoded string with their proper 1252 equivalents.
|
|
321
|
+
if (codePage === 1252 && /[\u0080-\u009F]/.test(result)) {
|
|
322
|
+
const map = {
|
|
323
|
+
'\u0080': '€', '\u0082': '‚', '\u0083': 'ƒ', '\u0084': '„', '\u0085': '…',
|
|
324
|
+
'\u0086': '†', '\u0087': '‡', '\u0088': 'ˆ', '\u0089': '‰', '\u008A': 'Š',
|
|
325
|
+
'\u008B': '‹', '\u008C': 'Œ', '\u008E': 'Ž', '\u0091': '‘', '\u0092': '’',
|
|
326
|
+
'\u0093': '“', '\u0094': '”', '\u0095': '•', '\u0096': '–', '\u0097': '—',
|
|
327
|
+
'\u0098': '˜', '\u0099': '™', '\u009A': 'š', '\u009B': '›', '\u009C': 'œ',
|
|
328
|
+
'\u009E': 'ž', '\u009F': 'Ÿ'
|
|
329
|
+
};
|
|
330
|
+
return result.replace(/[\u0080-\u009F]/g, m => map[m] || m);
|
|
331
|
+
}
|
|
332
|
+
return result;
|
|
264
333
|
}
|
|
265
334
|
}
|
|
266
335
|
exports.SimpleRtfParser = SimpleRtfParser;
|
|
@@ -1510,10 +1579,10 @@ const parseRtf = async (buffer, config) => {
|
|
|
1510
1579
|
// Passing base64 string directly would be interpreted as a file path,
|
|
1511
1580
|
// causing ENAMETOOLONG error for large images.
|
|
1512
1581
|
const imageBuffer = Buffer.from(attachment.data, 'base64');
|
|
1513
|
-
attachment.ocrText = (await (0,
|
|
1582
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
1514
1583
|
}
|
|
1515
1584
|
catch (e) {
|
|
1516
|
-
(0,
|
|
1585
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
1517
1586
|
}
|
|
1518
1587
|
}
|
|
1519
1588
|
}
|
|
@@ -39,6 +39,7 @@
|
|
|
39
39
|
* - `<w:p>` - Paragraph
|
|
40
40
|
* - `<w:r>` - Run (contiguous text with same formatting)
|
|
41
41
|
* - `<w:t>` - Text content
|
|
42
|
+
* - `<w:br>` - Line or page break
|
|
42
43
|
* - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
|
|
43
44
|
* - `<w:pStyle>` - Paragraph style (for headings)
|
|
44
45
|
* - `<w:numPr>` - List numbering properties
|
|
@@ -58,7 +59,7 @@
|
|
|
58
59
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
|
|
59
60
|
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
|
|
60
61
|
*/
|
|
61
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
62
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
62
63
|
/**
|
|
63
64
|
* Parses a Word document (.docx) and extracts content, formatting, and metadata.
|
|
64
65
|
*
|