officeparser 6.1.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +284 -86
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -28
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +107 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +878 -5
- package/dist/officeparser.browser.iife.js +703 -49
- package/dist/officeparser.browser.mjs +703 -49
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +237 -128
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +132 -123
- package/dist/parsers/RtfParser.d.ts +22 -2
- package/dist/parsers/RtfParser.js +1398 -1282
- package/dist/parsers/WordParser.d.ts +3 -2
- package/dist/parsers/WordParser.js +333 -115
- package/dist/sbom.cdx.json +103 -103
- package/dist/types.d.ts +833 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +28 -9
|
@@ -42,6 +42,8 @@
|
|
|
42
42
|
*/
|
|
43
43
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
44
44
|
exports.parseRtf = exports.SimpleRtfParser = void 0;
|
|
45
|
+
const types_js_1 = require("../types.js");
|
|
46
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
45
47
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
46
48
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
47
49
|
/**
|
|
@@ -92,6 +94,12 @@ class SimpleRtfParser {
|
|
|
92
94
|
index = 0;
|
|
93
95
|
/** The RTF content as a Buffer */
|
|
94
96
|
buffer;
|
|
97
|
+
/** Current code page for character decoding (default is Windows-1252) */
|
|
98
|
+
codePage = 1252;
|
|
99
|
+
/** Cached TextDecoders for different code pages */
|
|
100
|
+
decoders = {};
|
|
101
|
+
/** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
|
|
102
|
+
pendingBytes = [];
|
|
95
103
|
/** Total length of the buffer */
|
|
96
104
|
length;
|
|
97
105
|
/**
|
|
@@ -110,12 +118,14 @@ class SimpleRtfParser {
|
|
|
110
118
|
const currentGroup = stack[stack.length - 1];
|
|
111
119
|
if (char === 0x7B) { // '{'
|
|
112
120
|
this.index++;
|
|
121
|
+
this.flushPendingText(currentGroup);
|
|
113
122
|
const newGroup = { type: 'group', content: [] };
|
|
114
123
|
currentGroup.content.push(newGroup);
|
|
115
124
|
stack.push(newGroup);
|
|
116
125
|
}
|
|
117
126
|
else if (char === 0x7D) { // '}'
|
|
118
127
|
this.index++;
|
|
128
|
+
this.flushPendingText(currentGroup);
|
|
119
129
|
if (stack.length > 1) {
|
|
120
130
|
stack.pop();
|
|
121
131
|
}
|
|
@@ -133,6 +143,7 @@ class SimpleRtfParser {
|
|
|
133
143
|
this.parseText(currentGroup);
|
|
134
144
|
}
|
|
135
145
|
}
|
|
146
|
+
this.flushPendingText(root);
|
|
136
147
|
return root;
|
|
137
148
|
}
|
|
138
149
|
parseControl(group) {
|
|
@@ -141,55 +152,23 @@ class SimpleRtfParser {
|
|
|
141
152
|
const char = this.buffer[this.index];
|
|
142
153
|
// Special control symbols
|
|
143
154
|
if (char === 0x7B || char === 0x7D || char === 0x5C) { // \{ \} \\
|
|
144
|
-
|
|
155
|
+
this.pendingBytes.push(char);
|
|
145
156
|
this.index++;
|
|
146
157
|
return;
|
|
147
158
|
}
|
|
148
159
|
if (char === 0x27) { // \'xx (hex)
|
|
149
160
|
this.index++;
|
|
150
161
|
if (this.index + 1 < this.length) {
|
|
151
|
-
const hex = this.buffer
|
|
162
|
+
const hex = String.fromCharCode(this.buffer[this.index], this.buffer[this.index + 1]);
|
|
152
163
|
const code = parseInt(hex, 16);
|
|
153
164
|
if (!isNaN(code)) {
|
|
154
|
-
|
|
155
|
-
// Characters 0x80-0x9F in Windows-1252 don't map directly to Unicode
|
|
156
|
-
// We need to convert them properly
|
|
157
|
-
const windows1252ToUnicode = {
|
|
158
|
-
0x80: 0x20AC, // €
|
|
159
|
-
0x82: 0x201A, // ‚
|
|
160
|
-
0x83: 0x0192, // ƒ
|
|
161
|
-
0x84: 0x201E, // „
|
|
162
|
-
0x85: 0x2026, // …
|
|
163
|
-
0x86: 0x2020, // †
|
|
164
|
-
0x87: 0x2021, // ‡
|
|
165
|
-
0x88: 0x02C6, // ˆ
|
|
166
|
-
0x89: 0x2030, // ‰
|
|
167
|
-
0x8A: 0x0160, // Š
|
|
168
|
-
0x8B: 0x2039, // ‹
|
|
169
|
-
0x8C: 0x0152, // Œ
|
|
170
|
-
0x8E: 0x017D, // Ž
|
|
171
|
-
0x91: 0x2018, // '
|
|
172
|
-
0x92: 0x2019, // '
|
|
173
|
-
0x93: 0x201C, // "
|
|
174
|
-
0x94: 0x201D, // "
|
|
175
|
-
0x95: 0x2022, // •
|
|
176
|
-
0x96: 0x2013, // –
|
|
177
|
-
0x97: 0x2014, // —
|
|
178
|
-
0x98: 0x02DC, // ˜
|
|
179
|
-
0x99: 0x2122, // ™
|
|
180
|
-
0x9A: 0x0161, // š
|
|
181
|
-
0x9B: 0x203A, // ›
|
|
182
|
-
0x9C: 0x0153, // œ
|
|
183
|
-
0x9E: 0x017E, // ž
|
|
184
|
-
0x9F: 0x0178 // Ÿ
|
|
185
|
-
};
|
|
186
|
-
const unicodeCode = windows1252ToUnicode[code] || code;
|
|
187
|
-
group.content.push({ type: 'text', value: String.fromCharCode(unicodeCode) });
|
|
165
|
+
this.pendingBytes.push(code);
|
|
188
166
|
}
|
|
189
167
|
this.index += 2;
|
|
190
168
|
}
|
|
191
169
|
return;
|
|
192
170
|
}
|
|
171
|
+
this.flushPendingText(group);
|
|
193
172
|
if (char === 0x2A) { // \* (ignorable destination)
|
|
194
173
|
// We treat this as a control word named '*'
|
|
195
174
|
group.content.push({ type: 'control', value: '*' });
|
|
@@ -231,7 +210,7 @@ class SimpleRtfParser {
|
|
|
231
210
|
param = parseInt(paramStr, 10);
|
|
232
211
|
}
|
|
233
212
|
// Space after control word is consumed
|
|
234
|
-
if (this.index < this.length && this.buffer[this.index] === 0x20) {
|
|
213
|
+
if (name !== '' && this.index < this.length && this.buffer[this.index] === 0x20) {
|
|
235
214
|
this.index++;
|
|
236
215
|
}
|
|
237
216
|
// Handle \binN
|
|
@@ -241,6 +220,22 @@ class SimpleRtfParser {
|
|
|
241
220
|
// \binN is not added to content as we want to ignore it
|
|
242
221
|
return;
|
|
243
222
|
}
|
|
223
|
+
// Handle encoding control words
|
|
224
|
+
if (name === 'ansicpg' && param !== undefined) {
|
|
225
|
+
this.codePage = param;
|
|
226
|
+
}
|
|
227
|
+
else if (name === 'ansi') {
|
|
228
|
+
this.codePage = 1252;
|
|
229
|
+
}
|
|
230
|
+
else if (name === 'mac') {
|
|
231
|
+
this.codePage = 10000;
|
|
232
|
+
}
|
|
233
|
+
else if (name === 'pc') {
|
|
234
|
+
this.codePage = 437;
|
|
235
|
+
}
|
|
236
|
+
else if (name === 'pca') {
|
|
237
|
+
this.codePage = 850;
|
|
238
|
+
}
|
|
244
239
|
group.content.push({ type: 'control', value: name, param });
|
|
245
240
|
// If this is the first control word in the group, it might be the destination
|
|
246
241
|
if (group.content.length === 1 && group.type === 'group') {
|
|
@@ -252,20 +247,92 @@ class SimpleRtfParser {
|
|
|
252
247
|
}
|
|
253
248
|
}
|
|
254
249
|
parseText(group) {
|
|
255
|
-
let text = '';
|
|
256
250
|
while (this.index < this.length) {
|
|
257
251
|
const char = this.buffer[this.index];
|
|
252
|
+
if (char === undefined)
|
|
253
|
+
break;
|
|
258
254
|
if (char === 0x7B || char === 0x7D || char === 0x5C || char === 0x0D || char === 0x0A) {
|
|
259
255
|
break;
|
|
260
256
|
}
|
|
261
|
-
|
|
262
|
-
text += String.fromCharCode(char);
|
|
257
|
+
this.pendingBytes.push(char);
|
|
263
258
|
this.index++;
|
|
264
259
|
}
|
|
265
|
-
|
|
266
|
-
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* Flushes the pending bytes buffer as a text node to the current group.
|
|
263
|
+
* @param group The group to append the text node to
|
|
264
|
+
*/
|
|
265
|
+
flushPendingText(group) {
|
|
266
|
+
if (this.pendingBytes.length > 0) {
|
|
267
|
+
group.content.push({ type: 'text', value: this.decodeBytes(this.pendingBytes, this.codePage) });
|
|
268
|
+
this.pendingBytes = [];
|
|
267
269
|
}
|
|
268
270
|
}
|
|
271
|
+
/**
|
|
272
|
+
* Decodes a byte array using a "UTF-8 first" strategy.
|
|
273
|
+
* If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
|
|
274
|
+
* Otherwise, falls back to the specified code page.
|
|
275
|
+
* @param bytes The bytes to decode
|
|
276
|
+
* @param codePage The RTF code page ID
|
|
277
|
+
* @returns The decoded string
|
|
278
|
+
*/
|
|
279
|
+
decodeBytes(bytes, codePage) {
|
|
280
|
+
const uint8 = new Uint8Array(bytes);
|
|
281
|
+
// Try UTF-8 first if there are any non-ASCII bytes.
|
|
282
|
+
// Many modern RTF generators (like calibre or web-based tools) dump UTF-8 bytes
|
|
283
|
+
// into the RTF even if the header claims a different code page.
|
|
284
|
+
if (bytes.some(b => b > 127)) {
|
|
285
|
+
try {
|
|
286
|
+
// Use fatal: true to ensure we fall back on invalid UTF-8 sequences
|
|
287
|
+
const utf8Decoder = new TextDecoder('utf-8', { fatal: true });
|
|
288
|
+
return utf8Decoder.decode(uint8);
|
|
289
|
+
}
|
|
290
|
+
catch (e) {
|
|
291
|
+
// Not valid UTF-8, continue to code page fallback
|
|
292
|
+
}
|
|
293
|
+
}
|
|
294
|
+
// Fallback to specified code page
|
|
295
|
+
if (!this.decoders[codePage]) {
|
|
296
|
+
let encoding = `windows-${codePage}`;
|
|
297
|
+
if (codePage === 10000)
|
|
298
|
+
encoding = 'macintosh';
|
|
299
|
+
else if (codePage === 437)
|
|
300
|
+
encoding = 'ibm437';
|
|
301
|
+
else if (codePage === 850)
|
|
302
|
+
encoding = 'ibm850';
|
|
303
|
+
try {
|
|
304
|
+
this.decoders[codePage] = new TextDecoder(encoding);
|
|
305
|
+
}
|
|
306
|
+
catch (e) {
|
|
307
|
+
if (codePage !== 1252) {
|
|
308
|
+
try {
|
|
309
|
+
this.decoders[codePage] = new TextDecoder('windows-1252');
|
|
310
|
+
}
|
|
311
|
+
catch (e2) {
|
|
312
|
+
return String.fromCharCode(...bytes);
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
else {
|
|
316
|
+
return String.fromCharCode(...bytes);
|
|
317
|
+
}
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
let result = this.decoders[codePage].decode(uint8);
|
|
321
|
+
// Safety override for Windows-1252 0x80-0x9F range if TextDecoder behaves like Latin-1.
|
|
322
|
+
// We replace control characters in the decoded string with their proper 1252 equivalents.
|
|
323
|
+
if (codePage === 1252 && /[\u0080-\u009F]/.test(result)) {
|
|
324
|
+
const map = {
|
|
325
|
+
'\u0080': '€', '\u0082': '‚', '\u0083': 'ƒ', '\u0084': '„', '\u0085': '…',
|
|
326
|
+
'\u0086': '†', '\u0087': '‡', '\u0088': 'ˆ', '\u0089': '‰', '\u008A': 'Š',
|
|
327
|
+
'\u008B': '‹', '\u008C': 'Œ', '\u008E': 'Ž', '\u0091': '‘', '\u0092': '’',
|
|
328
|
+
'\u0093': '“', '\u0094': '”', '\u0095': '•', '\u0096': '–', '\u0097': '—',
|
|
329
|
+
'\u0098': '˜', '\u0099': '™', '\u009A': 'š', '\u009B': '›', '\u009C': 'œ',
|
|
330
|
+
'\u009E': 'ž', '\u009F': 'Ÿ'
|
|
331
|
+
};
|
|
332
|
+
return result.replace(/[\u0080-\u009F]/g, m => map[m] || m);
|
|
333
|
+
}
|
|
334
|
+
return result;
|
|
335
|
+
}
|
|
269
336
|
}
|
|
270
337
|
exports.SimpleRtfParser = SimpleRtfParser;
|
|
271
338
|
/**
|
|
@@ -292,1388 +359,1437 @@ exports.SimpleRtfParser = SimpleRtfParser;
|
|
|
292
359
|
* @returns The parsed AST.
|
|
293
360
|
*/
|
|
294
361
|
const parseRtf = async (buffer, config) => {
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
362
|
+
const parser = new SimpleRtfParser(buffer);
|
|
363
|
+
const doc = parser.parse();
|
|
364
|
+
// Extract font and color tables
|
|
365
|
+
const fontTable = extractFontTable(doc);
|
|
366
|
+
const colorTable = extractColorTable(doc);
|
|
367
|
+
const content = [];
|
|
368
|
+
const notes = [];
|
|
369
|
+
const attachments = [];
|
|
370
|
+
// State for paragraph construction
|
|
371
|
+
let currentParagraphTextChunks = [];
|
|
372
|
+
let currentParagraphChildren = [];
|
|
373
|
+
let currentParagraphRawChunks = [];
|
|
374
|
+
// State for text run construction
|
|
375
|
+
let currentRunTextChunks = [];
|
|
376
|
+
let currentFormatting = {};
|
|
377
|
+
// Target for content (main body or notes)
|
|
378
|
+
let currentTarget = content;
|
|
379
|
+
// Paragraph-level state
|
|
380
|
+
let paragraphIndent = 0;
|
|
381
|
+
let paragraphAlignment = 'left';
|
|
382
|
+
let isListItem = false;
|
|
383
|
+
let listType;
|
|
384
|
+
let headingLevel;
|
|
385
|
+
let currentListId;
|
|
386
|
+
let currentAnchorIds = [];
|
|
387
|
+
// Persistent list state for listtext/pntext detection
|
|
388
|
+
// These persist across paragraphs to allow list items without explicit \ls
|
|
389
|
+
let lastKnownListId;
|
|
390
|
+
let lastKnownListType;
|
|
391
|
+
let tableStack = [];
|
|
392
|
+
let currentFootnoteId = 0;
|
|
393
|
+
// ═══════════════════════════════════════════════════════════════════
|
|
394
|
+
// Note type tracking (footnotes vs endnotes)
|
|
395
|
+
// ═══════════════════════════════════════════════════════════════════
|
|
396
|
+
// RTF uses \fet to distinguish note types:
|
|
397
|
+
// \fet0 = footnotes only (default)
|
|
398
|
+
// \fet1 = endnotes only
|
|
399
|
+
// \fet2 = both footnotes and endnotes
|
|
400
|
+
let fetValue = 0; // Default to footnotes only
|
|
401
|
+
// Helper to get current table context
|
|
402
|
+
const getCurrentTable = () => tableStack.length > 0 ? tableStack[tableStack.length - 1] : undefined;
|
|
403
|
+
// Helper to ensure a table context exists (for top-level tables)
|
|
404
|
+
const ensureTableContext = () => {
|
|
405
|
+
if (tableStack.length === 0) {
|
|
406
|
+
tableStack.push({
|
|
407
|
+
rows: [],
|
|
408
|
+
currentCells: [],
|
|
409
|
+
currentCellContent: [],
|
|
410
|
+
rowIndex: 0
|
|
411
|
+
});
|
|
412
|
+
}
|
|
413
|
+
};
|
|
414
|
+
let inTable = false;
|
|
415
|
+
let paragraphInTable = false;
|
|
416
|
+
let tableId = 0;
|
|
417
|
+
let rowCellProps = [];
|
|
418
|
+
let currentCellDefinitionProps = { isMergedContinuation: false };
|
|
419
|
+
let cellContentIndex = 0;
|
|
420
|
+
// ═══════════════════════════════════════════════════════════════════
|
|
421
|
+
// List state tracking (Word 97+ uses \ls for list style ID)
|
|
422
|
+
// ═══════════════════════════════════════════════════════════════════
|
|
423
|
+
let listIdCounter = 0;
|
|
424
|
+
const listStyleIdMap = {};
|
|
425
|
+
// List definition state for parsing \listtable
|
|
426
|
+
let parsingListTable = false;
|
|
427
|
+
let parsingListDefinition = false;
|
|
428
|
+
let currentDefinedListId;
|
|
429
|
+
let currentDefinedListType;
|
|
430
|
+
const listTypeMap = {};
|
|
431
|
+
// List override state for parsing \listoverridetable
|
|
432
|
+
let parsingListOverrideTable = false;
|
|
433
|
+
let currentListOverrideListId;
|
|
434
|
+
let currentListOverrideLs;
|
|
435
|
+
const listOverrideMap = {}; // Maps \ls ID to \listid
|
|
436
|
+
// List counters for itemIndex tracking
|
|
437
|
+
// Map: listId -> indentation level -> count
|
|
438
|
+
const listCounters = {};
|
|
439
|
+
// ═══════════════════════════════════════════════════════════════════
|
|
440
|
+
// Hyperlink state (RTF uses \field{\*\fldinst HYPERLINK "url"})
|
|
441
|
+
// ═══════════════════════════════════════════════════════════════════
|
|
442
|
+
let currentLinkUrl;
|
|
443
|
+
// Helper to check if formatting changed
|
|
444
|
+
const formattingChanged = (a, b) => {
|
|
445
|
+
return a.bold !== b.bold ||
|
|
446
|
+
a.italic !== b.italic ||
|
|
447
|
+
a.underline !== b.underline ||
|
|
448
|
+
a.strikethrough !== b.strikethrough ||
|
|
449
|
+
a.size !== b.size ||
|
|
450
|
+
a.font !== b.font ||
|
|
451
|
+
a.color !== b.color ||
|
|
452
|
+
a.backgroundColor !== b.backgroundColor ||
|
|
453
|
+
a.subscript !== b.subscript ||
|
|
454
|
+
a.superscript !== b.superscript;
|
|
455
|
+
};
|
|
456
|
+
// Helper to flush current run to paragraph children
|
|
457
|
+
const flushRun = () => {
|
|
458
|
+
if (currentRunTextChunks.length > 0) {
|
|
459
|
+
const currentRunText = currentRunTextChunks.join('');
|
|
460
|
+
const node = {
|
|
461
|
+
type: 'text',
|
|
462
|
+
text: currentRunText,
|
|
463
|
+
formatting: { ...currentFormatting }
|
|
464
|
+
};
|
|
465
|
+
// Use TextMetadata.link for hyperlinks
|
|
466
|
+
if (currentLinkUrl) {
|
|
467
|
+
node.metadata = {
|
|
468
|
+
link: currentLinkUrl,
|
|
469
|
+
linkType: classifyLinkType(currentLinkUrl)
|
|
396
470
|
};
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
471
|
+
}
|
|
472
|
+
currentParagraphChildren.push(node);
|
|
473
|
+
currentParagraphTextChunks.push(currentRunText);
|
|
474
|
+
currentRunTextChunks = [];
|
|
475
|
+
}
|
|
476
|
+
};
|
|
477
|
+
let isFlushingTable = false;
|
|
478
|
+
// Helper to flush current paragraph
|
|
479
|
+
const flushParagraph = () => {
|
|
480
|
+
flushRun(); // Ensure last run is added
|
|
481
|
+
// Check if we need to end the table
|
|
482
|
+
// If we were in a table, but this paragraph is NOT marked as in-table,
|
|
483
|
+
// and we have content, then the table has ended.
|
|
484
|
+
const hasContent = currentParagraphTextChunks.length > 0 || currentParagraphChildren.length > 0;
|
|
485
|
+
if (inTable && !paragraphInTable && !isFlushingTable && hasContent) {
|
|
486
|
+
// CRITICAL: Save current paragraph content before flushing table
|
|
487
|
+
// because flushTable() -> flushRow() -> flushCell() -> flushParagraph()
|
|
488
|
+
// would otherwise process this content during the table flush
|
|
489
|
+
const savedParagraphTextChunks = [...currentParagraphTextChunks];
|
|
490
|
+
const savedParagraphChildren = [...currentParagraphChildren];
|
|
491
|
+
const savedParagraphRawChunks = [...currentParagraphRawChunks];
|
|
492
|
+
// Clear buffers so nested flushParagraph() doesn't process them
|
|
493
|
+
currentParagraphTextChunks = [];
|
|
494
|
+
currentParagraphChildren = [];
|
|
495
|
+
currentParagraphRawChunks = [];
|
|
496
|
+
flushTable();
|
|
497
|
+
// Restore the saved content for processing after the table
|
|
498
|
+
currentParagraphTextChunks = savedParagraphTextChunks;
|
|
499
|
+
currentParagraphChildren = savedParagraphChildren;
|
|
500
|
+
currentParagraphRawChunks = savedParagraphRawChunks;
|
|
501
|
+
}
|
|
502
|
+
if (hasContent) {
|
|
503
|
+
const currentParagraphText = currentParagraphTextChunks.join('');
|
|
504
|
+
let nodeType = 'paragraph';
|
|
505
|
+
let metadata = undefined;
|
|
506
|
+
// Heuristic heading detection if no explicit \s style was found
|
|
507
|
+
if (headingLevel === undefined && currentParagraphChildren.length > 0) {
|
|
508
|
+
// Check if the first child is bold and larger than default (12pt)
|
|
509
|
+
const firstChild = currentParagraphChildren[0];
|
|
510
|
+
if (firstChild.type === 'text' && firstChild.formatting?.bold) {
|
|
511
|
+
const size = parseInt(firstChild.formatting.size || '12');
|
|
512
|
+
if (size >= 14 && currentParagraphText.length < 300) {
|
|
513
|
+
if (size >= 22)
|
|
514
|
+
headingLevel = 1;
|
|
515
|
+
else if (size >= 18)
|
|
516
|
+
headingLevel = 2;
|
|
517
|
+
else if (size >= 16)
|
|
518
|
+
headingLevel = 3;
|
|
519
|
+
else
|
|
520
|
+
headingLevel = 4;
|
|
521
|
+
}
|
|
403
522
|
}
|
|
404
|
-
currentParagraphChildren.push(node);
|
|
405
|
-
currentParagraphText += currentRunText;
|
|
406
|
-
currentRunText = '';
|
|
407
523
|
}
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
if (
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
//
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
}
|
|
444
|
-
else if (isListItem) {
|
|
445
|
-
nodeType = 'list';
|
|
446
|
-
// Use lastKnownListId if currentListId is not set
|
|
447
|
-
// (happens when list item is detected via listtext/pntext)
|
|
448
|
-
const effectiveListId = currentListId || lastKnownListId;
|
|
449
|
-
const effectiveListType = listType || lastKnownListType || 'unordered';
|
|
450
|
-
// Calculate itemIndex
|
|
451
|
-
let itemIndex = 0;
|
|
452
|
-
if (effectiveListId) {
|
|
453
|
-
if (!listCounters[effectiveListId]) {
|
|
454
|
-
listCounters[effectiveListId] = {};
|
|
455
|
-
}
|
|
456
|
-
if (listCounters[effectiveListId][paragraphIndent] === undefined) {
|
|
457
|
-
listCounters[effectiveListId][paragraphIndent] = 0;
|
|
458
|
-
}
|
|
459
|
-
else {
|
|
460
|
-
listCounters[effectiveListId][paragraphIndent]++;
|
|
524
|
+
// Heuristic list detection if no explicit list control words were found
|
|
525
|
+
if (!isListItem && paragraphIndent > 300) { // RTF indents are in twips (1440 = 1 inch)
|
|
526
|
+
const trimmed = currentParagraphText.trim();
|
|
527
|
+
// Check for bullet characters or digits followed by period
|
|
528
|
+
if (/^[\u2022\u00b7\-\u25cf\u25cb]/.test(trimmed) || /^\d+[.\)]/.test(trimmed)) {
|
|
529
|
+
isListItem = true;
|
|
530
|
+
listType = /^\d+[.\)]/.test(trimmed) ? 'ordered' : 'unordered';
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
if (headingLevel !== undefined && headingLevel > 0) {
|
|
534
|
+
nodeType = 'heading';
|
|
535
|
+
metadata = { level: headingLevel };
|
|
536
|
+
// Reset list context when we encounter a heading
|
|
537
|
+
lastKnownListId = undefined;
|
|
538
|
+
lastKnownListType = undefined;
|
|
539
|
+
}
|
|
540
|
+
else if (isListItem) {
|
|
541
|
+
nodeType = 'list';
|
|
542
|
+
// Use lastKnownListId if currentListId is not set
|
|
543
|
+
// (happens when list item is detected via listtext/pntext)
|
|
544
|
+
const effectiveListId = currentListId || lastKnownListId;
|
|
545
|
+
const effectiveListType = listType || lastKnownListType || 'unordered';
|
|
546
|
+
// Calculate itemIndex
|
|
547
|
+
let itemIndex = 0;
|
|
548
|
+
if (effectiveListId) {
|
|
549
|
+
if (!listCounters[effectiveListId]) {
|
|
550
|
+
listCounters[effectiveListId] = {};
|
|
551
|
+
}
|
|
552
|
+
// Reset all sub-level counters when returning to a shallower level
|
|
553
|
+
// This ensures that if we go from level 2 to level 0 and back to level 2,
|
|
554
|
+
// the new level 2 sequence starts from 0.
|
|
555
|
+
const levels = Object.keys(listCounters[effectiveListId]).map(l => parseInt(l));
|
|
556
|
+
for (const level of levels) {
|
|
557
|
+
if (level > paragraphIndent) {
|
|
558
|
+
delete listCounters[effectiveListId][level];
|
|
461
559
|
}
|
|
462
|
-
itemIndex = listCounters[effectiveListId][paragraphIndent];
|
|
463
560
|
}
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
indentation: paragraphIndent,
|
|
467
|
-
listId: effectiveListId || '',
|
|
468
|
-
itemIndex: itemIndex,
|
|
469
|
-
alignment: paragraphAlignment,
|
|
470
|
-
};
|
|
471
|
-
// Save for next listtext detection
|
|
472
|
-
if (effectiveListId) {
|
|
473
|
-
lastKnownListId = effectiveListId;
|
|
561
|
+
if (listCounters[effectiveListId][paragraphIndent] === undefined) {
|
|
562
|
+
listCounters[effectiveListId][paragraphIndent] = 0;
|
|
474
563
|
}
|
|
475
|
-
|
|
476
|
-
|
|
564
|
+
else {
|
|
565
|
+
listCounters[effectiveListId][paragraphIndent]++;
|
|
477
566
|
}
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
567
|
+
itemIndex = listCounters[effectiveListId][paragraphIndent];
|
|
568
|
+
}
|
|
569
|
+
metadata = {
|
|
570
|
+
listType: effectiveListType,
|
|
571
|
+
indentation: paragraphIndent,
|
|
572
|
+
listId: effectiveListId || '',
|
|
573
|
+
itemIndex: itemIndex,
|
|
574
|
+
alignment: paragraphAlignment,
|
|
485
575
|
};
|
|
486
|
-
|
|
487
|
-
|
|
576
|
+
// Save for next listtext detection
|
|
577
|
+
if (effectiveListId) {
|
|
578
|
+
lastKnownListId = effectiveListId;
|
|
488
579
|
}
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
if (inTable && paragraphInTable) {
|
|
492
|
-
ensureTableContext();
|
|
493
|
-
getCurrentTable().currentCellContent.push(node);
|
|
494
|
-
}
|
|
495
|
-
else {
|
|
496
|
-
currentTarget.push(node);
|
|
580
|
+
if (effectiveListType) {
|
|
581
|
+
lastKnownListType = effectiveListType;
|
|
497
582
|
}
|
|
498
|
-
currentParagraphText = '';
|
|
499
|
-
currentParagraphChildren = [];
|
|
500
|
-
currentParagraphRaw = '';
|
|
501
|
-
// Reset paragraph-level state
|
|
502
|
-
paragraphIndent = 0;
|
|
503
|
-
isListItem = false;
|
|
504
|
-
listType = undefined;
|
|
505
|
-
headingLevel = undefined;
|
|
506
|
-
currentListId = undefined;
|
|
507
583
|
}
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
if (!ctx)
|
|
514
|
-
return undefined;
|
|
515
|
-
// Always return a cell node, even if empty, to preserve table structure (grid)
|
|
516
|
-
const cellNode = {
|
|
517
|
-
type: 'cell',
|
|
518
|
-
text: ctx.currentCellContent.map(c => c.text).join('\n'),
|
|
519
|
-
children: [...ctx.currentCellContent],
|
|
584
|
+
const node = {
|
|
585
|
+
type: nodeType,
|
|
586
|
+
text: currentParagraphText,
|
|
587
|
+
children: currentParagraphChildren,
|
|
588
|
+
formatting: undefined,
|
|
520
589
|
metadata: {
|
|
521
|
-
|
|
522
|
-
|
|
590
|
+
...metadata,
|
|
591
|
+
anchorIds: currentAnchorIds.length > 0 ? [...currentAnchorIds] : undefined
|
|
523
592
|
}
|
|
524
593
|
};
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
};
|
|
528
|
-
// Helper to flush current row - creates a row node from collected cells
|
|
529
|
-
const flushRow = (tableCtx) => {
|
|
530
|
-
const ctx = tableCtx || getCurrentTable();
|
|
531
|
-
if (!ctx)
|
|
532
|
-
return;
|
|
533
|
-
const cell = flushCell(ctx);
|
|
534
|
-
// Only add cell if it has content (prevents phantom empty cells during cleanup)
|
|
535
|
-
if (cell && (cell.children && cell.children.length > 0 || cell.text)) {
|
|
536
|
-
ctx.currentCells.push(cell);
|
|
537
|
-
}
|
|
538
|
-
if (ctx.currentCells.length > 0) {
|
|
539
|
-
const rowNode = {
|
|
540
|
-
type: 'row',
|
|
541
|
-
text: ctx.currentCells.map(c => c.text).filter(t => t !== '').join(config.newlineDelimiter ?? '\n'),
|
|
542
|
-
children: [...ctx.currentCells]
|
|
543
|
-
};
|
|
544
|
-
ctx.rows.push(rowNode);
|
|
545
|
-
ctx.currentCells = [];
|
|
546
|
-
ctx.rowIndex++;
|
|
594
|
+
if (config.includeRawContent && currentParagraphRawChunks.length > 0) {
|
|
595
|
+
node.rawContent = currentParagraphRawChunks.join('');
|
|
547
596
|
}
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
isFlushingTable = true;
|
|
554
|
-
const ctx = getCurrentTable();
|
|
555
|
-
if (!ctx) {
|
|
556
|
-
isFlushingTable = false;
|
|
557
|
-
return;
|
|
597
|
+
// If we're building a table, add to current cell
|
|
598
|
+
// but ONLY if this paragraph was actually marked as in-table
|
|
599
|
+
if (inTable && paragraphInTable) {
|
|
600
|
+
ensureTableContext();
|
|
601
|
+
getCurrentTable().currentCellContent.push(node);
|
|
558
602
|
}
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
tableId++;
|
|
562
|
-
const tableNode = {
|
|
563
|
-
type: 'table',
|
|
564
|
-
text: ctx.rows.map(r => r.text).join('\n'), // Aggregate text from rows
|
|
565
|
-
children: [...ctx.rows]
|
|
566
|
-
};
|
|
567
|
-
// If we have a parent table, add this table to the parent's current cell
|
|
568
|
-
if (tableStack.length > 1) {
|
|
569
|
-
const parentCtx = tableStack[tableStack.length - 2];
|
|
570
|
-
parentCtx.currentCellContent.push(tableNode);
|
|
571
|
-
}
|
|
572
|
-
else {
|
|
573
|
-
currentTarget.push(tableNode);
|
|
574
|
-
}
|
|
603
|
+
else {
|
|
604
|
+
currentTarget.push(node);
|
|
575
605
|
}
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
606
|
+
currentParagraphTextChunks = [];
|
|
607
|
+
currentParagraphChildren = [];
|
|
608
|
+
currentParagraphRawChunks = [];
|
|
609
|
+
// Reset paragraph-level state that should NOT persist
|
|
610
|
+
// Note: list properties (\ls, \ilvl, \li) and alignment (\ql, etc.)
|
|
611
|
+
// persist in RTF until \pard or a new value is set.
|
|
612
|
+
currentAnchorIds = []; // Reset anchors
|
|
613
|
+
}
|
|
614
|
+
};
|
|
615
|
+
// Helper to flush current cell
|
|
616
|
+
const flushCell = (tableCtx) => {
|
|
617
|
+
flushParagraph();
|
|
618
|
+
const ctx = tableCtx || getCurrentTable();
|
|
619
|
+
if (!ctx)
|
|
620
|
+
return undefined;
|
|
621
|
+
// Always return a cell node, even if empty, to preserve table structure (grid)
|
|
622
|
+
const cellNode = {
|
|
623
|
+
type: 'cell',
|
|
624
|
+
text: ctx.currentCellContent.map(c => c.text).join('\n'),
|
|
625
|
+
children: [...ctx.currentCellContent],
|
|
626
|
+
metadata: {
|
|
627
|
+
row: ctx.rowIndex,
|
|
628
|
+
col: ctx.currentCells.length
|
|
582
629
|
}
|
|
583
|
-
isFlushingTable = false;
|
|
584
630
|
};
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
631
|
+
ctx.currentCellContent = [];
|
|
632
|
+
return cellNode;
|
|
633
|
+
};
|
|
634
|
+
// Helper to flush current row - creates a row node from collected cells
|
|
635
|
+
const flushRow = (tableCtx) => {
|
|
636
|
+
const ctx = tableCtx || getCurrentTable();
|
|
637
|
+
if (!ctx)
|
|
638
|
+
return;
|
|
639
|
+
const cell = flushCell(ctx);
|
|
640
|
+
// Only add cell if it has content (prevents phantom empty cells during cleanup)
|
|
641
|
+
if (cell && (cell.children && cell.children.length > 0 || cell.text)) {
|
|
642
|
+
ctx.currentCells.push(cell);
|
|
643
|
+
}
|
|
644
|
+
if (ctx.currentCells.length > 0) {
|
|
645
|
+
const rowNode = {
|
|
646
|
+
type: 'row',
|
|
647
|
+
text: ctx.currentCells.map(c => c.text).filter(t => t !== '').join(config.newlineDelimiter),
|
|
648
|
+
children: [...ctx.currentCells]
|
|
649
|
+
};
|
|
650
|
+
ctx.rows.push(rowNode);
|
|
651
|
+
ctx.currentCells = [];
|
|
652
|
+
ctx.rowIndex++;
|
|
653
|
+
}
|
|
654
|
+
};
|
|
655
|
+
// Helper to flush table
|
|
656
|
+
const flushTable = () => {
|
|
657
|
+
if (isFlushingTable)
|
|
658
|
+
return;
|
|
659
|
+
isFlushingTable = true;
|
|
660
|
+
const ctx = getCurrentTable();
|
|
661
|
+
if (!ctx) {
|
|
662
|
+
isFlushingTable = false;
|
|
663
|
+
return;
|
|
664
|
+
}
|
|
665
|
+
flushRow(ctx);
|
|
666
|
+
if (ctx.rows.length > 0) {
|
|
667
|
+
tableId++;
|
|
668
|
+
const tableNode = {
|
|
669
|
+
type: 'table',
|
|
670
|
+
text: ctx.rows.map(r => r.text).join('\n'), // Aggregate text from rows
|
|
671
|
+
children: [...ctx.rows]
|
|
672
|
+
};
|
|
673
|
+
// If we have a parent table, add this table to the parent's current cell
|
|
674
|
+
if (tableStack.length > 1) {
|
|
675
|
+
const parentCtx = tableStack[tableStack.length - 2];
|
|
676
|
+
parentCtx.currentCellContent.push(tableNode);
|
|
677
|
+
}
|
|
678
|
+
else {
|
|
679
|
+
currentTarget.push(tableNode);
|
|
680
|
+
}
|
|
681
|
+
}
|
|
682
|
+
// Pop the table from stack
|
|
683
|
+
tableStack.pop();
|
|
684
|
+
// If stack is empty, we are out of table mode
|
|
685
|
+
if (tableStack.length === 0) {
|
|
686
|
+
inTable = false;
|
|
687
|
+
paragraphInTable = false;
|
|
688
|
+
}
|
|
689
|
+
isFlushingTable = false;
|
|
690
|
+
};
|
|
691
|
+
// Extract hyperlink URL from field instruction group
|
|
692
|
+
// Recursively searches for HYPERLINK "url" pattern in nested groups
|
|
693
|
+
const extractHyperlinkUrl = (group) => {
|
|
694
|
+
let url;
|
|
695
|
+
// Helper to recursively find hyperlink URL
|
|
696
|
+
const findUrl = (node) => {
|
|
697
|
+
if (node.type === 'text') {
|
|
698
|
+
// Check for HYPERLINK "url" pattern
|
|
699
|
+
const text = node.value;
|
|
700
|
+
const match = text.match(/HYPERLINK\s+"([^"]+)"/i);
|
|
701
|
+
if (match) {
|
|
702
|
+
return match[1];
|
|
703
|
+
}
|
|
704
|
+
// Also check for URL after HYPERLINK on same or separate text node
|
|
705
|
+
const urlOnlyMatch = text.match(/"(https?:\/\/[^"]+|mailto:[^"]+|#[^"]+)"/);
|
|
706
|
+
if (urlOnlyMatch) {
|
|
707
|
+
return urlOnlyMatch[1];
|
|
603
708
|
}
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
const urlMatch = child.value.match(/"([^"]+)"/);
|
|
616
|
-
if (urlMatch && foundHyperlink) {
|
|
617
|
-
foundUrl = urlMatch[1];
|
|
618
|
-
}
|
|
709
|
+
}
|
|
710
|
+
else if (node.type === 'group') {
|
|
711
|
+
// Check if this is a fldinst group (contains field instruction)
|
|
712
|
+
let foundHyperlink = false;
|
|
713
|
+
let foundUrl;
|
|
714
|
+
let isLocal = false;
|
|
715
|
+
for (const child of node.content) {
|
|
716
|
+
if (child.type === 'text') {
|
|
717
|
+
const text = child.value.toUpperCase();
|
|
718
|
+
if (text.includes('HYPERLINK')) {
|
|
719
|
+
foundHyperlink = true;
|
|
619
720
|
}
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
721
|
+
if (text.includes('\\L')) {
|
|
722
|
+
isLocal = true;
|
|
723
|
+
}
|
|
724
|
+
// Look for quoted URL/Anchor
|
|
725
|
+
const urlMatch = child.value.match(/"([^"]+)"/);
|
|
726
|
+
if (urlMatch && foundHyperlink) {
|
|
727
|
+
foundUrl = urlMatch[1];
|
|
728
|
+
}
|
|
729
|
+
}
|
|
730
|
+
else if (child.type === 'group') {
|
|
731
|
+
const nestedUrl = findUrl(child);
|
|
732
|
+
if (nestedUrl) {
|
|
733
|
+
foundUrl = nestedUrl;
|
|
625
734
|
}
|
|
626
735
|
}
|
|
627
|
-
if (foundUrl)
|
|
628
|
-
return foundUrl;
|
|
629
736
|
}
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
const foundUrl = findUrl(child);
|
|
636
|
-
if (foundUrl)
|
|
637
|
-
return foundUrl;
|
|
737
|
+
if (foundUrl) {
|
|
738
|
+
if (isLocal && !foundUrl.startsWith('#')) {
|
|
739
|
+
return '#' + foundUrl;
|
|
740
|
+
}
|
|
741
|
+
return foundUrl;
|
|
638
742
|
}
|
|
639
743
|
}
|
|
640
|
-
return
|
|
744
|
+
return undefined;
|
|
641
745
|
};
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
746
|
+
// Search the field group for fldinst
|
|
747
|
+
for (const child of group.content) {
|
|
748
|
+
if (child.type === 'group') {
|
|
749
|
+
const foundUrl = findUrl(child);
|
|
750
|
+
if (foundUrl)
|
|
751
|
+
return foundUrl;
|
|
752
|
+
}
|
|
753
|
+
}
|
|
754
|
+
return url;
|
|
755
|
+
};
|
|
756
|
+
/**
|
|
757
|
+
* Determines whether a hyperlink URL is internal or external.
|
|
758
|
+
* Internal = bookmark references (no scheme or starts with "#").
|
|
759
|
+
* External = any scheme like http, https, mailto, ftp, file, etc.
|
|
760
|
+
*/
|
|
761
|
+
const classifyLinkType = (url) => {
|
|
762
|
+
// Trim whitespace
|
|
763
|
+
const clean = url.trim();
|
|
764
|
+
// Internal pattern 1: starts with "#"
|
|
765
|
+
if (clean.startsWith('#')) {
|
|
766
|
+
return 'internal';
|
|
767
|
+
}
|
|
768
|
+
// Internal pattern 2: no scheme at all (pure bookmark)
|
|
769
|
+
// Detect schemes by checking "something:" prefix
|
|
770
|
+
if (!/^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(clean)) {
|
|
771
|
+
return 'internal';
|
|
772
|
+
}
|
|
773
|
+
// Everything else is external
|
|
774
|
+
return 'external';
|
|
775
|
+
};
|
|
776
|
+
// Helper to extract text content from a group (for list marker detection)
|
|
777
|
+
// Recursively collects all text content from a group
|
|
778
|
+
const extractTextFromGroup = (group) => {
|
|
779
|
+
let text = '';
|
|
780
|
+
const collectText = (node) => {
|
|
781
|
+
if (node.type === 'text') {
|
|
782
|
+
text += node.value;
|
|
783
|
+
}
|
|
784
|
+
else if (node.type === 'group') {
|
|
785
|
+
for (const child of node.content) {
|
|
786
|
+
collectText(child);
|
|
674
787
|
}
|
|
675
|
-
};
|
|
676
|
-
for (const child of group.content) {
|
|
677
|
-
collectText(child);
|
|
678
788
|
}
|
|
679
|
-
return text;
|
|
680
789
|
};
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
}
|
|
709
|
-
else if (child.type === 'group') {
|
|
710
|
-
// Skip nested structures like picprop or blipuid
|
|
790
|
+
for (const child of group.content) {
|
|
791
|
+
collectText(child);
|
|
792
|
+
}
|
|
793
|
+
return text;
|
|
794
|
+
};
|
|
795
|
+
/**
|
|
796
|
+
* Extracts an image attachment from an RTF \pict group.
|
|
797
|
+
* Uses lookup tables for clean and safe handling of all formats.
|
|
798
|
+
*
|
|
799
|
+
* @param pictGroup The RtfGroup node that represents a \pict group.
|
|
800
|
+
* @returns An OfficeAttachment or undefined when unsupported or invalid.
|
|
801
|
+
*/
|
|
802
|
+
const extractPictAttachment = (pictGroup) => {
|
|
803
|
+
/** Internal format detected from the RTF pict group */
|
|
804
|
+
let imageFormat;
|
|
805
|
+
/** Hexadecimal string chunks extracted from the pict binary section */
|
|
806
|
+
const hexDataChunks = [];
|
|
807
|
+
// -------------------------------------------------------------
|
|
808
|
+
// Walk through pict group content to detect the blip type and gather hex data
|
|
809
|
+
// -------------------------------------------------------------
|
|
810
|
+
for (const child of pictGroup.content) {
|
|
811
|
+
// If the node is a control word, we try to resolve it from lookup map
|
|
812
|
+
if (child.type === 'control') {
|
|
813
|
+
// Lookup directly instead of if/else
|
|
814
|
+
const mapped = RTF_BLIP_MAP[child.value];
|
|
815
|
+
if (mapped) {
|
|
816
|
+
imageFormat = mapped;
|
|
711
817
|
}
|
|
712
818
|
}
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
819
|
+
else if (child.type === 'text') {
|
|
820
|
+
// Append only valid hex characters
|
|
821
|
+
hexDataChunks.push(child.value.replace(/[^0-9a-fA-F]/g, ''));
|
|
716
822
|
}
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
if (!mimeType) {
|
|
720
|
-
return undefined;
|
|
823
|
+
else if (child.type === 'group') {
|
|
824
|
+
// Skip nested structures like picprop or blipuid
|
|
721
825
|
}
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
826
|
+
}
|
|
827
|
+
// Missing or unknown format → stop
|
|
828
|
+
const hexData = hexDataChunks.join('');
|
|
829
|
+
if (!imageFormat || hexData.length === 0) {
|
|
830
|
+
return undefined;
|
|
831
|
+
}
|
|
832
|
+
// Map internal format to MIME
|
|
833
|
+
const mimeType = IMAGE_MIME_MAP[imageFormat];
|
|
834
|
+
if (!mimeType) {
|
|
835
|
+
return undefined;
|
|
836
|
+
}
|
|
837
|
+
// -------------------------------------------------------------
|
|
838
|
+
// Convert hex → binary and construct the attachment object
|
|
839
|
+
// -------------------------------------------------------------
|
|
840
|
+
try {
|
|
841
|
+
// Convert hex into a raw buffer
|
|
842
|
+
const buffer = Buffer.from(hexData, 'hex');
|
|
843
|
+
// Derive file extension directly from format
|
|
844
|
+
const extension = imageFormat;
|
|
845
|
+
// Generate a stable incremental filename
|
|
846
|
+
const name = `image_${attachments.length + 1}.${extension}`;
|
|
847
|
+
// Build and return final attachment
|
|
848
|
+
return {
|
|
849
|
+
type: 'image',
|
|
850
|
+
mimeType: mimeType,
|
|
851
|
+
data: buffer.toString('base64'),
|
|
852
|
+
name: name,
|
|
853
|
+
extension: extension
|
|
854
|
+
};
|
|
855
|
+
}
|
|
856
|
+
catch {
|
|
857
|
+
// If conversion fails, ignore this image
|
|
858
|
+
return undefined;
|
|
859
|
+
}
|
|
860
|
+
};
|
|
861
|
+
// Helper to serialize RTF control word
|
|
862
|
+
const serializeRtfControl = (node) => {
|
|
863
|
+
// Symbol control words (non-alpha)
|
|
864
|
+
if (!/^[a-zA-Z]/.test(node.value)) {
|
|
865
|
+
return `\\${node.value}`;
|
|
866
|
+
}
|
|
867
|
+
// Alpha control words
|
|
868
|
+
let res = `\\${node.value}`;
|
|
869
|
+
if (node.param !== undefined) {
|
|
870
|
+
res += node.param;
|
|
871
|
+
}
|
|
872
|
+
// Add space delimiter for safety
|
|
873
|
+
res += ' ';
|
|
874
|
+
return res;
|
|
875
|
+
};
|
|
876
|
+
// Helper to extract text from a group (for bookmark names, etc.)
|
|
877
|
+
const extractGroupText = (group) => {
|
|
878
|
+
let text = '';
|
|
879
|
+
for (const item of group.content) {
|
|
880
|
+
if (item.type === 'text') {
|
|
881
|
+
text += item.value;
|
|
740
882
|
}
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
return undefined;
|
|
883
|
+
else if (item.type === 'group') {
|
|
884
|
+
text += extractGroupText(item);
|
|
744
885
|
}
|
|
745
|
-
}
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
if (node.destination) {
|
|
801
|
-
if (node.destination === 'footnote') {
|
|
802
|
-
isFootnote = true;
|
|
803
|
-
}
|
|
804
|
-
else if (node.destination === 'field') {
|
|
805
|
-
// Check if this is a hyperlink field
|
|
806
|
-
const url = extractHyperlinkUrl(node);
|
|
807
|
-
if (url) {
|
|
808
|
-
// Flush any pending text before starting the link context
|
|
809
|
-
// This prevents previous text from inheriting the link
|
|
810
|
-
flushRun();
|
|
811
|
-
isHyperlinkField = true;
|
|
812
|
-
currentLinkUrl = url;
|
|
813
|
-
}
|
|
814
|
-
}
|
|
815
|
-
else if (node.destination === 'listtable') {
|
|
816
|
-
// We want to parse list definitions
|
|
817
|
-
parsingListTable = true;
|
|
886
|
+
}
|
|
887
|
+
return text.trim();
|
|
888
|
+
};
|
|
889
|
+
// Helper to serialize RTF text
|
|
890
|
+
const serializeRtfText = (node) => {
|
|
891
|
+
// Escape special characters: \, {, }
|
|
892
|
+
return node.value.replace(/([\\{}])/g, '\\$1');
|
|
893
|
+
};
|
|
894
|
+
// Recursive function to traverse the RTF tree
|
|
895
|
+
const traverse = (node, formatting, depth = 0) => {
|
|
896
|
+
if (node.type === 'group') {
|
|
897
|
+
const ignoreList = [
|
|
898
|
+
'fonttbl', 'colortbl', 'stylesheet', 'info', 'macpict',
|
|
899
|
+
'pmmetafile', 'wmetafile', 'dibitmap', 'bitmap', 'object',
|
|
900
|
+
'nextGenerator', 'header', 'footer', 'nonshppict', 'xml', 'private',
|
|
901
|
+
'upnp', 'ud', 'filetbl', 'operator', 'author', 'creatim', 'revtim', 'printim', 'comment',
|
|
902
|
+
'fldinst', 'listtext', 'pntext' // Ignore list marker text (handled separately)
|
|
903
|
+
];
|
|
904
|
+
let isIgnored = false;
|
|
905
|
+
let isFootnote = false;
|
|
906
|
+
let isHyperlinkField = false;
|
|
907
|
+
let isPict = false;
|
|
908
|
+
// Add group start to raw content
|
|
909
|
+
// Note: We don't add ignored groups to rawContent to keep it clean
|
|
910
|
+
// But we might want to if we want full fidelity.
|
|
911
|
+
// For now, let's include everything in rawContent except truly skipped stuff?
|
|
912
|
+
// The user asked for "raw content which is probably the rtf group".
|
|
913
|
+
// If we skip 'fonttbl', it's fine as it's not part of the content.
|
|
914
|
+
// But 'listtext' IS part of the content structure even if we parse it separately.
|
|
915
|
+
// Let's stick to the plan: if ignored, we might skip it in rawContent too,
|
|
916
|
+
// OR we include it.
|
|
917
|
+
// If I include it, `currentParagraphRaw` might get huge with font tables if they were inside the paragraph (unlikely).
|
|
918
|
+
// Usually font tables are at document root.
|
|
919
|
+
// `traverse` is called on `doc`.
|
|
920
|
+
// `currentParagraphRaw` is reset on `flushParagraph`.
|
|
921
|
+
// So if we are at root level, `currentParagraphRaw` accumulates everything until the first paragraph ends.
|
|
922
|
+
// This might include the header/fonttbl if they are before the first \par.
|
|
923
|
+
// That seems correct for "raw content" of the first node?
|
|
924
|
+
// Actually, `fonttbl` is usually before any text.
|
|
925
|
+
// If we include it, the first paragraph node will contain the entire font table in its rawContent.
|
|
926
|
+
// That might be annoying.
|
|
927
|
+
// Let's ONLY add to `currentParagraphRaw` if NOT ignored.
|
|
928
|
+
if (node.destination) {
|
|
929
|
+
if (node.destination === 'footnote') {
|
|
930
|
+
isFootnote = true;
|
|
931
|
+
}
|
|
932
|
+
else if (node.destination === 'field') {
|
|
933
|
+
// Check if this is a hyperlink field
|
|
934
|
+
const url = extractHyperlinkUrl(node);
|
|
935
|
+
if (url) {
|
|
936
|
+
// Flush any pending text before starting the link context
|
|
937
|
+
// This prevents previous text from inheriting the link
|
|
938
|
+
flushRun();
|
|
939
|
+
isHyperlinkField = true;
|
|
940
|
+
currentLinkUrl = url;
|
|
818
941
|
}
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
942
|
+
}
|
|
943
|
+
else if (node.destination === 'listtable') {
|
|
944
|
+
// We want to parse list definitions
|
|
945
|
+
parsingListTable = true;
|
|
946
|
+
}
|
|
947
|
+
else if (parsingListTable && node.destination === 'list') {
|
|
948
|
+
parsingListDefinition = true;
|
|
949
|
+
currentDefinedListId = undefined;
|
|
950
|
+
currentDefinedListType = undefined;
|
|
951
|
+
}
|
|
952
|
+
else if (node.destination === 'listoverridetable') {
|
|
953
|
+
parsingListOverrideTable = true;
|
|
954
|
+
}
|
|
955
|
+
else if (parsingListOverrideTable && node.destination === 'listoverride') {
|
|
956
|
+
// Reset per override group
|
|
957
|
+
currentListOverrideListId = undefined;
|
|
958
|
+
currentListOverrideLs = undefined;
|
|
959
|
+
}
|
|
960
|
+
else if (node.destination === 'pict') {
|
|
961
|
+
// Handle picture extraction
|
|
962
|
+
if (config.extractAttachments) {
|
|
963
|
+
isPict = true;
|
|
823
964
|
}
|
|
824
|
-
else
|
|
825
|
-
|
|
965
|
+
else {
|
|
966
|
+
isIgnored = true;
|
|
826
967
|
}
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
968
|
+
}
|
|
969
|
+
else if (ignoreList.includes(node.destination)) {
|
|
970
|
+
isIgnored = true;
|
|
971
|
+
}
|
|
972
|
+
else if (node.content.length > 0 && node.content[0].type === 'control' && node.content[0].value === '*') {
|
|
973
|
+
// Ignorable destination, but allow certain ones for:
|
|
974
|
+
// - fldinst: hyperlinks
|
|
975
|
+
// - nesttableprops: nested tables
|
|
976
|
+
// - shppict: shape pictures (contain pict groups)
|
|
977
|
+
const allowedIgnorable = ['fldinst', 'nesttableprops'];
|
|
978
|
+
if (config.extractAttachments) {
|
|
979
|
+
allowedIgnorable.push('shppict', 'listpicture');
|
|
831
980
|
}
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
981
|
+
if (!allowedIgnorable.includes(node.destination || '')) {
|
|
982
|
+
if (node.destination === 'bkmkstart') {
|
|
983
|
+
const name = extractGroupText(node);
|
|
984
|
+
if (name && !currentAnchorIds.includes(name)) {
|
|
985
|
+
currentAnchorIds.push(name);
|
|
986
|
+
}
|
|
838
987
|
isIgnored = true;
|
|
839
988
|
}
|
|
840
|
-
|
|
841
|
-
else if (ignoreList.includes(node.destination)) {
|
|
842
|
-
isIgnored = true;
|
|
843
|
-
}
|
|
844
|
-
else if (node.content.length > 0 && node.content[0].type === 'control' && node.content[0].value === '*') {
|
|
845
|
-
// Ignorable destination, but allow certain ones for:
|
|
846
|
-
// - fldinst: hyperlinks
|
|
847
|
-
// - nesttableprops: nested tables
|
|
848
|
-
// - shppict: shape pictures (contain pict groups)
|
|
849
|
-
const allowedIgnorable = ['fldinst', 'nesttableprops'];
|
|
850
|
-
if (config.extractAttachments) {
|
|
851
|
-
allowedIgnorable.push('shppict', 'listpicture');
|
|
852
|
-
}
|
|
853
|
-
if (!allowedIgnorable.includes(node.destination || '')) {
|
|
989
|
+
else if (node.destination === 'bkmkend') {
|
|
854
990
|
isIgnored = true;
|
|
855
991
|
}
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
// ═══════════════════════════════════════════════════════════
|
|
859
|
-
// Handle listtext and pntext: These indicate the current paragraph
|
|
860
|
-
// is a list item. We extract list type info before ignoring content.
|
|
861
|
-
// ═══════════════════════════════════════════════════════════
|
|
862
|
-
if (node.destination === 'listtext' || node.destination === 'pntext') {
|
|
863
|
-
// If this is the first list item, reset the indent
|
|
864
|
-
if (!isListItem)
|
|
865
|
-
paragraphIndent = 0;
|
|
866
|
-
isListItem = true;
|
|
867
|
-
// Try to determine list type from the marker content
|
|
868
|
-
// Bullets (unordered): '·', '•', 'o', '§', etc.
|
|
869
|
-
// Numbers (ordered): '1.', '2.', 'i.', 'ii.', 'a.', 'A.', etc.
|
|
870
|
-
const markerText = extractTextFromGroup(node);
|
|
871
|
-
if (markerText) {
|
|
872
|
-
const trimmed = markerText.trim();
|
|
873
|
-
// Check for common bullet characters
|
|
874
|
-
const bulletChars = ['·', '•', 'o', '§', '■', '□', '●', '○', '◆', '◇', '►', '▸', '\u00b7', '\u2022', '\u25cf', '\u25cb'];
|
|
875
|
-
const isBullet = bulletChars.some(b => trimmed.includes(b)) ||
|
|
876
|
-
// Font symbol bullets often use characters from Symbol font
|
|
877
|
-
(trimmed.length === 1 && !/[0-9a-zA-Z]/.test(trimmed));
|
|
878
|
-
if (isBullet) {
|
|
879
|
-
listType = 'unordered';
|
|
880
|
-
}
|
|
881
|
-
else if (/^[0-9ivxlcdm]+[\.\)]/i.test(trimmed) || /^[a-z][\.\)]/i.test(trimmed)) {
|
|
882
|
-
// Matches: 1., 2), i., ii., a., A), etc.
|
|
883
|
-
listType = 'ordered';
|
|
992
|
+
else {
|
|
993
|
+
isIgnored = true;
|
|
884
994
|
}
|
|
885
|
-
// If we can't determine, leave listType as is (might be set by \ls/\levelnfc)
|
|
886
995
|
}
|
|
887
|
-
// Still mark as ignored to skip the marker text content
|
|
888
|
-
isIgnored = true;
|
|
889
996
|
}
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
// We still traverse pict content to reconstruct raw RTF?
|
|
914
|
-
// No, extractPictAttachment consumes it.
|
|
915
|
-
// But we want it in rawContent?
|
|
916
|
-
// If we return here, we miss the closing '}'.
|
|
917
|
-
// And we miss the content in rawContent.
|
|
918
|
-
// Let's traverse it purely for rawContent if needed, but `extractPictAttachment` doesn't modify the tree.
|
|
919
|
-
// But `extractPictAttachment` does not return the raw string.
|
|
920
|
-
// So we should probably continue traversal but suppress text extraction?
|
|
921
|
-
// The original code returned here: `return; // Don't traverse pict content as text`
|
|
922
|
-
// So we should do the same, but we need to append the content to `currentParagraphRaw`.
|
|
923
|
-
// We can manually serialize the group content here.
|
|
924
|
-
for (const child of node.content) {
|
|
925
|
-
if (child.type === 'control')
|
|
926
|
-
currentParagraphRaw += serializeRtfControl(child);
|
|
927
|
-
else if (child.type === 'text')
|
|
928
|
-
currentParagraphRaw += serializeRtfText(child);
|
|
929
|
-
else if (child.type === 'group') {
|
|
930
|
-
// Recursive serialization for nested groups in pict (e.g. blipuid)
|
|
931
|
-
// We can't easily recurse `traverse` because it has side effects (text extraction).
|
|
932
|
-
// We need a pure serializer or just let `traverse` run but with a flag?
|
|
933
|
-
// Or just ignore the raw content of the image binary data?
|
|
934
|
-
// Image binary data can be huge.
|
|
935
|
-
// Maybe we shouldn't include the full hex dump in `rawContent`?
|
|
936
|
-
// The user said "raw content which is probably the rtf group".
|
|
937
|
-
// Including 5MB of hex data in the JSON AST might be bad.
|
|
938
|
-
// But for consistency, it is the raw content.
|
|
939
|
-
// Let's include it for now.
|
|
940
|
-
// To do this without side effects, we need a separate serialize function?
|
|
941
|
-
// Or just call traverse and ensure `isPict` logic prevents text extraction.
|
|
942
|
-
// Wait, `isPict` is true for this node.
|
|
943
|
-
// If we recurse, `isPict` will be false for children (unless they are also pict).
|
|
944
|
-
// But we want to suppress text extraction for children of pict.
|
|
945
|
-
// The original code did `return`.
|
|
946
|
-
// So we should manually serialize children here.
|
|
947
|
-
// Let's define a simple recursive serializer.
|
|
948
|
-
const serializeGroupContent = (g) => {
|
|
949
|
-
for (const c of g.content) {
|
|
950
|
-
if (c.type === 'control')
|
|
951
|
-
currentParagraphRaw += serializeRtfControl(c);
|
|
952
|
-
else if (c.type === 'text')
|
|
953
|
-
currentParagraphRaw += serializeRtfText(c);
|
|
954
|
-
else if (c.type === 'group') {
|
|
955
|
-
currentParagraphRaw += '{';
|
|
956
|
-
serializeGroupContent(c);
|
|
957
|
-
currentParagraphRaw += '}';
|
|
958
|
-
}
|
|
959
|
-
}
|
|
960
|
-
};
|
|
961
|
-
serializeGroupContent(node);
|
|
962
|
-
}
|
|
963
|
-
}
|
|
964
|
-
currentParagraphRaw += '}';
|
|
965
|
-
return;
|
|
966
|
-
}
|
|
967
|
-
// Handle footnote: switch target to notes
|
|
968
|
-
const previousTarget = currentTarget;
|
|
969
|
-
if (isFootnote) {
|
|
970
|
-
if (config.ignoreNotes) {
|
|
971
|
-
return; // Skip footnote content entirely
|
|
997
|
+
}
|
|
998
|
+
// ═══════════════════════════════════════════════════════════
|
|
999
|
+
// Handle listtext and pntext: These indicate the current paragraph
|
|
1000
|
+
// is a list item. We extract list type info before ignoring content.
|
|
1001
|
+
// ═══════════════════════════════════════════════════════════
|
|
1002
|
+
if (node.destination === 'listtext' || node.destination === 'pntext') {
|
|
1003
|
+
// If this is the first list item, reset the indent
|
|
1004
|
+
if (!isListItem)
|
|
1005
|
+
paragraphIndent = 0;
|
|
1006
|
+
isListItem = true;
|
|
1007
|
+
// Try to determine list type from the marker content
|
|
1008
|
+
// Bullets (unordered): '·', '•', 'o', '§', etc.
|
|
1009
|
+
// Numbers (ordered): '1.', '2.', 'i.', 'ii.', 'a.', 'A.', etc.
|
|
1010
|
+
const markerText = extractTextFromGroup(node);
|
|
1011
|
+
if (markerText) {
|
|
1012
|
+
const trimmed = markerText.trim();
|
|
1013
|
+
// Check for common bullet characters
|
|
1014
|
+
const bulletChars = ['·', '•', 'o', '§', '■', '□', '●', '○', '◆', '◇', '►', '▸', '\u00b7', '\u2022', '\u25cf', '\u25cb'];
|
|
1015
|
+
const isBullet = bulletChars.some(b => trimmed.includes(b)) ||
|
|
1016
|
+
// Font symbol bullets often use characters from Symbol font
|
|
1017
|
+
(trimmed.length === 1 && !/[0-9a-zA-Z]/.test(trimmed));
|
|
1018
|
+
if (isBullet) {
|
|
1019
|
+
listType = 'unordered';
|
|
972
1020
|
}
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
let noteType = 'footnote';
|
|
977
|
-
if (fetValue === 1) {
|
|
978
|
-
// \fet1 means all notes are endnotes
|
|
979
|
-
noteType = 'endnote';
|
|
1021
|
+
else if (/^[0-9ivxlcdm]+[\.\)]/i.test(trimmed) || /^[a-z][\.\)]/i.test(trimmed)) {
|
|
1022
|
+
// Matches: 1., 2), i., ii., a., A), etc.
|
|
1023
|
+
listType = 'ordered';
|
|
980
1024
|
}
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
1025
|
+
// If we can't determine, leave listType as is (might be set by \ls/\levelnfc)
|
|
1026
|
+
}
|
|
1027
|
+
// Still mark as ignored to skip the marker text content
|
|
1028
|
+
isIgnored = true;
|
|
1029
|
+
}
|
|
1030
|
+
if (isIgnored)
|
|
1031
|
+
return;
|
|
1032
|
+
// Append group start to raw content
|
|
1033
|
+
currentParagraphRawChunks.push('{');
|
|
1034
|
+
// Handle pict group: extract image and add to content tree
|
|
1035
|
+
if (isPict) {
|
|
1036
|
+
const attachment = extractPictAttachment(node);
|
|
1037
|
+
if (attachment) {
|
|
1038
|
+
attachments.push(attachment);
|
|
1039
|
+
// Only add image node to content if this is NOT a list definition picture
|
|
1040
|
+
// List pictures (bullets) should not appear in content, only as attachments
|
|
1041
|
+
if (!parsingListTable && !parsingListDefinition) {
|
|
1042
|
+
// Also add an image node to the content tree (like DOCX)
|
|
1043
|
+
flushParagraph();
|
|
1044
|
+
currentTarget.push({
|
|
1045
|
+
type: 'image',
|
|
1046
|
+
text: '',
|
|
1047
|
+
metadata: {
|
|
1048
|
+
attachmentName: attachment.name || `image_${attachments.length}`
|
|
1049
|
+
}
|
|
1050
|
+
});
|
|
987
1051
|
}
|
|
988
|
-
// fetValue === 0 (default) means footnotes only
|
|
989
|
-
const noteNode = {
|
|
990
|
-
type: 'note',
|
|
991
|
-
children: [],
|
|
992
|
-
metadata: {
|
|
993
|
-
noteId: currentFootnoteId.toString(),
|
|
994
|
-
noteType: noteType
|
|
995
|
-
}
|
|
996
|
-
};
|
|
997
|
-
notes.push(noteNode);
|
|
998
|
-
currentTarget = noteNode.children;
|
|
999
1052
|
}
|
|
1000
|
-
//
|
|
1001
|
-
|
|
1053
|
+
// We still traverse pict content to reconstruct raw RTF?
|
|
1054
|
+
// No, extractPictAttachment consumes it.
|
|
1055
|
+
// But we want it in rawContent?
|
|
1056
|
+
// If we return here, we miss the closing '}'.
|
|
1057
|
+
// And we miss the content in rawContent.
|
|
1058
|
+
// Let's traverse it purely for rawContent if needed, but `extractPictAttachment` doesn't modify the tree.
|
|
1059
|
+
// But `extractPictAttachment` does not return the raw string.
|
|
1060
|
+
// So we should probably continue traversal but suppress text extraction?
|
|
1061
|
+
// The original code returned here: `return; // Don't traverse pict content as text`
|
|
1062
|
+
// So we should do the same, but we need to append the content to `currentParagraphRaw`.
|
|
1063
|
+
// We can manually serialize the group content here.
|
|
1002
1064
|
for (const child of node.content) {
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
//
|
|
1009
|
-
//
|
|
1010
|
-
//
|
|
1011
|
-
//
|
|
1012
|
-
//
|
|
1013
|
-
//
|
|
1014
|
-
//
|
|
1015
|
-
|
|
1016
|
-
//
|
|
1065
|
+
if (child.type === 'control')
|
|
1066
|
+
currentParagraphRawChunks.push(serializeRtfControl(child));
|
|
1067
|
+
else if (child.type === 'text')
|
|
1068
|
+
currentParagraphRawChunks.push(serializeRtfText(child));
|
|
1069
|
+
else if (child.type === 'group') {
|
|
1070
|
+
// Recursive serialization for nested groups in pict (e.g. blipuid)
|
|
1071
|
+
// We can't easily recurse `traverse` because it has side effects (text extraction).
|
|
1072
|
+
// We need a pure serializer or just let `traverse` run but with a flag?
|
|
1073
|
+
// Or just ignore the raw content of the image binary data?
|
|
1074
|
+
// Image binary data can be huge.
|
|
1075
|
+
// Maybe we shouldn't include the full hex dump in `rawContent`?
|
|
1076
|
+
// The user said "raw content which is probably the rtf group".
|
|
1077
|
+
// Including 5MB of hex data in the JSON AST might be bad.
|
|
1078
|
+
// But for consistency, it is the raw content.
|
|
1079
|
+
// Let's include it for now.
|
|
1080
|
+
// To do this without side effects, we need a separate serialize function?
|
|
1081
|
+
// Or just call traverse and ensure `isPict` logic prevents text extraction.
|
|
1082
|
+
// Wait, `isPict` is true for this node.
|
|
1083
|
+
// If we recurse, `isPict` will be false for children (unless they are also pict).
|
|
1084
|
+
// But we want to suppress text extraction for children of pict.
|
|
1085
|
+
// The original code did `return`.
|
|
1086
|
+
// So we should manually serialize children here.
|
|
1087
|
+
// Let's define a simple recursive serializer.
|
|
1017
1088
|
const serializeGroupContent = (g) => {
|
|
1018
1089
|
for (const c of g.content) {
|
|
1019
1090
|
if (c.type === 'control')
|
|
1020
|
-
|
|
1091
|
+
currentParagraphRawChunks.push(serializeRtfControl(c));
|
|
1021
1092
|
else if (c.type === 'text')
|
|
1022
|
-
|
|
1093
|
+
currentParagraphRawChunks.push(serializeRtfText(c));
|
|
1023
1094
|
else if (c.type === 'group') {
|
|
1024
|
-
|
|
1095
|
+
currentParagraphRawChunks.push('{');
|
|
1025
1096
|
serializeGroupContent(c);
|
|
1026
|
-
|
|
1097
|
+
currentParagraphRawChunks.push('}');
|
|
1027
1098
|
}
|
|
1028
1099
|
}
|
|
1029
1100
|
};
|
|
1030
|
-
serializeGroupContent(
|
|
1031
|
-
currentParagraphRaw += '}';
|
|
1032
|
-
continue;
|
|
1101
|
+
serializeGroupContent(node);
|
|
1033
1102
|
}
|
|
1034
|
-
traverse(child, groupFormatting, depth + 1);
|
|
1035
|
-
}
|
|
1036
|
-
if (node.destination === 'listtable') {
|
|
1037
|
-
parsingListTable = false;
|
|
1038
1103
|
}
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1104
|
+
currentParagraphRawChunks.push('}');
|
|
1105
|
+
return;
|
|
1106
|
+
}
|
|
1107
|
+
// Handle footnote: switch target to notes
|
|
1108
|
+
const previousTarget = currentTarget;
|
|
1109
|
+
if (isFootnote) {
|
|
1110
|
+
if (config.ignoreNotes) {
|
|
1111
|
+
return; // Skip footnote content entirely
|
|
1112
|
+
}
|
|
1113
|
+
flushParagraph();
|
|
1114
|
+
currentFootnoteId++;
|
|
1115
|
+
// Determine note type based on \fet value
|
|
1116
|
+
let noteType = 'footnote';
|
|
1117
|
+
if (fetValue === 1) {
|
|
1118
|
+
// \fet1 means all notes are endnotes
|
|
1119
|
+
noteType = 'endnote';
|
|
1120
|
+
}
|
|
1121
|
+
else if (fetValue === 2) {
|
|
1122
|
+
// \fet2 means both types exist
|
|
1123
|
+
// Check for \ftnalt marker to distinguish endnotes from footnotes
|
|
1124
|
+
// \footnote\ftnalt indicates an endnote
|
|
1125
|
+
const hasFtnalt = node.content.some(child => child.type === 'control' && child.value === 'ftnalt');
|
|
1126
|
+
noteType = hasFtnalt ? 'endnote' : 'footnote';
|
|
1127
|
+
}
|
|
1128
|
+
// fetValue === 0 (default) means footnotes only
|
|
1129
|
+
const noteNode = {
|
|
1130
|
+
type: 'note',
|
|
1131
|
+
children: [],
|
|
1132
|
+
metadata: {
|
|
1133
|
+
noteId: currentFootnoteId.toString(),
|
|
1134
|
+
noteType: noteType
|
|
1043
1135
|
}
|
|
1136
|
+
};
|
|
1137
|
+
notes.push(noteNode);
|
|
1138
|
+
currentTarget = noteNode.children;
|
|
1139
|
+
}
|
|
1140
|
+
// Create a new formatting context for the group
|
|
1141
|
+
const groupFormatting = { ...formatting };
|
|
1142
|
+
for (const child of node.content) {
|
|
1143
|
+
// Skip fldinst groups (we already extracted the URL)
|
|
1144
|
+
if (child.type === 'group' && child.destination === 'fldinst') {
|
|
1145
|
+
// We still want it in rawContent!
|
|
1146
|
+
// So we should traverse it but suppress text extraction?
|
|
1147
|
+
// Or just serialize it?
|
|
1148
|
+
// `fldinst` contains the URL.
|
|
1149
|
+
// If we skip it in `traverse`, we miss it in `rawContent`.
|
|
1150
|
+
// Let's traverse it but maybe the `fldinst` logic inside `traverse` handles it?
|
|
1151
|
+
// The original code:
|
|
1152
|
+
// if (child.type === 'group' && child.destination === 'fldinst') { continue; }
|
|
1153
|
+
// This skips the child entirely.
|
|
1154
|
+
// So we need to manually serialize it if we want it in rawContent.
|
|
1155
|
+
currentParagraphRawChunks.push('{');
|
|
1156
|
+
// We need to serialize the content of fldinst
|
|
1157
|
+
const serializeGroupContent = (g) => {
|
|
1158
|
+
for (const c of g.content) {
|
|
1159
|
+
if (c.type === 'control')
|
|
1160
|
+
currentParagraphRawChunks.push(serializeRtfControl(c));
|
|
1161
|
+
else if (c.type === 'text')
|
|
1162
|
+
currentParagraphRawChunks.push(serializeRtfText(c));
|
|
1163
|
+
else if (c.type === 'group') {
|
|
1164
|
+
currentParagraphRawChunks.push('{');
|
|
1165
|
+
serializeGroupContent(c);
|
|
1166
|
+
currentParagraphRawChunks.push('}');
|
|
1167
|
+
}
|
|
1168
|
+
}
|
|
1169
|
+
};
|
|
1170
|
+
serializeGroupContent(child);
|
|
1171
|
+
currentParagraphRawChunks.push('}');
|
|
1172
|
+
continue;
|
|
1044
1173
|
}
|
|
1045
|
-
|
|
1046
|
-
|
|
1174
|
+
traverse(child, groupFormatting, depth + 1);
|
|
1175
|
+
}
|
|
1176
|
+
if (node.destination === 'listtable') {
|
|
1177
|
+
parsingListTable = false;
|
|
1178
|
+
}
|
|
1179
|
+
else if (node.destination === 'list') {
|
|
1180
|
+
parsingListDefinition = false;
|
|
1181
|
+
if (currentDefinedListId !== undefined && currentDefinedListType !== undefined) {
|
|
1182
|
+
listTypeMap[currentDefinedListId] = currentDefinedListType;
|
|
1047
1183
|
}
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
|
|
1184
|
+
}
|
|
1185
|
+
else if (node.destination === 'listoverridetable') {
|
|
1186
|
+
parsingListOverrideTable = false;
|
|
1187
|
+
}
|
|
1188
|
+
else if (parsingListOverrideTable && node.destination === 'listoverride') {
|
|
1189
|
+
// End of listoverride group - populate map
|
|
1190
|
+
if (currentListOverrideLs !== undefined && currentListOverrideListId !== undefined) {
|
|
1191
|
+
listOverrideMap[currentListOverrideLs] = currentListOverrideListId;
|
|
1192
|
+
}
|
|
1193
|
+
}
|
|
1194
|
+
if (isFootnote) {
|
|
1195
|
+
flushParagraph();
|
|
1196
|
+
currentTarget = previousTarget;
|
|
1197
|
+
}
|
|
1198
|
+
// Clear link URL after processing the field group
|
|
1199
|
+
if (isHyperlinkField) {
|
|
1200
|
+
flushRun();
|
|
1201
|
+
currentLinkUrl = undefined;
|
|
1202
|
+
}
|
|
1203
|
+
// Append group end to raw content
|
|
1204
|
+
currentParagraphRawChunks.push('}');
|
|
1205
|
+
}
|
|
1206
|
+
else if (node.type === 'text') {
|
|
1207
|
+
if (parsingListTable || parsingListOverrideTable) {
|
|
1208
|
+
// Even if we don't extract text, we might want it in rawContent?
|
|
1209
|
+
// Yes, rawContent should reflect the source.
|
|
1210
|
+
currentParagraphRawChunks.push(serializeRtfText(node));
|
|
1211
|
+
return;
|
|
1212
|
+
}
|
|
1213
|
+
if (formattingChanged(currentFormatting, formatting)) {
|
|
1214
|
+
flushRun();
|
|
1215
|
+
currentFormatting = { ...formatting };
|
|
1216
|
+
}
|
|
1217
|
+
currentRunTextChunks.push(node.value);
|
|
1218
|
+
currentParagraphRawChunks.push(serializeRtfText(node));
|
|
1219
|
+
}
|
|
1220
|
+
else if (node.type === 'control') {
|
|
1221
|
+
// Append control to raw content
|
|
1222
|
+
currentParagraphRawChunks.push(serializeRtfControl(node));
|
|
1223
|
+
// Handle list definition control words
|
|
1224
|
+
if (parsingListTable) {
|
|
1225
|
+
if (node.value === 'listid') {
|
|
1226
|
+
currentDefinedListId = node.param;
|
|
1227
|
+
}
|
|
1228
|
+
else if (node.value === 'levelnfc' || node.value === 'levelnfcn') {
|
|
1229
|
+
// 0 = Arabic, 1 = Upper Roman, 2 = Lower Roman, 3 = Upper Alpha, 4 = Lower Alpha -> Ordered
|
|
1230
|
+
// 23 = Bullet, 255 = None -> Unordered
|
|
1231
|
+
const isOrdered = node.param !== undefined && (node.param === 0 || (node.param >= 0 && node.param <= 4));
|
|
1232
|
+
// Only set if not already set (or prioritize ordered if mixed?)
|
|
1233
|
+
// We'll assume if any level is ordered, it's ordered.
|
|
1234
|
+
// Or if we haven't set it yet.
|
|
1235
|
+
if (!currentDefinedListType || (currentDefinedListType === 'unordered' && isOrdered)) {
|
|
1236
|
+
currentDefinedListType = isOrdered ? 'ordered' : 'unordered';
|
|
1052
1237
|
}
|
|
1053
1238
|
}
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
|
|
1239
|
+
return;
|
|
1240
|
+
}
|
|
1241
|
+
// Handle list override control words
|
|
1242
|
+
if (parsingListOverrideTable) {
|
|
1243
|
+
if (node.value === 'listid') {
|
|
1244
|
+
currentListOverrideListId = node.param;
|
|
1057
1245
|
}
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
flushRun();
|
|
1061
|
-
currentLinkUrl = undefined;
|
|
1246
|
+
else if (node.value === 'ls') {
|
|
1247
|
+
currentListOverrideLs = node.param;
|
|
1062
1248
|
}
|
|
1063
|
-
|
|
1064
|
-
currentParagraphRaw += '}';
|
|
1249
|
+
return;
|
|
1065
1250
|
}
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1251
|
+
// Paragraph control words
|
|
1252
|
+
if (node.value === 'par') {
|
|
1253
|
+
flushParagraph();
|
|
1254
|
+
currentFormatting = { ...formatting };
|
|
1255
|
+
}
|
|
1256
|
+
// Table control words
|
|
1257
|
+
else if (node.value === 'trowd') {
|
|
1258
|
+
// Table row definition - start of a new row
|
|
1259
|
+
// Check if we are starting a nested table
|
|
1260
|
+
// If we are already in a table, and we have content in the current cell,
|
|
1261
|
+
// then this trowd implies a nested table start.
|
|
1262
|
+
const ctx = getCurrentTable();
|
|
1263
|
+
if (inTable && ctx && ctx.currentCellContent.length > 0) {
|
|
1264
|
+
// Start nested table
|
|
1265
|
+
ensureTableContext(); // Should already exist if inTable is true
|
|
1266
|
+
// Push new table context
|
|
1267
|
+
tableStack.push({
|
|
1268
|
+
rows: [],
|
|
1269
|
+
currentCells: [],
|
|
1270
|
+
currentCellContent: [],
|
|
1271
|
+
rowIndex: 0
|
|
1272
|
+
});
|
|
1072
1273
|
}
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1274
|
+
else {
|
|
1275
|
+
if (!inTable) {
|
|
1276
|
+
inTable = true;
|
|
1277
|
+
ensureTableContext();
|
|
1278
|
+
}
|
|
1076
1279
|
}
|
|
1077
|
-
|
|
1078
|
-
|
|
1280
|
+
// After \trowd we are inside a table row, so content should go to table cells.
|
|
1281
|
+
// Many RTF files don't use \intbl, relying solely on \trowd...\cell...\row structure.
|
|
1282
|
+
paragraphInTable = true;
|
|
1283
|
+
// Reset cell properties for the new row definition
|
|
1284
|
+
rowCellProps = [];
|
|
1285
|
+
currentCellDefinitionProps = { isMergedContinuation: false };
|
|
1286
|
+
cellContentIndex = 0;
|
|
1079
1287
|
}
|
|
1080
|
-
else if (node.
|
|
1081
|
-
//
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1288
|
+
else if (node.value === 'clvmrg') {
|
|
1289
|
+
// Vertical merge continuation
|
|
1290
|
+
currentCellDefinitionProps.isMergedContinuation = true;
|
|
1291
|
+
}
|
|
1292
|
+
else if (node.value === 'clmgf') {
|
|
1293
|
+
// Vertical merge first cell (reset continuation flag if set, though usually mutually exclusive)
|
|
1294
|
+
currentCellDefinitionProps.isMergedContinuation = false;
|
|
1295
|
+
}
|
|
1296
|
+
else if (node.value === 'cellx') {
|
|
1297
|
+
// End of cell definition
|
|
1298
|
+
rowCellProps.push({ ...currentCellDefinitionProps });
|
|
1299
|
+
// Reset for next cell
|
|
1300
|
+
currentCellDefinitionProps = { isMergedContinuation: false };
|
|
1301
|
+
}
|
|
1302
|
+
else if (node.value === 'cell') {
|
|
1303
|
+
// End of cell - add it to current row
|
|
1304
|
+
// Force paragraphInTable = true because \cell implies we are in a table cell
|
|
1305
|
+
paragraphInTable = true;
|
|
1306
|
+
// Check if this cell is a merged continuation
|
|
1307
|
+
let isMergedContinuation = false;
|
|
1308
|
+
if (cellContentIndex < rowCellProps.length) {
|
|
1309
|
+
isMergedContinuation = rowCellProps[cellContentIndex].isMergedContinuation;
|
|
1310
|
+
}
|
|
1311
|
+
cellContentIndex++;
|
|
1312
|
+
const cell = flushCell();
|
|
1313
|
+
// Only add if not a merged continuation
|
|
1314
|
+
if (cell) {
|
|
1315
|
+
if (!isMergedContinuation) {
|
|
1316
|
+
const ctx = getCurrentTable();
|
|
1317
|
+
if (ctx)
|
|
1318
|
+
ctx.currentCells.push(cell);
|
|
1098
1319
|
}
|
|
1099
|
-
return;
|
|
1100
1320
|
}
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1321
|
+
currentFormatting = { ...formatting };
|
|
1322
|
+
}
|
|
1323
|
+
else if (node.value === 'nestcell') {
|
|
1324
|
+
// End of cell in outer table (nested context)
|
|
1325
|
+
// If we are in an inner table, we need to close it and return to outer
|
|
1326
|
+
// First, flush the current cell of the inner table (if any pending)
|
|
1327
|
+
// Actually, nestcell ends the OUTER cell.
|
|
1328
|
+
// So the inner table should have been finished by now?
|
|
1329
|
+
// Usually inner table ends with \row.
|
|
1330
|
+
// If we are in a nested table (stack > 1), we should pop until we are at the outer table?
|
|
1331
|
+
// Or maybe just pop one level?
|
|
1332
|
+
if (tableStack.length > 1) {
|
|
1333
|
+
// Flush the inner table if it has pending rows
|
|
1334
|
+
const innerCtx = getCurrentTable();
|
|
1335
|
+
if (innerCtx && (innerCtx.rows.length > 0 || innerCtx.currentCells.length > 0)) {
|
|
1336
|
+
flushTable(); // This pops the stack
|
|
1108
1337
|
}
|
|
1109
|
-
return;
|
|
1110
|
-
}
|
|
1111
|
-
// Paragraph control words
|
|
1112
|
-
if (node.value === 'par') {
|
|
1113
|
-
flushParagraph();
|
|
1114
|
-
currentFormatting = { ...formatting };
|
|
1115
1338
|
}
|
|
1116
|
-
//
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
// then this trowd implies a nested table start.
|
|
1339
|
+
// Now we are (hopefully) at the outer table level
|
|
1340
|
+
// Treat as a regular cell end for the outer table
|
|
1341
|
+
paragraphInTable = true;
|
|
1342
|
+
const cell = flushCell();
|
|
1343
|
+
if (cell) {
|
|
1122
1344
|
const ctx = getCurrentTable();
|
|
1123
|
-
if (
|
|
1124
|
-
|
|
1125
|
-
ensureTableContext(); // Should already exist if inTable is true
|
|
1126
|
-
// Push new table context
|
|
1127
|
-
tableStack.push({
|
|
1128
|
-
rows: [],
|
|
1129
|
-
currentCells: [],
|
|
1130
|
-
currentCellContent: [],
|
|
1131
|
-
rowIndex: 0
|
|
1132
|
-
});
|
|
1133
|
-
}
|
|
1134
|
-
else {
|
|
1135
|
-
if (!inTable) {
|
|
1136
|
-
inTable = true;
|
|
1137
|
-
ensureTableContext();
|
|
1138
|
-
}
|
|
1139
|
-
}
|
|
1140
|
-
// After \trowd we are inside a table row, so content should go to table cells.
|
|
1141
|
-
// Many RTF files don't use \intbl, relying solely on \trowd...\cell...\row structure.
|
|
1142
|
-
paragraphInTable = true;
|
|
1143
|
-
// Reset cell properties for the new row definition
|
|
1144
|
-
rowCellProps = [];
|
|
1145
|
-
currentCellDefinitionProps = { isMergedContinuation: false };
|
|
1146
|
-
cellContentIndex = 0;
|
|
1147
|
-
}
|
|
1148
|
-
else if (node.value === 'clvmrg') {
|
|
1149
|
-
// Vertical merge continuation
|
|
1150
|
-
currentCellDefinitionProps.isMergedContinuation = true;
|
|
1151
|
-
}
|
|
1152
|
-
else if (node.value === 'clmgf') {
|
|
1153
|
-
// Vertical merge first cell (reset continuation flag if set, though usually mutually exclusive)
|
|
1154
|
-
currentCellDefinitionProps.isMergedContinuation = false;
|
|
1155
|
-
}
|
|
1156
|
-
else if (node.value === 'cellx') {
|
|
1157
|
-
// End of cell definition
|
|
1158
|
-
rowCellProps.push({ ...currentCellDefinitionProps });
|
|
1159
|
-
// Reset for next cell
|
|
1160
|
-
currentCellDefinitionProps = { isMergedContinuation: false };
|
|
1161
|
-
}
|
|
1162
|
-
else if (node.value === 'cell') {
|
|
1163
|
-
// End of cell - add it to current row
|
|
1164
|
-
// Force paragraphInTable = true because \cell implies we are in a table cell
|
|
1165
|
-
paragraphInTable = true;
|
|
1166
|
-
// Check if this cell is a merged continuation
|
|
1167
|
-
let isMergedContinuation = false;
|
|
1168
|
-
if (cellContentIndex < rowCellProps.length) {
|
|
1169
|
-
isMergedContinuation = rowCellProps[cellContentIndex].isMergedContinuation;
|
|
1170
|
-
}
|
|
1171
|
-
cellContentIndex++;
|
|
1172
|
-
const cell = flushCell();
|
|
1173
|
-
// Only add if not a merged continuation
|
|
1174
|
-
if (cell) {
|
|
1175
|
-
if (!isMergedContinuation) {
|
|
1176
|
-
const ctx = getCurrentTable();
|
|
1177
|
-
if (ctx)
|
|
1178
|
-
ctx.currentCells.push(cell);
|
|
1179
|
-
}
|
|
1180
|
-
}
|
|
1181
|
-
currentFormatting = { ...formatting };
|
|
1345
|
+
if (ctx)
|
|
1346
|
+
ctx.currentCells.push(cell);
|
|
1182
1347
|
}
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
paragraphInTable = true;
|
|
1202
|
-
const cell = flushCell();
|
|
1203
|
-
if (cell) {
|
|
1204
|
-
const ctx = getCurrentTable();
|
|
1205
|
-
if (ctx)
|
|
1206
|
-
ctx.currentCells.push(cell);
|
|
1207
|
-
}
|
|
1208
|
-
currentFormatting = { ...formatting };
|
|
1348
|
+
currentFormatting = { ...formatting };
|
|
1349
|
+
}
|
|
1350
|
+
else if (node.value === 'row') {
|
|
1351
|
+
// End of row
|
|
1352
|
+
flushRow();
|
|
1353
|
+
currentFormatting = { ...formatting };
|
|
1354
|
+
// Reset content index for safety (though trowd usually does it)
|
|
1355
|
+
cellContentIndex = 0;
|
|
1356
|
+
// Critical: Reset paragraphInTable after row ends.
|
|
1357
|
+
// Subsequent paragraphs must explicitly use \intbl to be part of the table.
|
|
1358
|
+
// Without this, content after the last \row gets incorrectly merged.
|
|
1359
|
+
paragraphInTable = false;
|
|
1360
|
+
}
|
|
1361
|
+
else if (node.value === 'nestrow') {
|
|
1362
|
+
// End of row in outer table
|
|
1363
|
+
// If we are still in inner table context, flush it
|
|
1364
|
+
if (tableStack.length > 1) {
|
|
1365
|
+
flushTable();
|
|
1209
1366
|
}
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1367
|
+
flushRow();
|
|
1368
|
+
currentFormatting = { ...formatting };
|
|
1369
|
+
cellContentIndex = 0;
|
|
1370
|
+
}
|
|
1371
|
+
else if (node.value === 'intbl') {
|
|
1372
|
+
// Paragraph is in a table
|
|
1373
|
+
inTable = true;
|
|
1374
|
+
paragraphInTable = true;
|
|
1375
|
+
ensureTableContext();
|
|
1376
|
+
}
|
|
1377
|
+
else if (node.value === 'pard') {
|
|
1378
|
+
// Reset paragraph properties
|
|
1379
|
+
paragraphInTable = false;
|
|
1380
|
+
// Reset other props...
|
|
1381
|
+
paragraphIndent = 0;
|
|
1382
|
+
paragraphAlignment = 'left';
|
|
1383
|
+
isListItem = false;
|
|
1384
|
+
listType = undefined;
|
|
1385
|
+
headingLevel = undefined;
|
|
1386
|
+
currentListId = undefined;
|
|
1387
|
+
// Reset paragraph-level background (cbpat) to prevent leaking to next paragraph
|
|
1388
|
+
formatting.backgroundColor = undefined;
|
|
1389
|
+
}
|
|
1390
|
+
// Text flow control
|
|
1391
|
+
else if (node.value === 'tab') {
|
|
1392
|
+
if (formattingChanged(currentFormatting, formatting)) {
|
|
1393
|
+
flushRun();
|
|
1213
1394
|
currentFormatting = { ...formatting };
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
}
|
|
1221
|
-
else if (node.value === 'nestrow') {
|
|
1222
|
-
// End of row in outer table
|
|
1223
|
-
// If we are still in inner table context, flush it
|
|
1224
|
-
if (tableStack.length > 1) {
|
|
1225
|
-
flushTable();
|
|
1226
|
-
}
|
|
1227
|
-
flushRow();
|
|
1395
|
+
}
|
|
1396
|
+
currentRunTextChunks.push('\t');
|
|
1397
|
+
}
|
|
1398
|
+
else if (node.value === 'line') {
|
|
1399
|
+
if (formattingChanged(currentFormatting, formatting)) {
|
|
1400
|
+
flushRun();
|
|
1228
1401
|
currentFormatting = { ...formatting };
|
|
1229
|
-
cellContentIndex = 0;
|
|
1230
|
-
}
|
|
1231
|
-
else if (node.value === 'intbl') {
|
|
1232
|
-
// Paragraph is in a table
|
|
1233
|
-
inTable = true;
|
|
1234
|
-
paragraphInTable = true;
|
|
1235
|
-
ensureTableContext();
|
|
1236
|
-
}
|
|
1237
|
-
else if (node.value === 'pard') {
|
|
1238
|
-
// Reset paragraph properties
|
|
1239
|
-
paragraphInTable = false;
|
|
1240
|
-
// Reset other props...
|
|
1241
|
-
paragraphIndent = 0;
|
|
1242
|
-
paragraphAlignment = 'left';
|
|
1243
|
-
isListItem = false;
|
|
1244
|
-
listType = undefined;
|
|
1245
|
-
headingLevel = undefined;
|
|
1246
|
-
currentListId = undefined;
|
|
1247
|
-
// Reset paragraph-level background (cbpat) to prevent leaking to next paragraph
|
|
1248
|
-
formatting.backgroundColor = undefined;
|
|
1249
|
-
}
|
|
1250
|
-
// Text flow control
|
|
1251
|
-
else if (node.value === 'tab') {
|
|
1252
|
-
if (formattingChanged(currentFormatting, formatting)) {
|
|
1253
|
-
flushRun();
|
|
1254
|
-
currentFormatting = { ...formatting };
|
|
1255
|
-
}
|
|
1256
|
-
currentRunText += '\t';
|
|
1257
1402
|
}
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1403
|
+
currentRunTextChunks.push('\n');
|
|
1404
|
+
}
|
|
1405
|
+
// Quote characters
|
|
1406
|
+
else if (node.value === 'lquote') {
|
|
1407
|
+
// Left single quotation mark (U+2018)
|
|
1408
|
+
if (formattingChanged(currentFormatting, formatting)) {
|
|
1409
|
+
flushRun();
|
|
1410
|
+
currentFormatting = { ...formatting };
|
|
1264
1411
|
}
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
}
|
|
1272
|
-
currentRunText += '\u2018';
|
|
1412
|
+
currentRunTextChunks.push('\u2018');
|
|
1413
|
+
}
|
|
1414
|
+
else if (node.value === 'rquote') {
|
|
1415
|
+
// Right single quotation mark (U+2019)
|
|
1416
|
+
if (formattingChanged(currentFormatting, formatting)) {
|
|
1417
|
+
flushRun();
|
|
1418
|
+
currentFormatting = { ...formatting };
|
|
1273
1419
|
}
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1420
|
+
currentRunTextChunks.push('\u2019');
|
|
1421
|
+
}
|
|
1422
|
+
else if (node.value === 'ldblquote') {
|
|
1423
|
+
// Left double quotation mark (U+201C)
|
|
1424
|
+
if (formattingChanged(currentFormatting, formatting)) {
|
|
1425
|
+
flushRun();
|
|
1426
|
+
currentFormatting = { ...formatting };
|
|
1281
1427
|
}
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1428
|
+
currentRunTextChunks.push('\u201C');
|
|
1429
|
+
}
|
|
1430
|
+
else if (node.value === 'rdblquote') {
|
|
1431
|
+
// Right double quotation mark (U+201D)
|
|
1432
|
+
if (formattingChanged(currentFormatting, formatting)) {
|
|
1433
|
+
flushRun();
|
|
1434
|
+
currentFormatting = { ...formatting };
|
|
1289
1435
|
}
|
|
1290
|
-
|
|
1291
|
-
|
|
1436
|
+
currentRunTextChunks.push('\u201D');
|
|
1437
|
+
}
|
|
1438
|
+
// Unicode character
|
|
1439
|
+
else if (node.value === 'u') {
|
|
1440
|
+
if (node.param !== undefined) {
|
|
1441
|
+
let code = node.param;
|
|
1442
|
+
if (code < 0)
|
|
1443
|
+
code += 65536;
|
|
1292
1444
|
if (formattingChanged(currentFormatting, formatting)) {
|
|
1293
1445
|
flushRun();
|
|
1294
1446
|
currentFormatting = { ...formatting };
|
|
1295
1447
|
}
|
|
1296
|
-
|
|
1297
|
-
}
|
|
1298
|
-
// Unicode character
|
|
1299
|
-
else if (node.value === 'u') {
|
|
1300
|
-
if (node.param !== undefined) {
|
|
1301
|
-
let code = node.param;
|
|
1302
|
-
if (code < 0)
|
|
1303
|
-
code += 65536;
|
|
1304
|
-
if (formattingChanged(currentFormatting, formatting)) {
|
|
1305
|
-
flushRun();
|
|
1306
|
-
currentFormatting = { ...formatting };
|
|
1307
|
-
}
|
|
1308
|
-
currentRunText += String.fromCharCode(code);
|
|
1309
|
-
}
|
|
1310
|
-
}
|
|
1311
|
-
// Character formatting
|
|
1312
|
-
else if (node.value === 'b') {
|
|
1313
|
-
formatting.bold = (node.param !== 0);
|
|
1314
|
-
}
|
|
1315
|
-
else if (node.value === 'i') {
|
|
1316
|
-
formatting.italic = (node.param !== 0);
|
|
1317
|
-
}
|
|
1318
|
-
else if (node.value === 'ul') {
|
|
1319
|
-
formatting.underline = (node.param !== 0);
|
|
1320
|
-
}
|
|
1321
|
-
else if (node.value === 'ulnone') {
|
|
1322
|
-
formatting.underline = false;
|
|
1323
|
-
}
|
|
1324
|
-
else if (node.value === 'strike') {
|
|
1325
|
-
formatting.strikethrough = (node.param !== 0);
|
|
1326
|
-
}
|
|
1327
|
-
else if (node.value === 'plain') {
|
|
1328
|
-
// Reset all character formatting
|
|
1329
|
-
formatting.bold = false;
|
|
1330
|
-
formatting.italic = false;
|
|
1331
|
-
formatting.underline = false;
|
|
1332
|
-
formatting.strikethrough = false;
|
|
1333
|
-
formatting.subscript = false;
|
|
1334
|
-
formatting.superscript = false;
|
|
1335
|
-
formatting.size = undefined;
|
|
1336
|
-
formatting.font = undefined;
|
|
1337
|
-
formatting.color = undefined;
|
|
1338
|
-
formatting.backgroundColor = undefined;
|
|
1339
|
-
}
|
|
1340
|
-
// Font size (\fs - in half-points)
|
|
1341
|
-
else if (node.value === 'fs') {
|
|
1342
|
-
if (node.param !== undefined) {
|
|
1343
|
-
formatting.size = (node.param / 2).toString() + 'pt';
|
|
1344
|
-
}
|
|
1345
|
-
}
|
|
1346
|
-
// Font family (\f)
|
|
1347
|
-
else if (node.value === 'f') {
|
|
1348
|
-
if (node.param !== undefined && fontTable[node.param]) {
|
|
1349
|
-
formatting.font = fontTable[node.param];
|
|
1350
|
-
}
|
|
1351
|
-
}
|
|
1352
|
-
// Text color (\cf)
|
|
1353
|
-
else if (node.value === 'cf') {
|
|
1354
|
-
if (node.param !== undefined && colorTable[node.param]) {
|
|
1355
|
-
formatting.color = colorTable[node.param];
|
|
1356
|
-
}
|
|
1357
|
-
}
|
|
1358
|
-
// Note type (\fet)
|
|
1359
|
-
else if (node.value === 'fet') {
|
|
1360
|
-
// \fet0 = footnotes only (default)
|
|
1361
|
-
// \fet1 = endnotes only
|
|
1362
|
-
// \fet2 = both footnotes and endnotes
|
|
1363
|
-
if (node.param !== undefined) {
|
|
1364
|
-
fetValue = node.param;
|
|
1365
|
-
}
|
|
1448
|
+
currentRunTextChunks.push(String.fromCharCode(code));
|
|
1366
1449
|
}
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1450
|
+
}
|
|
1451
|
+
// Character formatting
|
|
1452
|
+
else if (node.value === 'b') {
|
|
1453
|
+
formatting.bold = (node.param !== 0);
|
|
1454
|
+
}
|
|
1455
|
+
else if (node.value === 'i') {
|
|
1456
|
+
formatting.italic = (node.param !== 0);
|
|
1457
|
+
}
|
|
1458
|
+
else if (node.value === 'ul') {
|
|
1459
|
+
formatting.underline = (node.param !== 0);
|
|
1460
|
+
}
|
|
1461
|
+
else if (node.value === 'ulnone') {
|
|
1462
|
+
formatting.underline = false;
|
|
1463
|
+
}
|
|
1464
|
+
else if (node.value === 'strike') {
|
|
1465
|
+
formatting.strikethrough = (node.param !== 0);
|
|
1466
|
+
}
|
|
1467
|
+
else if (node.value === 'plain') {
|
|
1468
|
+
// Reset all character formatting
|
|
1469
|
+
formatting.bold = false;
|
|
1470
|
+
formatting.italic = false;
|
|
1471
|
+
formatting.underline = false;
|
|
1472
|
+
formatting.strikethrough = false;
|
|
1473
|
+
formatting.subscript = false;
|
|
1474
|
+
formatting.superscript = false;
|
|
1475
|
+
formatting.size = undefined;
|
|
1476
|
+
formatting.font = undefined;
|
|
1477
|
+
formatting.color = undefined;
|
|
1478
|
+
formatting.backgroundColor = undefined;
|
|
1479
|
+
}
|
|
1480
|
+
// Font size (\fs - in half-points)
|
|
1481
|
+
else if (node.value === 'fs') {
|
|
1482
|
+
if (node.param !== undefined) {
|
|
1483
|
+
formatting.size = (node.param / 2).toString() + 'pt';
|
|
1374
1484
|
}
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
// Superscript
|
|
1381
|
-
else if (node.value === 'super') {
|
|
1382
|
-
formatting.superscript = true;
|
|
1383
|
-
formatting.subscript = false;
|
|
1384
|
-
}
|
|
1385
|
-
// No subscript/superscript
|
|
1386
|
-
else if (node.value === 'nosupersub') {
|
|
1387
|
-
formatting.subscript = false;
|
|
1388
|
-
formatting.superscript = false;
|
|
1389
|
-
}
|
|
1390
|
-
// ═══════════════════════════════════════════════════════════
|
|
1391
|
-
// List control words
|
|
1392
|
-
// ═══════════════════════════════════════════════════════════
|
|
1393
|
-
// Paragraph indentation (\li - left indent in twips)
|
|
1394
|
-
else if (node.value === 'li') {
|
|
1395
|
-
if (node.param !== undefined) {
|
|
1396
|
-
// Convert twips to a simpler unit (720 twips = 1 inch, ~0.5 inch per level)
|
|
1397
|
-
paragraphIndent = Math.floor(node.param / 360);
|
|
1398
|
-
}
|
|
1485
|
+
}
|
|
1486
|
+
// Font family (\f)
|
|
1487
|
+
else if (node.value === 'f') {
|
|
1488
|
+
if (node.param !== undefined && fontTable[node.param]) {
|
|
1489
|
+
formatting.font = fontTable[node.param];
|
|
1399
1490
|
}
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
paragraphIndent = 0;
|
|
1406
|
-
isListItem = true;
|
|
1407
|
-
// Generate or retrieve list ID
|
|
1408
|
-
if (!listStyleIdMap[node.param]) {
|
|
1409
|
-
listIdCounter++;
|
|
1410
|
-
listStyleIdMap[node.param] = `rtf-list-${listIdCounter}`;
|
|
1411
|
-
}
|
|
1412
|
-
currentListId = listStyleIdMap[node.param];
|
|
1413
|
-
// Look up type from list definition
|
|
1414
|
-
// First check override map to get real list ID
|
|
1415
|
-
const realListId = listOverrideMap[node.param] !== undefined ? listOverrideMap[node.param] : node.param;
|
|
1416
|
-
if (listTypeMap[realListId]) {
|
|
1417
|
-
listType = listTypeMap[realListId];
|
|
1418
|
-
}
|
|
1419
|
-
}
|
|
1491
|
+
}
|
|
1492
|
+
// Text color (\cf)
|
|
1493
|
+
else if (node.value === 'cf') {
|
|
1494
|
+
if (node.param !== undefined && colorTable[node.param]) {
|
|
1495
|
+
formatting.color = colorTable[node.param];
|
|
1420
1496
|
}
|
|
1421
|
-
|
|
1422
|
-
|
|
1423
|
-
|
|
1424
|
-
|
|
1425
|
-
|
|
1426
|
-
|
|
1497
|
+
}
|
|
1498
|
+
// Note type (\fet)
|
|
1499
|
+
else if (node.value === 'fet') {
|
|
1500
|
+
// \fet0 = footnotes only (default)
|
|
1501
|
+
// \fet1 = endnotes only
|
|
1502
|
+
// \fet2 = both footnotes and endnotes
|
|
1503
|
+
if (node.param !== undefined) {
|
|
1504
|
+
fetValue = node.param;
|
|
1427
1505
|
}
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1506
|
+
}
|
|
1507
|
+
// Background/highlight color (\cb, \highlight, \chcbpat, \cbpat)
|
|
1508
|
+
// \chcbpat = character background pattern color (used for shading)
|
|
1509
|
+
// \cbpat = paragraph background pattern color
|
|
1510
|
+
else if (node.value === 'cb' || node.value === 'highlight' || node.value === 'chcbpat' || node.value === 'cbpat') {
|
|
1511
|
+
if (node.param !== undefined && colorTable[node.param]) {
|
|
1512
|
+
formatting.backgroundColor = colorTable[node.param];
|
|
1434
1513
|
}
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1441
|
-
|
|
1442
|
-
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1514
|
+
}
|
|
1515
|
+
// Subscript
|
|
1516
|
+
else if (node.value === 'sub') {
|
|
1517
|
+
formatting.subscript = true;
|
|
1518
|
+
formatting.superscript = false;
|
|
1519
|
+
}
|
|
1520
|
+
// Superscript
|
|
1521
|
+
else if (node.value === 'super') {
|
|
1522
|
+
formatting.superscript = true;
|
|
1523
|
+
formatting.subscript = false;
|
|
1524
|
+
}
|
|
1525
|
+
// No subscript/superscript
|
|
1526
|
+
else if (node.value === 'nosupersub') {
|
|
1527
|
+
formatting.subscript = false;
|
|
1528
|
+
formatting.superscript = false;
|
|
1529
|
+
}
|
|
1530
|
+
// ═══════════════════════════════════════════════════════════
|
|
1531
|
+
// List control words
|
|
1532
|
+
// ═══════════════════════════════════════════════════════════
|
|
1533
|
+
// Paragraph indentation (\li - left indent in twips)
|
|
1534
|
+
else if (node.value === 'li') {
|
|
1535
|
+
if (node.param !== undefined) {
|
|
1536
|
+
// Standard level indent is 720 twips (0.5 inch)
|
|
1537
|
+
// Using a slightly more flexible divisor to account for different generators
|
|
1538
|
+
const level = Math.round(node.param / 720);
|
|
1539
|
+
// Only update if not already explicitly set by ilvl (Word 97+)
|
|
1540
|
+
if (!isListItem) {
|
|
1541
|
+
paragraphIndent = level;
|
|
1447
1542
|
}
|
|
1448
1543
|
}
|
|
1449
|
-
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
paragraphIndent = 0;
|
|
1454
|
-
isListItem = true;
|
|
1455
|
-
listType = 'ordered';
|
|
1456
|
-
}
|
|
1457
|
-
// Unordered list indicator
|
|
1458
|
-
else if (node.value === 'pnbullet' || node.value === 'pncard') {
|
|
1544
|
+
}
|
|
1545
|
+
// List style ID (Word 97+)
|
|
1546
|
+
else if (node.value === 'ls') {
|
|
1547
|
+
if (node.param !== undefined) {
|
|
1459
1548
|
// If this is the first list item, reset the indent
|
|
1460
1549
|
if (!isListItem)
|
|
1461
1550
|
paragraphIndent = 0;
|
|
1462
1551
|
isListItem = true;
|
|
1463
|
-
|
|
1464
|
-
|
|
1465
|
-
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1552
|
+
// Generate or retrieve list ID
|
|
1553
|
+
if (!listStyleIdMap[node.param]) {
|
|
1554
|
+
listIdCounter++;
|
|
1555
|
+
listStyleIdMap[node.param] = `rtf-list-${listIdCounter}`;
|
|
1556
|
+
}
|
|
1557
|
+
currentListId = listStyleIdMap[node.param];
|
|
1558
|
+
// Look up type from list definition
|
|
1559
|
+
// First check override map to get real list ID
|
|
1560
|
+
const realListId = listOverrideMap[node.param] !== undefined ? listOverrideMap[node.param] : node.param;
|
|
1561
|
+
if (listTypeMap[realListId]) {
|
|
1562
|
+
listType = listTypeMap[realListId];
|
|
1472
1563
|
}
|
|
1473
1564
|
}
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
}
|
|
1481
|
-
else if (node.value === 'qr') {
|
|
1482
|
-
paragraphAlignment = 'right';
|
|
1565
|
+
}
|
|
1566
|
+
// List indent level (Word 97+)
|
|
1567
|
+
else if (node.value === 'ilvl') {
|
|
1568
|
+
if (node.param !== undefined) {
|
|
1569
|
+
isListItem = true;
|
|
1570
|
+
paragraphIndent = node.param;
|
|
1483
1571
|
}
|
|
1484
|
-
|
|
1485
|
-
|
|
1572
|
+
}
|
|
1573
|
+
// List numbering level (\pnlvl)
|
|
1574
|
+
else if (node.value === 'pnlvl') {
|
|
1575
|
+
isListItem = true;
|
|
1576
|
+
if (node.param !== undefined) {
|
|
1577
|
+
paragraphIndent = node.param;
|
|
1486
1578
|
}
|
|
1487
1579
|
}
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
// Notes handling:
|
|
1497
|
-
// - If putNotesAtLast is false, notes should be added inline during traversal
|
|
1498
|
-
// (currently they go to 'notes' array, then we append them here - this is wrong)
|
|
1499
|
-
// - If putNotesAtLast is true, notes are appended at the very end (see below)
|
|
1500
|
-
//
|
|
1501
|
-
// For now, when putNotesAtLast is false, we append notes immediately after content
|
|
1502
|
-
// This isn't truly "inline" but it's better than at the end
|
|
1503
|
-
// TODO: Implement true inline placement during traversal
|
|
1504
|
-
if (!config.putNotesAtLast && notes.length > 0) {
|
|
1505
|
-
content.push(...notes);
|
|
1506
|
-
notes.length = 0; // Clear so they don't get appended again
|
|
1507
|
-
}
|
|
1508
|
-
// Perform OCR if enabled
|
|
1509
|
-
if (config.ocr && config.extractAttachments) {
|
|
1510
|
-
for (const attachment of attachments) {
|
|
1511
|
-
if (attachment.mimeType.startsWith('image/')) {
|
|
1512
|
-
try {
|
|
1513
|
-
// Convert base64 data back to Buffer for Tesseract.js
|
|
1514
|
-
// Passing base64 string directly would be interpreted as a file path,
|
|
1515
|
-
// causing ENAMETOOLONG error for large images.
|
|
1516
|
-
const imageBuffer = Buffer.from(attachment.data, 'base64');
|
|
1517
|
-
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
1580
|
+
// List numbering format
|
|
1581
|
+
else if (node.value === 'levelnfc' || node.value === 'pnf') {
|
|
1582
|
+
// 0 = Arabic (1, 2, 3), 1 = Roman upper, 2 = Roman lower,
|
|
1583
|
+
// 3 = Letter upper, 4 = Letter lower, 23 = Bullet
|
|
1584
|
+
if (node.param !== undefined) {
|
|
1585
|
+
isListItem = true;
|
|
1586
|
+
if (node.param === 23) {
|
|
1587
|
+
listType = 'unordered';
|
|
1518
1588
|
}
|
|
1519
|
-
|
|
1520
|
-
|
|
1589
|
+
else {
|
|
1590
|
+
listType = 'ordered';
|
|
1521
1591
|
}
|
|
1522
1592
|
}
|
|
1523
1593
|
}
|
|
1524
|
-
//
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
|
|
1538
|
-
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
|
|
1542
|
-
|
|
1594
|
+
// Ordered list indicator
|
|
1595
|
+
else if (node.value === 'pndec' || node.value === 'pnord' || node.value === 'pnlcltr' || node.value === 'pnucltr') {
|
|
1596
|
+
// If this is the first list item, reset the indent
|
|
1597
|
+
if (!isListItem)
|
|
1598
|
+
paragraphIndent = 0;
|
|
1599
|
+
isListItem = true;
|
|
1600
|
+
listType = 'ordered';
|
|
1601
|
+
}
|
|
1602
|
+
// Unordered list indicator
|
|
1603
|
+
else if (node.value === 'pnbullet' || node.value === 'pncard') {
|
|
1604
|
+
// If this is the first list item, reset the indent
|
|
1605
|
+
if (!isListItem)
|
|
1606
|
+
paragraphIndent = 0;
|
|
1607
|
+
isListItem = true;
|
|
1608
|
+
listType = 'unordered';
|
|
1609
|
+
}
|
|
1610
|
+
// Style-based heading detection (\s)
|
|
1611
|
+
else if (node.value === 's') {
|
|
1612
|
+
if (node.param !== undefined) {
|
|
1613
|
+
// Common heading styles: s1-s9 (though this varies by document)
|
|
1614
|
+
if (node.param >= 1 && node.param <= 9) {
|
|
1615
|
+
headingLevel = node.param;
|
|
1543
1616
|
}
|
|
1544
1617
|
}
|
|
1545
|
-
}
|
|
1546
|
-
|
|
1618
|
+
}
|
|
1619
|
+
// Paragraph alignment
|
|
1620
|
+
else if (node.value === 'ql') {
|
|
1621
|
+
paragraphAlignment = 'left';
|
|
1622
|
+
}
|
|
1623
|
+
else if (node.value === 'qc') {
|
|
1624
|
+
paragraphAlignment = 'center';
|
|
1625
|
+
}
|
|
1626
|
+
else if (node.value === 'qr') {
|
|
1627
|
+
paragraphAlignment = 'right';
|
|
1628
|
+
}
|
|
1629
|
+
else if (node.value === 'qj') {
|
|
1630
|
+
paragraphAlignment = 'justify';
|
|
1631
|
+
}
|
|
1547
1632
|
}
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
-
|
|
1633
|
+
};
|
|
1634
|
+
traverse(doc, {});
|
|
1635
|
+
// Flush any remaining table
|
|
1636
|
+
const finalCtx = getCurrentTable();
|
|
1637
|
+
if (inTable || (finalCtx && (finalCtx.rows.length > 0 || finalCtx.currentCells.length > 0))) {
|
|
1638
|
+
flushTable();
|
|
1639
|
+
}
|
|
1640
|
+
flushParagraph();
|
|
1641
|
+
// Notes handling:
|
|
1642
|
+
// - If putNotesAtLast is false, notes should be added inline during traversal
|
|
1643
|
+
// (currently they go to 'notes' array, then we append them here - this is wrong)
|
|
1644
|
+
// - If putNotesAtLast is true, notes are appended at the very end (see below)
|
|
1645
|
+
//
|
|
1646
|
+
// For now, when putNotesAtLast is false, we append notes immediately after content
|
|
1647
|
+
// This isn't truly "inline" but it's better than at the end
|
|
1648
|
+
// TODO: Implement true inline placement during traversal
|
|
1649
|
+
if (!config.putNotesAtLast && notes.length > 0) {
|
|
1650
|
+
content.push(...notes);
|
|
1651
|
+
notes.length = 0; // Clear so they don't get appended again
|
|
1652
|
+
}
|
|
1653
|
+
// Perform OCR if enabled
|
|
1654
|
+
if (config.ocr && config.extractAttachments) {
|
|
1655
|
+
for (const attachment of attachments) {
|
|
1656
|
+
if (attachment.mimeType.startsWith('image/')) {
|
|
1657
|
+
try {
|
|
1658
|
+
// Convert base64 data back to Buffer for Tesseract.js
|
|
1659
|
+
// Passing base64 string directly would be interpreted as a file path,
|
|
1660
|
+
// causing ENAMETOOLONG error for large images.
|
|
1661
|
+
const imageBuffer = Buffer.from(attachment.data, 'base64');
|
|
1662
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { ...config.ocrConfig })).trim();
|
|
1663
|
+
}
|
|
1664
|
+
catch (e) {
|
|
1665
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
|
|
1666
|
+
}
|
|
1667
|
+
}
|
|
1668
|
+
}
|
|
1669
|
+
// Link OCR text and altText to image nodes in content
|
|
1670
|
+
const assignOcr = (nodes) => {
|
|
1551
1671
|
for (const node of nodes) {
|
|
1552
|
-
if (node.type === '
|
|
1553
|
-
const
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1672
|
+
if (node.type === 'image' && node.metadata && 'attachmentName' in node.metadata) {
|
|
1673
|
+
const meta = node.metadata;
|
|
1674
|
+
const attachment = attachments.find(a => a.name === meta.attachmentName);
|
|
1675
|
+
if (attachment) {
|
|
1676
|
+
// Propagate OCR text to image node
|
|
1677
|
+
if (attachment.ocrText) {
|
|
1678
|
+
node.text = attachment.ocrText;
|
|
1679
|
+
}
|
|
1680
|
+
// Propagate altText if available
|
|
1681
|
+
if (attachment.altText) {
|
|
1682
|
+
meta.altText = attachment.altText;
|
|
1683
|
+
}
|
|
1684
|
+
}
|
|
1559
1685
|
}
|
|
1560
1686
|
if (node.children) {
|
|
1561
|
-
|
|
1687
|
+
assignOcr(node.children);
|
|
1562
1688
|
}
|
|
1563
1689
|
}
|
|
1564
1690
|
};
|
|
1565
|
-
|
|
1566
|
-
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
|
|
1572
|
-
|
|
1573
|
-
|
|
1574
|
-
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1578
|
-
}
|
|
1579
|
-
return text;
|
|
1691
|
+
assignOcr(content);
|
|
1692
|
+
}
|
|
1693
|
+
// Final pass to ensure all 'note' nodes have their 'text' property populated
|
|
1694
|
+
// (This supports the simple toText implementation)
|
|
1695
|
+
const populateNoteText = (nodes) => {
|
|
1696
|
+
for (const node of nodes) {
|
|
1697
|
+
if (node.type === 'note' && node.children) {
|
|
1698
|
+
const getText = (n) => {
|
|
1699
|
+
if (n.children && n.children.length > 0)
|
|
1700
|
+
return n.children.map(getText).join('');
|
|
1701
|
+
return n.text || '';
|
|
1702
|
+
};
|
|
1703
|
+
node.text = node.children.map(getText).join('').trim();
|
|
1580
1704
|
}
|
|
1581
|
-
|
|
1582
|
-
|
|
1705
|
+
if (node.children) {
|
|
1706
|
+
populateNoteText(node.children);
|
|
1707
|
+
}
|
|
1708
|
+
}
|
|
1709
|
+
};
|
|
1710
|
+
populateNoteText(content);
|
|
1711
|
+
populateNoteText(notes);
|
|
1712
|
+
const toTextSync = () => {
|
|
1713
|
+
let text = content.map(c => c.text).join(config.newlineDelimiter);
|
|
1583
1714
|
if (config.putNotesAtLast && notes.length > 0) {
|
|
1584
|
-
|
|
1715
|
+
text += config.newlineDelimiter + notes.map(c => c.text).join(config.newlineDelimiter);
|
|
1585
1716
|
}
|
|
1586
|
-
return
|
|
1587
|
-
}
|
|
1588
|
-
|
|
1589
|
-
|
|
1717
|
+
return text;
|
|
1718
|
+
};
|
|
1719
|
+
const result = (0, astUtils_js_1.createAST)('rtf', {
|
|
1720
|
+
// RTF Limitation: No style map available (RTF uses inline styles)
|
|
1721
|
+
}, content, attachments, // PNG and JPEG images extracted from \\pict groups
|
|
1722
|
+
config, toTextSync);
|
|
1723
|
+
// If putNotesAtLast is true, append notes to the end of the content array
|
|
1724
|
+
if (config.putNotesAtLast && notes.length > 0) {
|
|
1725
|
+
content.push(...notes);
|
|
1590
1726
|
}
|
|
1727
|
+
return result;
|
|
1591
1728
|
};
|
|
1592
1729
|
exports.parseRtf = parseRtf;
|
|
1730
|
+
// Helper to find an RTF group by destination name
|
|
1731
|
+
function findRtfGroup(group, destination) {
|
|
1732
|
+
for (const node of group.content) {
|
|
1733
|
+
if (node.type === 'group') {
|
|
1734
|
+
if (node.destination === destination)
|
|
1735
|
+
return node;
|
|
1736
|
+
const found = findRtfGroup(node, destination);
|
|
1737
|
+
if (found)
|
|
1738
|
+
return found;
|
|
1739
|
+
}
|
|
1740
|
+
}
|
|
1741
|
+
return null;
|
|
1742
|
+
}
|
|
1593
1743
|
// Helper function to extract font table from RTF document
|
|
1594
1744
|
function extractFontTable(doc) {
|
|
1595
1745
|
const fontTable = {};
|
|
1596
|
-
|
|
1597
|
-
|
|
1598
|
-
for (const
|
|
1599
|
-
if (
|
|
1600
|
-
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
|
|
1607
|
-
|
|
1608
|
-
fontIndex = item.param;
|
|
1609
|
-
}
|
|
1610
|
-
else if (item.type === 'text') {
|
|
1611
|
-
// Font name (may have trailing semicolon)
|
|
1612
|
-
fontName += item.value;
|
|
1613
|
-
}
|
|
1614
|
-
}
|
|
1615
|
-
if (fontIndex !== undefined && fontName) {
|
|
1616
|
-
// Remove trailing semicolon and whitespace
|
|
1617
|
-
fontName = fontName.replace(/;$/, '').trim();
|
|
1618
|
-
fontTable[fontIndex] = fontName;
|
|
1619
|
-
}
|
|
1620
|
-
}
|
|
1746
|
+
const tableGroup = findRtfGroup(doc, 'fonttbl');
|
|
1747
|
+
if (tableGroup) {
|
|
1748
|
+
for (const fontNode of tableGroup.content) {
|
|
1749
|
+
if (fontNode.type === 'group') {
|
|
1750
|
+
let fontIndex;
|
|
1751
|
+
let fontName = '';
|
|
1752
|
+
for (const item of fontNode.content) {
|
|
1753
|
+
if (item.type === 'control' && item.value === 'f') {
|
|
1754
|
+
fontIndex = item.param;
|
|
1755
|
+
}
|
|
1756
|
+
else if (item.type === 'text') {
|
|
1757
|
+
fontName += item.value;
|
|
1621
1758
|
}
|
|
1622
|
-
return true; // Found and parsed
|
|
1623
1759
|
}
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
return true;
|
|
1760
|
+
if (fontIndex !== undefined && fontName) {
|
|
1761
|
+
fontTable[fontIndex] = fontName.replace(/;$/, '').trim();
|
|
1627
1762
|
}
|
|
1628
1763
|
}
|
|
1629
1764
|
}
|
|
1630
|
-
|
|
1631
|
-
};
|
|
1632
|
-
findAndParseFontTable(doc);
|
|
1765
|
+
}
|
|
1633
1766
|
return fontTable;
|
|
1634
1767
|
}
|
|
1635
1768
|
// Helper function to extract color table from RTF document
|
|
1636
1769
|
function extractColorTable(doc) {
|
|
1637
1770
|
const colorTable = {};
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1656
|
-
|
|
1657
|
-
|
|
1658
|
-
// Semicolon marks end of color definition
|
|
1659
|
-
const hex = `#${red.toString(16).padStart(2, '0')}${green.toString(16).padStart(2, '0')}${blue.toString(16).padStart(2, '0')}`;
|
|
1660
|
-
colorTable[colorIndex] = hex;
|
|
1661
|
-
colorIndex++;
|
|
1662
|
-
red = 0;
|
|
1663
|
-
green = 0;
|
|
1664
|
-
blue = 0;
|
|
1665
|
-
}
|
|
1666
|
-
}
|
|
1667
|
-
return true; // Found and parsed
|
|
1668
|
-
}
|
|
1669
|
-
// Recurse into child groups
|
|
1670
|
-
if (findAndParseColorTable(node)) {
|
|
1671
|
-
return true;
|
|
1672
|
-
}
|
|
1771
|
+
const tableGroup = findRtfGroup(doc, 'colortbl');
|
|
1772
|
+
if (tableGroup) {
|
|
1773
|
+
let colorIndex = 0;
|
|
1774
|
+
let red = 0, green = 0, blue = 0;
|
|
1775
|
+
for (const item of tableGroup.content) {
|
|
1776
|
+
if (item.type === 'control') {
|
|
1777
|
+
if (item.value === 'red' && item.param !== undefined)
|
|
1778
|
+
red = item.param;
|
|
1779
|
+
else if (item.value === 'green' && item.param !== undefined)
|
|
1780
|
+
green = item.param;
|
|
1781
|
+
else if (item.value === 'blue' && item.param !== undefined)
|
|
1782
|
+
blue = item.param;
|
|
1783
|
+
}
|
|
1784
|
+
else if (item.type === 'text' && item.value === ';') {
|
|
1785
|
+
const hex = `#${red.toString(16).padStart(2, '0')}${green.toString(16).padStart(2, '0')}${blue.toString(16).padStart(2, '0')}`;
|
|
1786
|
+
colorTable[colorIndex] = hex;
|
|
1787
|
+
colorIndex++;
|
|
1788
|
+
red = 0;
|
|
1789
|
+
green = 0;
|
|
1790
|
+
blue = 0;
|
|
1673
1791
|
}
|
|
1674
1792
|
}
|
|
1675
|
-
|
|
1676
|
-
};
|
|
1677
|
-
findAndParseColorTable(doc);
|
|
1793
|
+
}
|
|
1678
1794
|
return colorTable;
|
|
1679
1795
|
}
|