officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -42,6 +42,8 @@
42
42
  */
43
43
  Object.defineProperty(exports, "__esModule", { value: true });
44
44
  exports.parseRtf = exports.SimpleRtfParser = void 0;
45
+ const types_js_1 = require("../types.js");
46
+ const astUtils_js_1 = require("../utils/astUtils.js");
45
47
  const errorUtils_js_1 = require("../utils/errorUtils.js");
46
48
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
47
49
  /**
@@ -92,6 +94,12 @@ class SimpleRtfParser {
92
94
  index = 0;
93
95
  /** The RTF content as a Buffer */
94
96
  buffer;
97
+ /** Current code page for character decoding (default is Windows-1252) */
98
+ codePage = 1252;
99
+ /** Cached TextDecoders for different code pages */
100
+ decoders = {};
101
+ /** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
102
+ pendingBytes = [];
95
103
  /** Total length of the buffer */
96
104
  length;
97
105
  /**
@@ -110,12 +118,14 @@ class SimpleRtfParser {
110
118
  const currentGroup = stack[stack.length - 1];
111
119
  if (char === 0x7B) { // '{'
112
120
  this.index++;
121
+ this.flushPendingText(currentGroup);
113
122
  const newGroup = { type: 'group', content: [] };
114
123
  currentGroup.content.push(newGroup);
115
124
  stack.push(newGroup);
116
125
  }
117
126
  else if (char === 0x7D) { // '}'
118
127
  this.index++;
128
+ this.flushPendingText(currentGroup);
119
129
  if (stack.length > 1) {
120
130
  stack.pop();
121
131
  }
@@ -133,6 +143,7 @@ class SimpleRtfParser {
133
143
  this.parseText(currentGroup);
134
144
  }
135
145
  }
146
+ this.flushPendingText(root);
136
147
  return root;
137
148
  }
138
149
  parseControl(group) {
@@ -141,55 +152,23 @@ class SimpleRtfParser {
141
152
  const char = this.buffer[this.index];
142
153
  // Special control symbols
143
154
  if (char === 0x7B || char === 0x7D || char === 0x5C) { // \{ \} \\
144
- group.content.push({ type: 'text', value: String.fromCharCode(char) });
155
+ this.pendingBytes.push(char);
145
156
  this.index++;
146
157
  return;
147
158
  }
148
159
  if (char === 0x27) { // \'xx (hex)
149
160
  this.index++;
150
161
  if (this.index + 1 < this.length) {
151
- const hex = this.buffer.toString('utf8', this.index, this.index + 2);
162
+ const hex = String.fromCharCode(this.buffer[this.index], this.buffer[this.index + 1]);
152
163
  const code = parseInt(hex, 16);
153
164
  if (!isNaN(code)) {
154
- // RTF hex escapes represent bytes in the document's code page (usually Windows-1252)
155
- // Characters 0x80-0x9F in Windows-1252 don't map directly to Unicode
156
- // We need to convert them properly
157
- const windows1252ToUnicode = {
158
- 0x80: 0x20AC, // €
159
- 0x82: 0x201A, // ‚
160
- 0x83: 0x0192, // ƒ
161
- 0x84: 0x201E, // „
162
- 0x85: 0x2026, // …
163
- 0x86: 0x2020, // †
164
- 0x87: 0x2021, // ‡
165
- 0x88: 0x02C6, // ˆ
166
- 0x89: 0x2030, // ‰
167
- 0x8A: 0x0160, // Š
168
- 0x8B: 0x2039, // ‹
169
- 0x8C: 0x0152, // Œ
170
- 0x8E: 0x017D, // Ž
171
- 0x91: 0x2018, // '
172
- 0x92: 0x2019, // '
173
- 0x93: 0x201C, // "
174
- 0x94: 0x201D, // "
175
- 0x95: 0x2022, // •
176
- 0x96: 0x2013, // –
177
- 0x97: 0x2014, // —
178
- 0x98: 0x02DC, // ˜
179
- 0x99: 0x2122, // ™
180
- 0x9A: 0x0161, // š
181
- 0x9B: 0x203A, // ›
182
- 0x9C: 0x0153, // œ
183
- 0x9E: 0x017E, // ž
184
- 0x9F: 0x0178 // Ÿ
185
- };
186
- const unicodeCode = windows1252ToUnicode[code] || code;
187
- group.content.push({ type: 'text', value: String.fromCharCode(unicodeCode) });
165
+ this.pendingBytes.push(code);
188
166
  }
189
167
  this.index += 2;
190
168
  }
191
169
  return;
192
170
  }
171
+ this.flushPendingText(group);
193
172
  if (char === 0x2A) { // \* (ignorable destination)
194
173
  // We treat this as a control word named '*'
195
174
  group.content.push({ type: 'control', value: '*' });
@@ -231,7 +210,7 @@ class SimpleRtfParser {
231
210
  param = parseInt(paramStr, 10);
232
211
  }
233
212
  // Space after control word is consumed
234
- if (this.index < this.length && this.buffer[this.index] === 0x20) {
213
+ if (name !== '' && this.index < this.length && this.buffer[this.index] === 0x20) {
235
214
  this.index++;
236
215
  }
237
216
  // Handle \binN
@@ -241,6 +220,22 @@ class SimpleRtfParser {
241
220
  // \binN is not added to content as we want to ignore it
242
221
  return;
243
222
  }
223
+ // Handle encoding control words
224
+ if (name === 'ansicpg' && param !== undefined) {
225
+ this.codePage = param;
226
+ }
227
+ else if (name === 'ansi') {
228
+ this.codePage = 1252;
229
+ }
230
+ else if (name === 'mac') {
231
+ this.codePage = 10000;
232
+ }
233
+ else if (name === 'pc') {
234
+ this.codePage = 437;
235
+ }
236
+ else if (name === 'pca') {
237
+ this.codePage = 850;
238
+ }
244
239
  group.content.push({ type: 'control', value: name, param });
245
240
  // If this is the first control word in the group, it might be the destination
246
241
  if (group.content.length === 1 && group.type === 'group') {
@@ -252,20 +247,92 @@ class SimpleRtfParser {
252
247
  }
253
248
  }
254
249
  parseText(group) {
255
- let text = '';
256
250
  while (this.index < this.length) {
257
251
  const char = this.buffer[this.index];
252
+ if (char === undefined)
253
+ break;
258
254
  if (char === 0x7B || char === 0x7D || char === 0x5C || char === 0x0D || char === 0x0A) {
259
255
  break;
260
256
  }
261
- // Basic ASCII text.
262
- text += String.fromCharCode(char);
257
+ this.pendingBytes.push(char);
263
258
  this.index++;
264
259
  }
265
- if (text.length > 0) {
266
- group.content.push({ type: 'text', value: text });
260
+ }
261
+ /**
262
+ * Flushes the pending bytes buffer as a text node to the current group.
263
+ * @param group The group to append the text node to
264
+ */
265
+ flushPendingText(group) {
266
+ if (this.pendingBytes.length > 0) {
267
+ group.content.push({ type: 'text', value: this.decodeBytes(this.pendingBytes, this.codePage) });
268
+ this.pendingBytes = [];
267
269
  }
268
270
  }
271
+ /**
272
+ * Decodes a byte array using a "UTF-8 first" strategy.
273
+ * If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
274
+ * Otherwise, falls back to the specified code page.
275
+ * @param bytes The bytes to decode
276
+ * @param codePage The RTF code page ID
277
+ * @returns The decoded string
278
+ */
279
+ decodeBytes(bytes, codePage) {
280
+ const uint8 = new Uint8Array(bytes);
281
+ // Try UTF-8 first if there are any non-ASCII bytes.
282
+ // Many modern RTF generators (like calibre or web-based tools) dump UTF-8 bytes
283
+ // into the RTF even if the header claims a different code page.
284
+ if (bytes.some(b => b > 127)) {
285
+ try {
286
+ // Use fatal: true to ensure we fall back on invalid UTF-8 sequences
287
+ const utf8Decoder = new TextDecoder('utf-8', { fatal: true });
288
+ return utf8Decoder.decode(uint8);
289
+ }
290
+ catch (e) {
291
+ // Not valid UTF-8, continue to code page fallback
292
+ }
293
+ }
294
+ // Fallback to specified code page
295
+ if (!this.decoders[codePage]) {
296
+ let encoding = `windows-${codePage}`;
297
+ if (codePage === 10000)
298
+ encoding = 'macintosh';
299
+ else if (codePage === 437)
300
+ encoding = 'ibm437';
301
+ else if (codePage === 850)
302
+ encoding = 'ibm850';
303
+ try {
304
+ this.decoders[codePage] = new TextDecoder(encoding);
305
+ }
306
+ catch (e) {
307
+ if (codePage !== 1252) {
308
+ try {
309
+ this.decoders[codePage] = new TextDecoder('windows-1252');
310
+ }
311
+ catch (e2) {
312
+ return String.fromCharCode(...bytes);
313
+ }
314
+ }
315
+ else {
316
+ return String.fromCharCode(...bytes);
317
+ }
318
+ }
319
+ }
320
+ let result = this.decoders[codePage].decode(uint8);
321
+ // Safety override for Windows-1252 0x80-0x9F range if TextDecoder behaves like Latin-1.
322
+ // We replace control characters in the decoded string with their proper 1252 equivalents.
323
+ if (codePage === 1252 && /[\u0080-\u009F]/.test(result)) {
324
+ const map = {
325
+ '\u0080': '€', '\u0082': '‚', '\u0083': 'ƒ', '\u0084': '„', '\u0085': '…',
326
+ '\u0086': '†', '\u0087': '‡', '\u0088': 'ˆ', '\u0089': '‰', '\u008A': 'Š',
327
+ '\u008B': '‹', '\u008C': 'Œ', '\u008E': 'Ž', '\u0091': '‘', '\u0092': '’',
328
+ '\u0093': '“', '\u0094': '”', '\u0095': '•', '\u0096': '–', '\u0097': '—',
329
+ '\u0098': '˜', '\u0099': '™', '\u009A': 'š', '\u009B': '›', '\u009C': 'œ',
330
+ '\u009E': 'ž', '\u009F': 'Ÿ'
331
+ };
332
+ return result.replace(/[\u0080-\u009F]/g, m => map[m] || m);
333
+ }
334
+ return result;
335
+ }
269
336
  }
270
337
  exports.SimpleRtfParser = SimpleRtfParser;
271
338
  /**
@@ -292,1388 +359,1437 @@ exports.SimpleRtfParser = SimpleRtfParser;
292
359
  * @returns The parsed AST.
293
360
  */
294
361
  const parseRtf = async (buffer, config) => {
295
- try {
296
- const parser = new SimpleRtfParser(buffer);
297
- const doc = parser.parse();
298
- // Extract font and color tables
299
- const fontTable = extractFontTable(doc);
300
- const colorTable = extractColorTable(doc);
301
- const content = [];
302
- const notes = [];
303
- const attachments = [];
304
- // State for paragraph construction
305
- let currentParagraphText = '';
306
- let currentParagraphChildren = [];
307
- let currentParagraphRaw = '';
308
- // State for text run construction
309
- let currentRunText = '';
310
- let currentFormatting = {};
311
- // Target for content (main body or notes)
312
- let currentTarget = content;
313
- // Paragraph-level state
314
- let paragraphIndent = 0;
315
- let paragraphAlignment = 'left';
316
- let isListItem = false;
317
- let listType;
318
- let headingLevel;
319
- let currentListId;
320
- // Persistent list state for listtext/pntext detection
321
- // These persist across paragraphs to allow list items without explicit \ls
322
- let lastKnownListId;
323
- let lastKnownListType;
324
- let tableStack = [];
325
- let currentFootnoteId = 0;
326
- // ═══════════════════════════════════════════════════════════════════
327
- // Note type tracking (footnotes vs endnotes)
328
- // ═══════════════════════════════════════════════════════════════════
329
- // RTF uses \fet to distinguish note types:
330
- // \fet0 = footnotes only (default)
331
- // \fet1 = endnotes only
332
- // \fet2 = both footnotes and endnotes
333
- let fetValue = 0; // Default to footnotes only
334
- // Helper to get current table context
335
- const getCurrentTable = () => tableStack.length > 0 ? tableStack[tableStack.length - 1] : undefined;
336
- // Helper to ensure a table context exists (for top-level tables)
337
- const ensureTableContext = () => {
338
- if (tableStack.length === 0) {
339
- tableStack.push({
340
- rows: [],
341
- currentCells: [],
342
- currentCellContent: [],
343
- rowIndex: 0
344
- });
345
- }
346
- };
347
- let inTable = false;
348
- let paragraphInTable = false;
349
- let tableId = 0;
350
- let rowCellProps = [];
351
- let currentCellDefinitionProps = { isMergedContinuation: false };
352
- let cellContentIndex = 0;
353
- // ═══════════════════════════════════════════════════════════════════
354
- // List state tracking (Word 97+ uses \ls for list style ID)
355
- // ═══════════════════════════════════════════════════════════════════
356
- let listIdCounter = 0;
357
- const listStyleIdMap = {};
358
- // List definition state for parsing \listtable
359
- let parsingListTable = false;
360
- let parsingListDefinition = false;
361
- let currentDefinedListId;
362
- let currentDefinedListType;
363
- const listTypeMap = {};
364
- // List override state for parsing \listoverridetable
365
- let parsingListOverrideTable = false;
366
- let currentListOverrideListId;
367
- let currentListOverrideLs;
368
- const listOverrideMap = {}; // Maps \ls ID to \listid
369
- // List counters for itemIndex tracking
370
- // Map: listId -> indentation level -> count
371
- const listCounters = {};
372
- // ═══════════════════════════════════════════════════════════════════
373
- // Hyperlink state (RTF uses \field{\*\fldinst HYPERLINK "url"})
374
- // ═══════════════════════════════════════════════════════════════════
375
- let currentLinkUrl;
376
- // Helper to check if formatting changed
377
- const formattingChanged = (a, b) => {
378
- return a.bold !== b.bold ||
379
- a.italic !== b.italic ||
380
- a.underline !== b.underline ||
381
- a.strikethrough !== b.strikethrough ||
382
- a.size !== b.size ||
383
- a.font !== b.font ||
384
- a.color !== b.color ||
385
- a.backgroundColor !== b.backgroundColor ||
386
- a.subscript !== b.subscript ||
387
- a.superscript !== b.superscript;
388
- };
389
- // Helper to flush current run to paragraph children
390
- const flushRun = () => {
391
- if (currentRunText) {
392
- const node = {
393
- type: 'text',
394
- text: currentRunText,
395
- formatting: { ...currentFormatting }
362
+ const parser = new SimpleRtfParser(buffer);
363
+ const doc = parser.parse();
364
+ // Extract font and color tables
365
+ const fontTable = extractFontTable(doc);
366
+ const colorTable = extractColorTable(doc);
367
+ const content = [];
368
+ const notes = [];
369
+ const attachments = [];
370
+ // State for paragraph construction
371
+ let currentParagraphTextChunks = [];
372
+ let currentParagraphChildren = [];
373
+ let currentParagraphRawChunks = [];
374
+ // State for text run construction
375
+ let currentRunTextChunks = [];
376
+ let currentFormatting = {};
377
+ // Target for content (main body or notes)
378
+ let currentTarget = content;
379
+ // Paragraph-level state
380
+ let paragraphIndent = 0;
381
+ let paragraphAlignment = 'left';
382
+ let isListItem = false;
383
+ let listType;
384
+ let headingLevel;
385
+ let currentListId;
386
+ let currentAnchorIds = [];
387
+ // Persistent list state for listtext/pntext detection
388
+ // These persist across paragraphs to allow list items without explicit \ls
389
+ let lastKnownListId;
390
+ let lastKnownListType;
391
+ let tableStack = [];
392
+ let currentFootnoteId = 0;
393
+ // ═══════════════════════════════════════════════════════════════════
394
+ // Note type tracking (footnotes vs endnotes)
395
+ // ═══════════════════════════════════════════════════════════════════
396
+ // RTF uses \fet to distinguish note types:
397
+ // \fet0 = footnotes only (default)
398
+ // \fet1 = endnotes only
399
+ // \fet2 = both footnotes and endnotes
400
+ let fetValue = 0; // Default to footnotes only
401
+ // Helper to get current table context
402
+ const getCurrentTable = () => tableStack.length > 0 ? tableStack[tableStack.length - 1] : undefined;
403
+ // Helper to ensure a table context exists (for top-level tables)
404
+ const ensureTableContext = () => {
405
+ if (tableStack.length === 0) {
406
+ tableStack.push({
407
+ rows: [],
408
+ currentCells: [],
409
+ currentCellContent: [],
410
+ rowIndex: 0
411
+ });
412
+ }
413
+ };
414
+ let inTable = false;
415
+ let paragraphInTable = false;
416
+ let tableId = 0;
417
+ let rowCellProps = [];
418
+ let currentCellDefinitionProps = { isMergedContinuation: false };
419
+ let cellContentIndex = 0;
420
+ // ═══════════════════════════════════════════════════════════════════
421
+ // List state tracking (Word 97+ uses \ls for list style ID)
422
+ // ═══════════════════════════════════════════════════════════════════
423
+ let listIdCounter = 0;
424
+ const listStyleIdMap = {};
425
+ // List definition state for parsing \listtable
426
+ let parsingListTable = false;
427
+ let parsingListDefinition = false;
428
+ let currentDefinedListId;
429
+ let currentDefinedListType;
430
+ const listTypeMap = {};
431
+ // List override state for parsing \listoverridetable
432
+ let parsingListOverrideTable = false;
433
+ let currentListOverrideListId;
434
+ let currentListOverrideLs;
435
+ const listOverrideMap = {}; // Maps \ls ID to \listid
436
+ // List counters for itemIndex tracking
437
+ // Map: listId -> indentation level -> count
438
+ const listCounters = {};
439
+ // ═══════════════════════════════════════════════════════════════════
440
+ // Hyperlink state (RTF uses \field{\*\fldinst HYPERLINK "url"})
441
+ // ═══════════════════════════════════════════════════════════════════
442
+ let currentLinkUrl;
443
+ // Helper to check if formatting changed
444
+ const formattingChanged = (a, b) => {
445
+ return a.bold !== b.bold ||
446
+ a.italic !== b.italic ||
447
+ a.underline !== b.underline ||
448
+ a.strikethrough !== b.strikethrough ||
449
+ a.size !== b.size ||
450
+ a.font !== b.font ||
451
+ a.color !== b.color ||
452
+ a.backgroundColor !== b.backgroundColor ||
453
+ a.subscript !== b.subscript ||
454
+ a.superscript !== b.superscript;
455
+ };
456
+ // Helper to flush current run to paragraph children
457
+ const flushRun = () => {
458
+ if (currentRunTextChunks.length > 0) {
459
+ const currentRunText = currentRunTextChunks.join('');
460
+ const node = {
461
+ type: 'text',
462
+ text: currentRunText,
463
+ formatting: { ...currentFormatting }
464
+ };
465
+ // Use TextMetadata.link for hyperlinks
466
+ if (currentLinkUrl) {
467
+ node.metadata = {
468
+ link: currentLinkUrl,
469
+ linkType: classifyLinkType(currentLinkUrl)
396
470
  };
397
- // Use TextMetadata.link for hyperlinks
398
- if (currentLinkUrl) {
399
- node.metadata = {
400
- link: currentLinkUrl,
401
- linkType: classifyLinkType(currentLinkUrl)
402
- };
471
+ }
472
+ currentParagraphChildren.push(node);
473
+ currentParagraphTextChunks.push(currentRunText);
474
+ currentRunTextChunks = [];
475
+ }
476
+ };
477
+ let isFlushingTable = false;
478
+ // Helper to flush current paragraph
479
+ const flushParagraph = () => {
480
+ flushRun(); // Ensure last run is added
481
+ // Check if we need to end the table
482
+ // If we were in a table, but this paragraph is NOT marked as in-table,
483
+ // and we have content, then the table has ended.
484
+ const hasContent = currentParagraphTextChunks.length > 0 || currentParagraphChildren.length > 0;
485
+ if (inTable && !paragraphInTable && !isFlushingTable && hasContent) {
486
+ // CRITICAL: Save current paragraph content before flushing table
487
+ // because flushTable() -> flushRow() -> flushCell() -> flushParagraph()
488
+ // would otherwise process this content during the table flush
489
+ const savedParagraphTextChunks = [...currentParagraphTextChunks];
490
+ const savedParagraphChildren = [...currentParagraphChildren];
491
+ const savedParagraphRawChunks = [...currentParagraphRawChunks];
492
+ // Clear buffers so nested flushParagraph() doesn't process them
493
+ currentParagraphTextChunks = [];
494
+ currentParagraphChildren = [];
495
+ currentParagraphRawChunks = [];
496
+ flushTable();
497
+ // Restore the saved content for processing after the table
498
+ currentParagraphTextChunks = savedParagraphTextChunks;
499
+ currentParagraphChildren = savedParagraphChildren;
500
+ currentParagraphRawChunks = savedParagraphRawChunks;
501
+ }
502
+ if (hasContent) {
503
+ const currentParagraphText = currentParagraphTextChunks.join('');
504
+ let nodeType = 'paragraph';
505
+ let metadata = undefined;
506
+ // Heuristic heading detection if no explicit \s style was found
507
+ if (headingLevel === undefined && currentParagraphChildren.length > 0) {
508
+ // Check if the first child is bold and larger than default (12pt)
509
+ const firstChild = currentParagraphChildren[0];
510
+ if (firstChild.type === 'text' && firstChild.formatting?.bold) {
511
+ const size = parseInt(firstChild.formatting.size || '12');
512
+ if (size >= 14 && currentParagraphText.length < 300) {
513
+ if (size >= 22)
514
+ headingLevel = 1;
515
+ else if (size >= 18)
516
+ headingLevel = 2;
517
+ else if (size >= 16)
518
+ headingLevel = 3;
519
+ else
520
+ headingLevel = 4;
521
+ }
403
522
  }
404
- currentParagraphChildren.push(node);
405
- currentParagraphText += currentRunText;
406
- currentRunText = '';
407
523
  }
408
- };
409
- let isFlushingTable = false;
410
- // Helper to flush current paragraph
411
- const flushParagraph = () => {
412
- flushRun(); // Ensure last run is added
413
- // Check if we need to end the table
414
- // If we were in a table, but this paragraph is NOT marked as in-table,
415
- // and we have content, then the table has ended.
416
- const hasContent = currentParagraphText || currentParagraphChildren.length > 0;
417
- if (inTable && !paragraphInTable && !isFlushingTable && hasContent) {
418
- // CRITICAL: Save current paragraph content before flushing table
419
- // because flushTable() -> flushRow() -> flushCell() -> flushParagraph()
420
- // would otherwise process this content during the table flush
421
- const savedParagraphText = currentParagraphText;
422
- const savedParagraphChildren = [...currentParagraphChildren];
423
- const savedParagraphRaw = currentParagraphRaw;
424
- // Clear buffers so nested flushParagraph() doesn't process them
425
- currentParagraphText = '';
426
- currentParagraphChildren = [];
427
- currentParagraphRaw = '';
428
- flushTable();
429
- // Restore the saved content for processing after the table
430
- currentParagraphText = savedParagraphText;
431
- currentParagraphChildren = savedParagraphChildren;
432
- currentParagraphRaw = savedParagraphRaw;
433
- }
434
- if (hasContent) {
435
- let nodeType = 'paragraph';
436
- let metadata = undefined;
437
- if (headingLevel !== undefined && headingLevel > 0) {
438
- nodeType = 'heading';
439
- metadata = { level: headingLevel };
440
- // Reset list context when we encounter a heading
441
- lastKnownListId = undefined;
442
- lastKnownListType = undefined;
443
- }
444
- else if (isListItem) {
445
- nodeType = 'list';
446
- // Use lastKnownListId if currentListId is not set
447
- // (happens when list item is detected via listtext/pntext)
448
- const effectiveListId = currentListId || lastKnownListId;
449
- const effectiveListType = listType || lastKnownListType || 'unordered';
450
- // Calculate itemIndex
451
- let itemIndex = 0;
452
- if (effectiveListId) {
453
- if (!listCounters[effectiveListId]) {
454
- listCounters[effectiveListId] = {};
455
- }
456
- if (listCounters[effectiveListId][paragraphIndent] === undefined) {
457
- listCounters[effectiveListId][paragraphIndent] = 0;
458
- }
459
- else {
460
- listCounters[effectiveListId][paragraphIndent]++;
524
+ // Heuristic list detection if no explicit list control words were found
525
+ if (!isListItem && paragraphIndent > 300) { // RTF indents are in twips (1440 = 1 inch)
526
+ const trimmed = currentParagraphText.trim();
527
+ // Check for bullet characters or digits followed by period
528
+ if (/^[\u2022\u00b7\-\u25cf\u25cb]/.test(trimmed) || /^\d+[.\)]/.test(trimmed)) {
529
+ isListItem = true;
530
+ listType = /^\d+[.\)]/.test(trimmed) ? 'ordered' : 'unordered';
531
+ }
532
+ }
533
+ if (headingLevel !== undefined && headingLevel > 0) {
534
+ nodeType = 'heading';
535
+ metadata = { level: headingLevel };
536
+ // Reset list context when we encounter a heading
537
+ lastKnownListId = undefined;
538
+ lastKnownListType = undefined;
539
+ }
540
+ else if (isListItem) {
541
+ nodeType = 'list';
542
+ // Use lastKnownListId if currentListId is not set
543
+ // (happens when list item is detected via listtext/pntext)
544
+ const effectiveListId = currentListId || lastKnownListId;
545
+ const effectiveListType = listType || lastKnownListType || 'unordered';
546
+ // Calculate itemIndex
547
+ let itemIndex = 0;
548
+ if (effectiveListId) {
549
+ if (!listCounters[effectiveListId]) {
550
+ listCounters[effectiveListId] = {};
551
+ }
552
+ // Reset all sub-level counters when returning to a shallower level
553
+ // This ensures that if we go from level 2 to level 0 and back to level 2,
554
+ // the new level 2 sequence starts from 0.
555
+ const levels = Object.keys(listCounters[effectiveListId]).map(l => parseInt(l));
556
+ for (const level of levels) {
557
+ if (level > paragraphIndent) {
558
+ delete listCounters[effectiveListId][level];
461
559
  }
462
- itemIndex = listCounters[effectiveListId][paragraphIndent];
463
560
  }
464
- metadata = {
465
- listType: effectiveListType,
466
- indentation: paragraphIndent,
467
- listId: effectiveListId || '',
468
- itemIndex: itemIndex,
469
- alignment: paragraphAlignment,
470
- };
471
- // Save for next listtext detection
472
- if (effectiveListId) {
473
- lastKnownListId = effectiveListId;
561
+ if (listCounters[effectiveListId][paragraphIndent] === undefined) {
562
+ listCounters[effectiveListId][paragraphIndent] = 0;
474
563
  }
475
- if (effectiveListType) {
476
- lastKnownListType = effectiveListType;
564
+ else {
565
+ listCounters[effectiveListId][paragraphIndent]++;
477
566
  }
478
- }
479
- const node = {
480
- type: nodeType,
481
- text: currentParagraphText,
482
- children: currentParagraphChildren,
483
- formatting: undefined,
484
- metadata: metadata
567
+ itemIndex = listCounters[effectiveListId][paragraphIndent];
568
+ }
569
+ metadata = {
570
+ listType: effectiveListType,
571
+ indentation: paragraphIndent,
572
+ listId: effectiveListId || '',
573
+ itemIndex: itemIndex,
574
+ alignment: paragraphAlignment,
485
575
  };
486
- if (config.includeRawContent && currentParagraphRaw) {
487
- node.rawContent = currentParagraphRaw;
576
+ // Save for next listtext detection
577
+ if (effectiveListId) {
578
+ lastKnownListId = effectiveListId;
488
579
  }
489
- // If we're building a table, add to current cell
490
- // but ONLY if this paragraph was actually marked as in-table
491
- if (inTable && paragraphInTable) {
492
- ensureTableContext();
493
- getCurrentTable().currentCellContent.push(node);
494
- }
495
- else {
496
- currentTarget.push(node);
580
+ if (effectiveListType) {
581
+ lastKnownListType = effectiveListType;
497
582
  }
498
- currentParagraphText = '';
499
- currentParagraphChildren = [];
500
- currentParagraphRaw = '';
501
- // Reset paragraph-level state
502
- paragraphIndent = 0;
503
- isListItem = false;
504
- listType = undefined;
505
- headingLevel = undefined;
506
- currentListId = undefined;
507
583
  }
508
- };
509
- // Helper to flush current cell
510
- const flushCell = (tableCtx) => {
511
- flushParagraph();
512
- const ctx = tableCtx || getCurrentTable();
513
- if (!ctx)
514
- return undefined;
515
- // Always return a cell node, even if empty, to preserve table structure (grid)
516
- const cellNode = {
517
- type: 'cell',
518
- text: ctx.currentCellContent.map(c => c.text).join('\n'),
519
- children: [...ctx.currentCellContent],
584
+ const node = {
585
+ type: nodeType,
586
+ text: currentParagraphText,
587
+ children: currentParagraphChildren,
588
+ formatting: undefined,
520
589
  metadata: {
521
- row: ctx.rowIndex,
522
- col: ctx.currentCells.length
590
+ ...metadata,
591
+ anchorIds: currentAnchorIds.length > 0 ? [...currentAnchorIds] : undefined
523
592
  }
524
593
  };
525
- ctx.currentCellContent = [];
526
- return cellNode;
527
- };
528
- // Helper to flush current row - creates a row node from collected cells
529
- const flushRow = (tableCtx) => {
530
- const ctx = tableCtx || getCurrentTable();
531
- if (!ctx)
532
- return;
533
- const cell = flushCell(ctx);
534
- // Only add cell if it has content (prevents phantom empty cells during cleanup)
535
- if (cell && (cell.children && cell.children.length > 0 || cell.text)) {
536
- ctx.currentCells.push(cell);
537
- }
538
- if (ctx.currentCells.length > 0) {
539
- const rowNode = {
540
- type: 'row',
541
- text: ctx.currentCells.map(c => c.text).filter(t => t !== '').join(config.newlineDelimiter ?? '\n'),
542
- children: [...ctx.currentCells]
543
- };
544
- ctx.rows.push(rowNode);
545
- ctx.currentCells = [];
546
- ctx.rowIndex++;
594
+ if (config.includeRawContent && currentParagraphRawChunks.length > 0) {
595
+ node.rawContent = currentParagraphRawChunks.join('');
547
596
  }
548
- };
549
- // Helper to flush table
550
- const flushTable = () => {
551
- if (isFlushingTable)
552
- return;
553
- isFlushingTable = true;
554
- const ctx = getCurrentTable();
555
- if (!ctx) {
556
- isFlushingTable = false;
557
- return;
597
+ // If we're building a table, add to current cell
598
+ // but ONLY if this paragraph was actually marked as in-table
599
+ if (inTable && paragraphInTable) {
600
+ ensureTableContext();
601
+ getCurrentTable().currentCellContent.push(node);
558
602
  }
559
- flushRow(ctx);
560
- if (ctx.rows.length > 0) {
561
- tableId++;
562
- const tableNode = {
563
- type: 'table',
564
- text: ctx.rows.map(r => r.text).join('\n'), // Aggregate text from rows
565
- children: [...ctx.rows]
566
- };
567
- // If we have a parent table, add this table to the parent's current cell
568
- if (tableStack.length > 1) {
569
- const parentCtx = tableStack[tableStack.length - 2];
570
- parentCtx.currentCellContent.push(tableNode);
571
- }
572
- else {
573
- currentTarget.push(tableNode);
574
- }
603
+ else {
604
+ currentTarget.push(node);
575
605
  }
576
- // Pop the table from stack
577
- tableStack.pop();
578
- // If stack is empty, we are out of table mode
579
- if (tableStack.length === 0) {
580
- inTable = false;
581
- paragraphInTable = false;
606
+ currentParagraphTextChunks = [];
607
+ currentParagraphChildren = [];
608
+ currentParagraphRawChunks = [];
609
+ // Reset paragraph-level state that should NOT persist
610
+ // Note: list properties (\ls, \ilvl, \li) and alignment (\ql, etc.)
611
+ // persist in RTF until \pard or a new value is set.
612
+ currentAnchorIds = []; // Reset anchors
613
+ }
614
+ };
615
+ // Helper to flush current cell
616
+ const flushCell = (tableCtx) => {
617
+ flushParagraph();
618
+ const ctx = tableCtx || getCurrentTable();
619
+ if (!ctx)
620
+ return undefined;
621
+ // Always return a cell node, even if empty, to preserve table structure (grid)
622
+ const cellNode = {
623
+ type: 'cell',
624
+ text: ctx.currentCellContent.map(c => c.text).join('\n'),
625
+ children: [...ctx.currentCellContent],
626
+ metadata: {
627
+ row: ctx.rowIndex,
628
+ col: ctx.currentCells.length
582
629
  }
583
- isFlushingTable = false;
584
630
  };
585
- // Extract hyperlink URL from field instruction group
586
- // Recursively searches for HYPERLINK "url" pattern in nested groups
587
- const extractHyperlinkUrl = (group) => {
588
- let url;
589
- // Helper to recursively find hyperlink URL
590
- const findUrl = (node) => {
591
- if (node.type === 'text') {
592
- // Check for HYPERLINK "url" pattern
593
- const text = node.value;
594
- const match = text.match(/HYPERLINK\s+"([^"]+)"/i);
595
- if (match) {
596
- return match[1];
597
- }
598
- // Also check for URL after HYPERLINK on same or separate text node
599
- const urlOnlyMatch = text.match(/"(https?:\/\/[^"]+|mailto:[^"]+|#[^"]+)"/);
600
- if (urlOnlyMatch) {
601
- return urlOnlyMatch[1];
602
- }
631
+ ctx.currentCellContent = [];
632
+ return cellNode;
633
+ };
634
+ // Helper to flush current row - creates a row node from collected cells
635
+ const flushRow = (tableCtx) => {
636
+ const ctx = tableCtx || getCurrentTable();
637
+ if (!ctx)
638
+ return;
639
+ const cell = flushCell(ctx);
640
+ // Only add cell if it has content (prevents phantom empty cells during cleanup)
641
+ if (cell && (cell.children && cell.children.length > 0 || cell.text)) {
642
+ ctx.currentCells.push(cell);
643
+ }
644
+ if (ctx.currentCells.length > 0) {
645
+ const rowNode = {
646
+ type: 'row',
647
+ text: ctx.currentCells.map(c => c.text).filter(t => t !== '').join(config.newlineDelimiter),
648
+ children: [...ctx.currentCells]
649
+ };
650
+ ctx.rows.push(rowNode);
651
+ ctx.currentCells = [];
652
+ ctx.rowIndex++;
653
+ }
654
+ };
655
+ // Helper to flush table
656
+ const flushTable = () => {
657
+ if (isFlushingTable)
658
+ return;
659
+ isFlushingTable = true;
660
+ const ctx = getCurrentTable();
661
+ if (!ctx) {
662
+ isFlushingTable = false;
663
+ return;
664
+ }
665
+ flushRow(ctx);
666
+ if (ctx.rows.length > 0) {
667
+ tableId++;
668
+ const tableNode = {
669
+ type: 'table',
670
+ text: ctx.rows.map(r => r.text).join('\n'), // Aggregate text from rows
671
+ children: [...ctx.rows]
672
+ };
673
+ // If we have a parent table, add this table to the parent's current cell
674
+ if (tableStack.length > 1) {
675
+ const parentCtx = tableStack[tableStack.length - 2];
676
+ parentCtx.currentCellContent.push(tableNode);
677
+ }
678
+ else {
679
+ currentTarget.push(tableNode);
680
+ }
681
+ }
682
+ // Pop the table from stack
683
+ tableStack.pop();
684
+ // If stack is empty, we are out of table mode
685
+ if (tableStack.length === 0) {
686
+ inTable = false;
687
+ paragraphInTable = false;
688
+ }
689
+ isFlushingTable = false;
690
+ };
691
+ // Extract hyperlink URL from field instruction group
692
+ // Recursively searches for HYPERLINK "url" pattern in nested groups
693
+ const extractHyperlinkUrl = (group) => {
694
+ let url;
695
+ // Helper to recursively find hyperlink URL
696
+ const findUrl = (node) => {
697
+ if (node.type === 'text') {
698
+ // Check for HYPERLINK "url" pattern
699
+ const text = node.value;
700
+ const match = text.match(/HYPERLINK\s+"([^"]+)"/i);
701
+ if (match) {
702
+ return match[1];
703
+ }
704
+ // Also check for URL after HYPERLINK on same or separate text node
705
+ const urlOnlyMatch = text.match(/"(https?:\/\/[^"]+|mailto:[^"]+|#[^"]+)"/);
706
+ if (urlOnlyMatch) {
707
+ return urlOnlyMatch[1];
603
708
  }
604
- else if (node.type === 'group') {
605
- // Check if this is a fldinst group (contains field instruction)
606
- let foundHyperlink = false;
607
- let foundUrl;
608
- for (const child of node.content) {
609
- if (child.type === 'text') {
610
- const text = child.value.toUpperCase();
611
- if (text.includes('HYPERLINK')) {
612
- foundHyperlink = true;
613
- }
614
- // Look for quoted URL
615
- const urlMatch = child.value.match(/"([^"]+)"/);
616
- if (urlMatch && foundHyperlink) {
617
- foundUrl = urlMatch[1];
618
- }
709
+ }
710
+ else if (node.type === 'group') {
711
+ // Check if this is a fldinst group (contains field instruction)
712
+ let foundHyperlink = false;
713
+ let foundUrl;
714
+ let isLocal = false;
715
+ for (const child of node.content) {
716
+ if (child.type === 'text') {
717
+ const text = child.value.toUpperCase();
718
+ if (text.includes('HYPERLINK')) {
719
+ foundHyperlink = true;
619
720
  }
620
- else if (child.type === 'group') {
621
- const nestedUrl = findUrl(child);
622
- if (nestedUrl) {
623
- foundUrl = nestedUrl;
624
- }
721
+ if (text.includes('\\L')) {
722
+ isLocal = true;
723
+ }
724
+ // Look for quoted URL/Anchor
725
+ const urlMatch = child.value.match(/"([^"]+)"/);
726
+ if (urlMatch && foundHyperlink) {
727
+ foundUrl = urlMatch[1];
728
+ }
729
+ }
730
+ else if (child.type === 'group') {
731
+ const nestedUrl = findUrl(child);
732
+ if (nestedUrl) {
733
+ foundUrl = nestedUrl;
625
734
  }
626
735
  }
627
- if (foundUrl)
628
- return foundUrl;
629
736
  }
630
- return undefined;
631
- };
632
- // Search the field group for fldinst
633
- for (const child of group.content) {
634
- if (child.type === 'group') {
635
- const foundUrl = findUrl(child);
636
- if (foundUrl)
637
- return foundUrl;
737
+ if (foundUrl) {
738
+ if (isLocal && !foundUrl.startsWith('#')) {
739
+ return '#' + foundUrl;
740
+ }
741
+ return foundUrl;
638
742
  }
639
743
  }
640
- return url;
744
+ return undefined;
641
745
  };
642
- /**
643
- * Determines whether a hyperlink URL is internal or external.
644
- * Internal = bookmark references (no scheme or starts with "#").
645
- * External = any scheme like http, https, mailto, ftp, file, etc.
646
- */
647
- const classifyLinkType = (url) => {
648
- // Trim whitespace
649
- const clean = url.trim();
650
- // Internal pattern 1: starts with "#"
651
- if (clean.startsWith('#')) {
652
- return 'internal';
653
- }
654
- // Internal pattern 2: no scheme at all (pure bookmark)
655
- // Detect schemes by checking "something:" prefix
656
- if (!/^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(clean)) {
657
- return 'internal';
658
- }
659
- // Everything else is external
660
- return 'external';
661
- };
662
- // Helper to extract text content from a group (for list marker detection)
663
- // Recursively collects all text content from a group
664
- const extractTextFromGroup = (group) => {
665
- let text = '';
666
- const collectText = (node) => {
667
- if (node.type === 'text') {
668
- text += node.value;
669
- }
670
- else if (node.type === 'group') {
671
- for (const child of node.content) {
672
- collectText(child);
673
- }
746
+ // Search the field group for fldinst
747
+ for (const child of group.content) {
748
+ if (child.type === 'group') {
749
+ const foundUrl = findUrl(child);
750
+ if (foundUrl)
751
+ return foundUrl;
752
+ }
753
+ }
754
+ return url;
755
+ };
756
+ /**
757
+ * Determines whether a hyperlink URL is internal or external.
758
+ * Internal = bookmark references (no scheme or starts with "#").
759
+ * External = any scheme like http, https, mailto, ftp, file, etc.
760
+ */
761
+ const classifyLinkType = (url) => {
762
+ // Trim whitespace
763
+ const clean = url.trim();
764
+ // Internal pattern 1: starts with "#"
765
+ if (clean.startsWith('#')) {
766
+ return 'internal';
767
+ }
768
+ // Internal pattern 2: no scheme at all (pure bookmark)
769
+ // Detect schemes by checking "something:" prefix
770
+ if (!/^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(clean)) {
771
+ return 'internal';
772
+ }
773
+ // Everything else is external
774
+ return 'external';
775
+ };
776
+ // Helper to extract text content from a group (for list marker detection)
777
+ // Recursively collects all text content from a group
778
+ const extractTextFromGroup = (group) => {
779
+ let text = '';
780
+ const collectText = (node) => {
781
+ if (node.type === 'text') {
782
+ text += node.value;
783
+ }
784
+ else if (node.type === 'group') {
785
+ for (const child of node.content) {
786
+ collectText(child);
674
787
  }
675
- };
676
- for (const child of group.content) {
677
- collectText(child);
678
788
  }
679
- return text;
680
789
  };
681
- /**
682
- * Extracts an image attachment from an RTF \pict group.
683
- * Uses lookup tables for clean and safe handling of all formats.
684
- *
685
- * @param pictGroup The RtfGroup node that represents a \pict group.
686
- * @returns An OfficeAttachment or undefined when unsupported or invalid.
687
- */
688
- const extractPictAttachment = (pictGroup) => {
689
- /** Internal format detected from the RTF pict group */
690
- let imageFormat;
691
- /** Hexadecimal string extracted from the pict binary section */
692
- let hexData = '';
693
- // -------------------------------------------------------------
694
- // Walk through pict group content to detect the blip type and gather hex data
695
- // -------------------------------------------------------------
696
- for (const child of pictGroup.content) {
697
- // If the node is a control word, we try to resolve it from lookup map
698
- if (child.type === 'control') {
699
- // Lookup directly instead of if/else
700
- const mapped = RTF_BLIP_MAP[child.value];
701
- if (mapped) {
702
- imageFormat = mapped;
703
- }
704
- }
705
- else if (child.type === 'text') {
706
- // Append only valid hex characters
707
- hexData += child.value.replace(/[^0-9a-fA-F]/g, '');
708
- }
709
- else if (child.type === 'group') {
710
- // Skip nested structures like picprop or blipuid
790
+ for (const child of group.content) {
791
+ collectText(child);
792
+ }
793
+ return text;
794
+ };
795
+ /**
796
+ * Extracts an image attachment from an RTF \pict group.
797
+ * Uses lookup tables for clean and safe handling of all formats.
798
+ *
799
+ * @param pictGroup The RtfGroup node that represents a \pict group.
800
+ * @returns An OfficeAttachment or undefined when unsupported or invalid.
801
+ */
802
+ const extractPictAttachment = (pictGroup) => {
803
+ /** Internal format detected from the RTF pict group */
804
+ let imageFormat;
805
+ /** Hexadecimal string chunks extracted from the pict binary section */
806
+ const hexDataChunks = [];
807
+ // -------------------------------------------------------------
808
+ // Walk through pict group content to detect the blip type and gather hex data
809
+ // -------------------------------------------------------------
810
+ for (const child of pictGroup.content) {
811
+ // If the node is a control word, we try to resolve it from lookup map
812
+ if (child.type === 'control') {
813
+ // Lookup directly instead of if/else
814
+ const mapped = RTF_BLIP_MAP[child.value];
815
+ if (mapped) {
816
+ imageFormat = mapped;
711
817
  }
712
818
  }
713
- // Missing or unknown format → stop
714
- if (!imageFormat || hexData.length === 0) {
715
- return undefined;
819
+ else if (child.type === 'text') {
820
+ // Append only valid hex characters
821
+ hexDataChunks.push(child.value.replace(/[^0-9a-fA-F]/g, ''));
716
822
  }
717
- // Map internal format to MIME
718
- const mimeType = IMAGE_MIME_MAP[imageFormat];
719
- if (!mimeType) {
720
- return undefined;
823
+ else if (child.type === 'group') {
824
+ // Skip nested structures like picprop or blipuid
721
825
  }
722
- // -------------------------------------------------------------
723
- // Convert hex → binary and construct the attachment object
724
- // -------------------------------------------------------------
725
- try {
726
- // Convert hex into a raw buffer
727
- const buffer = Buffer.from(hexData, 'hex');
728
- // Derive file extension directly from format
729
- const extension = imageFormat;
730
- // Generate a stable incremental filename
731
- const name = `image_${attachments.length + 1}.${extension}`;
732
- // Build and return final attachment
733
- return {
734
- type: 'image',
735
- mimeType: mimeType,
736
- data: buffer.toString('base64'),
737
- name: name,
738
- extension: extension
739
- };
826
+ }
827
+ // Missing or unknown format → stop
828
+ const hexData = hexDataChunks.join('');
829
+ if (!imageFormat || hexData.length === 0) {
830
+ return undefined;
831
+ }
832
+ // Map internal format to MIME
833
+ const mimeType = IMAGE_MIME_MAP[imageFormat];
834
+ if (!mimeType) {
835
+ return undefined;
836
+ }
837
+ // -------------------------------------------------------------
838
+ // Convert hex → binary and construct the attachment object
839
+ // -------------------------------------------------------------
840
+ try {
841
+ // Convert hex into a raw buffer
842
+ const buffer = Buffer.from(hexData, 'hex');
843
+ // Derive file extension directly from format
844
+ const extension = imageFormat;
845
+ // Generate a stable incremental filename
846
+ const name = `image_${attachments.length + 1}.${extension}`;
847
+ // Build and return final attachment
848
+ return {
849
+ type: 'image',
850
+ mimeType: mimeType,
851
+ data: buffer.toString('base64'),
852
+ name: name,
853
+ extension: extension
854
+ };
855
+ }
856
+ catch {
857
+ // If conversion fails, ignore this image
858
+ return undefined;
859
+ }
860
+ };
861
+ // Helper to serialize RTF control word
862
+ const serializeRtfControl = (node) => {
863
+ // Symbol control words (non-alpha)
864
+ if (!/^[a-zA-Z]/.test(node.value)) {
865
+ return `\\${node.value}`;
866
+ }
867
+ // Alpha control words
868
+ let res = `\\${node.value}`;
869
+ if (node.param !== undefined) {
870
+ res += node.param;
871
+ }
872
+ // Add space delimiter for safety
873
+ res += ' ';
874
+ return res;
875
+ };
876
+ // Helper to extract text from a group (for bookmark names, etc.)
877
+ const extractGroupText = (group) => {
878
+ let text = '';
879
+ for (const item of group.content) {
880
+ if (item.type === 'text') {
881
+ text += item.value;
740
882
  }
741
- catch {
742
- // If conversion fails, ignore this image
743
- return undefined;
883
+ else if (item.type === 'group') {
884
+ text += extractGroupText(item);
744
885
  }
745
- };
746
- // Helper to serialize RTF control word
747
- const serializeRtfControl = (node) => {
748
- // Symbol control words (non-alpha)
749
- if (!/^[a-zA-Z]/.test(node.value)) {
750
- return `\\${node.value}`;
751
- }
752
- // Alpha control words
753
- let res = `\\${node.value}`;
754
- if (node.param !== undefined) {
755
- res += node.param;
756
- }
757
- // Add space delimiter for safety
758
- res += ' ';
759
- return res;
760
- };
761
- // Helper to serialize RTF text
762
- const serializeRtfText = (node) => {
763
- // Escape special characters: \, {, }
764
- return node.value.replace(/([\\{}])/g, '\\$1');
765
- };
766
- // Recursive function to traverse the RTF tree
767
- const traverse = (node, formatting, depth = 0) => {
768
- if (node.type === 'group') {
769
- const ignoreList = [
770
- 'fonttbl', 'colortbl', 'stylesheet', 'info', 'macpict',
771
- 'pmmetafile', 'wmetafile', 'dibitmap', 'bitmap', 'object',
772
- 'nextGenerator', 'header', 'footer', 'nonshppict', 'xml', 'private',
773
- 'upnp', 'ud', 'filetbl', 'operator', 'author', 'creatim', 'revtim', 'printim', 'comment',
774
- 'fldinst', 'listtext', 'pntext' // Ignore list marker text (handled separately)
775
- ];
776
- let isIgnored = false;
777
- let isFootnote = false;
778
- let isHyperlinkField = false;
779
- let isPict = false;
780
- // Add group start to raw content
781
- // Note: We don't add ignored groups to rawContent to keep it clean
782
- // But we might want to if we want full fidelity.
783
- // For now, let's include everything in rawContent except truly skipped stuff?
784
- // The user asked for "raw content which is probably the rtf group".
785
- // If we skip 'fonttbl', it's fine as it's not part of the content.
786
- // But 'listtext' IS part of the content structure even if we parse it separately.
787
- // Let's stick to the plan: if ignored, we might skip it in rawContent too,
788
- // OR we include it.
789
- // If I include it, `currentParagraphRaw` might get huge with font tables if they were inside the paragraph (unlikely).
790
- // Usually font tables are at document root.
791
- // `traverse` is called on `doc`.
792
- // `currentParagraphRaw` is reset on `flushParagraph`.
793
- // So if we are at root level, `currentParagraphRaw` accumulates everything until the first paragraph ends.
794
- // This might include the header/fonttbl if they are before the first \par.
795
- // That seems correct for "raw content" of the first node?
796
- // Actually, `fonttbl` is usually before any text.
797
- // If we include it, the first paragraph node will contain the entire font table in its rawContent.
798
- // That might be annoying.
799
- // Let's ONLY add to `currentParagraphRaw` if NOT ignored.
800
- if (node.destination) {
801
- if (node.destination === 'footnote') {
802
- isFootnote = true;
803
- }
804
- else if (node.destination === 'field') {
805
- // Check if this is a hyperlink field
806
- const url = extractHyperlinkUrl(node);
807
- if (url) {
808
- // Flush any pending text before starting the link context
809
- // This prevents previous text from inheriting the link
810
- flushRun();
811
- isHyperlinkField = true;
812
- currentLinkUrl = url;
813
- }
814
- }
815
- else if (node.destination === 'listtable') {
816
- // We want to parse list definitions
817
- parsingListTable = true;
886
+ }
887
+ return text.trim();
888
+ };
889
+ // Helper to serialize RTF text
890
+ const serializeRtfText = (node) => {
891
+ // Escape special characters: \, {, }
892
+ return node.value.replace(/([\\{}])/g, '\\$1');
893
+ };
894
+ // Recursive function to traverse the RTF tree
895
+ const traverse = (node, formatting, depth = 0) => {
896
+ if (node.type === 'group') {
897
+ const ignoreList = [
898
+ 'fonttbl', 'colortbl', 'stylesheet', 'info', 'macpict',
899
+ 'pmmetafile', 'wmetafile', 'dibitmap', 'bitmap', 'object',
900
+ 'nextGenerator', 'header', 'footer', 'nonshppict', 'xml', 'private',
901
+ 'upnp', 'ud', 'filetbl', 'operator', 'author', 'creatim', 'revtim', 'printim', 'comment',
902
+ 'fldinst', 'listtext', 'pntext' // Ignore list marker text (handled separately)
903
+ ];
904
+ let isIgnored = false;
905
+ let isFootnote = false;
906
+ let isHyperlinkField = false;
907
+ let isPict = false;
908
+ // Add group start to raw content
909
+ // Note: We don't add ignored groups to rawContent to keep it clean
910
+ // But we might want to if we want full fidelity.
911
+ // For now, let's include everything in rawContent except truly skipped stuff?
912
+ // The user asked for "raw content which is probably the rtf group".
913
+ // If we skip 'fonttbl', it's fine as it's not part of the content.
914
+ // But 'listtext' IS part of the content structure even if we parse it separately.
915
+ // Let's stick to the plan: if ignored, we might skip it in rawContent too,
916
+ // OR we include it.
917
+ // If I include it, `currentParagraphRaw` might get huge with font tables if they were inside the paragraph (unlikely).
918
+ // Usually font tables are at document root.
919
+ // `traverse` is called on `doc`.
920
+ // `currentParagraphRaw` is reset on `flushParagraph`.
921
+ // So if we are at root level, `currentParagraphRaw` accumulates everything until the first paragraph ends.
922
+ // This might include the header/fonttbl if they are before the first \par.
923
+ // That seems correct for "raw content" of the first node?
924
+ // Actually, `fonttbl` is usually before any text.
925
+ // If we include it, the first paragraph node will contain the entire font table in its rawContent.
926
+ // That might be annoying.
927
+ // Let's ONLY add to `currentParagraphRaw` if NOT ignored.
928
+ if (node.destination) {
929
+ if (node.destination === 'footnote') {
930
+ isFootnote = true;
931
+ }
932
+ else if (node.destination === 'field') {
933
+ // Check if this is a hyperlink field
934
+ const url = extractHyperlinkUrl(node);
935
+ if (url) {
936
+ // Flush any pending text before starting the link context
937
+ // This prevents previous text from inheriting the link
938
+ flushRun();
939
+ isHyperlinkField = true;
940
+ currentLinkUrl = url;
818
941
  }
819
- else if (parsingListTable && node.destination === 'list') {
820
- parsingListDefinition = true;
821
- currentDefinedListId = undefined;
822
- currentDefinedListType = undefined;
942
+ }
943
+ else if (node.destination === 'listtable') {
944
+ // We want to parse list definitions
945
+ parsingListTable = true;
946
+ }
947
+ else if (parsingListTable && node.destination === 'list') {
948
+ parsingListDefinition = true;
949
+ currentDefinedListId = undefined;
950
+ currentDefinedListType = undefined;
951
+ }
952
+ else if (node.destination === 'listoverridetable') {
953
+ parsingListOverrideTable = true;
954
+ }
955
+ else if (parsingListOverrideTable && node.destination === 'listoverride') {
956
+ // Reset per override group
957
+ currentListOverrideListId = undefined;
958
+ currentListOverrideLs = undefined;
959
+ }
960
+ else if (node.destination === 'pict') {
961
+ // Handle picture extraction
962
+ if (config.extractAttachments) {
963
+ isPict = true;
823
964
  }
824
- else if (node.destination === 'listoverridetable') {
825
- parsingListOverrideTable = true;
965
+ else {
966
+ isIgnored = true;
826
967
  }
827
- else if (parsingListOverrideTable && node.destination === 'listoverride') {
828
- // Reset per override group
829
- currentListOverrideListId = undefined;
830
- currentListOverrideLs = undefined;
968
+ }
969
+ else if (ignoreList.includes(node.destination)) {
970
+ isIgnored = true;
971
+ }
972
+ else if (node.content.length > 0 && node.content[0].type === 'control' && node.content[0].value === '*') {
973
+ // Ignorable destination, but allow certain ones for:
974
+ // - fldinst: hyperlinks
975
+ // - nesttableprops: nested tables
976
+ // - shppict: shape pictures (contain pict groups)
977
+ const allowedIgnorable = ['fldinst', 'nesttableprops'];
978
+ if (config.extractAttachments) {
979
+ allowedIgnorable.push('shppict', 'listpicture');
831
980
  }
832
- else if (node.destination === 'pict') {
833
- // Handle picture extraction
834
- if (config.extractAttachments) {
835
- isPict = true;
836
- }
837
- else {
981
+ if (!allowedIgnorable.includes(node.destination || '')) {
982
+ if (node.destination === 'bkmkstart') {
983
+ const name = extractGroupText(node);
984
+ if (name && !currentAnchorIds.includes(name)) {
985
+ currentAnchorIds.push(name);
986
+ }
838
987
  isIgnored = true;
839
988
  }
840
- }
841
- else if (ignoreList.includes(node.destination)) {
842
- isIgnored = true;
843
- }
844
- else if (node.content.length > 0 && node.content[0].type === 'control' && node.content[0].value === '*') {
845
- // Ignorable destination, but allow certain ones for:
846
- // - fldinst: hyperlinks
847
- // - nesttableprops: nested tables
848
- // - shppict: shape pictures (contain pict groups)
849
- const allowedIgnorable = ['fldinst', 'nesttableprops'];
850
- if (config.extractAttachments) {
851
- allowedIgnorable.push('shppict', 'listpicture');
852
- }
853
- if (!allowedIgnorable.includes(node.destination || '')) {
989
+ else if (node.destination === 'bkmkend') {
854
990
  isIgnored = true;
855
991
  }
856
- }
857
- }
858
- // ═══════════════════════════════════════════════════════════
859
- // Handle listtext and pntext: These indicate the current paragraph
860
- // is a list item. We extract list type info before ignoring content.
861
- // ═══════════════════════════════════════════════════════════
862
- if (node.destination === 'listtext' || node.destination === 'pntext') {
863
- // If this is the first list item, reset the indent
864
- if (!isListItem)
865
- paragraphIndent = 0;
866
- isListItem = true;
867
- // Try to determine list type from the marker content
868
- // Bullets (unordered): '·', '•', 'o', '§', etc.
869
- // Numbers (ordered): '1.', '2.', 'i.', 'ii.', 'a.', 'A.', etc.
870
- const markerText = extractTextFromGroup(node);
871
- if (markerText) {
872
- const trimmed = markerText.trim();
873
- // Check for common bullet characters
874
- const bulletChars = ['·', '•', 'o', '§', '■', '□', '●', '○', '◆', '◇', '►', '▸', '\u00b7', '\u2022', '\u25cf', '\u25cb'];
875
- const isBullet = bulletChars.some(b => trimmed.includes(b)) ||
876
- // Font symbol bullets often use characters from Symbol font
877
- (trimmed.length === 1 && !/[0-9a-zA-Z]/.test(trimmed));
878
- if (isBullet) {
879
- listType = 'unordered';
880
- }
881
- else if (/^[0-9ivxlcdm]+[\.\)]/i.test(trimmed) || /^[a-z][\.\)]/i.test(trimmed)) {
882
- // Matches: 1., 2), i., ii., a., A), etc.
883
- listType = 'ordered';
992
+ else {
993
+ isIgnored = true;
884
994
  }
885
- // If we can't determine, leave listType as is (might be set by \ls/\levelnfc)
886
995
  }
887
- // Still mark as ignored to skip the marker text content
888
- isIgnored = true;
889
996
  }
890
- if (isIgnored)
891
- return;
892
- // Append group start to raw content
893
- currentParagraphRaw += '{';
894
- // Handle pict group: extract image and add to content tree
895
- if (isPict) {
896
- const attachment = extractPictAttachment(node);
897
- if (attachment) {
898
- attachments.push(attachment);
899
- // Only add image node to content if this is NOT a list definition picture
900
- // List pictures (bullets) should not appear in content, only as attachments
901
- if (!parsingListTable && !parsingListDefinition) {
902
- // Also add an image node to the content tree (like DOCX)
903
- flushParagraph();
904
- currentTarget.push({
905
- type: 'image',
906
- text: '',
907
- metadata: {
908
- attachmentName: attachment.name || `image_${attachments.length}`
909
- }
910
- });
911
- }
912
- }
913
- // We still traverse pict content to reconstruct raw RTF?
914
- // No, extractPictAttachment consumes it.
915
- // But we want it in rawContent?
916
- // If we return here, we miss the closing '}'.
917
- // And we miss the content in rawContent.
918
- // Let's traverse it purely for rawContent if needed, but `extractPictAttachment` doesn't modify the tree.
919
- // But `extractPictAttachment` does not return the raw string.
920
- // So we should probably continue traversal but suppress text extraction?
921
- // The original code returned here: `return; // Don't traverse pict content as text`
922
- // So we should do the same, but we need to append the content to `currentParagraphRaw`.
923
- // We can manually serialize the group content here.
924
- for (const child of node.content) {
925
- if (child.type === 'control')
926
- currentParagraphRaw += serializeRtfControl(child);
927
- else if (child.type === 'text')
928
- currentParagraphRaw += serializeRtfText(child);
929
- else if (child.type === 'group') {
930
- // Recursive serialization for nested groups in pict (e.g. blipuid)
931
- // We can't easily recurse `traverse` because it has side effects (text extraction).
932
- // We need a pure serializer or just let `traverse` run but with a flag?
933
- // Or just ignore the raw content of the image binary data?
934
- // Image binary data can be huge.
935
- // Maybe we shouldn't include the full hex dump in `rawContent`?
936
- // The user said "raw content which is probably the rtf group".
937
- // Including 5MB of hex data in the JSON AST might be bad.
938
- // But for consistency, it is the raw content.
939
- // Let's include it for now.
940
- // To do this without side effects, we need a separate serialize function?
941
- // Or just call traverse and ensure `isPict` logic prevents text extraction.
942
- // Wait, `isPict` is true for this node.
943
- // If we recurse, `isPict` will be false for children (unless they are also pict).
944
- // But we want to suppress text extraction for children of pict.
945
- // The original code did `return`.
946
- // So we should manually serialize children here.
947
- // Let's define a simple recursive serializer.
948
- const serializeGroupContent = (g) => {
949
- for (const c of g.content) {
950
- if (c.type === 'control')
951
- currentParagraphRaw += serializeRtfControl(c);
952
- else if (c.type === 'text')
953
- currentParagraphRaw += serializeRtfText(c);
954
- else if (c.type === 'group') {
955
- currentParagraphRaw += '{';
956
- serializeGroupContent(c);
957
- currentParagraphRaw += '}';
958
- }
959
- }
960
- };
961
- serializeGroupContent(node);
962
- }
963
- }
964
- currentParagraphRaw += '}';
965
- return;
966
- }
967
- // Handle footnote: switch target to notes
968
- const previousTarget = currentTarget;
969
- if (isFootnote) {
970
- if (config.ignoreNotes) {
971
- return; // Skip footnote content entirely
997
+ }
998
+ // ═══════════════════════════════════════════════════════════
999
+ // Handle listtext and pntext: These indicate the current paragraph
1000
+ // is a list item. We extract list type info before ignoring content.
1001
+ // ═══════════════════════════════════════════════════════════
1002
+ if (node.destination === 'listtext' || node.destination === 'pntext') {
1003
+ // If this is the first list item, reset the indent
1004
+ if (!isListItem)
1005
+ paragraphIndent = 0;
1006
+ isListItem = true;
1007
+ // Try to determine list type from the marker content
1008
+ // Bullets (unordered): '·', '•', 'o', '§', etc.
1009
+ // Numbers (ordered): '1.', '2.', 'i.', 'ii.', 'a.', 'A.', etc.
1010
+ const markerText = extractTextFromGroup(node);
1011
+ if (markerText) {
1012
+ const trimmed = markerText.trim();
1013
+ // Check for common bullet characters
1014
+ const bulletChars = ['·', '•', 'o', '§', '■', '□', '●', '○', '◆', '◇', '►', '▸', '\u00b7', '\u2022', '\u25cf', '\u25cb'];
1015
+ const isBullet = bulletChars.some(b => trimmed.includes(b)) ||
1016
+ // Font symbol bullets often use characters from Symbol font
1017
+ (trimmed.length === 1 && !/[0-9a-zA-Z]/.test(trimmed));
1018
+ if (isBullet) {
1019
+ listType = 'unordered';
972
1020
  }
973
- flushParagraph();
974
- currentFootnoteId++;
975
- // Determine note type based on \fet value
976
- let noteType = 'footnote';
977
- if (fetValue === 1) {
978
- // \fet1 means all notes are endnotes
979
- noteType = 'endnote';
1021
+ else if (/^[0-9ivxlcdm]+[\.\)]/i.test(trimmed) || /^[a-z][\.\)]/i.test(trimmed)) {
1022
+ // Matches: 1., 2), i., ii., a., A), etc.
1023
+ listType = 'ordered';
980
1024
  }
981
- else if (fetValue === 2) {
982
- // \fet2 means both types exist
983
- // Check for \ftnalt marker to distinguish endnotes from footnotes
984
- // \footnote\ftnalt indicates an endnote
985
- const hasFtnalt = node.content.some(child => child.type === 'control' && child.value === 'ftnalt');
986
- noteType = hasFtnalt ? 'endnote' : 'footnote';
1025
+ // If we can't determine, leave listType as is (might be set by \ls/\levelnfc)
1026
+ }
1027
+ // Still mark as ignored to skip the marker text content
1028
+ isIgnored = true;
1029
+ }
1030
+ if (isIgnored)
1031
+ return;
1032
+ // Append group start to raw content
1033
+ currentParagraphRawChunks.push('{');
1034
+ // Handle pict group: extract image and add to content tree
1035
+ if (isPict) {
1036
+ const attachment = extractPictAttachment(node);
1037
+ if (attachment) {
1038
+ attachments.push(attachment);
1039
+ // Only add image node to content if this is NOT a list definition picture
1040
+ // List pictures (bullets) should not appear in content, only as attachments
1041
+ if (!parsingListTable && !parsingListDefinition) {
1042
+ // Also add an image node to the content tree (like DOCX)
1043
+ flushParagraph();
1044
+ currentTarget.push({
1045
+ type: 'image',
1046
+ text: '',
1047
+ metadata: {
1048
+ attachmentName: attachment.name || `image_${attachments.length}`
1049
+ }
1050
+ });
987
1051
  }
988
- // fetValue === 0 (default) means footnotes only
989
- const noteNode = {
990
- type: 'note',
991
- children: [],
992
- metadata: {
993
- noteId: currentFootnoteId.toString(),
994
- noteType: noteType
995
- }
996
- };
997
- notes.push(noteNode);
998
- currentTarget = noteNode.children;
999
1052
  }
1000
- // Create a new formatting context for the group
1001
- const groupFormatting = { ...formatting };
1053
+ // We still traverse pict content to reconstruct raw RTF?
1054
+ // No, extractPictAttachment consumes it.
1055
+ // But we want it in rawContent?
1056
+ // If we return here, we miss the closing '}'.
1057
+ // And we miss the content in rawContent.
1058
+ // Let's traverse it purely for rawContent if needed, but `extractPictAttachment` doesn't modify the tree.
1059
+ // But `extractPictAttachment` does not return the raw string.
1060
+ // So we should probably continue traversal but suppress text extraction?
1061
+ // The original code returned here: `return; // Don't traverse pict content as text`
1062
+ // So we should do the same, but we need to append the content to `currentParagraphRaw`.
1063
+ // We can manually serialize the group content here.
1002
1064
  for (const child of node.content) {
1003
- // Skip fldinst groups (we already extracted the URL)
1004
- if (child.type === 'group' && child.destination === 'fldinst') {
1005
- // We still want it in rawContent!
1006
- // So we should traverse it but suppress text extraction?
1007
- // Or just serialize it?
1008
- // `fldinst` contains the URL.
1009
- // If we skip it in `traverse`, we miss it in `rawContent`.
1010
- // Let's traverse it but maybe the `fldinst` logic inside `traverse` handles it?
1011
- // The original code:
1012
- // if (child.type === 'group' && child.destination === 'fldinst') { continue; }
1013
- // This skips the child entirely.
1014
- // So we need to manually serialize it if we want it in rawContent.
1015
- currentParagraphRaw += '{';
1016
- // We need to serialize the content of fldinst
1065
+ if (child.type === 'control')
1066
+ currentParagraphRawChunks.push(serializeRtfControl(child));
1067
+ else if (child.type === 'text')
1068
+ currentParagraphRawChunks.push(serializeRtfText(child));
1069
+ else if (child.type === 'group') {
1070
+ // Recursive serialization for nested groups in pict (e.g. blipuid)
1071
+ // We can't easily recurse `traverse` because it has side effects (text extraction).
1072
+ // We need a pure serializer or just let `traverse` run but with a flag?
1073
+ // Or just ignore the raw content of the image binary data?
1074
+ // Image binary data can be huge.
1075
+ // Maybe we shouldn't include the full hex dump in `rawContent`?
1076
+ // The user said "raw content which is probably the rtf group".
1077
+ // Including 5MB of hex data in the JSON AST might be bad.
1078
+ // But for consistency, it is the raw content.
1079
+ // Let's include it for now.
1080
+ // To do this without side effects, we need a separate serialize function?
1081
+ // Or just call traverse and ensure `isPict` logic prevents text extraction.
1082
+ // Wait, `isPict` is true for this node.
1083
+ // If we recurse, `isPict` will be false for children (unless they are also pict).
1084
+ // But we want to suppress text extraction for children of pict.
1085
+ // The original code did `return`.
1086
+ // So we should manually serialize children here.
1087
+ // Let's define a simple recursive serializer.
1017
1088
  const serializeGroupContent = (g) => {
1018
1089
  for (const c of g.content) {
1019
1090
  if (c.type === 'control')
1020
- currentParagraphRaw += serializeRtfControl(c);
1091
+ currentParagraphRawChunks.push(serializeRtfControl(c));
1021
1092
  else if (c.type === 'text')
1022
- currentParagraphRaw += serializeRtfText(c);
1093
+ currentParagraphRawChunks.push(serializeRtfText(c));
1023
1094
  else if (c.type === 'group') {
1024
- currentParagraphRaw += '{';
1095
+ currentParagraphRawChunks.push('{');
1025
1096
  serializeGroupContent(c);
1026
- currentParagraphRaw += '}';
1097
+ currentParagraphRawChunks.push('}');
1027
1098
  }
1028
1099
  }
1029
1100
  };
1030
- serializeGroupContent(child);
1031
- currentParagraphRaw += '}';
1032
- continue;
1101
+ serializeGroupContent(node);
1033
1102
  }
1034
- traverse(child, groupFormatting, depth + 1);
1035
- }
1036
- if (node.destination === 'listtable') {
1037
- parsingListTable = false;
1038
1103
  }
1039
- else if (node.destination === 'list') {
1040
- parsingListDefinition = false;
1041
- if (currentDefinedListId !== undefined && currentDefinedListType !== undefined) {
1042
- listTypeMap[currentDefinedListId] = currentDefinedListType;
1104
+ currentParagraphRawChunks.push('}');
1105
+ return;
1106
+ }
1107
+ // Handle footnote: switch target to notes
1108
+ const previousTarget = currentTarget;
1109
+ if (isFootnote) {
1110
+ if (config.ignoreNotes) {
1111
+ return; // Skip footnote content entirely
1112
+ }
1113
+ flushParagraph();
1114
+ currentFootnoteId++;
1115
+ // Determine note type based on \fet value
1116
+ let noteType = 'footnote';
1117
+ if (fetValue === 1) {
1118
+ // \fet1 means all notes are endnotes
1119
+ noteType = 'endnote';
1120
+ }
1121
+ else if (fetValue === 2) {
1122
+ // \fet2 means both types exist
1123
+ // Check for \ftnalt marker to distinguish endnotes from footnotes
1124
+ // \footnote\ftnalt indicates an endnote
1125
+ const hasFtnalt = node.content.some(child => child.type === 'control' && child.value === 'ftnalt');
1126
+ noteType = hasFtnalt ? 'endnote' : 'footnote';
1127
+ }
1128
+ // fetValue === 0 (default) means footnotes only
1129
+ const noteNode = {
1130
+ type: 'note',
1131
+ children: [],
1132
+ metadata: {
1133
+ noteId: currentFootnoteId.toString(),
1134
+ noteType: noteType
1043
1135
  }
1136
+ };
1137
+ notes.push(noteNode);
1138
+ currentTarget = noteNode.children;
1139
+ }
1140
+ // Create a new formatting context for the group
1141
+ const groupFormatting = { ...formatting };
1142
+ for (const child of node.content) {
1143
+ // Skip fldinst groups (we already extracted the URL)
1144
+ if (child.type === 'group' && child.destination === 'fldinst') {
1145
+ // We still want it in rawContent!
1146
+ // So we should traverse it but suppress text extraction?
1147
+ // Or just serialize it?
1148
+ // `fldinst` contains the URL.
1149
+ // If we skip it in `traverse`, we miss it in `rawContent`.
1150
+ // Let's traverse it but maybe the `fldinst` logic inside `traverse` handles it?
1151
+ // The original code:
1152
+ // if (child.type === 'group' && child.destination === 'fldinst') { continue; }
1153
+ // This skips the child entirely.
1154
+ // So we need to manually serialize it if we want it in rawContent.
1155
+ currentParagraphRawChunks.push('{');
1156
+ // We need to serialize the content of fldinst
1157
+ const serializeGroupContent = (g) => {
1158
+ for (const c of g.content) {
1159
+ if (c.type === 'control')
1160
+ currentParagraphRawChunks.push(serializeRtfControl(c));
1161
+ else if (c.type === 'text')
1162
+ currentParagraphRawChunks.push(serializeRtfText(c));
1163
+ else if (c.type === 'group') {
1164
+ currentParagraphRawChunks.push('{');
1165
+ serializeGroupContent(c);
1166
+ currentParagraphRawChunks.push('}');
1167
+ }
1168
+ }
1169
+ };
1170
+ serializeGroupContent(child);
1171
+ currentParagraphRawChunks.push('}');
1172
+ continue;
1044
1173
  }
1045
- else if (node.destination === 'listoverridetable') {
1046
- parsingListOverrideTable = false;
1174
+ traverse(child, groupFormatting, depth + 1);
1175
+ }
1176
+ if (node.destination === 'listtable') {
1177
+ parsingListTable = false;
1178
+ }
1179
+ else if (node.destination === 'list') {
1180
+ parsingListDefinition = false;
1181
+ if (currentDefinedListId !== undefined && currentDefinedListType !== undefined) {
1182
+ listTypeMap[currentDefinedListId] = currentDefinedListType;
1047
1183
  }
1048
- else if (parsingListOverrideTable && node.destination === 'listoverride') {
1049
- // End of listoverride group - populate map
1050
- if (currentListOverrideLs !== undefined && currentListOverrideListId !== undefined) {
1051
- listOverrideMap[currentListOverrideLs] = currentListOverrideListId;
1184
+ }
1185
+ else if (node.destination === 'listoverridetable') {
1186
+ parsingListOverrideTable = false;
1187
+ }
1188
+ else if (parsingListOverrideTable && node.destination === 'listoverride') {
1189
+ // End of listoverride group - populate map
1190
+ if (currentListOverrideLs !== undefined && currentListOverrideListId !== undefined) {
1191
+ listOverrideMap[currentListOverrideLs] = currentListOverrideListId;
1192
+ }
1193
+ }
1194
+ if (isFootnote) {
1195
+ flushParagraph();
1196
+ currentTarget = previousTarget;
1197
+ }
1198
+ // Clear link URL after processing the field group
1199
+ if (isHyperlinkField) {
1200
+ flushRun();
1201
+ currentLinkUrl = undefined;
1202
+ }
1203
+ // Append group end to raw content
1204
+ currentParagraphRawChunks.push('}');
1205
+ }
1206
+ else if (node.type === 'text') {
1207
+ if (parsingListTable || parsingListOverrideTable) {
1208
+ // Even if we don't extract text, we might want it in rawContent?
1209
+ // Yes, rawContent should reflect the source.
1210
+ currentParagraphRawChunks.push(serializeRtfText(node));
1211
+ return;
1212
+ }
1213
+ if (formattingChanged(currentFormatting, formatting)) {
1214
+ flushRun();
1215
+ currentFormatting = { ...formatting };
1216
+ }
1217
+ currentRunTextChunks.push(node.value);
1218
+ currentParagraphRawChunks.push(serializeRtfText(node));
1219
+ }
1220
+ else if (node.type === 'control') {
1221
+ // Append control to raw content
1222
+ currentParagraphRawChunks.push(serializeRtfControl(node));
1223
+ // Handle list definition control words
1224
+ if (parsingListTable) {
1225
+ if (node.value === 'listid') {
1226
+ currentDefinedListId = node.param;
1227
+ }
1228
+ else if (node.value === 'levelnfc' || node.value === 'levelnfcn') {
1229
+ // 0 = Arabic, 1 = Upper Roman, 2 = Lower Roman, 3 = Upper Alpha, 4 = Lower Alpha -> Ordered
1230
+ // 23 = Bullet, 255 = None -> Unordered
1231
+ const isOrdered = node.param !== undefined && (node.param === 0 || (node.param >= 0 && node.param <= 4));
1232
+ // Only set if not already set (or prioritize ordered if mixed?)
1233
+ // We'll assume if any level is ordered, it's ordered.
1234
+ // Or if we haven't set it yet.
1235
+ if (!currentDefinedListType || (currentDefinedListType === 'unordered' && isOrdered)) {
1236
+ currentDefinedListType = isOrdered ? 'ordered' : 'unordered';
1052
1237
  }
1053
1238
  }
1054
- if (isFootnote) {
1055
- flushParagraph();
1056
- currentTarget = previousTarget;
1239
+ return;
1240
+ }
1241
+ // Handle list override control words
1242
+ if (parsingListOverrideTable) {
1243
+ if (node.value === 'listid') {
1244
+ currentListOverrideListId = node.param;
1057
1245
  }
1058
- // Clear link URL after processing the field group
1059
- if (isHyperlinkField) {
1060
- flushRun();
1061
- currentLinkUrl = undefined;
1246
+ else if (node.value === 'ls') {
1247
+ currentListOverrideLs = node.param;
1062
1248
  }
1063
- // Append group end to raw content
1064
- currentParagraphRaw += '}';
1249
+ return;
1065
1250
  }
1066
- else if (node.type === 'text') {
1067
- if (parsingListTable || parsingListOverrideTable) {
1068
- // Even if we don't extract text, we might want it in rawContent?
1069
- // Yes, rawContent should reflect the source.
1070
- currentParagraphRaw += serializeRtfText(node);
1071
- return;
1251
+ // Paragraph control words
1252
+ if (node.value === 'par') {
1253
+ flushParagraph();
1254
+ currentFormatting = { ...formatting };
1255
+ }
1256
+ // Table control words
1257
+ else if (node.value === 'trowd') {
1258
+ // Table row definition - start of a new row
1259
+ // Check if we are starting a nested table
1260
+ // If we are already in a table, and we have content in the current cell,
1261
+ // then this trowd implies a nested table start.
1262
+ const ctx = getCurrentTable();
1263
+ if (inTable && ctx && ctx.currentCellContent.length > 0) {
1264
+ // Start nested table
1265
+ ensureTableContext(); // Should already exist if inTable is true
1266
+ // Push new table context
1267
+ tableStack.push({
1268
+ rows: [],
1269
+ currentCells: [],
1270
+ currentCellContent: [],
1271
+ rowIndex: 0
1272
+ });
1072
1273
  }
1073
- if (formattingChanged(currentFormatting, formatting)) {
1074
- flushRun();
1075
- currentFormatting = { ...formatting };
1274
+ else {
1275
+ if (!inTable) {
1276
+ inTable = true;
1277
+ ensureTableContext();
1278
+ }
1076
1279
  }
1077
- currentRunText += node.value;
1078
- currentParagraphRaw += serializeRtfText(node);
1280
+ // After \trowd we are inside a table row, so content should go to table cells.
1281
+ // Many RTF files don't use \intbl, relying solely on \trowd...\cell...\row structure.
1282
+ paragraphInTable = true;
1283
+ // Reset cell properties for the new row definition
1284
+ rowCellProps = [];
1285
+ currentCellDefinitionProps = { isMergedContinuation: false };
1286
+ cellContentIndex = 0;
1079
1287
  }
1080
- else if (node.type === 'control') {
1081
- // Append control to raw content
1082
- currentParagraphRaw += serializeRtfControl(node);
1083
- // Handle list definition control words
1084
- if (parsingListTable) {
1085
- if (node.value === 'listid') {
1086
- currentDefinedListId = node.param;
1087
- }
1088
- else if (node.value === 'levelnfc' || node.value === 'levelnfcn') {
1089
- // 0 = Arabic, 1 = Upper Roman, 2 = Lower Roman, 3 = Upper Alpha, 4 = Lower Alpha -> Ordered
1090
- // 23 = Bullet, 255 = None -> Unordered
1091
- const isOrdered = node.param !== undefined && (node.param === 0 || (node.param >= 0 && node.param <= 4));
1092
- // Only set if not already set (or prioritize ordered if mixed?)
1093
- // We'll assume if any level is ordered, it's ordered.
1094
- // Or if we haven't set it yet.
1095
- if (!currentDefinedListType || (currentDefinedListType === 'unordered' && isOrdered)) {
1096
- currentDefinedListType = isOrdered ? 'ordered' : 'unordered';
1097
- }
1288
+ else if (node.value === 'clvmrg') {
1289
+ // Vertical merge continuation
1290
+ currentCellDefinitionProps.isMergedContinuation = true;
1291
+ }
1292
+ else if (node.value === 'clmgf') {
1293
+ // Vertical merge first cell (reset continuation flag if set, though usually mutually exclusive)
1294
+ currentCellDefinitionProps.isMergedContinuation = false;
1295
+ }
1296
+ else if (node.value === 'cellx') {
1297
+ // End of cell definition
1298
+ rowCellProps.push({ ...currentCellDefinitionProps });
1299
+ // Reset for next cell
1300
+ currentCellDefinitionProps = { isMergedContinuation: false };
1301
+ }
1302
+ else if (node.value === 'cell') {
1303
+ // End of cell - add it to current row
1304
+ // Force paragraphInTable = true because \cell implies we are in a table cell
1305
+ paragraphInTable = true;
1306
+ // Check if this cell is a merged continuation
1307
+ let isMergedContinuation = false;
1308
+ if (cellContentIndex < rowCellProps.length) {
1309
+ isMergedContinuation = rowCellProps[cellContentIndex].isMergedContinuation;
1310
+ }
1311
+ cellContentIndex++;
1312
+ const cell = flushCell();
1313
+ // Only add if not a merged continuation
1314
+ if (cell) {
1315
+ if (!isMergedContinuation) {
1316
+ const ctx = getCurrentTable();
1317
+ if (ctx)
1318
+ ctx.currentCells.push(cell);
1098
1319
  }
1099
- return;
1100
1320
  }
1101
- // Handle list override control words
1102
- if (parsingListOverrideTable) {
1103
- if (node.value === 'listid') {
1104
- currentListOverrideListId = node.param;
1105
- }
1106
- else if (node.value === 'ls') {
1107
- currentListOverrideLs = node.param;
1321
+ currentFormatting = { ...formatting };
1322
+ }
1323
+ else if (node.value === 'nestcell') {
1324
+ // End of cell in outer table (nested context)
1325
+ // If we are in an inner table, we need to close it and return to outer
1326
+ // First, flush the current cell of the inner table (if any pending)
1327
+ // Actually, nestcell ends the OUTER cell.
1328
+ // So the inner table should have been finished by now?
1329
+ // Usually inner table ends with \row.
1330
+ // If we are in a nested table (stack > 1), we should pop until we are at the outer table?
1331
+ // Or maybe just pop one level?
1332
+ if (tableStack.length > 1) {
1333
+ // Flush the inner table if it has pending rows
1334
+ const innerCtx = getCurrentTable();
1335
+ if (innerCtx && (innerCtx.rows.length > 0 || innerCtx.currentCells.length > 0)) {
1336
+ flushTable(); // This pops the stack
1108
1337
  }
1109
- return;
1110
- }
1111
- // Paragraph control words
1112
- if (node.value === 'par') {
1113
- flushParagraph();
1114
- currentFormatting = { ...formatting };
1115
1338
  }
1116
- // Table control words
1117
- else if (node.value === 'trowd') {
1118
- // Table row definition - start of a new row
1119
- // Check if we are starting a nested table
1120
- // If we are already in a table, and we have content in the current cell,
1121
- // then this trowd implies a nested table start.
1339
+ // Now we are (hopefully) at the outer table level
1340
+ // Treat as a regular cell end for the outer table
1341
+ paragraphInTable = true;
1342
+ const cell = flushCell();
1343
+ if (cell) {
1122
1344
  const ctx = getCurrentTable();
1123
- if (inTable && ctx && ctx.currentCellContent.length > 0) {
1124
- // Start nested table
1125
- ensureTableContext(); // Should already exist if inTable is true
1126
- // Push new table context
1127
- tableStack.push({
1128
- rows: [],
1129
- currentCells: [],
1130
- currentCellContent: [],
1131
- rowIndex: 0
1132
- });
1133
- }
1134
- else {
1135
- if (!inTable) {
1136
- inTable = true;
1137
- ensureTableContext();
1138
- }
1139
- }
1140
- // After \trowd we are inside a table row, so content should go to table cells.
1141
- // Many RTF files don't use \intbl, relying solely on \trowd...\cell...\row structure.
1142
- paragraphInTable = true;
1143
- // Reset cell properties for the new row definition
1144
- rowCellProps = [];
1145
- currentCellDefinitionProps = { isMergedContinuation: false };
1146
- cellContentIndex = 0;
1147
- }
1148
- else if (node.value === 'clvmrg') {
1149
- // Vertical merge continuation
1150
- currentCellDefinitionProps.isMergedContinuation = true;
1151
- }
1152
- else if (node.value === 'clmgf') {
1153
- // Vertical merge first cell (reset continuation flag if set, though usually mutually exclusive)
1154
- currentCellDefinitionProps.isMergedContinuation = false;
1155
- }
1156
- else if (node.value === 'cellx') {
1157
- // End of cell definition
1158
- rowCellProps.push({ ...currentCellDefinitionProps });
1159
- // Reset for next cell
1160
- currentCellDefinitionProps = { isMergedContinuation: false };
1161
- }
1162
- else if (node.value === 'cell') {
1163
- // End of cell - add it to current row
1164
- // Force paragraphInTable = true because \cell implies we are in a table cell
1165
- paragraphInTable = true;
1166
- // Check if this cell is a merged continuation
1167
- let isMergedContinuation = false;
1168
- if (cellContentIndex < rowCellProps.length) {
1169
- isMergedContinuation = rowCellProps[cellContentIndex].isMergedContinuation;
1170
- }
1171
- cellContentIndex++;
1172
- const cell = flushCell();
1173
- // Only add if not a merged continuation
1174
- if (cell) {
1175
- if (!isMergedContinuation) {
1176
- const ctx = getCurrentTable();
1177
- if (ctx)
1178
- ctx.currentCells.push(cell);
1179
- }
1180
- }
1181
- currentFormatting = { ...formatting };
1345
+ if (ctx)
1346
+ ctx.currentCells.push(cell);
1182
1347
  }
1183
- else if (node.value === 'nestcell') {
1184
- // End of cell in outer table (nested context)
1185
- // If we are in an inner table, we need to close it and return to outer
1186
- // First, flush the current cell of the inner table (if any pending)
1187
- // Actually, nestcell ends the OUTER cell.
1188
- // So the inner table should have been finished by now?
1189
- // Usually inner table ends with \row.
1190
- // If we are in a nested table (stack > 1), we should pop until we are at the outer table?
1191
- // Or maybe just pop one level?
1192
- if (tableStack.length > 1) {
1193
- // Flush the inner table if it has pending rows
1194
- const innerCtx = getCurrentTable();
1195
- if (innerCtx && (innerCtx.rows.length > 0 || innerCtx.currentCells.length > 0)) {
1196
- flushTable(); // This pops the stack
1197
- }
1198
- }
1199
- // Now we are (hopefully) at the outer table level
1200
- // Treat as a regular cell end for the outer table
1201
- paragraphInTable = true;
1202
- const cell = flushCell();
1203
- if (cell) {
1204
- const ctx = getCurrentTable();
1205
- if (ctx)
1206
- ctx.currentCells.push(cell);
1207
- }
1208
- currentFormatting = { ...formatting };
1348
+ currentFormatting = { ...formatting };
1349
+ }
1350
+ else if (node.value === 'row') {
1351
+ // End of row
1352
+ flushRow();
1353
+ currentFormatting = { ...formatting };
1354
+ // Reset content index for safety (though trowd usually does it)
1355
+ cellContentIndex = 0;
1356
+ // Critical: Reset paragraphInTable after row ends.
1357
+ // Subsequent paragraphs must explicitly use \intbl to be part of the table.
1358
+ // Without this, content after the last \row gets incorrectly merged.
1359
+ paragraphInTable = false;
1360
+ }
1361
+ else if (node.value === 'nestrow') {
1362
+ // End of row in outer table
1363
+ // If we are still in inner table context, flush it
1364
+ if (tableStack.length > 1) {
1365
+ flushTable();
1209
1366
  }
1210
- else if (node.value === 'row') {
1211
- // End of row
1212
- flushRow();
1367
+ flushRow();
1368
+ currentFormatting = { ...formatting };
1369
+ cellContentIndex = 0;
1370
+ }
1371
+ else if (node.value === 'intbl') {
1372
+ // Paragraph is in a table
1373
+ inTable = true;
1374
+ paragraphInTable = true;
1375
+ ensureTableContext();
1376
+ }
1377
+ else if (node.value === 'pard') {
1378
+ // Reset paragraph properties
1379
+ paragraphInTable = false;
1380
+ // Reset other props...
1381
+ paragraphIndent = 0;
1382
+ paragraphAlignment = 'left';
1383
+ isListItem = false;
1384
+ listType = undefined;
1385
+ headingLevel = undefined;
1386
+ currentListId = undefined;
1387
+ // Reset paragraph-level background (cbpat) to prevent leaking to next paragraph
1388
+ formatting.backgroundColor = undefined;
1389
+ }
1390
+ // Text flow control
1391
+ else if (node.value === 'tab') {
1392
+ if (formattingChanged(currentFormatting, formatting)) {
1393
+ flushRun();
1213
1394
  currentFormatting = { ...formatting };
1214
- // Reset content index for safety (though trowd usually does it)
1215
- cellContentIndex = 0;
1216
- // Critical: Reset paragraphInTable after row ends.
1217
- // Subsequent paragraphs must explicitly use \intbl to be part of the table.
1218
- // Without this, content after the last \row gets incorrectly merged.
1219
- paragraphInTable = false;
1220
- }
1221
- else if (node.value === 'nestrow') {
1222
- // End of row in outer table
1223
- // If we are still in inner table context, flush it
1224
- if (tableStack.length > 1) {
1225
- flushTable();
1226
- }
1227
- flushRow();
1395
+ }
1396
+ currentRunTextChunks.push('\t');
1397
+ }
1398
+ else if (node.value === 'line') {
1399
+ if (formattingChanged(currentFormatting, formatting)) {
1400
+ flushRun();
1228
1401
  currentFormatting = { ...formatting };
1229
- cellContentIndex = 0;
1230
- }
1231
- else if (node.value === 'intbl') {
1232
- // Paragraph is in a table
1233
- inTable = true;
1234
- paragraphInTable = true;
1235
- ensureTableContext();
1236
- }
1237
- else if (node.value === 'pard') {
1238
- // Reset paragraph properties
1239
- paragraphInTable = false;
1240
- // Reset other props...
1241
- paragraphIndent = 0;
1242
- paragraphAlignment = 'left';
1243
- isListItem = false;
1244
- listType = undefined;
1245
- headingLevel = undefined;
1246
- currentListId = undefined;
1247
- // Reset paragraph-level background (cbpat) to prevent leaking to next paragraph
1248
- formatting.backgroundColor = undefined;
1249
- }
1250
- // Text flow control
1251
- else if (node.value === 'tab') {
1252
- if (formattingChanged(currentFormatting, formatting)) {
1253
- flushRun();
1254
- currentFormatting = { ...formatting };
1255
- }
1256
- currentRunText += '\t';
1257
1402
  }
1258
- else if (node.value === 'line') {
1259
- if (formattingChanged(currentFormatting, formatting)) {
1260
- flushRun();
1261
- currentFormatting = { ...formatting };
1262
- }
1263
- currentRunText += '\n';
1403
+ currentRunTextChunks.push('\n');
1404
+ }
1405
+ // Quote characters
1406
+ else if (node.value === 'lquote') {
1407
+ // Left single quotation mark (U+2018)
1408
+ if (formattingChanged(currentFormatting, formatting)) {
1409
+ flushRun();
1410
+ currentFormatting = { ...formatting };
1264
1411
  }
1265
- // Quote characters
1266
- else if (node.value === 'lquote') {
1267
- // Left single quotation mark (U+2018)
1268
- if (formattingChanged(currentFormatting, formatting)) {
1269
- flushRun();
1270
- currentFormatting = { ...formatting };
1271
- }
1272
- currentRunText += '\u2018';
1412
+ currentRunTextChunks.push('\u2018');
1413
+ }
1414
+ else if (node.value === 'rquote') {
1415
+ // Right single quotation mark (U+2019)
1416
+ if (formattingChanged(currentFormatting, formatting)) {
1417
+ flushRun();
1418
+ currentFormatting = { ...formatting };
1273
1419
  }
1274
- else if (node.value === 'rquote') {
1275
- // Right single quotation mark (U+2019)
1276
- if (formattingChanged(currentFormatting, formatting)) {
1277
- flushRun();
1278
- currentFormatting = { ...formatting };
1279
- }
1280
- currentRunText += '\u2019';
1420
+ currentRunTextChunks.push('\u2019');
1421
+ }
1422
+ else if (node.value === 'ldblquote') {
1423
+ // Left double quotation mark (U+201C)
1424
+ if (formattingChanged(currentFormatting, formatting)) {
1425
+ flushRun();
1426
+ currentFormatting = { ...formatting };
1281
1427
  }
1282
- else if (node.value === 'ldblquote') {
1283
- // Left double quotation mark (U+201C)
1284
- if (formattingChanged(currentFormatting, formatting)) {
1285
- flushRun();
1286
- currentFormatting = { ...formatting };
1287
- }
1288
- currentRunText += '\u201C';
1428
+ currentRunTextChunks.push('\u201C');
1429
+ }
1430
+ else if (node.value === 'rdblquote') {
1431
+ // Right double quotation mark (U+201D)
1432
+ if (formattingChanged(currentFormatting, formatting)) {
1433
+ flushRun();
1434
+ currentFormatting = { ...formatting };
1289
1435
  }
1290
- else if (node.value === 'rdblquote') {
1291
- // Right double quotation mark (U+201D)
1436
+ currentRunTextChunks.push('\u201D');
1437
+ }
1438
+ // Unicode character
1439
+ else if (node.value === 'u') {
1440
+ if (node.param !== undefined) {
1441
+ let code = node.param;
1442
+ if (code < 0)
1443
+ code += 65536;
1292
1444
  if (formattingChanged(currentFormatting, formatting)) {
1293
1445
  flushRun();
1294
1446
  currentFormatting = { ...formatting };
1295
1447
  }
1296
- currentRunText += '\u201D';
1297
- }
1298
- // Unicode character
1299
- else if (node.value === 'u') {
1300
- if (node.param !== undefined) {
1301
- let code = node.param;
1302
- if (code < 0)
1303
- code += 65536;
1304
- if (formattingChanged(currentFormatting, formatting)) {
1305
- flushRun();
1306
- currentFormatting = { ...formatting };
1307
- }
1308
- currentRunText += String.fromCharCode(code);
1309
- }
1310
- }
1311
- // Character formatting
1312
- else if (node.value === 'b') {
1313
- formatting.bold = (node.param !== 0);
1314
- }
1315
- else if (node.value === 'i') {
1316
- formatting.italic = (node.param !== 0);
1317
- }
1318
- else if (node.value === 'ul') {
1319
- formatting.underline = (node.param !== 0);
1320
- }
1321
- else if (node.value === 'ulnone') {
1322
- formatting.underline = false;
1323
- }
1324
- else if (node.value === 'strike') {
1325
- formatting.strikethrough = (node.param !== 0);
1326
- }
1327
- else if (node.value === 'plain') {
1328
- // Reset all character formatting
1329
- formatting.bold = false;
1330
- formatting.italic = false;
1331
- formatting.underline = false;
1332
- formatting.strikethrough = false;
1333
- formatting.subscript = false;
1334
- formatting.superscript = false;
1335
- formatting.size = undefined;
1336
- formatting.font = undefined;
1337
- formatting.color = undefined;
1338
- formatting.backgroundColor = undefined;
1339
- }
1340
- // Font size (\fs - in half-points)
1341
- else if (node.value === 'fs') {
1342
- if (node.param !== undefined) {
1343
- formatting.size = (node.param / 2).toString() + 'pt';
1344
- }
1345
- }
1346
- // Font family (\f)
1347
- else if (node.value === 'f') {
1348
- if (node.param !== undefined && fontTable[node.param]) {
1349
- formatting.font = fontTable[node.param];
1350
- }
1351
- }
1352
- // Text color (\cf)
1353
- else if (node.value === 'cf') {
1354
- if (node.param !== undefined && colorTable[node.param]) {
1355
- formatting.color = colorTable[node.param];
1356
- }
1357
- }
1358
- // Note type (\fet)
1359
- else if (node.value === 'fet') {
1360
- // \fet0 = footnotes only (default)
1361
- // \fet1 = endnotes only
1362
- // \fet2 = both footnotes and endnotes
1363
- if (node.param !== undefined) {
1364
- fetValue = node.param;
1365
- }
1448
+ currentRunTextChunks.push(String.fromCharCode(code));
1366
1449
  }
1367
- // Background/highlight color (\cb, \highlight, \chcbpat, \cbpat)
1368
- // \chcbpat = character background pattern color (used for shading)
1369
- // \cbpat = paragraph background pattern color
1370
- else if (node.value === 'cb' || node.value === 'highlight' || node.value === 'chcbpat' || node.value === 'cbpat') {
1371
- if (node.param !== undefined && colorTable[node.param]) {
1372
- formatting.backgroundColor = colorTable[node.param];
1373
- }
1450
+ }
1451
+ // Character formatting
1452
+ else if (node.value === 'b') {
1453
+ formatting.bold = (node.param !== 0);
1454
+ }
1455
+ else if (node.value === 'i') {
1456
+ formatting.italic = (node.param !== 0);
1457
+ }
1458
+ else if (node.value === 'ul') {
1459
+ formatting.underline = (node.param !== 0);
1460
+ }
1461
+ else if (node.value === 'ulnone') {
1462
+ formatting.underline = false;
1463
+ }
1464
+ else if (node.value === 'strike') {
1465
+ formatting.strikethrough = (node.param !== 0);
1466
+ }
1467
+ else if (node.value === 'plain') {
1468
+ // Reset all character formatting
1469
+ formatting.bold = false;
1470
+ formatting.italic = false;
1471
+ formatting.underline = false;
1472
+ formatting.strikethrough = false;
1473
+ formatting.subscript = false;
1474
+ formatting.superscript = false;
1475
+ formatting.size = undefined;
1476
+ formatting.font = undefined;
1477
+ formatting.color = undefined;
1478
+ formatting.backgroundColor = undefined;
1479
+ }
1480
+ // Font size (\fs - in half-points)
1481
+ else if (node.value === 'fs') {
1482
+ if (node.param !== undefined) {
1483
+ formatting.size = (node.param / 2).toString() + 'pt';
1374
1484
  }
1375
- // Subscript
1376
- else if (node.value === 'sub') {
1377
- formatting.subscript = true;
1378
- formatting.superscript = false;
1379
- }
1380
- // Superscript
1381
- else if (node.value === 'super') {
1382
- formatting.superscript = true;
1383
- formatting.subscript = false;
1384
- }
1385
- // No subscript/superscript
1386
- else if (node.value === 'nosupersub') {
1387
- formatting.subscript = false;
1388
- formatting.superscript = false;
1389
- }
1390
- // ═══════════════════════════════════════════════════════════
1391
- // List control words
1392
- // ═══════════════════════════════════════════════════════════
1393
- // Paragraph indentation (\li - left indent in twips)
1394
- else if (node.value === 'li') {
1395
- if (node.param !== undefined) {
1396
- // Convert twips to a simpler unit (720 twips = 1 inch, ~0.5 inch per level)
1397
- paragraphIndent = Math.floor(node.param / 360);
1398
- }
1485
+ }
1486
+ // Font family (\f)
1487
+ else if (node.value === 'f') {
1488
+ if (node.param !== undefined && fontTable[node.param]) {
1489
+ formatting.font = fontTable[node.param];
1399
1490
  }
1400
- // List style ID (Word 97+)
1401
- else if (node.value === 'ls') {
1402
- if (node.param !== undefined) {
1403
- // If this is the first list item, reset the indent
1404
- if (!isListItem)
1405
- paragraphIndent = 0;
1406
- isListItem = true;
1407
- // Generate or retrieve list ID
1408
- if (!listStyleIdMap[node.param]) {
1409
- listIdCounter++;
1410
- listStyleIdMap[node.param] = `rtf-list-${listIdCounter}`;
1411
- }
1412
- currentListId = listStyleIdMap[node.param];
1413
- // Look up type from list definition
1414
- // First check override map to get real list ID
1415
- const realListId = listOverrideMap[node.param] !== undefined ? listOverrideMap[node.param] : node.param;
1416
- if (listTypeMap[realListId]) {
1417
- listType = listTypeMap[realListId];
1418
- }
1419
- }
1491
+ }
1492
+ // Text color (\cf)
1493
+ else if (node.value === 'cf') {
1494
+ if (node.param !== undefined && colorTable[node.param]) {
1495
+ formatting.color = colorTable[node.param];
1420
1496
  }
1421
- // List indent level (Word 97+)
1422
- else if (node.value === 'ilvl') {
1423
- if (node.param !== undefined) {
1424
- isListItem = true;
1425
- paragraphIndent = node.param;
1426
- }
1497
+ }
1498
+ // Note type (\fet)
1499
+ else if (node.value === 'fet') {
1500
+ // \fet0 = footnotes only (default)
1501
+ // \fet1 = endnotes only
1502
+ // \fet2 = both footnotes and endnotes
1503
+ if (node.param !== undefined) {
1504
+ fetValue = node.param;
1427
1505
  }
1428
- // List numbering level (\pnlvl)
1429
- else if (node.value === 'pnlvl') {
1430
- isListItem = true;
1431
- if (node.param !== undefined) {
1432
- paragraphIndent = node.param;
1433
- }
1506
+ }
1507
+ // Background/highlight color (\cb, \highlight, \chcbpat, \cbpat)
1508
+ // \chcbpat = character background pattern color (used for shading)
1509
+ // \cbpat = paragraph background pattern color
1510
+ else if (node.value === 'cb' || node.value === 'highlight' || node.value === 'chcbpat' || node.value === 'cbpat') {
1511
+ if (node.param !== undefined && colorTable[node.param]) {
1512
+ formatting.backgroundColor = colorTable[node.param];
1434
1513
  }
1435
- // List numbering format
1436
- else if (node.value === 'levelnfc' || node.value === 'pnf') {
1437
- // 0 = Arabic (1, 2, 3), 1 = Roman upper, 2 = Roman lower,
1438
- // 3 = Letter upper, 4 = Letter lower, 23 = Bullet
1439
- if (node.param !== undefined) {
1440
- isListItem = true;
1441
- if (node.param === 23) {
1442
- listType = 'unordered';
1443
- }
1444
- else {
1445
- listType = 'ordered';
1446
- }
1514
+ }
1515
+ // Subscript
1516
+ else if (node.value === 'sub') {
1517
+ formatting.subscript = true;
1518
+ formatting.superscript = false;
1519
+ }
1520
+ // Superscript
1521
+ else if (node.value === 'super') {
1522
+ formatting.superscript = true;
1523
+ formatting.subscript = false;
1524
+ }
1525
+ // No subscript/superscript
1526
+ else if (node.value === 'nosupersub') {
1527
+ formatting.subscript = false;
1528
+ formatting.superscript = false;
1529
+ }
1530
+ // ═══════════════════════════════════════════════════════════
1531
+ // List control words
1532
+ // ═══════════════════════════════════════════════════════════
1533
+ // Paragraph indentation (\li - left indent in twips)
1534
+ else if (node.value === 'li') {
1535
+ if (node.param !== undefined) {
1536
+ // Standard level indent is 720 twips (0.5 inch)
1537
+ // Using a slightly more flexible divisor to account for different generators
1538
+ const level = Math.round(node.param / 720);
1539
+ // Only update if not already explicitly set by ilvl (Word 97+)
1540
+ if (!isListItem) {
1541
+ paragraphIndent = level;
1447
1542
  }
1448
1543
  }
1449
- // Ordered list indicator
1450
- else if (node.value === 'pndec' || node.value === 'pnord' || node.value === 'pnlcltr' || node.value === 'pnucltr') {
1451
- // If this is the first list item, reset the indent
1452
- if (!isListItem)
1453
- paragraphIndent = 0;
1454
- isListItem = true;
1455
- listType = 'ordered';
1456
- }
1457
- // Unordered list indicator
1458
- else if (node.value === 'pnbullet' || node.value === 'pncard') {
1544
+ }
1545
+ // List style ID (Word 97+)
1546
+ else if (node.value === 'ls') {
1547
+ if (node.param !== undefined) {
1459
1548
  // If this is the first list item, reset the indent
1460
1549
  if (!isListItem)
1461
1550
  paragraphIndent = 0;
1462
1551
  isListItem = true;
1463
- listType = 'unordered';
1464
- }
1465
- // Style-based heading detection (\s)
1466
- else if (node.value === 's') {
1467
- if (node.param !== undefined) {
1468
- // Common heading styles: s1-s9 (though this varies by document)
1469
- if (node.param >= 1 && node.param <= 9) {
1470
- headingLevel = node.param;
1471
- }
1552
+ // Generate or retrieve list ID
1553
+ if (!listStyleIdMap[node.param]) {
1554
+ listIdCounter++;
1555
+ listStyleIdMap[node.param] = `rtf-list-${listIdCounter}`;
1556
+ }
1557
+ currentListId = listStyleIdMap[node.param];
1558
+ // Look up type from list definition
1559
+ // First check override map to get real list ID
1560
+ const realListId = listOverrideMap[node.param] !== undefined ? listOverrideMap[node.param] : node.param;
1561
+ if (listTypeMap[realListId]) {
1562
+ listType = listTypeMap[realListId];
1472
1563
  }
1473
1564
  }
1474
- // Paragraph alignment
1475
- else if (node.value === 'ql') {
1476
- paragraphAlignment = 'left';
1477
- }
1478
- else if (node.value === 'qc') {
1479
- paragraphAlignment = 'center';
1480
- }
1481
- else if (node.value === 'qr') {
1482
- paragraphAlignment = 'right';
1565
+ }
1566
+ // List indent level (Word 97+)
1567
+ else if (node.value === 'ilvl') {
1568
+ if (node.param !== undefined) {
1569
+ isListItem = true;
1570
+ paragraphIndent = node.param;
1483
1571
  }
1484
- else if (node.value === 'qj') {
1485
- paragraphAlignment = 'justify';
1572
+ }
1573
+ // List numbering level (\pnlvl)
1574
+ else if (node.value === 'pnlvl') {
1575
+ isListItem = true;
1576
+ if (node.param !== undefined) {
1577
+ paragraphIndent = node.param;
1486
1578
  }
1487
1579
  }
1488
- };
1489
- traverse(doc, {});
1490
- // Flush any remaining table
1491
- const finalCtx = getCurrentTable();
1492
- if (inTable || (finalCtx && (finalCtx.rows.length > 0 || finalCtx.currentCells.length > 0))) {
1493
- flushTable();
1494
- }
1495
- flushParagraph();
1496
- // Notes handling:
1497
- // - If putNotesAtLast is false, notes should be added inline during traversal
1498
- // (currently they go to 'notes' array, then we append them here - this is wrong)
1499
- // - If putNotesAtLast is true, notes are appended at the very end (see below)
1500
- //
1501
- // For now, when putNotesAtLast is false, we append notes immediately after content
1502
- // This isn't truly "inline" but it's better than at the end
1503
- // TODO: Implement true inline placement during traversal
1504
- if (!config.putNotesAtLast && notes.length > 0) {
1505
- content.push(...notes);
1506
- notes.length = 0; // Clear so they don't get appended again
1507
- }
1508
- // Perform OCR if enabled
1509
- if (config.ocr && config.extractAttachments) {
1510
- for (const attachment of attachments) {
1511
- if (attachment.mimeType.startsWith('image/')) {
1512
- try {
1513
- // Convert base64 data back to Buffer for Tesseract.js
1514
- // Passing base64 string directly would be interpreted as a file path,
1515
- // causing ENAMETOOLONG error for large images.
1516
- const imageBuffer = Buffer.from(attachment.data, 'base64');
1517
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1580
+ // List numbering format
1581
+ else if (node.value === 'levelnfc' || node.value === 'pnf') {
1582
+ // 0 = Arabic (1, 2, 3), 1 = Roman upper, 2 = Roman lower,
1583
+ // 3 = Letter upper, 4 = Letter lower, 23 = Bullet
1584
+ if (node.param !== undefined) {
1585
+ isListItem = true;
1586
+ if (node.param === 23) {
1587
+ listType = 'unordered';
1518
1588
  }
1519
- catch (e) {
1520
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1589
+ else {
1590
+ listType = 'ordered';
1521
1591
  }
1522
1592
  }
1523
1593
  }
1524
- // Link OCR text and altText to image nodes in content
1525
- const assignOcr = (nodes) => {
1526
- for (const node of nodes) {
1527
- if (node.type === 'image' && node.metadata && 'attachmentName' in node.metadata) {
1528
- const meta = node.metadata;
1529
- const attachment = attachments.find(a => a.name === meta.attachmentName);
1530
- if (attachment) {
1531
- // Propagate OCR text to image node
1532
- if (attachment.ocrText) {
1533
- node.text = attachment.ocrText;
1534
- }
1535
- // Propagate altText if available
1536
- if (attachment.altText) {
1537
- meta.altText = attachment.altText;
1538
- }
1539
- }
1540
- }
1541
- if (node.children) {
1542
- assignOcr(node.children);
1594
+ // Ordered list indicator
1595
+ else if (node.value === 'pndec' || node.value === 'pnord' || node.value === 'pnlcltr' || node.value === 'pnucltr') {
1596
+ // If this is the first list item, reset the indent
1597
+ if (!isListItem)
1598
+ paragraphIndent = 0;
1599
+ isListItem = true;
1600
+ listType = 'ordered';
1601
+ }
1602
+ // Unordered list indicator
1603
+ else if (node.value === 'pnbullet' || node.value === 'pncard') {
1604
+ // If this is the first list item, reset the indent
1605
+ if (!isListItem)
1606
+ paragraphIndent = 0;
1607
+ isListItem = true;
1608
+ listType = 'unordered';
1609
+ }
1610
+ // Style-based heading detection (\s)
1611
+ else if (node.value === 's') {
1612
+ if (node.param !== undefined) {
1613
+ // Common heading styles: s1-s9 (though this varies by document)
1614
+ if (node.param >= 1 && node.param <= 9) {
1615
+ headingLevel = node.param;
1543
1616
  }
1544
1617
  }
1545
- };
1546
- assignOcr(content);
1618
+ }
1619
+ // Paragraph alignment
1620
+ else if (node.value === 'ql') {
1621
+ paragraphAlignment = 'left';
1622
+ }
1623
+ else if (node.value === 'qc') {
1624
+ paragraphAlignment = 'center';
1625
+ }
1626
+ else if (node.value === 'qr') {
1627
+ paragraphAlignment = 'right';
1628
+ }
1629
+ else if (node.value === 'qj') {
1630
+ paragraphAlignment = 'justify';
1631
+ }
1547
1632
  }
1548
- // Final pass to ensure all 'note' nodes have their 'text' property populated
1549
- // (This supports the simple toText implementation)
1550
- const populateNoteText = (nodes) => {
1633
+ };
1634
+ traverse(doc, {});
1635
+ // Flush any remaining table
1636
+ const finalCtx = getCurrentTable();
1637
+ if (inTable || (finalCtx && (finalCtx.rows.length > 0 || finalCtx.currentCells.length > 0))) {
1638
+ flushTable();
1639
+ }
1640
+ flushParagraph();
1641
+ // Notes handling:
1642
+ // - If putNotesAtLast is false, notes should be added inline during traversal
1643
+ // (currently they go to 'notes' array, then we append them here - this is wrong)
1644
+ // - If putNotesAtLast is true, notes are appended at the very end (see below)
1645
+ //
1646
+ // For now, when putNotesAtLast is false, we append notes immediately after content
1647
+ // This isn't truly "inline" but it's better than at the end
1648
+ // TODO: Implement true inline placement during traversal
1649
+ if (!config.putNotesAtLast && notes.length > 0) {
1650
+ content.push(...notes);
1651
+ notes.length = 0; // Clear so they don't get appended again
1652
+ }
1653
+ // Perform OCR if enabled
1654
+ if (config.ocr && config.extractAttachments) {
1655
+ for (const attachment of attachments) {
1656
+ if (attachment.mimeType.startsWith('image/')) {
1657
+ try {
1658
+ // Convert base64 data back to Buffer for Tesseract.js
1659
+ // Passing base64 string directly would be interpreted as a file path,
1660
+ // causing ENAMETOOLONG error for large images.
1661
+ const imageBuffer = Buffer.from(attachment.data, 'base64');
1662
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { ...config.ocrConfig })).trim();
1663
+ }
1664
+ catch (e) {
1665
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
1666
+ }
1667
+ }
1668
+ }
1669
+ // Link OCR text and altText to image nodes in content
1670
+ const assignOcr = (nodes) => {
1551
1671
  for (const node of nodes) {
1552
- if (node.type === 'note' && node.children) {
1553
- const getText = (n) => {
1554
- if (n.children && n.children.length > 0)
1555
- return n.children.map(getText).join('');
1556
- return n.text || '';
1557
- };
1558
- node.text = node.children.map(getText).join('').trim();
1672
+ if (node.type === 'image' && node.metadata && 'attachmentName' in node.metadata) {
1673
+ const meta = node.metadata;
1674
+ const attachment = attachments.find(a => a.name === meta.attachmentName);
1675
+ if (attachment) {
1676
+ // Propagate OCR text to image node
1677
+ if (attachment.ocrText) {
1678
+ node.text = attachment.ocrText;
1679
+ }
1680
+ // Propagate altText if available
1681
+ if (attachment.altText) {
1682
+ meta.altText = attachment.altText;
1683
+ }
1684
+ }
1559
1685
  }
1560
1686
  if (node.children) {
1561
- populateNoteText(node.children);
1687
+ assignOcr(node.children);
1562
1688
  }
1563
1689
  }
1564
1690
  };
1565
- populateNoteText(content);
1566
- populateNoteText(notes);
1567
- const result = {
1568
- type: 'rtf',
1569
- metadata: {
1570
- // RTF Limitation: No style map available (RTF uses inline styles)
1571
- },
1572
- content: content,
1573
- attachments: attachments, // PNG and JPEG images extracted from \\pict groups
1574
- toText: () => {
1575
- let text = content.map(c => c.text).join(config.newlineDelimiter ?? '\n');
1576
- if (config.putNotesAtLast && notes.length > 0) {
1577
- text += (config.newlineDelimiter ?? '\n') + notes.map(c => c.text).join(config.newlineDelimiter ?? '\n');
1578
- }
1579
- return text;
1691
+ assignOcr(content);
1692
+ }
1693
+ // Final pass to ensure all 'note' nodes have their 'text' property populated
1694
+ // (This supports the simple toText implementation)
1695
+ const populateNoteText = (nodes) => {
1696
+ for (const node of nodes) {
1697
+ if (node.type === 'note' && node.children) {
1698
+ const getText = (n) => {
1699
+ if (n.children && n.children.length > 0)
1700
+ return n.children.map(getText).join('');
1701
+ return n.text || '';
1702
+ };
1703
+ node.text = node.children.map(getText).join('').trim();
1580
1704
  }
1581
- };
1582
- // If putNotesAtLast is true, append notes to the end of the content array
1705
+ if (node.children) {
1706
+ populateNoteText(node.children);
1707
+ }
1708
+ }
1709
+ };
1710
+ populateNoteText(content);
1711
+ populateNoteText(notes);
1712
+ const toTextSync = () => {
1713
+ let text = content.map(c => c.text).join(config.newlineDelimiter);
1583
1714
  if (config.putNotesAtLast && notes.length > 0) {
1584
- content.push(...notes);
1715
+ text += config.newlineDelimiter + notes.map(c => c.text).join(config.newlineDelimiter);
1585
1716
  }
1586
- return result;
1587
- }
1588
- catch (err) {
1589
- throw err;
1717
+ return text;
1718
+ };
1719
+ const result = (0, astUtils_js_1.createAST)('rtf', {
1720
+ // RTF Limitation: No style map available (RTF uses inline styles)
1721
+ }, content, attachments, // PNG and JPEG images extracted from \\pict groups
1722
+ config, toTextSync);
1723
+ // If putNotesAtLast is true, append notes to the end of the content array
1724
+ if (config.putNotesAtLast && notes.length > 0) {
1725
+ content.push(...notes);
1590
1726
  }
1727
+ return result;
1591
1728
  };
1592
1729
  exports.parseRtf = parseRtf;
1730
+ // Helper to find an RTF group by destination name
1731
+ function findRtfGroup(group, destination) {
1732
+ for (const node of group.content) {
1733
+ if (node.type === 'group') {
1734
+ if (node.destination === destination)
1735
+ return node;
1736
+ const found = findRtfGroup(node, destination);
1737
+ if (found)
1738
+ return found;
1739
+ }
1740
+ }
1741
+ return null;
1742
+ }
1593
1743
  // Helper function to extract font table from RTF document
1594
1744
  function extractFontTable(doc) {
1595
1745
  const fontTable = {};
1596
- // Recursive helper to find the font table group at any depth
1597
- const findAndParseFontTable = (group) => {
1598
- for (const node of group.content) {
1599
- if (node.type === 'group') {
1600
- if (node.destination === 'fonttbl') {
1601
- // Iterate through font definitions
1602
- for (const fontNode of node.content) {
1603
- if (fontNode.type === 'group') {
1604
- let fontIndex;
1605
- let fontName = '';
1606
- for (const item of fontNode.content) {
1607
- if (item.type === 'control' && item.value === 'f') {
1608
- fontIndex = item.param;
1609
- }
1610
- else if (item.type === 'text') {
1611
- // Font name (may have trailing semicolon)
1612
- fontName += item.value;
1613
- }
1614
- }
1615
- if (fontIndex !== undefined && fontName) {
1616
- // Remove trailing semicolon and whitespace
1617
- fontName = fontName.replace(/;$/, '').trim();
1618
- fontTable[fontIndex] = fontName;
1619
- }
1620
- }
1746
+ const tableGroup = findRtfGroup(doc, 'fonttbl');
1747
+ if (tableGroup) {
1748
+ for (const fontNode of tableGroup.content) {
1749
+ if (fontNode.type === 'group') {
1750
+ let fontIndex;
1751
+ let fontName = '';
1752
+ for (const item of fontNode.content) {
1753
+ if (item.type === 'control' && item.value === 'f') {
1754
+ fontIndex = item.param;
1755
+ }
1756
+ else if (item.type === 'text') {
1757
+ fontName += item.value;
1621
1758
  }
1622
- return true; // Found and parsed
1623
1759
  }
1624
- // Recurse into child groups
1625
- if (findAndParseFontTable(node)) {
1626
- return true;
1760
+ if (fontIndex !== undefined && fontName) {
1761
+ fontTable[fontIndex] = fontName.replace(/;$/, '').trim();
1627
1762
  }
1628
1763
  }
1629
1764
  }
1630
- return false;
1631
- };
1632
- findAndParseFontTable(doc);
1765
+ }
1633
1766
  return fontTable;
1634
1767
  }
1635
1768
  // Helper function to extract color table from RTF document
1636
1769
  function extractColorTable(doc) {
1637
1770
  const colorTable = {};
1638
- // Recursive helper to find the color table group at any depth
1639
- const findAndParseColorTable = (group) => {
1640
- for (const node of group.content) {
1641
- if (node.type === 'group') {
1642
- if (node.destination === 'colortbl') {
1643
- let colorIndex = 0;
1644
- let red = 0, green = 0, blue = 0;
1645
- for (const item of node.content) {
1646
- if (item.type === 'control') {
1647
- if (item.value === 'red' && item.param !== undefined) {
1648
- red = item.param;
1649
- }
1650
- else if (item.value === 'green' && item.param !== undefined) {
1651
- green = item.param;
1652
- }
1653
- else if (item.value === 'blue' && item.param !== undefined) {
1654
- blue = item.param;
1655
- }
1656
- }
1657
- else if (item.type === 'text' && item.value === ';') {
1658
- // Semicolon marks end of color definition
1659
- const hex = `#${red.toString(16).padStart(2, '0')}${green.toString(16).padStart(2, '0')}${blue.toString(16).padStart(2, '0')}`;
1660
- colorTable[colorIndex] = hex;
1661
- colorIndex++;
1662
- red = 0;
1663
- green = 0;
1664
- blue = 0;
1665
- }
1666
- }
1667
- return true; // Found and parsed
1668
- }
1669
- // Recurse into child groups
1670
- if (findAndParseColorTable(node)) {
1671
- return true;
1672
- }
1771
+ const tableGroup = findRtfGroup(doc, 'colortbl');
1772
+ if (tableGroup) {
1773
+ let colorIndex = 0;
1774
+ let red = 0, green = 0, blue = 0;
1775
+ for (const item of tableGroup.content) {
1776
+ if (item.type === 'control') {
1777
+ if (item.value === 'red' && item.param !== undefined)
1778
+ red = item.param;
1779
+ else if (item.value === 'green' && item.param !== undefined)
1780
+ green = item.param;
1781
+ else if (item.value === 'blue' && item.param !== undefined)
1782
+ blue = item.param;
1783
+ }
1784
+ else if (item.type === 'text' && item.value === ';') {
1785
+ const hex = `#${red.toString(16).padStart(2, '0')}${green.toString(16).padStart(2, '0')}${blue.toString(16).padStart(2, '0')}`;
1786
+ colorTable[colorIndex] = hex;
1787
+ colorIndex++;
1788
+ red = 0;
1789
+ green = 0;
1790
+ blue = 0;
1673
1791
  }
1674
1792
  }
1675
- return false;
1676
- };
1677
- findAndParseColorTable(doc);
1793
+ }
1678
1794
  return colorTable;
1679
1795
  }