officeparser 6.1.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +219 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -29
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +826 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +781 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +27 -8
@@ -42,6 +42,8 @@
42
42
  */
43
43
  Object.defineProperty(exports, "__esModule", { value: true });
44
44
  exports.parseRtf = exports.SimpleRtfParser = void 0;
45
+ const types_js_1 = require("../types.js");
46
+ const astUtils_js_1 = require("../utils/astUtils.js");
45
47
  const errorUtils_js_1 = require("../utils/errorUtils.js");
46
48
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
47
49
  /**
@@ -357,1388 +359,1437 @@ exports.SimpleRtfParser = SimpleRtfParser;
357
359
  * @returns The parsed AST.
358
360
  */
359
361
  const parseRtf = async (buffer, config) => {
360
- try {
361
- const parser = new SimpleRtfParser(buffer);
362
- const doc = parser.parse();
363
- // Extract font and color tables
364
- const fontTable = extractFontTable(doc);
365
- const colorTable = extractColorTable(doc);
366
- const content = [];
367
- const notes = [];
368
- const attachments = [];
369
- // State for paragraph construction
370
- let currentParagraphText = '';
371
- let currentParagraphChildren = [];
372
- let currentParagraphRaw = '';
373
- // State for text run construction
374
- let currentRunText = '';
375
- let currentFormatting = {};
376
- // Target for content (main body or notes)
377
- let currentTarget = content;
378
- // Paragraph-level state
379
- let paragraphIndent = 0;
380
- let paragraphAlignment = 'left';
381
- let isListItem = false;
382
- let listType;
383
- let headingLevel;
384
- let currentListId;
385
- // Persistent list state for listtext/pntext detection
386
- // These persist across paragraphs to allow list items without explicit \ls
387
- let lastKnownListId;
388
- let lastKnownListType;
389
- let tableStack = [];
390
- let currentFootnoteId = 0;
391
- // ═══════════════════════════════════════════════════════════════════
392
- // Note type tracking (footnotes vs endnotes)
393
- // ═══════════════════════════════════════════════════════════════════
394
- // RTF uses \fet to distinguish note types:
395
- // \fet0 = footnotes only (default)
396
- // \fet1 = endnotes only
397
- // \fet2 = both footnotes and endnotes
398
- let fetValue = 0; // Default to footnotes only
399
- // Helper to get current table context
400
- const getCurrentTable = () => tableStack.length > 0 ? tableStack[tableStack.length - 1] : undefined;
401
- // Helper to ensure a table context exists (for top-level tables)
402
- const ensureTableContext = () => {
403
- if (tableStack.length === 0) {
404
- tableStack.push({
405
- rows: [],
406
- currentCells: [],
407
- currentCellContent: [],
408
- rowIndex: 0
409
- });
410
- }
411
- };
412
- let inTable = false;
413
- let paragraphInTable = false;
414
- let tableId = 0;
415
- let rowCellProps = [];
416
- let currentCellDefinitionProps = { isMergedContinuation: false };
417
- let cellContentIndex = 0;
418
- // ═══════════════════════════════════════════════════════════════════
419
- // List state tracking (Word 97+ uses \ls for list style ID)
420
- // ═══════════════════════════════════════════════════════════════════
421
- let listIdCounter = 0;
422
- const listStyleIdMap = {};
423
- // List definition state for parsing \listtable
424
- let parsingListTable = false;
425
- let parsingListDefinition = false;
426
- let currentDefinedListId;
427
- let currentDefinedListType;
428
- const listTypeMap = {};
429
- // List override state for parsing \listoverridetable
430
- let parsingListOverrideTable = false;
431
- let currentListOverrideListId;
432
- let currentListOverrideLs;
433
- const listOverrideMap = {}; // Maps \ls ID to \listid
434
- // List counters for itemIndex tracking
435
- // Map: listId -> indentation level -> count
436
- const listCounters = {};
437
- // ═══════════════════════════════════════════════════════════════════
438
- // Hyperlink state (RTF uses \field{\*\fldinst HYPERLINK "url"})
439
- // ═══════════════════════════════════════════════════════════════════
440
- let currentLinkUrl;
441
- // Helper to check if formatting changed
442
- const formattingChanged = (a, b) => {
443
- return a.bold !== b.bold ||
444
- a.italic !== b.italic ||
445
- a.underline !== b.underline ||
446
- a.strikethrough !== b.strikethrough ||
447
- a.size !== b.size ||
448
- a.font !== b.font ||
449
- a.color !== b.color ||
450
- a.backgroundColor !== b.backgroundColor ||
451
- a.subscript !== b.subscript ||
452
- a.superscript !== b.superscript;
453
- };
454
- // Helper to flush current run to paragraph children
455
- const flushRun = () => {
456
- if (currentRunText) {
457
- const node = {
458
- type: 'text',
459
- text: currentRunText,
460
- formatting: { ...currentFormatting }
362
+ const parser = new SimpleRtfParser(buffer);
363
+ const doc = parser.parse();
364
+ // Extract font and color tables
365
+ const fontTable = extractFontTable(doc);
366
+ const colorTable = extractColorTable(doc);
367
+ const content = [];
368
+ const notes = [];
369
+ const attachments = [];
370
+ // State for paragraph construction
371
+ let currentParagraphTextChunks = [];
372
+ let currentParagraphChildren = [];
373
+ let currentParagraphRawChunks = [];
374
+ // State for text run construction
375
+ let currentRunTextChunks = [];
376
+ let currentFormatting = {};
377
+ // Target for content (main body or notes)
378
+ let currentTarget = content;
379
+ // Paragraph-level state
380
+ let paragraphIndent = 0;
381
+ let paragraphAlignment = 'left';
382
+ let isListItem = false;
383
+ let listType;
384
+ let headingLevel;
385
+ let currentListId;
386
+ let currentAnchorIds = [];
387
+ // Persistent list state for listtext/pntext detection
388
+ // These persist across paragraphs to allow list items without explicit \ls
389
+ let lastKnownListId;
390
+ let lastKnownListType;
391
+ let tableStack = [];
392
+ let currentFootnoteId = 0;
393
+ // ═══════════════════════════════════════════════════════════════════
394
+ // Note type tracking (footnotes vs endnotes)
395
+ // ═══════════════════════════════════════════════════════════════════
396
+ // RTF uses \fet to distinguish note types:
397
+ // \fet0 = footnotes only (default)
398
+ // \fet1 = endnotes only
399
+ // \fet2 = both footnotes and endnotes
400
+ let fetValue = 0; // Default to footnotes only
401
+ // Helper to get current table context
402
+ const getCurrentTable = () => tableStack.length > 0 ? tableStack[tableStack.length - 1] : undefined;
403
+ // Helper to ensure a table context exists (for top-level tables)
404
+ const ensureTableContext = () => {
405
+ if (tableStack.length === 0) {
406
+ tableStack.push({
407
+ rows: [],
408
+ currentCells: [],
409
+ currentCellContent: [],
410
+ rowIndex: 0
411
+ });
412
+ }
413
+ };
414
+ let inTable = false;
415
+ let paragraphInTable = false;
416
+ let tableId = 0;
417
+ let rowCellProps = [];
418
+ let currentCellDefinitionProps = { isMergedContinuation: false };
419
+ let cellContentIndex = 0;
420
+ // ═══════════════════════════════════════════════════════════════════
421
+ // List state tracking (Word 97+ uses \ls for list style ID)
422
+ // ═══════════════════════════════════════════════════════════════════
423
+ let listIdCounter = 0;
424
+ const listStyleIdMap = {};
425
+ // List definition state for parsing \listtable
426
+ let parsingListTable = false;
427
+ let parsingListDefinition = false;
428
+ let currentDefinedListId;
429
+ let currentDefinedListType;
430
+ const listTypeMap = {};
431
+ // List override state for parsing \listoverridetable
432
+ let parsingListOverrideTable = false;
433
+ let currentListOverrideListId;
434
+ let currentListOverrideLs;
435
+ const listOverrideMap = {}; // Maps \ls ID to \listid
436
+ // List counters for itemIndex tracking
437
+ // Map: listId -> indentation level -> count
438
+ const listCounters = {};
439
+ // ═══════════════════════════════════════════════════════════════════
440
+ // Hyperlink state (RTF uses \field{\*\fldinst HYPERLINK "url"})
441
+ // ═══════════════════════════════════════════════════════════════════
442
+ let currentLinkUrl;
443
+ // Helper to check if formatting changed
444
+ const formattingChanged = (a, b) => {
445
+ return a.bold !== b.bold ||
446
+ a.italic !== b.italic ||
447
+ a.underline !== b.underline ||
448
+ a.strikethrough !== b.strikethrough ||
449
+ a.size !== b.size ||
450
+ a.font !== b.font ||
451
+ a.color !== b.color ||
452
+ a.backgroundColor !== b.backgroundColor ||
453
+ a.subscript !== b.subscript ||
454
+ a.superscript !== b.superscript;
455
+ };
456
+ // Helper to flush current run to paragraph children
457
+ const flushRun = () => {
458
+ if (currentRunTextChunks.length > 0) {
459
+ const currentRunText = currentRunTextChunks.join('');
460
+ const node = {
461
+ type: 'text',
462
+ text: currentRunText,
463
+ formatting: { ...currentFormatting }
464
+ };
465
+ // Use TextMetadata.link for hyperlinks
466
+ if (currentLinkUrl) {
467
+ node.metadata = {
468
+ link: currentLinkUrl,
469
+ linkType: classifyLinkType(currentLinkUrl)
461
470
  };
462
- // Use TextMetadata.link for hyperlinks
463
- if (currentLinkUrl) {
464
- node.metadata = {
465
- link: currentLinkUrl,
466
- linkType: classifyLinkType(currentLinkUrl)
467
- };
471
+ }
472
+ currentParagraphChildren.push(node);
473
+ currentParagraphTextChunks.push(currentRunText);
474
+ currentRunTextChunks = [];
475
+ }
476
+ };
477
+ let isFlushingTable = false;
478
+ // Helper to flush current paragraph
479
+ const flushParagraph = () => {
480
+ flushRun(); // Ensure last run is added
481
+ // Check if we need to end the table
482
+ // If we were in a table, but this paragraph is NOT marked as in-table,
483
+ // and we have content, then the table has ended.
484
+ const hasContent = currentParagraphTextChunks.length > 0 || currentParagraphChildren.length > 0;
485
+ if (inTable && !paragraphInTable && !isFlushingTable && hasContent) {
486
+ // CRITICAL: Save current paragraph content before flushing table
487
+ // because flushTable() -> flushRow() -> flushCell() -> flushParagraph()
488
+ // would otherwise process this content during the table flush
489
+ const savedParagraphTextChunks = [...currentParagraphTextChunks];
490
+ const savedParagraphChildren = [...currentParagraphChildren];
491
+ const savedParagraphRawChunks = [...currentParagraphRawChunks];
492
+ // Clear buffers so nested flushParagraph() doesn't process them
493
+ currentParagraphTextChunks = [];
494
+ currentParagraphChildren = [];
495
+ currentParagraphRawChunks = [];
496
+ flushTable();
497
+ // Restore the saved content for processing after the table
498
+ currentParagraphTextChunks = savedParagraphTextChunks;
499
+ currentParagraphChildren = savedParagraphChildren;
500
+ currentParagraphRawChunks = savedParagraphRawChunks;
501
+ }
502
+ if (hasContent) {
503
+ const currentParagraphText = currentParagraphTextChunks.join('');
504
+ let nodeType = 'paragraph';
505
+ let metadata = undefined;
506
+ // Heuristic heading detection if no explicit \s style was found
507
+ if (headingLevel === undefined && currentParagraphChildren.length > 0) {
508
+ // Check if the first child is bold and larger than default (12pt)
509
+ const firstChild = currentParagraphChildren[0];
510
+ if (firstChild.type === 'text' && firstChild.formatting?.bold) {
511
+ const size = parseInt(firstChild.formatting.size || '12');
512
+ if (size >= 14 && currentParagraphText.length < 300) {
513
+ if (size >= 22)
514
+ headingLevel = 1;
515
+ else if (size >= 18)
516
+ headingLevel = 2;
517
+ else if (size >= 16)
518
+ headingLevel = 3;
519
+ else
520
+ headingLevel = 4;
521
+ }
468
522
  }
469
- currentParagraphChildren.push(node);
470
- currentParagraphText += currentRunText;
471
- currentRunText = '';
472
523
  }
473
- };
474
- let isFlushingTable = false;
475
- // Helper to flush current paragraph
476
- const flushParagraph = () => {
477
- flushRun(); // Ensure last run is added
478
- // Check if we need to end the table
479
- // If we were in a table, but this paragraph is NOT marked as in-table,
480
- // and we have content, then the table has ended.
481
- const hasContent = currentParagraphText || currentParagraphChildren.length > 0;
482
- if (inTable && !paragraphInTable && !isFlushingTable && hasContent) {
483
- // CRITICAL: Save current paragraph content before flushing table
484
- // because flushTable() -> flushRow() -> flushCell() -> flushParagraph()
485
- // would otherwise process this content during the table flush
486
- const savedParagraphText = currentParagraphText;
487
- const savedParagraphChildren = [...currentParagraphChildren];
488
- const savedParagraphRaw = currentParagraphRaw;
489
- // Clear buffers so nested flushParagraph() doesn't process them
490
- currentParagraphText = '';
491
- currentParagraphChildren = [];
492
- currentParagraphRaw = '';
493
- flushTable();
494
- // Restore the saved content for processing after the table
495
- currentParagraphText = savedParagraphText;
496
- currentParagraphChildren = savedParagraphChildren;
497
- currentParagraphRaw = savedParagraphRaw;
498
- }
499
- if (hasContent) {
500
- let nodeType = 'paragraph';
501
- let metadata = undefined;
502
- if (headingLevel !== undefined && headingLevel > 0) {
503
- nodeType = 'heading';
504
- metadata = { level: headingLevel };
505
- // Reset list context when we encounter a heading
506
- lastKnownListId = undefined;
507
- lastKnownListType = undefined;
508
- }
509
- else if (isListItem) {
510
- nodeType = 'list';
511
- // Use lastKnownListId if currentListId is not set
512
- // (happens when list item is detected via listtext/pntext)
513
- const effectiveListId = currentListId || lastKnownListId;
514
- const effectiveListType = listType || lastKnownListType || 'unordered';
515
- // Calculate itemIndex
516
- let itemIndex = 0;
517
- if (effectiveListId) {
518
- if (!listCounters[effectiveListId]) {
519
- listCounters[effectiveListId] = {};
520
- }
521
- if (listCounters[effectiveListId][paragraphIndent] === undefined) {
522
- listCounters[effectiveListId][paragraphIndent] = 0;
523
- }
524
- else {
525
- listCounters[effectiveListId][paragraphIndent]++;
524
+ // Heuristic list detection if no explicit list control words were found
525
+ if (!isListItem && paragraphIndent > 300) { // RTF indents are in twips (1440 = 1 inch)
526
+ const trimmed = currentParagraphText.trim();
527
+ // Check for bullet characters or digits followed by period
528
+ if (/^[\u2022\u00b7\-\u25cf\u25cb]/.test(trimmed) || /^\d+[.\)]/.test(trimmed)) {
529
+ isListItem = true;
530
+ listType = /^\d+[.\)]/.test(trimmed) ? 'ordered' : 'unordered';
531
+ }
532
+ }
533
+ if (headingLevel !== undefined && headingLevel > 0) {
534
+ nodeType = 'heading';
535
+ metadata = { level: headingLevel };
536
+ // Reset list context when we encounter a heading
537
+ lastKnownListId = undefined;
538
+ lastKnownListType = undefined;
539
+ }
540
+ else if (isListItem) {
541
+ nodeType = 'list';
542
+ // Use lastKnownListId if currentListId is not set
543
+ // (happens when list item is detected via listtext/pntext)
544
+ const effectiveListId = currentListId || lastKnownListId;
545
+ const effectiveListType = listType || lastKnownListType || 'unordered';
546
+ // Calculate itemIndex
547
+ let itemIndex = 0;
548
+ if (effectiveListId) {
549
+ if (!listCounters[effectiveListId]) {
550
+ listCounters[effectiveListId] = {};
551
+ }
552
+ // Reset all sub-level counters when returning to a shallower level
553
+ // This ensures that if we go from level 2 to level 0 and back to level 2,
554
+ // the new level 2 sequence starts from 0.
555
+ const levels = Object.keys(listCounters[effectiveListId]).map(l => parseInt(l));
556
+ for (const level of levels) {
557
+ if (level > paragraphIndent) {
558
+ delete listCounters[effectiveListId][level];
526
559
  }
527
- itemIndex = listCounters[effectiveListId][paragraphIndent];
528
560
  }
529
- metadata = {
530
- listType: effectiveListType,
531
- indentation: paragraphIndent,
532
- listId: effectiveListId || '',
533
- itemIndex: itemIndex,
534
- alignment: paragraphAlignment,
535
- };
536
- // Save for next listtext detection
537
- if (effectiveListId) {
538
- lastKnownListId = effectiveListId;
561
+ if (listCounters[effectiveListId][paragraphIndent] === undefined) {
562
+ listCounters[effectiveListId][paragraphIndent] = 0;
539
563
  }
540
- if (effectiveListType) {
541
- lastKnownListType = effectiveListType;
564
+ else {
565
+ listCounters[effectiveListId][paragraphIndent]++;
542
566
  }
543
- }
544
- const node = {
545
- type: nodeType,
546
- text: currentParagraphText,
547
- children: currentParagraphChildren,
548
- formatting: undefined,
549
- metadata: metadata
567
+ itemIndex = listCounters[effectiveListId][paragraphIndent];
568
+ }
569
+ metadata = {
570
+ listType: effectiveListType,
571
+ indentation: paragraphIndent,
572
+ listId: effectiveListId || '',
573
+ itemIndex: itemIndex,
574
+ alignment: paragraphAlignment,
550
575
  };
551
- if (config.includeRawContent && currentParagraphRaw) {
552
- node.rawContent = currentParagraphRaw;
576
+ // Save for next listtext detection
577
+ if (effectiveListId) {
578
+ lastKnownListId = effectiveListId;
553
579
  }
554
- // If we're building a table, add to current cell
555
- // but ONLY if this paragraph was actually marked as in-table
556
- if (inTable && paragraphInTable) {
557
- ensureTableContext();
558
- getCurrentTable().currentCellContent.push(node);
580
+ if (effectiveListType) {
581
+ lastKnownListType = effectiveListType;
559
582
  }
560
- else {
561
- currentTarget.push(node);
562
- }
563
- currentParagraphText = '';
564
- currentParagraphChildren = [];
565
- currentParagraphRaw = '';
566
- // Reset paragraph-level state
567
- paragraphIndent = 0;
568
- isListItem = false;
569
- listType = undefined;
570
- headingLevel = undefined;
571
- currentListId = undefined;
572
583
  }
573
- };
574
- // Helper to flush current cell
575
- const flushCell = (tableCtx) => {
576
- flushParagraph();
577
- const ctx = tableCtx || getCurrentTable();
578
- if (!ctx)
579
- return undefined;
580
- // Always return a cell node, even if empty, to preserve table structure (grid)
581
- const cellNode = {
582
- type: 'cell',
583
- text: ctx.currentCellContent.map(c => c.text).join('\n'),
584
- children: [...ctx.currentCellContent],
584
+ const node = {
585
+ type: nodeType,
586
+ text: currentParagraphText,
587
+ children: currentParagraphChildren,
588
+ formatting: undefined,
585
589
  metadata: {
586
- row: ctx.rowIndex,
587
- col: ctx.currentCells.length
590
+ ...metadata,
591
+ anchorIds: currentAnchorIds.length > 0 ? [...currentAnchorIds] : undefined
588
592
  }
589
593
  };
590
- ctx.currentCellContent = [];
591
- return cellNode;
592
- };
593
- // Helper to flush current row - creates a row node from collected cells
594
- const flushRow = (tableCtx) => {
595
- const ctx = tableCtx || getCurrentTable();
596
- if (!ctx)
597
- return;
598
- const cell = flushCell(ctx);
599
- // Only add cell if it has content (prevents phantom empty cells during cleanup)
600
- if (cell && (cell.children && cell.children.length > 0 || cell.text)) {
601
- ctx.currentCells.push(cell);
602
- }
603
- if (ctx.currentCells.length > 0) {
604
- const rowNode = {
605
- type: 'row',
606
- text: ctx.currentCells.map(c => c.text).filter(t => t !== '').join(config.newlineDelimiter ?? '\n'),
607
- children: [...ctx.currentCells]
608
- };
609
- ctx.rows.push(rowNode);
610
- ctx.currentCells = [];
611
- ctx.rowIndex++;
594
+ if (config.includeRawContent && currentParagraphRawChunks.length > 0) {
595
+ node.rawContent = currentParagraphRawChunks.join('');
612
596
  }
613
- };
614
- // Helper to flush table
615
- const flushTable = () => {
616
- if (isFlushingTable)
617
- return;
618
- isFlushingTable = true;
619
- const ctx = getCurrentTable();
620
- if (!ctx) {
621
- isFlushingTable = false;
622
- return;
597
+ // If we're building a table, add to current cell
598
+ // but ONLY if this paragraph was actually marked as in-table
599
+ if (inTable && paragraphInTable) {
600
+ ensureTableContext();
601
+ getCurrentTable().currentCellContent.push(node);
623
602
  }
624
- flushRow(ctx);
625
- if (ctx.rows.length > 0) {
626
- tableId++;
627
- const tableNode = {
628
- type: 'table',
629
- text: ctx.rows.map(r => r.text).join('\n'), // Aggregate text from rows
630
- children: [...ctx.rows]
631
- };
632
- // If we have a parent table, add this table to the parent's current cell
633
- if (tableStack.length > 1) {
634
- const parentCtx = tableStack[tableStack.length - 2];
635
- parentCtx.currentCellContent.push(tableNode);
636
- }
637
- else {
638
- currentTarget.push(tableNode);
639
- }
603
+ else {
604
+ currentTarget.push(node);
640
605
  }
641
- // Pop the table from stack
642
- tableStack.pop();
643
- // If stack is empty, we are out of table mode
644
- if (tableStack.length === 0) {
645
- inTable = false;
646
- paragraphInTable = false;
606
+ currentParagraphTextChunks = [];
607
+ currentParagraphChildren = [];
608
+ currentParagraphRawChunks = [];
609
+ // Reset paragraph-level state that should NOT persist
610
+ // Note: list properties (\ls, \ilvl, \li) and alignment (\ql, etc.)
611
+ // persist in RTF until \pard or a new value is set.
612
+ currentAnchorIds = []; // Reset anchors
613
+ }
614
+ };
615
+ // Helper to flush current cell
616
+ const flushCell = (tableCtx) => {
617
+ flushParagraph();
618
+ const ctx = tableCtx || getCurrentTable();
619
+ if (!ctx)
620
+ return undefined;
621
+ // Always return a cell node, even if empty, to preserve table structure (grid)
622
+ const cellNode = {
623
+ type: 'cell',
624
+ text: ctx.currentCellContent.map(c => c.text).join('\n'),
625
+ children: [...ctx.currentCellContent],
626
+ metadata: {
627
+ row: ctx.rowIndex,
628
+ col: ctx.currentCells.length
647
629
  }
648
- isFlushingTable = false;
649
630
  };
650
- // Extract hyperlink URL from field instruction group
651
- // Recursively searches for HYPERLINK "url" pattern in nested groups
652
- const extractHyperlinkUrl = (group) => {
653
- let url;
654
- // Helper to recursively find hyperlink URL
655
- const findUrl = (node) => {
656
- if (node.type === 'text') {
657
- // Check for HYPERLINK "url" pattern
658
- const text = node.value;
659
- const match = text.match(/HYPERLINK\s+"([^"]+)"/i);
660
- if (match) {
661
- return match[1];
662
- }
663
- // Also check for URL after HYPERLINK on same or separate text node
664
- const urlOnlyMatch = text.match(/"(https?:\/\/[^"]+|mailto:[^"]+|#[^"]+)"/);
665
- if (urlOnlyMatch) {
666
- return urlOnlyMatch[1];
667
- }
631
+ ctx.currentCellContent = [];
632
+ return cellNode;
633
+ };
634
+ // Helper to flush current row - creates a row node from collected cells
635
+ const flushRow = (tableCtx) => {
636
+ const ctx = tableCtx || getCurrentTable();
637
+ if (!ctx)
638
+ return;
639
+ const cell = flushCell(ctx);
640
+ // Only add cell if it has content (prevents phantom empty cells during cleanup)
641
+ if (cell && (cell.children && cell.children.length > 0 || cell.text)) {
642
+ ctx.currentCells.push(cell);
643
+ }
644
+ if (ctx.currentCells.length > 0) {
645
+ const rowNode = {
646
+ type: 'row',
647
+ text: ctx.currentCells.map(c => c.text).filter(t => t !== '').join(config.newlineDelimiter),
648
+ children: [...ctx.currentCells]
649
+ };
650
+ ctx.rows.push(rowNode);
651
+ ctx.currentCells = [];
652
+ ctx.rowIndex++;
653
+ }
654
+ };
655
+ // Helper to flush table
656
+ const flushTable = () => {
657
+ if (isFlushingTable)
658
+ return;
659
+ isFlushingTable = true;
660
+ const ctx = getCurrentTable();
661
+ if (!ctx) {
662
+ isFlushingTable = false;
663
+ return;
664
+ }
665
+ flushRow(ctx);
666
+ if (ctx.rows.length > 0) {
667
+ tableId++;
668
+ const tableNode = {
669
+ type: 'table',
670
+ text: ctx.rows.map(r => r.text).join('\n'), // Aggregate text from rows
671
+ children: [...ctx.rows]
672
+ };
673
+ // If we have a parent table, add this table to the parent's current cell
674
+ if (tableStack.length > 1) {
675
+ const parentCtx = tableStack[tableStack.length - 2];
676
+ parentCtx.currentCellContent.push(tableNode);
677
+ }
678
+ else {
679
+ currentTarget.push(tableNode);
680
+ }
681
+ }
682
+ // Pop the table from stack
683
+ tableStack.pop();
684
+ // If stack is empty, we are out of table mode
685
+ if (tableStack.length === 0) {
686
+ inTable = false;
687
+ paragraphInTable = false;
688
+ }
689
+ isFlushingTable = false;
690
+ };
691
+ // Extract hyperlink URL from field instruction group
692
+ // Recursively searches for HYPERLINK "url" pattern in nested groups
693
+ const extractHyperlinkUrl = (group) => {
694
+ let url;
695
+ // Helper to recursively find hyperlink URL
696
+ const findUrl = (node) => {
697
+ if (node.type === 'text') {
698
+ // Check for HYPERLINK "url" pattern
699
+ const text = node.value;
700
+ const match = text.match(/HYPERLINK\s+"([^"]+)"/i);
701
+ if (match) {
702
+ return match[1];
703
+ }
704
+ // Also check for URL after HYPERLINK on same or separate text node
705
+ const urlOnlyMatch = text.match(/"(https?:\/\/[^"]+|mailto:[^"]+|#[^"]+)"/);
706
+ if (urlOnlyMatch) {
707
+ return urlOnlyMatch[1];
668
708
  }
669
- else if (node.type === 'group') {
670
- // Check if this is a fldinst group (contains field instruction)
671
- let foundHyperlink = false;
672
- let foundUrl;
673
- for (const child of node.content) {
674
- if (child.type === 'text') {
675
- const text = child.value.toUpperCase();
676
- if (text.includes('HYPERLINK')) {
677
- foundHyperlink = true;
678
- }
679
- // Look for quoted URL
680
- const urlMatch = child.value.match(/"([^"]+)"/);
681
- if (urlMatch && foundHyperlink) {
682
- foundUrl = urlMatch[1];
683
- }
709
+ }
710
+ else if (node.type === 'group') {
711
+ // Check if this is a fldinst group (contains field instruction)
712
+ let foundHyperlink = false;
713
+ let foundUrl;
714
+ let isLocal = false;
715
+ for (const child of node.content) {
716
+ if (child.type === 'text') {
717
+ const text = child.value.toUpperCase();
718
+ if (text.includes('HYPERLINK')) {
719
+ foundHyperlink = true;
684
720
  }
685
- else if (child.type === 'group') {
686
- const nestedUrl = findUrl(child);
687
- if (nestedUrl) {
688
- foundUrl = nestedUrl;
689
- }
721
+ if (text.includes('\\L')) {
722
+ isLocal = true;
723
+ }
724
+ // Look for quoted URL/Anchor
725
+ const urlMatch = child.value.match(/"([^"]+)"/);
726
+ if (urlMatch && foundHyperlink) {
727
+ foundUrl = urlMatch[1];
728
+ }
729
+ }
730
+ else if (child.type === 'group') {
731
+ const nestedUrl = findUrl(child);
732
+ if (nestedUrl) {
733
+ foundUrl = nestedUrl;
690
734
  }
691
735
  }
692
- if (foundUrl)
693
- return foundUrl;
694
736
  }
695
- return undefined;
696
- };
697
- // Search the field group for fldinst
698
- for (const child of group.content) {
699
- if (child.type === 'group') {
700
- const foundUrl = findUrl(child);
701
- if (foundUrl)
702
- return foundUrl;
737
+ if (foundUrl) {
738
+ if (isLocal && !foundUrl.startsWith('#')) {
739
+ return '#' + foundUrl;
740
+ }
741
+ return foundUrl;
703
742
  }
704
743
  }
705
- return url;
706
- };
707
- /**
708
- * Determines whether a hyperlink URL is internal or external.
709
- * Internal = bookmark references (no scheme or starts with "#").
710
- * External = any scheme like http, https, mailto, ftp, file, etc.
711
- */
712
- const classifyLinkType = (url) => {
713
- // Trim whitespace
714
- const clean = url.trim();
715
- // Internal pattern 1: starts with "#"
716
- if (clean.startsWith('#')) {
717
- return 'internal';
718
- }
719
- // Internal pattern 2: no scheme at all (pure bookmark)
720
- // Detect schemes by checking "something:" prefix
721
- if (!/^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(clean)) {
722
- return 'internal';
723
- }
724
- // Everything else is external
725
- return 'external';
744
+ return undefined;
726
745
  };
727
- // Helper to extract text content from a group (for list marker detection)
728
- // Recursively collects all text content from a group
729
- const extractTextFromGroup = (group) => {
730
- let text = '';
731
- const collectText = (node) => {
732
- if (node.type === 'text') {
733
- text += node.value;
734
- }
735
- else if (node.type === 'group') {
736
- for (const child of node.content) {
737
- collectText(child);
738
- }
746
+ // Search the field group for fldinst
747
+ for (const child of group.content) {
748
+ if (child.type === 'group') {
749
+ const foundUrl = findUrl(child);
750
+ if (foundUrl)
751
+ return foundUrl;
752
+ }
753
+ }
754
+ return url;
755
+ };
756
+ /**
757
+ * Determines whether a hyperlink URL is internal or external.
758
+ * Internal = bookmark references (no scheme or starts with "#").
759
+ * External = any scheme like http, https, mailto, ftp, file, etc.
760
+ */
761
+ const classifyLinkType = (url) => {
762
+ // Trim whitespace
763
+ const clean = url.trim();
764
+ // Internal pattern 1: starts with "#"
765
+ if (clean.startsWith('#')) {
766
+ return 'internal';
767
+ }
768
+ // Internal pattern 2: no scheme at all (pure bookmark)
769
+ // Detect schemes by checking "something:" prefix
770
+ if (!/^[a-zA-Z][a-zA-Z0-9+.-]*:/.test(clean)) {
771
+ return 'internal';
772
+ }
773
+ // Everything else is external
774
+ return 'external';
775
+ };
776
+ // Helper to extract text content from a group (for list marker detection)
777
+ // Recursively collects all text content from a group
778
+ const extractTextFromGroup = (group) => {
779
+ let text = '';
780
+ const collectText = (node) => {
781
+ if (node.type === 'text') {
782
+ text += node.value;
783
+ }
784
+ else if (node.type === 'group') {
785
+ for (const child of node.content) {
786
+ collectText(child);
739
787
  }
740
- };
741
- for (const child of group.content) {
742
- collectText(child);
743
788
  }
744
- return text;
745
789
  };
746
- /**
747
- * Extracts an image attachment from an RTF \pict group.
748
- * Uses lookup tables for clean and safe handling of all formats.
749
- *
750
- * @param pictGroup The RtfGroup node that represents a \pict group.
751
- * @returns An OfficeAttachment or undefined when unsupported or invalid.
752
- */
753
- const extractPictAttachment = (pictGroup) => {
754
- /** Internal format detected from the RTF pict group */
755
- let imageFormat;
756
- /** Hexadecimal string extracted from the pict binary section */
757
- let hexData = '';
758
- // -------------------------------------------------------------
759
- // Walk through pict group content to detect the blip type and gather hex data
760
- // -------------------------------------------------------------
761
- for (const child of pictGroup.content) {
762
- // If the node is a control word, we try to resolve it from lookup map
763
- if (child.type === 'control') {
764
- // Lookup directly instead of if/else
765
- const mapped = RTF_BLIP_MAP[child.value];
766
- if (mapped) {
767
- imageFormat = mapped;
768
- }
769
- }
770
- else if (child.type === 'text') {
771
- // Append only valid hex characters
772
- hexData += child.value.replace(/[^0-9a-fA-F]/g, '');
773
- }
774
- else if (child.type === 'group') {
775
- // Skip nested structures like picprop or blipuid
790
+ for (const child of group.content) {
791
+ collectText(child);
792
+ }
793
+ return text;
794
+ };
795
+ /**
796
+ * Extracts an image attachment from an RTF \pict group.
797
+ * Uses lookup tables for clean and safe handling of all formats.
798
+ *
799
+ * @param pictGroup The RtfGroup node that represents a \pict group.
800
+ * @returns An OfficeAttachment or undefined when unsupported or invalid.
801
+ */
802
+ const extractPictAttachment = (pictGroup) => {
803
+ /** Internal format detected from the RTF pict group */
804
+ let imageFormat;
805
+ /** Hexadecimal string chunks extracted from the pict binary section */
806
+ const hexDataChunks = [];
807
+ // -------------------------------------------------------------
808
+ // Walk through pict group content to detect the blip type and gather hex data
809
+ // -------------------------------------------------------------
810
+ for (const child of pictGroup.content) {
811
+ // If the node is a control word, we try to resolve it from lookup map
812
+ if (child.type === 'control') {
813
+ // Lookup directly instead of if/else
814
+ const mapped = RTF_BLIP_MAP[child.value];
815
+ if (mapped) {
816
+ imageFormat = mapped;
776
817
  }
777
818
  }
778
- // Missing or unknown format → stop
779
- if (!imageFormat || hexData.length === 0) {
780
- return undefined;
819
+ else if (child.type === 'text') {
820
+ // Append only valid hex characters
821
+ hexDataChunks.push(child.value.replace(/[^0-9a-fA-F]/g, ''));
781
822
  }
782
- // Map internal format to MIME
783
- const mimeType = IMAGE_MIME_MAP[imageFormat];
784
- if (!mimeType) {
785
- return undefined;
823
+ else if (child.type === 'group') {
824
+ // Skip nested structures like picprop or blipuid
786
825
  }
787
- // -------------------------------------------------------------
788
- // Convert hex → binary and construct the attachment object
789
- // -------------------------------------------------------------
790
- try {
791
- // Convert hex into a raw buffer
792
- const buffer = Buffer.from(hexData, 'hex');
793
- // Derive file extension directly from format
794
- const extension = imageFormat;
795
- // Generate a stable incremental filename
796
- const name = `image_${attachments.length + 1}.${extension}`;
797
- // Build and return final attachment
798
- return {
799
- type: 'image',
800
- mimeType: mimeType,
801
- data: buffer.toString('base64'),
802
- name: name,
803
- extension: extension
804
- };
826
+ }
827
+ // Missing or unknown format → stop
828
+ const hexData = hexDataChunks.join('');
829
+ if (!imageFormat || hexData.length === 0) {
830
+ return undefined;
831
+ }
832
+ // Map internal format to MIME
833
+ const mimeType = IMAGE_MIME_MAP[imageFormat];
834
+ if (!mimeType) {
835
+ return undefined;
836
+ }
837
+ // -------------------------------------------------------------
838
+ // Convert hex → binary and construct the attachment object
839
+ // -------------------------------------------------------------
840
+ try {
841
+ // Convert hex into a raw buffer
842
+ const buffer = Buffer.from(hexData, 'hex');
843
+ // Derive file extension directly from format
844
+ const extension = imageFormat;
845
+ // Generate a stable incremental filename
846
+ const name = `image_${attachments.length + 1}.${extension}`;
847
+ // Build and return final attachment
848
+ return {
849
+ type: 'image',
850
+ mimeType: mimeType,
851
+ data: buffer.toString('base64'),
852
+ name: name,
853
+ extension: extension
854
+ };
855
+ }
856
+ catch {
857
+ // If conversion fails, ignore this image
858
+ return undefined;
859
+ }
860
+ };
861
+ // Helper to serialize RTF control word
862
+ const serializeRtfControl = (node) => {
863
+ // Symbol control words (non-alpha)
864
+ if (!/^[a-zA-Z]/.test(node.value)) {
865
+ return `\\${node.value}`;
866
+ }
867
+ // Alpha control words
868
+ let res = `\\${node.value}`;
869
+ if (node.param !== undefined) {
870
+ res += node.param;
871
+ }
872
+ // Add space delimiter for safety
873
+ res += ' ';
874
+ return res;
875
+ };
876
+ // Helper to extract text from a group (for bookmark names, etc.)
877
+ const extractGroupText = (group) => {
878
+ let text = '';
879
+ for (const item of group.content) {
880
+ if (item.type === 'text') {
881
+ text += item.value;
805
882
  }
806
- catch {
807
- // If conversion fails, ignore this image
808
- return undefined;
883
+ else if (item.type === 'group') {
884
+ text += extractGroupText(item);
809
885
  }
810
- };
811
- // Helper to serialize RTF control word
812
- const serializeRtfControl = (node) => {
813
- // Symbol control words (non-alpha)
814
- if (!/^[a-zA-Z]/.test(node.value)) {
815
- return `\\${node.value}`;
816
- }
817
- // Alpha control words
818
- let res = `\\${node.value}`;
819
- if (node.param !== undefined) {
820
- res += node.param;
821
- }
822
- // Add space delimiter for safety
823
- res += ' ';
824
- return res;
825
- };
826
- // Helper to serialize RTF text
827
- const serializeRtfText = (node) => {
828
- // Escape special characters: \, {, }
829
- return node.value.replace(/([\\{}])/g, '\\$1');
830
- };
831
- // Recursive function to traverse the RTF tree
832
- const traverse = (node, formatting, depth = 0) => {
833
- if (node.type === 'group') {
834
- const ignoreList = [
835
- 'fonttbl', 'colortbl', 'stylesheet', 'info', 'macpict',
836
- 'pmmetafile', 'wmetafile', 'dibitmap', 'bitmap', 'object',
837
- 'nextGenerator', 'header', 'footer', 'nonshppict', 'xml', 'private',
838
- 'upnp', 'ud', 'filetbl', 'operator', 'author', 'creatim', 'revtim', 'printim', 'comment',
839
- 'fldinst', 'listtext', 'pntext' // Ignore list marker text (handled separately)
840
- ];
841
- let isIgnored = false;
842
- let isFootnote = false;
843
- let isHyperlinkField = false;
844
- let isPict = false;
845
- // Add group start to raw content
846
- // Note: We don't add ignored groups to rawContent to keep it clean
847
- // But we might want to if we want full fidelity.
848
- // For now, let's include everything in rawContent except truly skipped stuff?
849
- // The user asked for "raw content which is probably the rtf group".
850
- // If we skip 'fonttbl', it's fine as it's not part of the content.
851
- // But 'listtext' IS part of the content structure even if we parse it separately.
852
- // Let's stick to the plan: if ignored, we might skip it in rawContent too,
853
- // OR we include it.
854
- // If I include it, `currentParagraphRaw` might get huge with font tables if they were inside the paragraph (unlikely).
855
- // Usually font tables are at document root.
856
- // `traverse` is called on `doc`.
857
- // `currentParagraphRaw` is reset on `flushParagraph`.
858
- // So if we are at root level, `currentParagraphRaw` accumulates everything until the first paragraph ends.
859
- // This might include the header/fonttbl if they are before the first \par.
860
- // That seems correct for "raw content" of the first node?
861
- // Actually, `fonttbl` is usually before any text.
862
- // If we include it, the first paragraph node will contain the entire font table in its rawContent.
863
- // That might be annoying.
864
- // Let's ONLY add to `currentParagraphRaw` if NOT ignored.
865
- if (node.destination) {
866
- if (node.destination === 'footnote') {
867
- isFootnote = true;
868
- }
869
- else if (node.destination === 'field') {
870
- // Check if this is a hyperlink field
871
- const url = extractHyperlinkUrl(node);
872
- if (url) {
873
- // Flush any pending text before starting the link context
874
- // This prevents previous text from inheriting the link
875
- flushRun();
876
- isHyperlinkField = true;
877
- currentLinkUrl = url;
878
- }
879
- }
880
- else if (node.destination === 'listtable') {
881
- // We want to parse list definitions
882
- parsingListTable = true;
886
+ }
887
+ return text.trim();
888
+ };
889
+ // Helper to serialize RTF text
890
+ const serializeRtfText = (node) => {
891
+ // Escape special characters: \, {, }
892
+ return node.value.replace(/([\\{}])/g, '\\$1');
893
+ };
894
+ // Recursive function to traverse the RTF tree
895
+ const traverse = (node, formatting, depth = 0) => {
896
+ if (node.type === 'group') {
897
+ const ignoreList = [
898
+ 'fonttbl', 'colortbl', 'stylesheet', 'info', 'macpict',
899
+ 'pmmetafile', 'wmetafile', 'dibitmap', 'bitmap', 'object',
900
+ 'nextGenerator', 'header', 'footer', 'nonshppict', 'xml', 'private',
901
+ 'upnp', 'ud', 'filetbl', 'operator', 'author', 'creatim', 'revtim', 'printim', 'comment',
902
+ 'fldinst', 'listtext', 'pntext' // Ignore list marker text (handled separately)
903
+ ];
904
+ let isIgnored = false;
905
+ let isFootnote = false;
906
+ let isHyperlinkField = false;
907
+ let isPict = false;
908
+ // Add group start to raw content
909
+ // Note: We don't add ignored groups to rawContent to keep it clean
910
+ // But we might want to if we want full fidelity.
911
+ // For now, let's include everything in rawContent except truly skipped stuff?
912
+ // The user asked for "raw content which is probably the rtf group".
913
+ // If we skip 'fonttbl', it's fine as it's not part of the content.
914
+ // But 'listtext' IS part of the content structure even if we parse it separately.
915
+ // Let's stick to the plan: if ignored, we might skip it in rawContent too,
916
+ // OR we include it.
917
+ // If I include it, `currentParagraphRaw` might get huge with font tables if they were inside the paragraph (unlikely).
918
+ // Usually font tables are at document root.
919
+ // `traverse` is called on `doc`.
920
+ // `currentParagraphRaw` is reset on `flushParagraph`.
921
+ // So if we are at root level, `currentParagraphRaw` accumulates everything until the first paragraph ends.
922
+ // This might include the header/fonttbl if they are before the first \par.
923
+ // That seems correct for "raw content" of the first node?
924
+ // Actually, `fonttbl` is usually before any text.
925
+ // If we include it, the first paragraph node will contain the entire font table in its rawContent.
926
+ // That might be annoying.
927
+ // Let's ONLY add to `currentParagraphRaw` if NOT ignored.
928
+ if (node.destination) {
929
+ if (node.destination === 'footnote') {
930
+ isFootnote = true;
931
+ }
932
+ else if (node.destination === 'field') {
933
+ // Check if this is a hyperlink field
934
+ const url = extractHyperlinkUrl(node);
935
+ if (url) {
936
+ // Flush any pending text before starting the link context
937
+ // This prevents previous text from inheriting the link
938
+ flushRun();
939
+ isHyperlinkField = true;
940
+ currentLinkUrl = url;
883
941
  }
884
- else if (parsingListTable && node.destination === 'list') {
885
- parsingListDefinition = true;
886
- currentDefinedListId = undefined;
887
- currentDefinedListType = undefined;
942
+ }
943
+ else if (node.destination === 'listtable') {
944
+ // We want to parse list definitions
945
+ parsingListTable = true;
946
+ }
947
+ else if (parsingListTable && node.destination === 'list') {
948
+ parsingListDefinition = true;
949
+ currentDefinedListId = undefined;
950
+ currentDefinedListType = undefined;
951
+ }
952
+ else if (node.destination === 'listoverridetable') {
953
+ parsingListOverrideTable = true;
954
+ }
955
+ else if (parsingListOverrideTable && node.destination === 'listoverride') {
956
+ // Reset per override group
957
+ currentListOverrideListId = undefined;
958
+ currentListOverrideLs = undefined;
959
+ }
960
+ else if (node.destination === 'pict') {
961
+ // Handle picture extraction
962
+ if (config.extractAttachments) {
963
+ isPict = true;
888
964
  }
889
- else if (node.destination === 'listoverridetable') {
890
- parsingListOverrideTable = true;
965
+ else {
966
+ isIgnored = true;
891
967
  }
892
- else if (parsingListOverrideTable && node.destination === 'listoverride') {
893
- // Reset per override group
894
- currentListOverrideListId = undefined;
895
- currentListOverrideLs = undefined;
968
+ }
969
+ else if (ignoreList.includes(node.destination)) {
970
+ isIgnored = true;
971
+ }
972
+ else if (node.content.length > 0 && node.content[0].type === 'control' && node.content[0].value === '*') {
973
+ // Ignorable destination, but allow certain ones for:
974
+ // - fldinst: hyperlinks
975
+ // - nesttableprops: nested tables
976
+ // - shppict: shape pictures (contain pict groups)
977
+ const allowedIgnorable = ['fldinst', 'nesttableprops'];
978
+ if (config.extractAttachments) {
979
+ allowedIgnorable.push('shppict', 'listpicture');
896
980
  }
897
- else if (node.destination === 'pict') {
898
- // Handle picture extraction
899
- if (config.extractAttachments) {
900
- isPict = true;
901
- }
902
- else {
981
+ if (!allowedIgnorable.includes(node.destination || '')) {
982
+ if (node.destination === 'bkmkstart') {
983
+ const name = extractGroupText(node);
984
+ if (name && !currentAnchorIds.includes(name)) {
985
+ currentAnchorIds.push(name);
986
+ }
903
987
  isIgnored = true;
904
988
  }
905
- }
906
- else if (ignoreList.includes(node.destination)) {
907
- isIgnored = true;
908
- }
909
- else if (node.content.length > 0 && node.content[0].type === 'control' && node.content[0].value === '*') {
910
- // Ignorable destination, but allow certain ones for:
911
- // - fldinst: hyperlinks
912
- // - nesttableprops: nested tables
913
- // - shppict: shape pictures (contain pict groups)
914
- const allowedIgnorable = ['fldinst', 'nesttableprops'];
915
- if (config.extractAttachments) {
916
- allowedIgnorable.push('shppict', 'listpicture');
917
- }
918
- if (!allowedIgnorable.includes(node.destination || '')) {
989
+ else if (node.destination === 'bkmkend') {
919
990
  isIgnored = true;
920
991
  }
921
- }
922
- }
923
- // ═══════════════════════════════════════════════════════════
924
- // Handle listtext and pntext: These indicate the current paragraph
925
- // is a list item. We extract list type info before ignoring content.
926
- // ═══════════════════════════════════════════════════════════
927
- if (node.destination === 'listtext' || node.destination === 'pntext') {
928
- // If this is the first list item, reset the indent
929
- if (!isListItem)
930
- paragraphIndent = 0;
931
- isListItem = true;
932
- // Try to determine list type from the marker content
933
- // Bullets (unordered): '·', '•', 'o', '§', etc.
934
- // Numbers (ordered): '1.', '2.', 'i.', 'ii.', 'a.', 'A.', etc.
935
- const markerText = extractTextFromGroup(node);
936
- if (markerText) {
937
- const trimmed = markerText.trim();
938
- // Check for common bullet characters
939
- const bulletChars = ['·', '•', 'o', '§', '■', '□', '●', '○', '◆', '◇', '►', '▸', '\u00b7', '\u2022', '\u25cf', '\u25cb'];
940
- const isBullet = bulletChars.some(b => trimmed.includes(b)) ||
941
- // Font symbol bullets often use characters from Symbol font
942
- (trimmed.length === 1 && !/[0-9a-zA-Z]/.test(trimmed));
943
- if (isBullet) {
944
- listType = 'unordered';
945
- }
946
- else if (/^[0-9ivxlcdm]+[\.\)]/i.test(trimmed) || /^[a-z][\.\)]/i.test(trimmed)) {
947
- // Matches: 1., 2), i., ii., a., A), etc.
948
- listType = 'ordered';
992
+ else {
993
+ isIgnored = true;
949
994
  }
950
- // If we can't determine, leave listType as is (might be set by \ls/\levelnfc)
951
995
  }
952
- // Still mark as ignored to skip the marker text content
953
- isIgnored = true;
954
996
  }
955
- if (isIgnored)
956
- return;
957
- // Append group start to raw content
958
- currentParagraphRaw += '{';
959
- // Handle pict group: extract image and add to content tree
960
- if (isPict) {
961
- const attachment = extractPictAttachment(node);
962
- if (attachment) {
963
- attachments.push(attachment);
964
- // Only add image node to content if this is NOT a list definition picture
965
- // List pictures (bullets) should not appear in content, only as attachments
966
- if (!parsingListTable && !parsingListDefinition) {
967
- // Also add an image node to the content tree (like DOCX)
968
- flushParagraph();
969
- currentTarget.push({
970
- type: 'image',
971
- text: '',
972
- metadata: {
973
- attachmentName: attachment.name || `image_${attachments.length}`
974
- }
975
- });
976
- }
977
- }
978
- // We still traverse pict content to reconstruct raw RTF?
979
- // No, extractPictAttachment consumes it.
980
- // But we want it in rawContent?
981
- // If we return here, we miss the closing '}'.
982
- // And we miss the content in rawContent.
983
- // Let's traverse it purely for rawContent if needed, but `extractPictAttachment` doesn't modify the tree.
984
- // But `extractPictAttachment` does not return the raw string.
985
- // So we should probably continue traversal but suppress text extraction?
986
- // The original code returned here: `return; // Don't traverse pict content as text`
987
- // So we should do the same, but we need to append the content to `currentParagraphRaw`.
988
- // We can manually serialize the group content here.
989
- for (const child of node.content) {
990
- if (child.type === 'control')
991
- currentParagraphRaw += serializeRtfControl(child);
992
- else if (child.type === 'text')
993
- currentParagraphRaw += serializeRtfText(child);
994
- else if (child.type === 'group') {
995
- // Recursive serialization for nested groups in pict (e.g. blipuid)
996
- // We can't easily recurse `traverse` because it has side effects (text extraction).
997
- // We need a pure serializer or just let `traverse` run but with a flag?
998
- // Or just ignore the raw content of the image binary data?
999
- // Image binary data can be huge.
1000
- // Maybe we shouldn't include the full hex dump in `rawContent`?
1001
- // The user said "raw content which is probably the rtf group".
1002
- // Including 5MB of hex data in the JSON AST might be bad.
1003
- // But for consistency, it is the raw content.
1004
- // Let's include it for now.
1005
- // To do this without side effects, we need a separate serialize function?
1006
- // Or just call traverse and ensure `isPict` logic prevents text extraction.
1007
- // Wait, `isPict` is true for this node.
1008
- // If we recurse, `isPict` will be false for children (unless they are also pict).
1009
- // But we want to suppress text extraction for children of pict.
1010
- // The original code did `return`.
1011
- // So we should manually serialize children here.
1012
- // Let's define a simple recursive serializer.
1013
- const serializeGroupContent = (g) => {
1014
- for (const c of g.content) {
1015
- if (c.type === 'control')
1016
- currentParagraphRaw += serializeRtfControl(c);
1017
- else if (c.type === 'text')
1018
- currentParagraphRaw += serializeRtfText(c);
1019
- else if (c.type === 'group') {
1020
- currentParagraphRaw += '{';
1021
- serializeGroupContent(c);
1022
- currentParagraphRaw += '}';
1023
- }
1024
- }
1025
- };
1026
- serializeGroupContent(node);
1027
- }
1028
- }
1029
- currentParagraphRaw += '}';
1030
- return;
1031
- }
1032
- // Handle footnote: switch target to notes
1033
- const previousTarget = currentTarget;
1034
- if (isFootnote) {
1035
- if (config.ignoreNotes) {
1036
- return; // Skip footnote content entirely
997
+ }
998
+ // ═══════════════════════════════════════════════════════════
999
+ // Handle listtext and pntext: These indicate the current paragraph
1000
+ // is a list item. We extract list type info before ignoring content.
1001
+ // ═══════════════════════════════════════════════════════════
1002
+ if (node.destination === 'listtext' || node.destination === 'pntext') {
1003
+ // If this is the first list item, reset the indent
1004
+ if (!isListItem)
1005
+ paragraphIndent = 0;
1006
+ isListItem = true;
1007
+ // Try to determine list type from the marker content
1008
+ // Bullets (unordered): '·', '•', 'o', '§', etc.
1009
+ // Numbers (ordered): '1.', '2.', 'i.', 'ii.', 'a.', 'A.', etc.
1010
+ const markerText = extractTextFromGroup(node);
1011
+ if (markerText) {
1012
+ const trimmed = markerText.trim();
1013
+ // Check for common bullet characters
1014
+ const bulletChars = ['·', '•', 'o', '§', '■', '□', '●', '○', '◆', '◇', '►', '▸', '\u00b7', '\u2022', '\u25cf', '\u25cb'];
1015
+ const isBullet = bulletChars.some(b => trimmed.includes(b)) ||
1016
+ // Font symbol bullets often use characters from Symbol font
1017
+ (trimmed.length === 1 && !/[0-9a-zA-Z]/.test(trimmed));
1018
+ if (isBullet) {
1019
+ listType = 'unordered';
1037
1020
  }
1038
- flushParagraph();
1039
- currentFootnoteId++;
1040
- // Determine note type based on \fet value
1041
- let noteType = 'footnote';
1042
- if (fetValue === 1) {
1043
- // \fet1 means all notes are endnotes
1044
- noteType = 'endnote';
1021
+ else if (/^[0-9ivxlcdm]+[\.\)]/i.test(trimmed) || /^[a-z][\.\)]/i.test(trimmed)) {
1022
+ // Matches: 1., 2), i., ii., a., A), etc.
1023
+ listType = 'ordered';
1045
1024
  }
1046
- else if (fetValue === 2) {
1047
- // \fet2 means both types exist
1048
- // Check for \ftnalt marker to distinguish endnotes from footnotes
1049
- // \footnote\ftnalt indicates an endnote
1050
- const hasFtnalt = node.content.some(child => child.type === 'control' && child.value === 'ftnalt');
1051
- noteType = hasFtnalt ? 'endnote' : 'footnote';
1025
+ // If we can't determine, leave listType as is (might be set by \ls/\levelnfc)
1026
+ }
1027
+ // Still mark as ignored to skip the marker text content
1028
+ isIgnored = true;
1029
+ }
1030
+ if (isIgnored)
1031
+ return;
1032
+ // Append group start to raw content
1033
+ currentParagraphRawChunks.push('{');
1034
+ // Handle pict group: extract image and add to content tree
1035
+ if (isPict) {
1036
+ const attachment = extractPictAttachment(node);
1037
+ if (attachment) {
1038
+ attachments.push(attachment);
1039
+ // Only add image node to content if this is NOT a list definition picture
1040
+ // List pictures (bullets) should not appear in content, only as attachments
1041
+ if (!parsingListTable && !parsingListDefinition) {
1042
+ // Also add an image node to the content tree (like DOCX)
1043
+ flushParagraph();
1044
+ currentTarget.push({
1045
+ type: 'image',
1046
+ text: '',
1047
+ metadata: {
1048
+ attachmentName: attachment.name || `image_${attachments.length}`
1049
+ }
1050
+ });
1052
1051
  }
1053
- // fetValue === 0 (default) means footnotes only
1054
- const noteNode = {
1055
- type: 'note',
1056
- children: [],
1057
- metadata: {
1058
- noteId: currentFootnoteId.toString(),
1059
- noteType: noteType
1060
- }
1061
- };
1062
- notes.push(noteNode);
1063
- currentTarget = noteNode.children;
1064
1052
  }
1065
- // Create a new formatting context for the group
1066
- const groupFormatting = { ...formatting };
1053
+ // We still traverse pict content to reconstruct raw RTF?
1054
+ // No, extractPictAttachment consumes it.
1055
+ // But we want it in rawContent?
1056
+ // If we return here, we miss the closing '}'.
1057
+ // And we miss the content in rawContent.
1058
+ // Let's traverse it purely for rawContent if needed, but `extractPictAttachment` doesn't modify the tree.
1059
+ // But `extractPictAttachment` does not return the raw string.
1060
+ // So we should probably continue traversal but suppress text extraction?
1061
+ // The original code returned here: `return; // Don't traverse pict content as text`
1062
+ // So we should do the same, but we need to append the content to `currentParagraphRaw`.
1063
+ // We can manually serialize the group content here.
1067
1064
  for (const child of node.content) {
1068
- // Skip fldinst groups (we already extracted the URL)
1069
- if (child.type === 'group' && child.destination === 'fldinst') {
1070
- // We still want it in rawContent!
1071
- // So we should traverse it but suppress text extraction?
1072
- // Or just serialize it?
1073
- // `fldinst` contains the URL.
1074
- // If we skip it in `traverse`, we miss it in `rawContent`.
1075
- // Let's traverse it but maybe the `fldinst` logic inside `traverse` handles it?
1076
- // The original code:
1077
- // if (child.type === 'group' && child.destination === 'fldinst') { continue; }
1078
- // This skips the child entirely.
1079
- // So we need to manually serialize it if we want it in rawContent.
1080
- currentParagraphRaw += '{';
1081
- // We need to serialize the content of fldinst
1065
+ if (child.type === 'control')
1066
+ currentParagraphRawChunks.push(serializeRtfControl(child));
1067
+ else if (child.type === 'text')
1068
+ currentParagraphRawChunks.push(serializeRtfText(child));
1069
+ else if (child.type === 'group') {
1070
+ // Recursive serialization for nested groups in pict (e.g. blipuid)
1071
+ // We can't easily recurse `traverse` because it has side effects (text extraction).
1072
+ // We need a pure serializer or just let `traverse` run but with a flag?
1073
+ // Or just ignore the raw content of the image binary data?
1074
+ // Image binary data can be huge.
1075
+ // Maybe we shouldn't include the full hex dump in `rawContent`?
1076
+ // The user said "raw content which is probably the rtf group".
1077
+ // Including 5MB of hex data in the JSON AST might be bad.
1078
+ // But for consistency, it is the raw content.
1079
+ // Let's include it for now.
1080
+ // To do this without side effects, we need a separate serialize function?
1081
+ // Or just call traverse and ensure `isPict` logic prevents text extraction.
1082
+ // Wait, `isPict` is true for this node.
1083
+ // If we recurse, `isPict` will be false for children (unless they are also pict).
1084
+ // But we want to suppress text extraction for children of pict.
1085
+ // The original code did `return`.
1086
+ // So we should manually serialize children here.
1087
+ // Let's define a simple recursive serializer.
1082
1088
  const serializeGroupContent = (g) => {
1083
1089
  for (const c of g.content) {
1084
1090
  if (c.type === 'control')
1085
- currentParagraphRaw += serializeRtfControl(c);
1091
+ currentParagraphRawChunks.push(serializeRtfControl(c));
1086
1092
  else if (c.type === 'text')
1087
- currentParagraphRaw += serializeRtfText(c);
1093
+ currentParagraphRawChunks.push(serializeRtfText(c));
1088
1094
  else if (c.type === 'group') {
1089
- currentParagraphRaw += '{';
1095
+ currentParagraphRawChunks.push('{');
1090
1096
  serializeGroupContent(c);
1091
- currentParagraphRaw += '}';
1097
+ currentParagraphRawChunks.push('}');
1092
1098
  }
1093
1099
  }
1094
1100
  };
1095
- serializeGroupContent(child);
1096
- currentParagraphRaw += '}';
1097
- continue;
1101
+ serializeGroupContent(node);
1098
1102
  }
1099
- traverse(child, groupFormatting, depth + 1);
1100
- }
1101
- if (node.destination === 'listtable') {
1102
- parsingListTable = false;
1103
1103
  }
1104
- else if (node.destination === 'list') {
1105
- parsingListDefinition = false;
1106
- if (currentDefinedListId !== undefined && currentDefinedListType !== undefined) {
1107
- listTypeMap[currentDefinedListId] = currentDefinedListType;
1104
+ currentParagraphRawChunks.push('}');
1105
+ return;
1106
+ }
1107
+ // Handle footnote: switch target to notes
1108
+ const previousTarget = currentTarget;
1109
+ if (isFootnote) {
1110
+ if (config.ignoreNotes) {
1111
+ return; // Skip footnote content entirely
1112
+ }
1113
+ flushParagraph();
1114
+ currentFootnoteId++;
1115
+ // Determine note type based on \fet value
1116
+ let noteType = 'footnote';
1117
+ if (fetValue === 1) {
1118
+ // \fet1 means all notes are endnotes
1119
+ noteType = 'endnote';
1120
+ }
1121
+ else if (fetValue === 2) {
1122
+ // \fet2 means both types exist
1123
+ // Check for \ftnalt marker to distinguish endnotes from footnotes
1124
+ // \footnote\ftnalt indicates an endnote
1125
+ const hasFtnalt = node.content.some(child => child.type === 'control' && child.value === 'ftnalt');
1126
+ noteType = hasFtnalt ? 'endnote' : 'footnote';
1127
+ }
1128
+ // fetValue === 0 (default) means footnotes only
1129
+ const noteNode = {
1130
+ type: 'note',
1131
+ children: [],
1132
+ metadata: {
1133
+ noteId: currentFootnoteId.toString(),
1134
+ noteType: noteType
1108
1135
  }
1136
+ };
1137
+ notes.push(noteNode);
1138
+ currentTarget = noteNode.children;
1139
+ }
1140
+ // Create a new formatting context for the group
1141
+ const groupFormatting = { ...formatting };
1142
+ for (const child of node.content) {
1143
+ // Skip fldinst groups (we already extracted the URL)
1144
+ if (child.type === 'group' && child.destination === 'fldinst') {
1145
+ // We still want it in rawContent!
1146
+ // So we should traverse it but suppress text extraction?
1147
+ // Or just serialize it?
1148
+ // `fldinst` contains the URL.
1149
+ // If we skip it in `traverse`, we miss it in `rawContent`.
1150
+ // Let's traverse it but maybe the `fldinst` logic inside `traverse` handles it?
1151
+ // The original code:
1152
+ // if (child.type === 'group' && child.destination === 'fldinst') { continue; }
1153
+ // This skips the child entirely.
1154
+ // So we need to manually serialize it if we want it in rawContent.
1155
+ currentParagraphRawChunks.push('{');
1156
+ // We need to serialize the content of fldinst
1157
+ const serializeGroupContent = (g) => {
1158
+ for (const c of g.content) {
1159
+ if (c.type === 'control')
1160
+ currentParagraphRawChunks.push(serializeRtfControl(c));
1161
+ else if (c.type === 'text')
1162
+ currentParagraphRawChunks.push(serializeRtfText(c));
1163
+ else if (c.type === 'group') {
1164
+ currentParagraphRawChunks.push('{');
1165
+ serializeGroupContent(c);
1166
+ currentParagraphRawChunks.push('}');
1167
+ }
1168
+ }
1169
+ };
1170
+ serializeGroupContent(child);
1171
+ currentParagraphRawChunks.push('}');
1172
+ continue;
1109
1173
  }
1110
- else if (node.destination === 'listoverridetable') {
1111
- parsingListOverrideTable = false;
1174
+ traverse(child, groupFormatting, depth + 1);
1175
+ }
1176
+ if (node.destination === 'listtable') {
1177
+ parsingListTable = false;
1178
+ }
1179
+ else if (node.destination === 'list') {
1180
+ parsingListDefinition = false;
1181
+ if (currentDefinedListId !== undefined && currentDefinedListType !== undefined) {
1182
+ listTypeMap[currentDefinedListId] = currentDefinedListType;
1112
1183
  }
1113
- else if (parsingListOverrideTable && node.destination === 'listoverride') {
1114
- // End of listoverride group - populate map
1115
- if (currentListOverrideLs !== undefined && currentListOverrideListId !== undefined) {
1116
- listOverrideMap[currentListOverrideLs] = currentListOverrideListId;
1184
+ }
1185
+ else if (node.destination === 'listoverridetable') {
1186
+ parsingListOverrideTable = false;
1187
+ }
1188
+ else if (parsingListOverrideTable && node.destination === 'listoverride') {
1189
+ // End of listoverride group - populate map
1190
+ if (currentListOverrideLs !== undefined && currentListOverrideListId !== undefined) {
1191
+ listOverrideMap[currentListOverrideLs] = currentListOverrideListId;
1192
+ }
1193
+ }
1194
+ if (isFootnote) {
1195
+ flushParagraph();
1196
+ currentTarget = previousTarget;
1197
+ }
1198
+ // Clear link URL after processing the field group
1199
+ if (isHyperlinkField) {
1200
+ flushRun();
1201
+ currentLinkUrl = undefined;
1202
+ }
1203
+ // Append group end to raw content
1204
+ currentParagraphRawChunks.push('}');
1205
+ }
1206
+ else if (node.type === 'text') {
1207
+ if (parsingListTable || parsingListOverrideTable) {
1208
+ // Even if we don't extract text, we might want it in rawContent?
1209
+ // Yes, rawContent should reflect the source.
1210
+ currentParagraphRawChunks.push(serializeRtfText(node));
1211
+ return;
1212
+ }
1213
+ if (formattingChanged(currentFormatting, formatting)) {
1214
+ flushRun();
1215
+ currentFormatting = { ...formatting };
1216
+ }
1217
+ currentRunTextChunks.push(node.value);
1218
+ currentParagraphRawChunks.push(serializeRtfText(node));
1219
+ }
1220
+ else if (node.type === 'control') {
1221
+ // Append control to raw content
1222
+ currentParagraphRawChunks.push(serializeRtfControl(node));
1223
+ // Handle list definition control words
1224
+ if (parsingListTable) {
1225
+ if (node.value === 'listid') {
1226
+ currentDefinedListId = node.param;
1227
+ }
1228
+ else if (node.value === 'levelnfc' || node.value === 'levelnfcn') {
1229
+ // 0 = Arabic, 1 = Upper Roman, 2 = Lower Roman, 3 = Upper Alpha, 4 = Lower Alpha -> Ordered
1230
+ // 23 = Bullet, 255 = None -> Unordered
1231
+ const isOrdered = node.param !== undefined && (node.param === 0 || (node.param >= 0 && node.param <= 4));
1232
+ // Only set if not already set (or prioritize ordered if mixed?)
1233
+ // We'll assume if any level is ordered, it's ordered.
1234
+ // Or if we haven't set it yet.
1235
+ if (!currentDefinedListType || (currentDefinedListType === 'unordered' && isOrdered)) {
1236
+ currentDefinedListType = isOrdered ? 'ordered' : 'unordered';
1117
1237
  }
1118
1238
  }
1119
- if (isFootnote) {
1120
- flushParagraph();
1121
- currentTarget = previousTarget;
1239
+ return;
1240
+ }
1241
+ // Handle list override control words
1242
+ if (parsingListOverrideTable) {
1243
+ if (node.value === 'listid') {
1244
+ currentListOverrideListId = node.param;
1122
1245
  }
1123
- // Clear link URL after processing the field group
1124
- if (isHyperlinkField) {
1125
- flushRun();
1126
- currentLinkUrl = undefined;
1246
+ else if (node.value === 'ls') {
1247
+ currentListOverrideLs = node.param;
1127
1248
  }
1128
- // Append group end to raw content
1129
- currentParagraphRaw += '}';
1249
+ return;
1250
+ }
1251
+ // Paragraph control words
1252
+ if (node.value === 'par') {
1253
+ flushParagraph();
1254
+ currentFormatting = { ...formatting };
1130
1255
  }
1131
- else if (node.type === 'text') {
1132
- if (parsingListTable || parsingListOverrideTable) {
1133
- // Even if we don't extract text, we might want it in rawContent?
1134
- // Yes, rawContent should reflect the source.
1135
- currentParagraphRaw += serializeRtfText(node);
1136
- return;
1256
+ // Table control words
1257
+ else if (node.value === 'trowd') {
1258
+ // Table row definition - start of a new row
1259
+ // Check if we are starting a nested table
1260
+ // If we are already in a table, and we have content in the current cell,
1261
+ // then this trowd implies a nested table start.
1262
+ const ctx = getCurrentTable();
1263
+ if (inTable && ctx && ctx.currentCellContent.length > 0) {
1264
+ // Start nested table
1265
+ ensureTableContext(); // Should already exist if inTable is true
1266
+ // Push new table context
1267
+ tableStack.push({
1268
+ rows: [],
1269
+ currentCells: [],
1270
+ currentCellContent: [],
1271
+ rowIndex: 0
1272
+ });
1137
1273
  }
1138
- if (formattingChanged(currentFormatting, formatting)) {
1139
- flushRun();
1140
- currentFormatting = { ...formatting };
1274
+ else {
1275
+ if (!inTable) {
1276
+ inTable = true;
1277
+ ensureTableContext();
1278
+ }
1141
1279
  }
1142
- currentRunText += node.value;
1143
- currentParagraphRaw += serializeRtfText(node);
1280
+ // After \trowd we are inside a table row, so content should go to table cells.
1281
+ // Many RTF files don't use \intbl, relying solely on \trowd...\cell...\row structure.
1282
+ paragraphInTable = true;
1283
+ // Reset cell properties for the new row definition
1284
+ rowCellProps = [];
1285
+ currentCellDefinitionProps = { isMergedContinuation: false };
1286
+ cellContentIndex = 0;
1144
1287
  }
1145
- else if (node.type === 'control') {
1146
- // Append control to raw content
1147
- currentParagraphRaw += serializeRtfControl(node);
1148
- // Handle list definition control words
1149
- if (parsingListTable) {
1150
- if (node.value === 'listid') {
1151
- currentDefinedListId = node.param;
1152
- }
1153
- else if (node.value === 'levelnfc' || node.value === 'levelnfcn') {
1154
- // 0 = Arabic, 1 = Upper Roman, 2 = Lower Roman, 3 = Upper Alpha, 4 = Lower Alpha -> Ordered
1155
- // 23 = Bullet, 255 = None -> Unordered
1156
- const isOrdered = node.param !== undefined && (node.param === 0 || (node.param >= 0 && node.param <= 4));
1157
- // Only set if not already set (or prioritize ordered if mixed?)
1158
- // We'll assume if any level is ordered, it's ordered.
1159
- // Or if we haven't set it yet.
1160
- if (!currentDefinedListType || (currentDefinedListType === 'unordered' && isOrdered)) {
1161
- currentDefinedListType = isOrdered ? 'ordered' : 'unordered';
1162
- }
1288
+ else if (node.value === 'clvmrg') {
1289
+ // Vertical merge continuation
1290
+ currentCellDefinitionProps.isMergedContinuation = true;
1291
+ }
1292
+ else if (node.value === 'clmgf') {
1293
+ // Vertical merge first cell (reset continuation flag if set, though usually mutually exclusive)
1294
+ currentCellDefinitionProps.isMergedContinuation = false;
1295
+ }
1296
+ else if (node.value === 'cellx') {
1297
+ // End of cell definition
1298
+ rowCellProps.push({ ...currentCellDefinitionProps });
1299
+ // Reset for next cell
1300
+ currentCellDefinitionProps = { isMergedContinuation: false };
1301
+ }
1302
+ else if (node.value === 'cell') {
1303
+ // End of cell - add it to current row
1304
+ // Force paragraphInTable = true because \cell implies we are in a table cell
1305
+ paragraphInTable = true;
1306
+ // Check if this cell is a merged continuation
1307
+ let isMergedContinuation = false;
1308
+ if (cellContentIndex < rowCellProps.length) {
1309
+ isMergedContinuation = rowCellProps[cellContentIndex].isMergedContinuation;
1310
+ }
1311
+ cellContentIndex++;
1312
+ const cell = flushCell();
1313
+ // Only add if not a merged continuation
1314
+ if (cell) {
1315
+ if (!isMergedContinuation) {
1316
+ const ctx = getCurrentTable();
1317
+ if (ctx)
1318
+ ctx.currentCells.push(cell);
1163
1319
  }
1164
- return;
1165
1320
  }
1166
- // Handle list override control words
1167
- if (parsingListOverrideTable) {
1168
- if (node.value === 'listid') {
1169
- currentListOverrideListId = node.param;
1170
- }
1171
- else if (node.value === 'ls') {
1172
- currentListOverrideLs = node.param;
1321
+ currentFormatting = { ...formatting };
1322
+ }
1323
+ else if (node.value === 'nestcell') {
1324
+ // End of cell in outer table (nested context)
1325
+ // If we are in an inner table, we need to close it and return to outer
1326
+ // First, flush the current cell of the inner table (if any pending)
1327
+ // Actually, nestcell ends the OUTER cell.
1328
+ // So the inner table should have been finished by now?
1329
+ // Usually inner table ends with \row.
1330
+ // If we are in a nested table (stack > 1), we should pop until we are at the outer table?
1331
+ // Or maybe just pop one level?
1332
+ if (tableStack.length > 1) {
1333
+ // Flush the inner table if it has pending rows
1334
+ const innerCtx = getCurrentTable();
1335
+ if (innerCtx && (innerCtx.rows.length > 0 || innerCtx.currentCells.length > 0)) {
1336
+ flushTable(); // This pops the stack
1173
1337
  }
1174
- return;
1175
1338
  }
1176
- // Paragraph control words
1177
- if (node.value === 'par') {
1178
- flushParagraph();
1179
- currentFormatting = { ...formatting };
1180
- }
1181
- // Table control words
1182
- else if (node.value === 'trowd') {
1183
- // Table row definition - start of a new row
1184
- // Check if we are starting a nested table
1185
- // If we are already in a table, and we have content in the current cell,
1186
- // then this trowd implies a nested table start.
1339
+ // Now we are (hopefully) at the outer table level
1340
+ // Treat as a regular cell end for the outer table
1341
+ paragraphInTable = true;
1342
+ const cell = flushCell();
1343
+ if (cell) {
1187
1344
  const ctx = getCurrentTable();
1188
- if (inTable && ctx && ctx.currentCellContent.length > 0) {
1189
- // Start nested table
1190
- ensureTableContext(); // Should already exist if inTable is true
1191
- // Push new table context
1192
- tableStack.push({
1193
- rows: [],
1194
- currentCells: [],
1195
- currentCellContent: [],
1196
- rowIndex: 0
1197
- });
1198
- }
1199
- else {
1200
- if (!inTable) {
1201
- inTable = true;
1202
- ensureTableContext();
1203
- }
1204
- }
1205
- // After \trowd we are inside a table row, so content should go to table cells.
1206
- // Many RTF files don't use \intbl, relying solely on \trowd...\cell...\row structure.
1207
- paragraphInTable = true;
1208
- // Reset cell properties for the new row definition
1209
- rowCellProps = [];
1210
- currentCellDefinitionProps = { isMergedContinuation: false };
1211
- cellContentIndex = 0;
1212
- }
1213
- else if (node.value === 'clvmrg') {
1214
- // Vertical merge continuation
1215
- currentCellDefinitionProps.isMergedContinuation = true;
1216
- }
1217
- else if (node.value === 'clmgf') {
1218
- // Vertical merge first cell (reset continuation flag if set, though usually mutually exclusive)
1219
- currentCellDefinitionProps.isMergedContinuation = false;
1220
- }
1221
- else if (node.value === 'cellx') {
1222
- // End of cell definition
1223
- rowCellProps.push({ ...currentCellDefinitionProps });
1224
- // Reset for next cell
1225
- currentCellDefinitionProps = { isMergedContinuation: false };
1226
- }
1227
- else if (node.value === 'cell') {
1228
- // End of cell - add it to current row
1229
- // Force paragraphInTable = true because \cell implies we are in a table cell
1230
- paragraphInTable = true;
1231
- // Check if this cell is a merged continuation
1232
- let isMergedContinuation = false;
1233
- if (cellContentIndex < rowCellProps.length) {
1234
- isMergedContinuation = rowCellProps[cellContentIndex].isMergedContinuation;
1235
- }
1236
- cellContentIndex++;
1237
- const cell = flushCell();
1238
- // Only add if not a merged continuation
1239
- if (cell) {
1240
- if (!isMergedContinuation) {
1241
- const ctx = getCurrentTable();
1242
- if (ctx)
1243
- ctx.currentCells.push(cell);
1244
- }
1245
- }
1246
- currentFormatting = { ...formatting };
1345
+ if (ctx)
1346
+ ctx.currentCells.push(cell);
1247
1347
  }
1248
- else if (node.value === 'nestcell') {
1249
- // End of cell in outer table (nested context)
1250
- // If we are in an inner table, we need to close it and return to outer
1251
- // First, flush the current cell of the inner table (if any pending)
1252
- // Actually, nestcell ends the OUTER cell.
1253
- // So the inner table should have been finished by now?
1254
- // Usually inner table ends with \row.
1255
- // If we are in a nested table (stack > 1), we should pop until we are at the outer table?
1256
- // Or maybe just pop one level?
1257
- if (tableStack.length > 1) {
1258
- // Flush the inner table if it has pending rows
1259
- const innerCtx = getCurrentTable();
1260
- if (innerCtx && (innerCtx.rows.length > 0 || innerCtx.currentCells.length > 0)) {
1261
- flushTable(); // This pops the stack
1262
- }
1263
- }
1264
- // Now we are (hopefully) at the outer table level
1265
- // Treat as a regular cell end for the outer table
1266
- paragraphInTable = true;
1267
- const cell = flushCell();
1268
- if (cell) {
1269
- const ctx = getCurrentTable();
1270
- if (ctx)
1271
- ctx.currentCells.push(cell);
1272
- }
1273
- currentFormatting = { ...formatting };
1348
+ currentFormatting = { ...formatting };
1349
+ }
1350
+ else if (node.value === 'row') {
1351
+ // End of row
1352
+ flushRow();
1353
+ currentFormatting = { ...formatting };
1354
+ // Reset content index for safety (though trowd usually does it)
1355
+ cellContentIndex = 0;
1356
+ // Critical: Reset paragraphInTable after row ends.
1357
+ // Subsequent paragraphs must explicitly use \intbl to be part of the table.
1358
+ // Without this, content after the last \row gets incorrectly merged.
1359
+ paragraphInTable = false;
1360
+ }
1361
+ else if (node.value === 'nestrow') {
1362
+ // End of row in outer table
1363
+ // If we are still in inner table context, flush it
1364
+ if (tableStack.length > 1) {
1365
+ flushTable();
1274
1366
  }
1275
- else if (node.value === 'row') {
1276
- // End of row
1277
- flushRow();
1367
+ flushRow();
1368
+ currentFormatting = { ...formatting };
1369
+ cellContentIndex = 0;
1370
+ }
1371
+ else if (node.value === 'intbl') {
1372
+ // Paragraph is in a table
1373
+ inTable = true;
1374
+ paragraphInTable = true;
1375
+ ensureTableContext();
1376
+ }
1377
+ else if (node.value === 'pard') {
1378
+ // Reset paragraph properties
1379
+ paragraphInTable = false;
1380
+ // Reset other props...
1381
+ paragraphIndent = 0;
1382
+ paragraphAlignment = 'left';
1383
+ isListItem = false;
1384
+ listType = undefined;
1385
+ headingLevel = undefined;
1386
+ currentListId = undefined;
1387
+ // Reset paragraph-level background (cbpat) to prevent leaking to next paragraph
1388
+ formatting.backgroundColor = undefined;
1389
+ }
1390
+ // Text flow control
1391
+ else if (node.value === 'tab') {
1392
+ if (formattingChanged(currentFormatting, formatting)) {
1393
+ flushRun();
1278
1394
  currentFormatting = { ...formatting };
1279
- // Reset content index for safety (though trowd usually does it)
1280
- cellContentIndex = 0;
1281
- // Critical: Reset paragraphInTable after row ends.
1282
- // Subsequent paragraphs must explicitly use \intbl to be part of the table.
1283
- // Without this, content after the last \row gets incorrectly merged.
1284
- paragraphInTable = false;
1285
- }
1286
- else if (node.value === 'nestrow') {
1287
- // End of row in outer table
1288
- // If we are still in inner table context, flush it
1289
- if (tableStack.length > 1) {
1290
- flushTable();
1291
- }
1292
- flushRow();
1395
+ }
1396
+ currentRunTextChunks.push('\t');
1397
+ }
1398
+ else if (node.value === 'line') {
1399
+ if (formattingChanged(currentFormatting, formatting)) {
1400
+ flushRun();
1293
1401
  currentFormatting = { ...formatting };
1294
- cellContentIndex = 0;
1295
- }
1296
- else if (node.value === 'intbl') {
1297
- // Paragraph is in a table
1298
- inTable = true;
1299
- paragraphInTable = true;
1300
- ensureTableContext();
1301
- }
1302
- else if (node.value === 'pard') {
1303
- // Reset paragraph properties
1304
- paragraphInTable = false;
1305
- // Reset other props...
1306
- paragraphIndent = 0;
1307
- paragraphAlignment = 'left';
1308
- isListItem = false;
1309
- listType = undefined;
1310
- headingLevel = undefined;
1311
- currentListId = undefined;
1312
- // Reset paragraph-level background (cbpat) to prevent leaking to next paragraph
1313
- formatting.backgroundColor = undefined;
1314
- }
1315
- // Text flow control
1316
- else if (node.value === 'tab') {
1317
- if (formattingChanged(currentFormatting, formatting)) {
1318
- flushRun();
1319
- currentFormatting = { ...formatting };
1320
- }
1321
- currentRunText += '\t';
1322
1402
  }
1323
- else if (node.value === 'line') {
1324
- if (formattingChanged(currentFormatting, formatting)) {
1325
- flushRun();
1326
- currentFormatting = { ...formatting };
1327
- }
1328
- currentRunText += '\n';
1403
+ currentRunTextChunks.push('\n');
1404
+ }
1405
+ // Quote characters
1406
+ else if (node.value === 'lquote') {
1407
+ // Left single quotation mark (U+2018)
1408
+ if (formattingChanged(currentFormatting, formatting)) {
1409
+ flushRun();
1410
+ currentFormatting = { ...formatting };
1329
1411
  }
1330
- // Quote characters
1331
- else if (node.value === 'lquote') {
1332
- // Left single quotation mark (U+2018)
1333
- if (formattingChanged(currentFormatting, formatting)) {
1334
- flushRun();
1335
- currentFormatting = { ...formatting };
1336
- }
1337
- currentRunText += '\u2018';
1412
+ currentRunTextChunks.push('\u2018');
1413
+ }
1414
+ else if (node.value === 'rquote') {
1415
+ // Right single quotation mark (U+2019)
1416
+ if (formattingChanged(currentFormatting, formatting)) {
1417
+ flushRun();
1418
+ currentFormatting = { ...formatting };
1338
1419
  }
1339
- else if (node.value === 'rquote') {
1340
- // Right single quotation mark (U+2019)
1341
- if (formattingChanged(currentFormatting, formatting)) {
1342
- flushRun();
1343
- currentFormatting = { ...formatting };
1344
- }
1345
- currentRunText += '\u2019';
1420
+ currentRunTextChunks.push('\u2019');
1421
+ }
1422
+ else if (node.value === 'ldblquote') {
1423
+ // Left double quotation mark (U+201C)
1424
+ if (formattingChanged(currentFormatting, formatting)) {
1425
+ flushRun();
1426
+ currentFormatting = { ...formatting };
1346
1427
  }
1347
- else if (node.value === 'ldblquote') {
1348
- // Left double quotation mark (U+201C)
1349
- if (formattingChanged(currentFormatting, formatting)) {
1350
- flushRun();
1351
- currentFormatting = { ...formatting };
1352
- }
1353
- currentRunText += '\u201C';
1428
+ currentRunTextChunks.push('\u201C');
1429
+ }
1430
+ else if (node.value === 'rdblquote') {
1431
+ // Right double quotation mark (U+201D)
1432
+ if (formattingChanged(currentFormatting, formatting)) {
1433
+ flushRun();
1434
+ currentFormatting = { ...formatting };
1354
1435
  }
1355
- else if (node.value === 'rdblquote') {
1356
- // Right double quotation mark (U+201D)
1436
+ currentRunTextChunks.push('\u201D');
1437
+ }
1438
+ // Unicode character
1439
+ else if (node.value === 'u') {
1440
+ if (node.param !== undefined) {
1441
+ let code = node.param;
1442
+ if (code < 0)
1443
+ code += 65536;
1357
1444
  if (formattingChanged(currentFormatting, formatting)) {
1358
1445
  flushRun();
1359
1446
  currentFormatting = { ...formatting };
1360
1447
  }
1361
- currentRunText += '\u201D';
1362
- }
1363
- // Unicode character
1364
- else if (node.value === 'u') {
1365
- if (node.param !== undefined) {
1366
- let code = node.param;
1367
- if (code < 0)
1368
- code += 65536;
1369
- if (formattingChanged(currentFormatting, formatting)) {
1370
- flushRun();
1371
- currentFormatting = { ...formatting };
1372
- }
1373
- currentRunText += String.fromCharCode(code);
1374
- }
1448
+ currentRunTextChunks.push(String.fromCharCode(code));
1375
1449
  }
1376
- // Character formatting
1377
- else if (node.value === 'b') {
1378
- formatting.bold = (node.param !== 0);
1379
- }
1380
- else if (node.value === 'i') {
1381
- formatting.italic = (node.param !== 0);
1382
- }
1383
- else if (node.value === 'ul') {
1384
- formatting.underline = (node.param !== 0);
1385
- }
1386
- else if (node.value === 'ulnone') {
1387
- formatting.underline = false;
1388
- }
1389
- else if (node.value === 'strike') {
1390
- formatting.strikethrough = (node.param !== 0);
1391
- }
1392
- else if (node.value === 'plain') {
1393
- // Reset all character formatting
1394
- formatting.bold = false;
1395
- formatting.italic = false;
1396
- formatting.underline = false;
1397
- formatting.strikethrough = false;
1398
- formatting.subscript = false;
1399
- formatting.superscript = false;
1400
- formatting.size = undefined;
1401
- formatting.font = undefined;
1402
- formatting.color = undefined;
1403
- formatting.backgroundColor = undefined;
1404
- }
1405
- // Font size (\fs - in half-points)
1406
- else if (node.value === 'fs') {
1407
- if (node.param !== undefined) {
1408
- formatting.size = (node.param / 2).toString() + 'pt';
1409
- }
1410
- }
1411
- // Font family (\f)
1412
- else if (node.value === 'f') {
1413
- if (node.param !== undefined && fontTable[node.param]) {
1414
- formatting.font = fontTable[node.param];
1415
- }
1416
- }
1417
- // Text color (\cf)
1418
- else if (node.value === 'cf') {
1419
- if (node.param !== undefined && colorTable[node.param]) {
1420
- formatting.color = colorTable[node.param];
1421
- }
1422
- }
1423
- // Note type (\fet)
1424
- else if (node.value === 'fet') {
1425
- // \fet0 = footnotes only (default)
1426
- // \fet1 = endnotes only
1427
- // \fet2 = both footnotes and endnotes
1428
- if (node.param !== undefined) {
1429
- fetValue = node.param;
1430
- }
1431
- }
1432
- // Background/highlight color (\cb, \highlight, \chcbpat, \cbpat)
1433
- // \chcbpat = character background pattern color (used for shading)
1434
- // \cbpat = paragraph background pattern color
1435
- else if (node.value === 'cb' || node.value === 'highlight' || node.value === 'chcbpat' || node.value === 'cbpat') {
1436
- if (node.param !== undefined && colorTable[node.param]) {
1437
- formatting.backgroundColor = colorTable[node.param];
1438
- }
1450
+ }
1451
+ // Character formatting
1452
+ else if (node.value === 'b') {
1453
+ formatting.bold = (node.param !== 0);
1454
+ }
1455
+ else if (node.value === 'i') {
1456
+ formatting.italic = (node.param !== 0);
1457
+ }
1458
+ else if (node.value === 'ul') {
1459
+ formatting.underline = (node.param !== 0);
1460
+ }
1461
+ else if (node.value === 'ulnone') {
1462
+ formatting.underline = false;
1463
+ }
1464
+ else if (node.value === 'strike') {
1465
+ formatting.strikethrough = (node.param !== 0);
1466
+ }
1467
+ else if (node.value === 'plain') {
1468
+ // Reset all character formatting
1469
+ formatting.bold = false;
1470
+ formatting.italic = false;
1471
+ formatting.underline = false;
1472
+ formatting.strikethrough = false;
1473
+ formatting.subscript = false;
1474
+ formatting.superscript = false;
1475
+ formatting.size = undefined;
1476
+ formatting.font = undefined;
1477
+ formatting.color = undefined;
1478
+ formatting.backgroundColor = undefined;
1479
+ }
1480
+ // Font size (\fs - in half-points)
1481
+ else if (node.value === 'fs') {
1482
+ if (node.param !== undefined) {
1483
+ formatting.size = (node.param / 2).toString() + 'pt';
1439
1484
  }
1440
- // Subscript
1441
- else if (node.value === 'sub') {
1442
- formatting.subscript = true;
1443
- formatting.superscript = false;
1444
- }
1445
- // Superscript
1446
- else if (node.value === 'super') {
1447
- formatting.superscript = true;
1448
- formatting.subscript = false;
1449
- }
1450
- // No subscript/superscript
1451
- else if (node.value === 'nosupersub') {
1452
- formatting.subscript = false;
1453
- formatting.superscript = false;
1454
- }
1455
- // ═══════════════════════════════════════════════════════════
1456
- // List control words
1457
- // ═══════════════════════════════════════════════════════════
1458
- // Paragraph indentation (\li - left indent in twips)
1459
- else if (node.value === 'li') {
1460
- if (node.param !== undefined) {
1461
- // Convert twips to a simpler unit (720 twips = 1 inch, ~0.5 inch per level)
1462
- paragraphIndent = Math.floor(node.param / 360);
1463
- }
1485
+ }
1486
+ // Font family (\f)
1487
+ else if (node.value === 'f') {
1488
+ if (node.param !== undefined && fontTable[node.param]) {
1489
+ formatting.font = fontTable[node.param];
1464
1490
  }
1465
- // List style ID (Word 97+)
1466
- else if (node.value === 'ls') {
1467
- if (node.param !== undefined) {
1468
- // If this is the first list item, reset the indent
1469
- if (!isListItem)
1470
- paragraphIndent = 0;
1471
- isListItem = true;
1472
- // Generate or retrieve list ID
1473
- if (!listStyleIdMap[node.param]) {
1474
- listIdCounter++;
1475
- listStyleIdMap[node.param] = `rtf-list-${listIdCounter}`;
1476
- }
1477
- currentListId = listStyleIdMap[node.param];
1478
- // Look up type from list definition
1479
- // First check override map to get real list ID
1480
- const realListId = listOverrideMap[node.param] !== undefined ? listOverrideMap[node.param] : node.param;
1481
- if (listTypeMap[realListId]) {
1482
- listType = listTypeMap[realListId];
1483
- }
1484
- }
1491
+ }
1492
+ // Text color (\cf)
1493
+ else if (node.value === 'cf') {
1494
+ if (node.param !== undefined && colorTable[node.param]) {
1495
+ formatting.color = colorTable[node.param];
1485
1496
  }
1486
- // List indent level (Word 97+)
1487
- else if (node.value === 'ilvl') {
1488
- if (node.param !== undefined) {
1489
- isListItem = true;
1490
- paragraphIndent = node.param;
1491
- }
1497
+ }
1498
+ // Note type (\fet)
1499
+ else if (node.value === 'fet') {
1500
+ // \fet0 = footnotes only (default)
1501
+ // \fet1 = endnotes only
1502
+ // \fet2 = both footnotes and endnotes
1503
+ if (node.param !== undefined) {
1504
+ fetValue = node.param;
1492
1505
  }
1493
- // List numbering level (\pnlvl)
1494
- else if (node.value === 'pnlvl') {
1495
- isListItem = true;
1496
- if (node.param !== undefined) {
1497
- paragraphIndent = node.param;
1498
- }
1506
+ }
1507
+ // Background/highlight color (\cb, \highlight, \chcbpat, \cbpat)
1508
+ // \chcbpat = character background pattern color (used for shading)
1509
+ // \cbpat = paragraph background pattern color
1510
+ else if (node.value === 'cb' || node.value === 'highlight' || node.value === 'chcbpat' || node.value === 'cbpat') {
1511
+ if (node.param !== undefined && colorTable[node.param]) {
1512
+ formatting.backgroundColor = colorTable[node.param];
1499
1513
  }
1500
- // List numbering format
1501
- else if (node.value === 'levelnfc' || node.value === 'pnf') {
1502
- // 0 = Arabic (1, 2, 3), 1 = Roman upper, 2 = Roman lower,
1503
- // 3 = Letter upper, 4 = Letter lower, 23 = Bullet
1504
- if (node.param !== undefined) {
1505
- isListItem = true;
1506
- if (node.param === 23) {
1507
- listType = 'unordered';
1508
- }
1509
- else {
1510
- listType = 'ordered';
1511
- }
1514
+ }
1515
+ // Subscript
1516
+ else if (node.value === 'sub') {
1517
+ formatting.subscript = true;
1518
+ formatting.superscript = false;
1519
+ }
1520
+ // Superscript
1521
+ else if (node.value === 'super') {
1522
+ formatting.superscript = true;
1523
+ formatting.subscript = false;
1524
+ }
1525
+ // No subscript/superscript
1526
+ else if (node.value === 'nosupersub') {
1527
+ formatting.subscript = false;
1528
+ formatting.superscript = false;
1529
+ }
1530
+ // ═══════════════════════════════════════════════════════════
1531
+ // List control words
1532
+ // ═══════════════════════════════════════════════════════════
1533
+ // Paragraph indentation (\li - left indent in twips)
1534
+ else if (node.value === 'li') {
1535
+ if (node.param !== undefined) {
1536
+ // Standard level indent is 720 twips (0.5 inch)
1537
+ // Using a slightly more flexible divisor to account for different generators
1538
+ const level = Math.round(node.param / 720);
1539
+ // Only update if not already explicitly set by ilvl (Word 97+)
1540
+ if (!isListItem) {
1541
+ paragraphIndent = level;
1512
1542
  }
1513
1543
  }
1514
- // Ordered list indicator
1515
- else if (node.value === 'pndec' || node.value === 'pnord' || node.value === 'pnlcltr' || node.value === 'pnucltr') {
1516
- // If this is the first list item, reset the indent
1517
- if (!isListItem)
1518
- paragraphIndent = 0;
1519
- isListItem = true;
1520
- listType = 'ordered';
1521
- }
1522
- // Unordered list indicator
1523
- else if (node.value === 'pnbullet' || node.value === 'pncard') {
1544
+ }
1545
+ // List style ID (Word 97+)
1546
+ else if (node.value === 'ls') {
1547
+ if (node.param !== undefined) {
1524
1548
  // If this is the first list item, reset the indent
1525
1549
  if (!isListItem)
1526
1550
  paragraphIndent = 0;
1527
1551
  isListItem = true;
1528
- listType = 'unordered';
1529
- }
1530
- // Style-based heading detection (\s)
1531
- else if (node.value === 's') {
1532
- if (node.param !== undefined) {
1533
- // Common heading styles: s1-s9 (though this varies by document)
1534
- if (node.param >= 1 && node.param <= 9) {
1535
- headingLevel = node.param;
1536
- }
1552
+ // Generate or retrieve list ID
1553
+ if (!listStyleIdMap[node.param]) {
1554
+ listIdCounter++;
1555
+ listStyleIdMap[node.param] = `rtf-list-${listIdCounter}`;
1556
+ }
1557
+ currentListId = listStyleIdMap[node.param];
1558
+ // Look up type from list definition
1559
+ // First check override map to get real list ID
1560
+ const realListId = listOverrideMap[node.param] !== undefined ? listOverrideMap[node.param] : node.param;
1561
+ if (listTypeMap[realListId]) {
1562
+ listType = listTypeMap[realListId];
1537
1563
  }
1538
1564
  }
1539
- // Paragraph alignment
1540
- else if (node.value === 'ql') {
1541
- paragraphAlignment = 'left';
1542
- }
1543
- else if (node.value === 'qc') {
1544
- paragraphAlignment = 'center';
1545
- }
1546
- else if (node.value === 'qr') {
1547
- paragraphAlignment = 'right';
1565
+ }
1566
+ // List indent level (Word 97+)
1567
+ else if (node.value === 'ilvl') {
1568
+ if (node.param !== undefined) {
1569
+ isListItem = true;
1570
+ paragraphIndent = node.param;
1548
1571
  }
1549
- else if (node.value === 'qj') {
1550
- paragraphAlignment = 'justify';
1572
+ }
1573
+ // List numbering level (\pnlvl)
1574
+ else if (node.value === 'pnlvl') {
1575
+ isListItem = true;
1576
+ if (node.param !== undefined) {
1577
+ paragraphIndent = node.param;
1551
1578
  }
1552
1579
  }
1553
- };
1554
- traverse(doc, {});
1555
- // Flush any remaining table
1556
- const finalCtx = getCurrentTable();
1557
- if (inTable || (finalCtx && (finalCtx.rows.length > 0 || finalCtx.currentCells.length > 0))) {
1558
- flushTable();
1559
- }
1560
- flushParagraph();
1561
- // Notes handling:
1562
- // - If putNotesAtLast is false, notes should be added inline during traversal
1563
- // (currently they go to 'notes' array, then we append them here - this is wrong)
1564
- // - If putNotesAtLast is true, notes are appended at the very end (see below)
1565
- //
1566
- // For now, when putNotesAtLast is false, we append notes immediately after content
1567
- // This isn't truly "inline" but it's better than at the end
1568
- // TODO: Implement true inline placement during traversal
1569
- if (!config.putNotesAtLast && notes.length > 0) {
1570
- content.push(...notes);
1571
- notes.length = 0; // Clear so they don't get appended again
1572
- }
1573
- // Perform OCR if enabled
1574
- if (config.ocr && config.extractAttachments) {
1575
- for (const attachment of attachments) {
1576
- if (attachment.mimeType.startsWith('image/')) {
1577
- try {
1578
- // Convert base64 data back to Buffer for Tesseract.js
1579
- // Passing base64 string directly would be interpreted as a file path,
1580
- // causing ENAMETOOLONG error for large images.
1581
- const imageBuffer = Buffer.from(attachment.data, 'base64');
1582
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1580
+ // List numbering format
1581
+ else if (node.value === 'levelnfc' || node.value === 'pnf') {
1582
+ // 0 = Arabic (1, 2, 3), 1 = Roman upper, 2 = Roman lower,
1583
+ // 3 = Letter upper, 4 = Letter lower, 23 = Bullet
1584
+ if (node.param !== undefined) {
1585
+ isListItem = true;
1586
+ if (node.param === 23) {
1587
+ listType = 'unordered';
1583
1588
  }
1584
- catch (e) {
1585
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1589
+ else {
1590
+ listType = 'ordered';
1586
1591
  }
1587
1592
  }
1588
1593
  }
1589
- // Link OCR text and altText to image nodes in content
1590
- const assignOcr = (nodes) => {
1591
- for (const node of nodes) {
1592
- if (node.type === 'image' && node.metadata && 'attachmentName' in node.metadata) {
1593
- const meta = node.metadata;
1594
- const attachment = attachments.find(a => a.name === meta.attachmentName);
1595
- if (attachment) {
1596
- // Propagate OCR text to image node
1597
- if (attachment.ocrText) {
1598
- node.text = attachment.ocrText;
1599
- }
1600
- // Propagate altText if available
1601
- if (attachment.altText) {
1602
- meta.altText = attachment.altText;
1603
- }
1604
- }
1605
- }
1606
- if (node.children) {
1607
- assignOcr(node.children);
1594
+ // Ordered list indicator
1595
+ else if (node.value === 'pndec' || node.value === 'pnord' || node.value === 'pnlcltr' || node.value === 'pnucltr') {
1596
+ // If this is the first list item, reset the indent
1597
+ if (!isListItem)
1598
+ paragraphIndent = 0;
1599
+ isListItem = true;
1600
+ listType = 'ordered';
1601
+ }
1602
+ // Unordered list indicator
1603
+ else if (node.value === 'pnbullet' || node.value === 'pncard') {
1604
+ // If this is the first list item, reset the indent
1605
+ if (!isListItem)
1606
+ paragraphIndent = 0;
1607
+ isListItem = true;
1608
+ listType = 'unordered';
1609
+ }
1610
+ // Style-based heading detection (\s)
1611
+ else if (node.value === 's') {
1612
+ if (node.param !== undefined) {
1613
+ // Common heading styles: s1-s9 (though this varies by document)
1614
+ if (node.param >= 1 && node.param <= 9) {
1615
+ headingLevel = node.param;
1608
1616
  }
1609
1617
  }
1610
- };
1611
- assignOcr(content);
1618
+ }
1619
+ // Paragraph alignment
1620
+ else if (node.value === 'ql') {
1621
+ paragraphAlignment = 'left';
1622
+ }
1623
+ else if (node.value === 'qc') {
1624
+ paragraphAlignment = 'center';
1625
+ }
1626
+ else if (node.value === 'qr') {
1627
+ paragraphAlignment = 'right';
1628
+ }
1629
+ else if (node.value === 'qj') {
1630
+ paragraphAlignment = 'justify';
1631
+ }
1612
1632
  }
1613
- // Final pass to ensure all 'note' nodes have their 'text' property populated
1614
- // (This supports the simple toText implementation)
1615
- const populateNoteText = (nodes) => {
1633
+ };
1634
+ traverse(doc, {});
1635
+ // Flush any remaining table
1636
+ const finalCtx = getCurrentTable();
1637
+ if (inTable || (finalCtx && (finalCtx.rows.length > 0 || finalCtx.currentCells.length > 0))) {
1638
+ flushTable();
1639
+ }
1640
+ flushParagraph();
1641
+ // Notes handling:
1642
+ // - If putNotesAtLast is false, notes should be added inline during traversal
1643
+ // (currently they go to 'notes' array, then we append them here - this is wrong)
1644
+ // - If putNotesAtLast is true, notes are appended at the very end (see below)
1645
+ //
1646
+ // For now, when putNotesAtLast is false, we append notes immediately after content
1647
+ // This isn't truly "inline" but it's better than at the end
1648
+ // TODO: Implement true inline placement during traversal
1649
+ if (!config.putNotesAtLast && notes.length > 0) {
1650
+ content.push(...notes);
1651
+ notes.length = 0; // Clear so they don't get appended again
1652
+ }
1653
+ // Perform OCR if enabled
1654
+ if (config.ocr && config.extractAttachments) {
1655
+ for (const attachment of attachments) {
1656
+ if (attachment.mimeType.startsWith('image/')) {
1657
+ try {
1658
+ // Convert base64 data back to Buffer for Tesseract.js
1659
+ // Passing base64 string directly would be interpreted as a file path,
1660
+ // causing ENAMETOOLONG error for large images.
1661
+ const imageBuffer = Buffer.from(attachment.data, 'base64');
1662
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { ...config.ocrConfig })).trim();
1663
+ }
1664
+ catch (e) {
1665
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
1666
+ }
1667
+ }
1668
+ }
1669
+ // Link OCR text and altText to image nodes in content
1670
+ const assignOcr = (nodes) => {
1616
1671
  for (const node of nodes) {
1617
- if (node.type === 'note' && node.children) {
1618
- const getText = (n) => {
1619
- if (n.children && n.children.length > 0)
1620
- return n.children.map(getText).join('');
1621
- return n.text || '';
1622
- };
1623
- node.text = node.children.map(getText).join('').trim();
1672
+ if (node.type === 'image' && node.metadata && 'attachmentName' in node.metadata) {
1673
+ const meta = node.metadata;
1674
+ const attachment = attachments.find(a => a.name === meta.attachmentName);
1675
+ if (attachment) {
1676
+ // Propagate OCR text to image node
1677
+ if (attachment.ocrText) {
1678
+ node.text = attachment.ocrText;
1679
+ }
1680
+ // Propagate altText if available
1681
+ if (attachment.altText) {
1682
+ meta.altText = attachment.altText;
1683
+ }
1684
+ }
1624
1685
  }
1625
1686
  if (node.children) {
1626
- populateNoteText(node.children);
1687
+ assignOcr(node.children);
1627
1688
  }
1628
1689
  }
1629
1690
  };
1630
- populateNoteText(content);
1631
- populateNoteText(notes);
1632
- const result = {
1633
- type: 'rtf',
1634
- metadata: {
1635
- // RTF Limitation: No style map available (RTF uses inline styles)
1636
- },
1637
- content: content,
1638
- attachments: attachments, // PNG and JPEG images extracted from \\pict groups
1639
- toText: () => {
1640
- let text = content.map(c => c.text).join(config.newlineDelimiter ?? '\n');
1641
- if (config.putNotesAtLast && notes.length > 0) {
1642
- text += (config.newlineDelimiter ?? '\n') + notes.map(c => c.text).join(config.newlineDelimiter ?? '\n');
1643
- }
1644
- return text;
1691
+ assignOcr(content);
1692
+ }
1693
+ // Final pass to ensure all 'note' nodes have their 'text' property populated
1694
+ // (This supports the simple toText implementation)
1695
+ const populateNoteText = (nodes) => {
1696
+ for (const node of nodes) {
1697
+ if (node.type === 'note' && node.children) {
1698
+ const getText = (n) => {
1699
+ if (n.children && n.children.length > 0)
1700
+ return n.children.map(getText).join('');
1701
+ return n.text || '';
1702
+ };
1703
+ node.text = node.children.map(getText).join('').trim();
1645
1704
  }
1646
- };
1647
- // If putNotesAtLast is true, append notes to the end of the content array
1705
+ if (node.children) {
1706
+ populateNoteText(node.children);
1707
+ }
1708
+ }
1709
+ };
1710
+ populateNoteText(content);
1711
+ populateNoteText(notes);
1712
+ const toTextSync = () => {
1713
+ let text = content.map(c => c.text).join(config.newlineDelimiter);
1648
1714
  if (config.putNotesAtLast && notes.length > 0) {
1649
- content.push(...notes);
1715
+ text += config.newlineDelimiter + notes.map(c => c.text).join(config.newlineDelimiter);
1650
1716
  }
1651
- return result;
1652
- }
1653
- catch (err) {
1654
- throw err;
1717
+ return text;
1718
+ };
1719
+ const result = (0, astUtils_js_1.createAST)('rtf', {
1720
+ // RTF Limitation: No style map available (RTF uses inline styles)
1721
+ }, content, attachments, // PNG and JPEG images extracted from \\pict groups
1722
+ config, toTextSync);
1723
+ // If putNotesAtLast is true, append notes to the end of the content array
1724
+ if (config.putNotesAtLast && notes.length > 0) {
1725
+ content.push(...notes);
1655
1726
  }
1727
+ return result;
1656
1728
  };
1657
1729
  exports.parseRtf = parseRtf;
1730
+ // Helper to find an RTF group by destination name
1731
+ function findRtfGroup(group, destination) {
1732
+ for (const node of group.content) {
1733
+ if (node.type === 'group') {
1734
+ if (node.destination === destination)
1735
+ return node;
1736
+ const found = findRtfGroup(node, destination);
1737
+ if (found)
1738
+ return found;
1739
+ }
1740
+ }
1741
+ return null;
1742
+ }
1658
1743
  // Helper function to extract font table from RTF document
1659
1744
  function extractFontTable(doc) {
1660
1745
  const fontTable = {};
1661
- // Recursive helper to find the font table group at any depth
1662
- const findAndParseFontTable = (group) => {
1663
- for (const node of group.content) {
1664
- if (node.type === 'group') {
1665
- if (node.destination === 'fonttbl') {
1666
- // Iterate through font definitions
1667
- for (const fontNode of node.content) {
1668
- if (fontNode.type === 'group') {
1669
- let fontIndex;
1670
- let fontName = '';
1671
- for (const item of fontNode.content) {
1672
- if (item.type === 'control' && item.value === 'f') {
1673
- fontIndex = item.param;
1674
- }
1675
- else if (item.type === 'text') {
1676
- // Font name (may have trailing semicolon)
1677
- fontName += item.value;
1678
- }
1679
- }
1680
- if (fontIndex !== undefined && fontName) {
1681
- // Remove trailing semicolon and whitespace
1682
- fontName = fontName.replace(/;$/, '').trim();
1683
- fontTable[fontIndex] = fontName;
1684
- }
1685
- }
1746
+ const tableGroup = findRtfGroup(doc, 'fonttbl');
1747
+ if (tableGroup) {
1748
+ for (const fontNode of tableGroup.content) {
1749
+ if (fontNode.type === 'group') {
1750
+ let fontIndex;
1751
+ let fontName = '';
1752
+ for (const item of fontNode.content) {
1753
+ if (item.type === 'control' && item.value === 'f') {
1754
+ fontIndex = item.param;
1755
+ }
1756
+ else if (item.type === 'text') {
1757
+ fontName += item.value;
1686
1758
  }
1687
- return true; // Found and parsed
1688
1759
  }
1689
- // Recurse into child groups
1690
- if (findAndParseFontTable(node)) {
1691
- return true;
1760
+ if (fontIndex !== undefined && fontName) {
1761
+ fontTable[fontIndex] = fontName.replace(/;$/, '').trim();
1692
1762
  }
1693
1763
  }
1694
1764
  }
1695
- return false;
1696
- };
1697
- findAndParseFontTable(doc);
1765
+ }
1698
1766
  return fontTable;
1699
1767
  }
1700
1768
  // Helper function to extract color table from RTF document
1701
1769
  function extractColorTable(doc) {
1702
1770
  const colorTable = {};
1703
- // Recursive helper to find the color table group at any depth
1704
- const findAndParseColorTable = (group) => {
1705
- for (const node of group.content) {
1706
- if (node.type === 'group') {
1707
- if (node.destination === 'colortbl') {
1708
- let colorIndex = 0;
1709
- let red = 0, green = 0, blue = 0;
1710
- for (const item of node.content) {
1711
- if (item.type === 'control') {
1712
- if (item.value === 'red' && item.param !== undefined) {
1713
- red = item.param;
1714
- }
1715
- else if (item.value === 'green' && item.param !== undefined) {
1716
- green = item.param;
1717
- }
1718
- else if (item.value === 'blue' && item.param !== undefined) {
1719
- blue = item.param;
1720
- }
1721
- }
1722
- else if (item.type === 'text' && item.value === ';') {
1723
- // Semicolon marks end of color definition
1724
- const hex = `#${red.toString(16).padStart(2, '0')}${green.toString(16).padStart(2, '0')}${blue.toString(16).padStart(2, '0')}`;
1725
- colorTable[colorIndex] = hex;
1726
- colorIndex++;
1727
- red = 0;
1728
- green = 0;
1729
- blue = 0;
1730
- }
1731
- }
1732
- return true; // Found and parsed
1733
- }
1734
- // Recurse into child groups
1735
- if (findAndParseColorTable(node)) {
1736
- return true;
1737
- }
1771
+ const tableGroup = findRtfGroup(doc, 'colortbl');
1772
+ if (tableGroup) {
1773
+ let colorIndex = 0;
1774
+ let red = 0, green = 0, blue = 0;
1775
+ for (const item of tableGroup.content) {
1776
+ if (item.type === 'control') {
1777
+ if (item.value === 'red' && item.param !== undefined)
1778
+ red = item.param;
1779
+ else if (item.value === 'green' && item.param !== undefined)
1780
+ green = item.param;
1781
+ else if (item.value === 'blue' && item.param !== undefined)
1782
+ blue = item.param;
1783
+ }
1784
+ else if (item.type === 'text' && item.value === ';') {
1785
+ const hex = `#${red.toString(16).padStart(2, '0')}${green.toString(16).padStart(2, '0')}${blue.toString(16).padStart(2, '0')}`;
1786
+ colorTable[colorIndex] = hex;
1787
+ colorIndex++;
1788
+ red = 0;
1789
+ green = 0;
1790
+ blue = 0;
1738
1791
  }
1739
1792
  }
1740
- return false;
1741
- };
1742
- findAndParseColorTable(doc);
1793
+ }
1743
1794
  return colorTable;
1744
1795
  }