officeparser 7.0.3 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +27 -1
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +31 -4
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +51 -10
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +377 -53
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +6 -1
- package/dist/parsers/ExcelParser.js +69 -21
- package/dist/parsers/HtmlParser.js +15 -1
- package/dist/parsers/MarkdownParser.js +18 -10
- package/dist/parsers/OpenOfficeParser.js +61 -34
- package/dist/parsers/PdfParser.js +26 -1
- package/dist/parsers/PowerPointParser.js +168 -40
- package/dist/parsers/RtfParser.js +30 -24
- package/dist/parsers/WordParser.js +158 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +383 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +69 -2
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +39 -3
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +17 -0
- package/dist/utils/xmlUtils.js +85 -1
- package/package.json +3 -2
|
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
40
40
|
* @returns A promise resolving to the parsed AST
|
|
41
41
|
*/
|
|
42
42
|
const parsePowerPoint = async (buffer, config) => {
|
|
43
|
+
// Honour cancellation requests immediately — before extracting the ZIP archive.
|
|
44
|
+
// PPTX presentations can have many slides with media/charts and optional OCR per image,
|
|
45
|
+
// so an early abort prevents decompressing and traversing data that will be discarded.
|
|
46
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
43
47
|
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
44
48
|
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
45
49
|
const slideRelsRegex = /ppt\/slides\/_rels\/slide\d+\.xml\.rels/;
|
|
@@ -48,10 +52,17 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
48
52
|
const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
|
|
49
53
|
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
50
54
|
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
55
|
+
const appPropsFileRegex = /docProps\/app\.xml/;
|
|
56
|
+
const commentsFileRegex = /ppt\/comments\/comment\d+\.xml/;
|
|
57
|
+
const commentAuthorsRegex = /ppt\/commentAuthors\.xml/;
|
|
58
|
+
const slideMastersRegex = /ppt\/slideMasters\/slideMaster\d+\.xml/;
|
|
51
59
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
|
|
52
60
|
!!x.match(corePropsFileRegex) ||
|
|
53
61
|
!!x.match(customPropsFileRegex) ||
|
|
62
|
+
!!x.match(appPropsFileRegex) ||
|
|
54
63
|
!!x.match(slideRelsRegex) ||
|
|
64
|
+
(!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
|
|
65
|
+
(!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
|
|
55
66
|
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
|
|
56
67
|
// Extract metadata
|
|
57
68
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
@@ -62,6 +73,12 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
62
73
|
if (Object.keys(customProperties).length > 0)
|
|
63
74
|
metadata.customProperties = customProperties;
|
|
64
75
|
}
|
|
76
|
+
const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
|
|
77
|
+
if (appPropsFile) {
|
|
78
|
+
const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
|
|
79
|
+
if (Object.keys(appProperties).length > 0)
|
|
80
|
+
metadata.nativeProperties = appProperties;
|
|
81
|
+
}
|
|
65
82
|
// Sort files
|
|
66
83
|
files.sort((a, b) => {
|
|
67
84
|
const aMatch = a.path.match(slideNumberRegex);
|
|
@@ -73,6 +90,23 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
73
90
|
const content = [];
|
|
74
91
|
const rawContents = [];
|
|
75
92
|
const slideRelsMap = {};
|
|
93
|
+
const authorMap = {};
|
|
94
|
+
if (!config.ignoreComments) {
|
|
95
|
+
const authorsFile = files.find(f => f.path === 'ppt/commentAuthors.xml');
|
|
96
|
+
if (authorsFile) {
|
|
97
|
+
const authorsXml = (0, xmlUtils_js_1.parseXmlString)(authorsFile.content.toString());
|
|
98
|
+
const authorNodes = (0, xmlUtils_js_1.getElementsByTagName)(authorsXml, "p:cmAuthor");
|
|
99
|
+
for (const aNode of authorNodes) {
|
|
100
|
+
const id = aNode.getAttribute("id");
|
|
101
|
+
if (id !== null) {
|
|
102
|
+
authorMap[id] = {
|
|
103
|
+
author: aNode.getAttribute("name") || undefined,
|
|
104
|
+
initials: aNode.getAttribute("initials") || undefined
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
}
|
|
76
110
|
let currentListId = 0;
|
|
77
111
|
let runningListIndex = 0;
|
|
78
112
|
let lastWasList = false;
|
|
@@ -155,11 +189,26 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
155
189
|
cellText += pNode.text;
|
|
156
190
|
}
|
|
157
191
|
}
|
|
192
|
+
let backgroundColor;
|
|
193
|
+
const tcPr = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "a:tcPr");
|
|
194
|
+
if (tcPr) {
|
|
195
|
+
for (const child of Array.from(tcPr.childNodes)) {
|
|
196
|
+
if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === "a:solidFill") {
|
|
197
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "a:srgbClr");
|
|
198
|
+
if (srgbClr) {
|
|
199
|
+
const val = srgbClr.getAttribute("val");
|
|
200
|
+
if (val)
|
|
201
|
+
backgroundColor = "#" + val;
|
|
202
|
+
}
|
|
203
|
+
break;
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
}
|
|
158
207
|
const cellNode = {
|
|
159
208
|
type: 'cell',
|
|
160
209
|
text: cellText,
|
|
161
210
|
children: cellChildren,
|
|
162
|
-
metadata: { row: rIndex, col: cIndex }
|
|
211
|
+
metadata: { row: rIndex, col: cIndex, ...(backgroundColor ? { backgroundColor } : {}) }
|
|
163
212
|
};
|
|
164
213
|
cells.push(cellNode);
|
|
165
214
|
}
|
|
@@ -280,12 +329,23 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
280
329
|
const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
|
|
281
330
|
for (let i = 0; i < paragraphs.length; i++) {
|
|
282
331
|
const p = paragraphs[i];
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
332
|
+
let pNode;
|
|
333
|
+
if (isTitle) {
|
|
334
|
+
pNode = {
|
|
335
|
+
type: 'heading',
|
|
336
|
+
text: '',
|
|
337
|
+
children: [],
|
|
338
|
+
metadata: { level: 1 }
|
|
339
|
+
};
|
|
340
|
+
}
|
|
341
|
+
else {
|
|
342
|
+
pNode = {
|
|
343
|
+
type: 'paragraph',
|
|
344
|
+
text: '',
|
|
345
|
+
children: [],
|
|
346
|
+
metadata: {}
|
|
347
|
+
};
|
|
348
|
+
}
|
|
289
349
|
// Paragraph Alignment and List Detection
|
|
290
350
|
const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(p, "a:pPr");
|
|
291
351
|
let isList = false;
|
|
@@ -326,7 +386,6 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
326
386
|
}
|
|
327
387
|
}
|
|
328
388
|
if (isList) {
|
|
329
|
-
pNode.type = 'list';
|
|
330
389
|
const ilvl = lvl;
|
|
331
390
|
// detect a new list when bullet type changes or previous was not a list
|
|
332
391
|
const newList = !lastWasList ||
|
|
@@ -374,13 +433,17 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
374
433
|
lastListType = listType;
|
|
375
434
|
lastListIndent = ilvl;
|
|
376
435
|
// metadata output
|
|
377
|
-
pNode
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
436
|
+
pNode = {
|
|
437
|
+
type: 'list',
|
|
438
|
+
text: pNode.text,
|
|
439
|
+
children: pNode.children,
|
|
440
|
+
metadata: {
|
|
441
|
+
listType,
|
|
442
|
+
indentation: ilvl,
|
|
443
|
+
listId: currentListId.toString(),
|
|
444
|
+
itemIndex: runningListIndex,
|
|
445
|
+
alignment: pNode.metadata?.alignment || 'left',
|
|
446
|
+
}
|
|
384
447
|
};
|
|
385
448
|
}
|
|
386
449
|
else {
|
|
@@ -388,7 +451,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
388
451
|
lastListType = null;
|
|
389
452
|
lastListIndent = 0;
|
|
390
453
|
}
|
|
391
|
-
if (isTitle) {
|
|
454
|
+
if (isTitle && pNode.type === 'heading') {
|
|
392
455
|
pNode.metadata = { ...pNode.metadata, level: 1 };
|
|
393
456
|
}
|
|
394
457
|
if (config.includeRawContent) {
|
|
@@ -498,7 +561,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
498
561
|
text: '',
|
|
499
562
|
children: [],
|
|
500
563
|
metadata: {
|
|
501
|
-
|
|
564
|
+
paragraphIndentation: { left: lvl },
|
|
502
565
|
alignment: pNode.metadata?.alignment || 'left'
|
|
503
566
|
}
|
|
504
567
|
};
|
|
@@ -634,6 +697,10 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
634
697
|
else if (typeAttr.includes("relationships/notesSlide")) {
|
|
635
698
|
simplifiedType = "notes";
|
|
636
699
|
}
|
|
700
|
+
// Check comments
|
|
701
|
+
else if (typeAttr.includes("relationships/comments")) {
|
|
702
|
+
simplifiedType = "comments";
|
|
703
|
+
}
|
|
637
704
|
// Now normalize the target only if it is a local file path.
|
|
638
705
|
// Hyperlinks are external and should not be normalized.
|
|
639
706
|
let normalizedTarget = targetRaw;
|
|
@@ -653,6 +720,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
653
720
|
}
|
|
654
721
|
}
|
|
655
722
|
}
|
|
723
|
+
const slidesMap = {};
|
|
724
|
+
const slideMasters = [];
|
|
656
725
|
// Now for processing all the other files - slides and notes.
|
|
657
726
|
for (const file of files) {
|
|
658
727
|
if (file.path.match(mediaFileRegex))
|
|
@@ -663,6 +732,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
663
732
|
continue;
|
|
664
733
|
if (file.path.match(corePropsFileRegex))
|
|
665
734
|
continue;
|
|
735
|
+
if (file.path.includes("comment"))
|
|
736
|
+
continue;
|
|
666
737
|
const xmlContentString = file.content.toString();
|
|
667
738
|
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
|
|
668
739
|
if (config.includeRawContent) {
|
|
@@ -670,30 +741,91 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
670
741
|
}
|
|
671
742
|
const slideMatch = file.path.match(slideNumberRegex);
|
|
672
743
|
const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
|
|
744
|
+
const masterMatch = file.path.match(/slideMaster(\d+)\.xml/);
|
|
745
|
+
const masterNumber = masterMatch ? parseInt(masterMatch[1]) : 0;
|
|
673
746
|
const isNote = file.path.includes("notesSlide");
|
|
674
|
-
const
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
747
|
+
const isMaster = file.path.includes("slideMaster");
|
|
748
|
+
const nodeType = isNote ? 'note' : (isMaster ? 'slideMaster' : 'slide');
|
|
749
|
+
const nodeNumber = isMaster ? masterNumber : slideNumber;
|
|
750
|
+
let slideNode;
|
|
751
|
+
if (isNote) {
|
|
752
|
+
slideNode = {
|
|
753
|
+
type: 'note',
|
|
754
|
+
children: [],
|
|
755
|
+
metadata: {
|
|
756
|
+
slideNumber: nodeNumber,
|
|
757
|
+
noteId: `slide-note-${slideNumber}`
|
|
758
|
+
}
|
|
759
|
+
};
|
|
760
|
+
}
|
|
761
|
+
else {
|
|
762
|
+
slideNode = {
|
|
763
|
+
type: isMaster ? 'slideMaster' : 'slide',
|
|
764
|
+
children: [],
|
|
765
|
+
metadata: {
|
|
766
|
+
slideNumber: nodeNumber
|
|
767
|
+
}
|
|
768
|
+
};
|
|
769
|
+
}
|
|
682
770
|
if (config.includeRawContent) {
|
|
683
771
|
slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(xml, xmlContentString, config);
|
|
684
772
|
}
|
|
685
|
-
/**
|
|
686
|
-
* Extract slide contents in correct document order by scanning p:spTree children.
|
|
687
|
-
* This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
|
|
688
|
-
*/
|
|
689
773
|
const spTree = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "p:spTree");
|
|
690
774
|
if (spTree) {
|
|
691
|
-
slideNode.children?.push(...traverseSpTree(spTree,
|
|
775
|
+
slideNode.children?.push(...traverseSpTree(spTree, nodeNumber, xmlContentString));
|
|
692
776
|
}
|
|
693
777
|
if (slideNode.children && slideNode.children.length > 0) {
|
|
694
|
-
|
|
778
|
+
if (isMaster) {
|
|
779
|
+
slideMasters.push(slideNode);
|
|
780
|
+
}
|
|
781
|
+
else if (isNote) {
|
|
782
|
+
if (!slidesMap[slideNumber])
|
|
783
|
+
slidesMap[slideNumber] = { type: 'slide', children: [], metadata: { slideNumber } };
|
|
784
|
+
if (!slidesMap[slideNumber].notes)
|
|
785
|
+
slidesMap[slideNumber].notes = [];
|
|
786
|
+
slidesMap[slideNumber].notes.push(slideNode);
|
|
787
|
+
}
|
|
788
|
+
else {
|
|
789
|
+
if (!slidesMap[slideNumber]) {
|
|
790
|
+
slidesMap[slideNumber] = slideNode;
|
|
791
|
+
}
|
|
792
|
+
else {
|
|
793
|
+
slidesMap[slideNumber].children = slideNode.children;
|
|
794
|
+
slidesMap[slideNumber].rawContent = slideNode.rawContent;
|
|
795
|
+
}
|
|
796
|
+
// Process comments
|
|
797
|
+
if (!config.ignoreComments && slideRelsMap[slideNumber]) {
|
|
798
|
+
const commentRels = Object.values(slideRelsMap[slideNumber]).filter(r => r.type === "comments");
|
|
799
|
+
for (const rel of commentRels) {
|
|
800
|
+
const cFile = files.find(f => f.path.endsWith(rel.target));
|
|
801
|
+
if (cFile) {
|
|
802
|
+
const cXml = (0, xmlUtils_js_1.parseXmlString)(cFile.content.toString());
|
|
803
|
+
const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "p:cm");
|
|
804
|
+
for (const cNode of commentNodes) {
|
|
805
|
+
const authorId = cNode.getAttribute("authorId");
|
|
806
|
+
const authorData = authorId !== null ? authorMap[authorId] : undefined;
|
|
807
|
+
const text = (0, xmlUtils_js_1.getElementsByTagName)(cNode, "a:t").map(t => t.textContent || '').join('');
|
|
808
|
+
if (text) {
|
|
809
|
+
if (!slidesMap[slideNumber].comments)
|
|
810
|
+
slidesMap[slideNumber].comments = [];
|
|
811
|
+
slidesMap[slideNumber].comments.push({
|
|
812
|
+
type: 'comment',
|
|
813
|
+
text,
|
|
814
|
+
children: [{ type: 'text', text, formatting: {} }],
|
|
815
|
+
metadata: authorData && authorData.author ? { author: authorData.author } : undefined
|
|
816
|
+
});
|
|
817
|
+
}
|
|
818
|
+
}
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
}
|
|
822
|
+
}
|
|
695
823
|
}
|
|
696
824
|
}
|
|
825
|
+
const sortedSlideNumbers = Object.keys(slidesMap).map(Number).sort((a, b) => a - b);
|
|
826
|
+
for (const num of sortedSlideNumbers) {
|
|
827
|
+
content.push(slidesMap[num]);
|
|
828
|
+
}
|
|
697
829
|
const attachments = [];
|
|
698
830
|
const mediaFiles = files.filter(f => f.path.match(/ppt\/media\/.*/));
|
|
699
831
|
const chartFiles = files.filter(f => f.path.match(/ppt\/charts\/chart\d+\.xml/));
|
|
@@ -758,14 +890,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
758
890
|
};
|
|
759
891
|
assignAttachmentData(content);
|
|
760
892
|
}
|
|
761
|
-
//
|
|
762
|
-
if (!config.ignoreNotes && config.putNotesAtLast) {
|
|
763
|
-
content.sort((a, b) => {
|
|
764
|
-
const aIsNote = a.type === 'note' ? 1 : 0;
|
|
765
|
-
const bIsNote = b.type === 'note' ? 1 : 0;
|
|
766
|
-
return aIsNote - bIsNote;
|
|
767
|
-
});
|
|
768
|
-
}
|
|
893
|
+
// putNotesAtLast is deprecated. Notes are now structurally attached to their respective slides.
|
|
769
894
|
const toTextSync = () => content.map(c => {
|
|
770
895
|
// Recursive text extraction
|
|
771
896
|
const getText = (node) => {
|
|
@@ -779,6 +904,9 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
779
904
|
};
|
|
780
905
|
return getText(c);
|
|
781
906
|
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
782
|
-
|
|
907
|
+
const auxiliaryContent = slideMasters.length > 0 ? {
|
|
908
|
+
slideMasters
|
|
909
|
+
} : undefined;
|
|
910
|
+
return (0, astUtils_js_1.createAST)('pptx', metadata, content, attachments, config, auxiliaryContent, toTextSync);
|
|
783
911
|
};
|
|
784
912
|
exports.parsePowerPoint = parsePowerPoint;
|
|
@@ -359,6 +359,7 @@ exports.SimpleRtfParser = SimpleRtfParser;
|
|
|
359
359
|
* @returns The parsed AST.
|
|
360
360
|
*/
|
|
361
361
|
const parseRtf = async (buffer, config) => {
|
|
362
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
362
363
|
const parser = new SimpleRtfParser(buffer);
|
|
363
364
|
const doc = parser.parse();
|
|
364
365
|
// Extract font and color tables
|
|
@@ -1106,11 +1107,14 @@ const parseRtf = async (buffer, config) => {
|
|
|
1106
1107
|
}
|
|
1107
1108
|
// Handle footnote: switch target to notes
|
|
1108
1109
|
const previousTarget = currentTarget;
|
|
1110
|
+
let savedParagraphTextChunks;
|
|
1111
|
+
let savedParagraphChildren;
|
|
1112
|
+
let savedParagraphRawChunks;
|
|
1109
1113
|
if (isFootnote) {
|
|
1110
1114
|
if (config.ignoreNotes) {
|
|
1111
1115
|
return; // Skip footnote content entirely
|
|
1112
1116
|
}
|
|
1113
|
-
|
|
1117
|
+
flushRun();
|
|
1114
1118
|
currentFootnoteId++;
|
|
1115
1119
|
// Determine note type based on \fet value
|
|
1116
1120
|
let noteType = 'footnote';
|
|
@@ -1134,8 +1138,25 @@ const parseRtf = async (buffer, config) => {
|
|
|
1134
1138
|
noteType: noteType
|
|
1135
1139
|
}
|
|
1136
1140
|
};
|
|
1137
|
-
|
|
1141
|
+
if (currentParagraphChildren.length > 0) {
|
|
1142
|
+
const precedingNode = currentParagraphChildren[currentParagraphChildren.length - 1];
|
|
1143
|
+
if (!precedingNode.notes)
|
|
1144
|
+
precedingNode.notes = [];
|
|
1145
|
+
precedingNode.notes.push(noteNode);
|
|
1146
|
+
}
|
|
1147
|
+
else {
|
|
1148
|
+
const emptyTextNode = { type: 'text', text: '' };
|
|
1149
|
+
emptyTextNode.notes = [noteNode];
|
|
1150
|
+
currentParagraphChildren.push(emptyTextNode);
|
|
1151
|
+
}
|
|
1138
1152
|
currentTarget = noteNode.children;
|
|
1153
|
+
// Save current paragraph state so we don't mix footnote paragraphs with main text
|
|
1154
|
+
savedParagraphTextChunks = [...currentParagraphTextChunks];
|
|
1155
|
+
savedParagraphChildren = [...currentParagraphChildren];
|
|
1156
|
+
savedParagraphRawChunks = [...currentParagraphRawChunks];
|
|
1157
|
+
currentParagraphTextChunks = [];
|
|
1158
|
+
currentParagraphChildren = [];
|
|
1159
|
+
currentParagraphRawChunks = [];
|
|
1139
1160
|
}
|
|
1140
1161
|
// Create a new formatting context for the group
|
|
1141
1162
|
const groupFormatting = { ...formatting };
|
|
@@ -1194,6 +1215,10 @@ const parseRtf = async (buffer, config) => {
|
|
|
1194
1215
|
if (isFootnote) {
|
|
1195
1216
|
flushParagraph();
|
|
1196
1217
|
currentTarget = previousTarget;
|
|
1218
|
+
// Restore the saved paragraph state
|
|
1219
|
+
currentParagraphTextChunks = savedParagraphTextChunks;
|
|
1220
|
+
currentParagraphChildren = savedParagraphChildren;
|
|
1221
|
+
currentParagraphRawChunks = savedParagraphRawChunks;
|
|
1197
1222
|
}
|
|
1198
1223
|
// Clear link URL after processing the field group
|
|
1199
1224
|
if (isHyperlinkField) {
|
|
@@ -1638,21 +1663,10 @@ const parseRtf = async (buffer, config) => {
|
|
|
1638
1663
|
flushTable();
|
|
1639
1664
|
}
|
|
1640
1665
|
flushParagraph();
|
|
1641
|
-
// Notes handling:
|
|
1642
|
-
// - If putNotesAtLast is false, notes should be added inline during traversal
|
|
1643
|
-
// (currently they go to 'notes' array, then we append them here - this is wrong)
|
|
1644
|
-
// - If putNotesAtLast is true, notes are appended at the very end (see below)
|
|
1645
|
-
//
|
|
1646
|
-
// For now, when putNotesAtLast is false, we append notes immediately after content
|
|
1647
|
-
// This isn't truly "inline" but it's better than at the end
|
|
1648
|
-
// TODO: Implement true inline placement during traversal
|
|
1649
|
-
if (!config.putNotesAtLast && notes.length > 0) {
|
|
1650
|
-
content.push(...notes);
|
|
1651
|
-
notes.length = 0; // Clear so they don't get appended again
|
|
1652
|
-
}
|
|
1653
1666
|
// Perform OCR if enabled
|
|
1654
1667
|
if (config.ocr && config.extractAttachments) {
|
|
1655
1668
|
for (const attachment of attachments) {
|
|
1669
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
1656
1670
|
if (attachment.mimeType.startsWith('image/')) {
|
|
1657
1671
|
try {
|
|
1658
1672
|
// Convert base64 data back to Buffer for Tesseract.js
|
|
@@ -1710,20 +1724,12 @@ const parseRtf = async (buffer, config) => {
|
|
|
1710
1724
|
populateNoteText(content);
|
|
1711
1725
|
populateNoteText(notes);
|
|
1712
1726
|
const toTextSync = () => {
|
|
1713
|
-
|
|
1714
|
-
if (config.putNotesAtLast && notes.length > 0) {
|
|
1715
|
-
text += config.newlineDelimiter + notes.map(c => c.text).join(config.newlineDelimiter);
|
|
1716
|
-
}
|
|
1717
|
-
return text;
|
|
1727
|
+
return content.map(c => c.text).join(config.newlineDelimiter);
|
|
1718
1728
|
};
|
|
1719
1729
|
const result = (0, astUtils_js_1.createAST)('rtf', {
|
|
1720
1730
|
// RTF Limitation: No style map available (RTF uses inline styles)
|
|
1721
1731
|
}, content, attachments, // PNG and JPEG images extracted from \\pict groups
|
|
1722
|
-
config, toTextSync);
|
|
1723
|
-
// If putNotesAtLast is true, append notes to the end of the content array
|
|
1724
|
-
if (config.putNotesAtLast && notes.length > 0) {
|
|
1725
|
-
content.push(...notes);
|
|
1726
|
-
}
|
|
1732
|
+
config, undefined, toTextSync);
|
|
1727
1733
|
return result;
|
|
1728
1734
|
};
|
|
1729
1735
|
exports.parseRtf = parseRtf;
|