officeparser 6.0.7 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -13
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +43 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +116 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +79 -1
- package/dist/officeparser.browser.iife.js +112 -0
- package/dist/officeparser.browser.mjs +111 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +71 -63
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +131 -114
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +85 -88
- package/dist/parsers/RtfParser.d.ts +1 -1
- package/dist/parsers/RtfParser.js +10 -6
- package/dist/parsers/WordParser.d.ts +1 -1
- package/dist/parsers/WordParser.js +109 -101
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +69 -1
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
|
@@ -24,13 +24,12 @@
|
|
|
24
24
|
*/
|
|
25
25
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
26
|
exports.parsePowerPoint = void 0;
|
|
27
|
-
const
|
|
28
|
-
const
|
|
29
|
-
const
|
|
30
|
-
const
|
|
31
|
-
const
|
|
32
|
-
const
|
|
33
|
-
const zipUtils_1 = require("../utils/zipUtils");
|
|
27
|
+
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
28
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
29
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
30
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
31
|
+
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
32
|
+
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
34
33
|
/**
|
|
35
34
|
* Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
|
|
36
35
|
*
|
|
@@ -46,14 +45,21 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
46
45
|
const mediaFileRegex = /ppt\/media\/.*/;
|
|
47
46
|
const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
|
|
48
47
|
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
49
|
-
const
|
|
50
|
-
const files = await (0,
|
|
48
|
+
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
49
|
+
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
|
|
51
50
|
!!x.match(corePropsFileRegex) ||
|
|
51
|
+
!!x.match(customPropsFileRegex) ||
|
|
52
52
|
!!x.match(slideRelsRegex) ||
|
|
53
53
|
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
|
|
54
54
|
// Extract metadata
|
|
55
55
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
56
|
-
const metadata = corePropsFile ? (0,
|
|
56
|
+
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
57
|
+
const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
|
|
58
|
+
if (customPropsFile) {
|
|
59
|
+
const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
|
|
60
|
+
if (Object.keys(customProperties).length > 0)
|
|
61
|
+
metadata.customProperties = customProperties;
|
|
62
|
+
}
|
|
57
63
|
// Sort files
|
|
58
64
|
files.sort((a, b) => {
|
|
59
65
|
const aMatch = a.path.match(slideNumberRegex);
|
|
@@ -73,21 +79,21 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
73
79
|
// per indent counters (for nested lists)
|
|
74
80
|
const levelCounters = {};
|
|
75
81
|
// Helper to parse a table node
|
|
76
|
-
const parseTable = (tblNode) => {
|
|
82
|
+
const parseTable = (tblNode, xmlContentString) => {
|
|
77
83
|
const rows = [];
|
|
78
|
-
const trNodes = (0,
|
|
84
|
+
const trNodes = (0, xmlUtils_js_1.getElementsByTagName)(tblNode, "a:tr");
|
|
79
85
|
for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
|
|
80
86
|
const trNode = trNodes[rIndex];
|
|
81
87
|
const cells = [];
|
|
82
|
-
const tcNodes = (0,
|
|
88
|
+
const tcNodes = (0, xmlUtils_js_1.getElementsByTagName)(trNode, "a:tc");
|
|
83
89
|
for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
|
|
84
90
|
const tcNode = tcNodes[cIndex];
|
|
85
91
|
const cellChildren = [];
|
|
86
92
|
let cellText = '';
|
|
87
93
|
// Cells contain text bodies (txBody) which contain paragraphs
|
|
88
|
-
const txBody = (0,
|
|
94
|
+
const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "a:txBody");
|
|
89
95
|
if (txBody) {
|
|
90
|
-
const paragraphs = (0,
|
|
96
|
+
const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
|
|
91
97
|
for (const p of paragraphs) {
|
|
92
98
|
// Reuse paragraph parsing logic if possible, or duplicate for now
|
|
93
99
|
// For simplicity, duplicating basic logic here as the main loop one is tied to shapes
|
|
@@ -98,15 +104,15 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
98
104
|
metadata: {}
|
|
99
105
|
};
|
|
100
106
|
if (config.includeRawContent) {
|
|
101
|
-
pNode.rawContent =
|
|
107
|
+
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
102
108
|
}
|
|
103
|
-
const runs = (0,
|
|
109
|
+
const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
|
|
104
110
|
for (const r of runs) {
|
|
105
|
-
const t = (0,
|
|
111
|
+
const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
|
|
106
112
|
if (t && t.childNodes[0]) {
|
|
107
113
|
const textContent = t.childNodes[0].nodeValue || '';
|
|
108
114
|
pNode.text += textContent;
|
|
109
|
-
const rPr = (0,
|
|
115
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
|
|
110
116
|
const formatting = {};
|
|
111
117
|
if (rPr) {
|
|
112
118
|
if (rPr.getAttribute("b") === "1")
|
|
@@ -120,16 +126,16 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
120
126
|
const sz = rPr.getAttribute("sz");
|
|
121
127
|
if (sz)
|
|
122
128
|
formatting.size = (parseInt(sz) / 100).toString() + 'pt';
|
|
123
|
-
const solidFill = (0,
|
|
129
|
+
const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
|
|
124
130
|
if (solidFill) {
|
|
125
|
-
const srgbClr = (0,
|
|
131
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
|
|
126
132
|
if (srgbClr) {
|
|
127
133
|
const val = srgbClr.getAttribute("val");
|
|
128
134
|
if (val)
|
|
129
135
|
formatting.color = '#' + val;
|
|
130
136
|
}
|
|
131
137
|
}
|
|
132
|
-
const latin = (0,
|
|
138
|
+
const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
|
|
133
139
|
if (latin) {
|
|
134
140
|
const typeface = latin.getAttribute("typeface");
|
|
135
141
|
if (typeface)
|
|
@@ -167,8 +173,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
167
173
|
};
|
|
168
174
|
};
|
|
169
175
|
/** Extract an AST node for p:pic */
|
|
170
|
-
const extractImageNode = (imageNode, slideNumber) => {
|
|
171
|
-
const blip = (0,
|
|
176
|
+
const extractImageNode = (imageNode, slideNumber, xmlContentString) => {
|
|
177
|
+
const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "a:blip");
|
|
172
178
|
if (!blip)
|
|
173
179
|
return null;
|
|
174
180
|
const rId = blip.getAttribute("r:embed");
|
|
@@ -178,8 +184,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
178
184
|
if (!rel || rel.type !== "image")
|
|
179
185
|
return null;
|
|
180
186
|
const attachmentName = rel.target;
|
|
181
|
-
const nvPicPr = (0,
|
|
182
|
-
const cNvPr = nvPicPr ? (0,
|
|
187
|
+
const nvPicPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "p:nvPicPr");
|
|
188
|
+
const cNvPr = nvPicPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPicPr, "p:cNvPr") : null;
|
|
183
189
|
const altText = cNvPr?.getAttribute("descr") || undefined;
|
|
184
190
|
return {
|
|
185
191
|
type: "image",
|
|
@@ -192,23 +198,11 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
192
198
|
};
|
|
193
199
|
/**
|
|
194
200
|
* Extract an AST node for p:graphicFrame that contains a chart.
|
|
195
|
-
*
|
|
196
|
-
*
|
|
197
|
-
* Steps:
|
|
198
|
-
* 1. Find <a:graphicData> inside the graphicFrame.
|
|
199
|
-
* 2. Check if uri is the chart namespace.
|
|
200
|
-
* 3. Extract <c:chart> and read r:id.
|
|
201
|
-
* 4. Resolve the relationship using slideRelsMap.
|
|
202
|
-
* 5. Produce a chart node with attachmentName (like images).
|
|
203
|
-
* 6. Chart text/data will be injected later in the pipeline.
|
|
204
|
-
*
|
|
205
|
-
* @param frameNode p:graphicFrame element
|
|
206
|
-
* @param slideNumber Current slide number for relationship resolution
|
|
207
|
-
* @returns A chart OfficeContentNode or null if not a chart.
|
|
201
|
+
* ... (comments omitted for brevity) ...
|
|
208
202
|
*/
|
|
209
|
-
const extractChartNode = (frameNode, slideNumber) => {
|
|
203
|
+
const extractChartNode = (frameNode, slideNumber, xmlContentString) => {
|
|
210
204
|
// Step 1: Find <a:graphicData>
|
|
211
|
-
const graphicData = (0,
|
|
205
|
+
const graphicData = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:graphicData");
|
|
212
206
|
if (!graphicData) {
|
|
213
207
|
return null;
|
|
214
208
|
}
|
|
@@ -220,7 +214,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
220
214
|
return null;
|
|
221
215
|
}
|
|
222
216
|
// Step 3: Find <c:chart>
|
|
223
|
-
const cChart = (0,
|
|
217
|
+
const cChart = (0, xmlUtils_js_1.getFirstElementByTagName)(graphicData, "c:chart");
|
|
224
218
|
if (!cChart) {
|
|
225
219
|
return null;
|
|
226
220
|
}
|
|
@@ -246,24 +240,24 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
246
240
|
};
|
|
247
241
|
// Optional: include raw XML of the whole frame
|
|
248
242
|
if (config.includeRawContent) {
|
|
249
|
-
chartNode.rawContent =
|
|
243
|
+
chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
|
|
250
244
|
}
|
|
251
245
|
return chartNode;
|
|
252
246
|
};
|
|
253
247
|
/** Extract an AST node for p:graphicFrame */
|
|
254
|
-
const extractGraphicFrameNode = (frameNode, slideNumber) => {
|
|
255
|
-
const tbl = (0,
|
|
248
|
+
const extractGraphicFrameNode = (frameNode, slideNumber, xmlContentString) => {
|
|
249
|
+
const tbl = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:tbl");
|
|
256
250
|
if (tbl) {
|
|
257
|
-
const tableNode = parseTable(tbl);
|
|
251
|
+
const tableNode = parseTable(tbl, xmlContentString);
|
|
258
252
|
if (config.includeRawContent) {
|
|
259
|
-
tableNode.rawContent =
|
|
253
|
+
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
|
|
260
254
|
}
|
|
261
255
|
if (tableNode.children && tableNode.children.length > 0) {
|
|
262
256
|
return tableNode;
|
|
263
257
|
}
|
|
264
258
|
}
|
|
265
259
|
if (frameNode.getElementsByTagName("c:chart").length > 0) {
|
|
266
|
-
const chartNode = extractChartNode(frameNode, slideNumber);
|
|
260
|
+
const chartNode = extractChartNode(frameNode, slideNumber, xmlContentString);
|
|
267
261
|
if (chartNode) {
|
|
268
262
|
return chartNode;
|
|
269
263
|
}
|
|
@@ -271,17 +265,17 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
271
265
|
return null;
|
|
272
266
|
};
|
|
273
267
|
/** Extract text and hyperlinks from a p:sp shape. */
|
|
274
|
-
const extractShapeNodes = (spNode, slideNumber) => {
|
|
268
|
+
const extractShapeNodes = (spNode, slideNumber, xmlContentString) => {
|
|
275
269
|
const nodes = [];
|
|
276
270
|
// Check for placeholder type (title, body, etc.)
|
|
277
|
-
const nvSpPr = (0,
|
|
278
|
-
const nvPr = nvSpPr ? (0,
|
|
279
|
-
const ph = nvPr ? (0,
|
|
271
|
+
const nvSpPr = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:nvSpPr");
|
|
272
|
+
const nvPr = nvSpPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvSpPr, "p:nvPr") : null;
|
|
273
|
+
const ph = nvPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPr, "p:ph") : null;
|
|
280
274
|
const type = ph ? ph.getAttribute("type") : "body";
|
|
281
275
|
const isTitle = type === "title" || type === "ctrTitle";
|
|
282
|
-
const txBody = (0,
|
|
276
|
+
const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:txBody");
|
|
283
277
|
if (txBody) {
|
|
284
|
-
const paragraphs = (0,
|
|
278
|
+
const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
|
|
285
279
|
for (let i = 0; i < paragraphs.length; i++) {
|
|
286
280
|
const p = paragraphs[i];
|
|
287
281
|
const pNode = {
|
|
@@ -291,7 +285,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
291
285
|
metadata: isTitle ? { level: 1 } : {}
|
|
292
286
|
};
|
|
293
287
|
// Paragraph Alignment and List Detection
|
|
294
|
-
const pPr = (0,
|
|
288
|
+
const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(p, "a:pPr");
|
|
295
289
|
let isList = false;
|
|
296
290
|
let listType = 'unordered';
|
|
297
291
|
let lvl = 0;
|
|
@@ -299,10 +293,10 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
299
293
|
const lvlAttr = pPr.getAttribute("lvl");
|
|
300
294
|
if (lvlAttr)
|
|
301
295
|
lvl = parseInt(lvlAttr);
|
|
302
|
-
const buAutoNum = (0,
|
|
303
|
-
const buChar = (0,
|
|
304
|
-
const buBlip = (0,
|
|
305
|
-
const buNode = (0,
|
|
296
|
+
const buAutoNum = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buAutoNum");
|
|
297
|
+
const buChar = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buChar");
|
|
298
|
+
const buBlip = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buBlip");
|
|
299
|
+
const buNode = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:bu");
|
|
306
300
|
if (buAutoNum) {
|
|
307
301
|
isList = true;
|
|
308
302
|
listType = 'ordered';
|
|
@@ -396,16 +390,16 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
396
390
|
pNode.metadata = { ...pNode.metadata, level: 1 };
|
|
397
391
|
}
|
|
398
392
|
if (config.includeRawContent) {
|
|
399
|
-
pNode.rawContent =
|
|
393
|
+
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
400
394
|
}
|
|
401
|
-
const runs = (0,
|
|
395
|
+
const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
|
|
402
396
|
for (let j = 0; j < runs.length; j++) {
|
|
403
397
|
const r = runs[j];
|
|
404
|
-
const t = (0,
|
|
398
|
+
const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
|
|
405
399
|
if (t && t.childNodes[0]) {
|
|
406
400
|
const textContent = t.childNodes[0].nodeValue || '';
|
|
407
401
|
pNode.text += textContent;
|
|
408
|
-
const rPr = (0,
|
|
402
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
|
|
409
403
|
const formatting = {};
|
|
410
404
|
if (rPr) {
|
|
411
405
|
if (rPr.getAttribute("b") === "1")
|
|
@@ -420,9 +414,9 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
420
414
|
if (sz)
|
|
421
415
|
formatting.size = (parseInt(sz) / 100).toString() + 'pt';
|
|
422
416
|
// Color extraction
|
|
423
|
-
const solidFill = (0,
|
|
417
|
+
const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
|
|
424
418
|
if (solidFill) {
|
|
425
|
-
const srgbClr = (0,
|
|
419
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
|
|
426
420
|
if (srgbClr) {
|
|
427
421
|
const val = srgbClr.getAttribute("val");
|
|
428
422
|
if (val)
|
|
@@ -430,9 +424,9 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
430
424
|
}
|
|
431
425
|
}
|
|
432
426
|
// Highlight extraction
|
|
433
|
-
const highlight = (0,
|
|
427
|
+
const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
|
|
434
428
|
if (highlight) {
|
|
435
|
-
const srgbClr = (0,
|
|
429
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
|
|
436
430
|
if (srgbClr) {
|
|
437
431
|
const val = srgbClr.getAttribute("val");
|
|
438
432
|
if (val)
|
|
@@ -440,7 +434,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
440
434
|
}
|
|
441
435
|
}
|
|
442
436
|
// Font family
|
|
443
|
-
const latin = (0,
|
|
437
|
+
const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
|
|
444
438
|
if (latin) {
|
|
445
439
|
const typeface = latin.getAttribute("typeface");
|
|
446
440
|
if (typeface)
|
|
@@ -462,7 +456,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
462
456
|
formatting: formatting
|
|
463
457
|
};
|
|
464
458
|
// Check for Hyperlinks
|
|
465
|
-
const hlinkClick = (0,
|
|
459
|
+
const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:hlinkClick");
|
|
466
460
|
// Check if this run has a hyperlink click action
|
|
467
461
|
if (hlinkClick) {
|
|
468
462
|
// Relationship ID for the link
|
|
@@ -520,31 +514,32 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
520
514
|
* transforms so nested groups inherit positional transforms.
|
|
521
515
|
*
|
|
522
516
|
* @param treeNode The XML node representing <p:spTree>
|
|
523
|
-
* @param
|
|
517
|
+
* @param slideNumber Current slide number for relationship resolution
|
|
518
|
+
* @param xmlContentString The source XML string for raw content extraction
|
|
524
519
|
*/
|
|
525
|
-
function traverseSpTree(treeNode, slideNumber) {
|
|
520
|
+
function traverseSpTree(treeNode, slideNumber, xmlContentString) {
|
|
526
521
|
const nodes = [];
|
|
527
522
|
// Process children in XML order (this preserves Z-order)
|
|
528
523
|
for (const child of Array.from(treeNode?.childNodes || [])) {
|
|
529
|
-
if (
|
|
524
|
+
if (!(0, xmlUtils_js_1.isElement)(child)) {
|
|
530
525
|
continue;
|
|
531
526
|
}
|
|
532
527
|
const element = child;
|
|
533
528
|
const tag = element.tagName;
|
|
534
529
|
// Case 1: Normal shape
|
|
535
530
|
if (tag === "p:sp") {
|
|
536
|
-
nodes.push(...extractShapeNodes(element, slideNumber));
|
|
531
|
+
nodes.push(...extractShapeNodes(element, slideNumber, xmlContentString));
|
|
537
532
|
}
|
|
538
533
|
// Case 2: Inline picture
|
|
539
534
|
else if (tag === "p:pic") {
|
|
540
|
-
const imageNode = extractImageNode(element, slideNumber);
|
|
535
|
+
const imageNode = extractImageNode(element, slideNumber, xmlContentString);
|
|
541
536
|
if (imageNode) {
|
|
542
537
|
nodes.push(imageNode);
|
|
543
538
|
}
|
|
544
539
|
}
|
|
545
540
|
// Case 3: Chart or other graphic frame
|
|
546
541
|
else if (tag === "p:graphicFrame") {
|
|
547
|
-
const tableNode = extractGraphicFrameNode(element, slideNumber);
|
|
542
|
+
const tableNode = extractGraphicFrameNode(element, slideNumber, xmlContentString);
|
|
548
543
|
if (tableNode) {
|
|
549
544
|
nodes.push(tableNode);
|
|
550
545
|
}
|
|
@@ -552,9 +547,11 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
552
547
|
// Case 4: Grouped shape (recursive!)
|
|
553
548
|
else if (tag === "p:grpSp") {
|
|
554
549
|
// Extract the nested <p:spTree> inside the group
|
|
555
|
-
const nestedTree =
|
|
550
|
+
const nestedTree = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "p:spTree");
|
|
556
551
|
// Recurse into the nested tree
|
|
557
|
-
|
|
552
|
+
if (nestedTree) {
|
|
553
|
+
nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
|
|
554
|
+
}
|
|
558
555
|
}
|
|
559
556
|
}
|
|
560
557
|
return nodes;
|
|
@@ -587,9 +584,9 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
587
584
|
// Prepare map for this slide
|
|
588
585
|
slideRelsMap[slideNum] = {};
|
|
589
586
|
// Parse the rels XML
|
|
590
|
-
const relsXml = (0,
|
|
587
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(file.content.toString());
|
|
591
588
|
// Get all Relationship nodes
|
|
592
|
-
const relationships = (0,
|
|
589
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
593
590
|
// Loop through each relationship node
|
|
594
591
|
for (let i = 0; i < relationships.length; i++) {
|
|
595
592
|
// Relationship ID, Example: "rId2"
|
|
@@ -653,7 +650,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
653
650
|
if (file.path.match(corePropsFileRegex))
|
|
654
651
|
continue;
|
|
655
652
|
const xmlContentString = file.content.toString();
|
|
656
|
-
const xml = (0,
|
|
653
|
+
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
|
|
657
654
|
if (config.includeRawContent) {
|
|
658
655
|
rawContents.push(xmlContentString);
|
|
659
656
|
}
|
|
@@ -669,15 +666,15 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
669
666
|
}
|
|
670
667
|
};
|
|
671
668
|
if (config.includeRawContent) {
|
|
672
|
-
slideNode.rawContent =
|
|
669
|
+
slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(xml, xmlContentString, config);
|
|
673
670
|
}
|
|
674
671
|
/**
|
|
675
672
|
* Extract slide contents in correct document order by scanning p:spTree children.
|
|
676
673
|
* This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
|
|
677
674
|
*/
|
|
678
|
-
const spTree = (0,
|
|
675
|
+
const spTree = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "p:spTree");
|
|
679
676
|
if (spTree) {
|
|
680
|
-
slideNode.children?.push(...traverseSpTree(spTree, slideNumber));
|
|
677
|
+
slideNode.children?.push(...traverseSpTree(spTree, slideNumber, xmlContentString));
|
|
681
678
|
}
|
|
682
679
|
if (slideNode.children && slideNode.children.length > 0) {
|
|
683
680
|
content.push(slideNode);
|
|
@@ -690,15 +687,15 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
690
687
|
if (config.extractAttachments) {
|
|
691
688
|
// Extract media files as attachments
|
|
692
689
|
for (const media of mediaFiles) {
|
|
693
|
-
const attachment = (0,
|
|
690
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
694
691
|
attachments.push(attachment);
|
|
695
692
|
if (config.ocr) {
|
|
696
693
|
if (attachment.mimeType.startsWith('image/')) {
|
|
697
694
|
try {
|
|
698
|
-
attachment.ocrText = (await (0,
|
|
695
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
699
696
|
}
|
|
700
697
|
catch (e) {
|
|
701
|
-
(0,
|
|
698
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
702
699
|
}
|
|
703
700
|
}
|
|
704
701
|
}
|
|
@@ -715,12 +712,12 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
715
712
|
attachments.push(attachment);
|
|
716
713
|
// Extract text from chart XML
|
|
717
714
|
try {
|
|
718
|
-
const chartData = await (0,
|
|
715
|
+
const chartData = await (0, chartUtils_js_1.extractChartData)(chart.content);
|
|
719
716
|
// Assign chartData to attachment
|
|
720
717
|
attachment.chartData = chartData;
|
|
721
718
|
}
|
|
722
719
|
catch (e) {
|
|
723
|
-
(0,
|
|
720
|
+
(0, errorUtils_js_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
|
|
724
721
|
}
|
|
725
722
|
}
|
|
726
723
|
// Loop through nodes to find images and charts and link their text and chartData
|
|
@@ -39,7 +39,7 @@
|
|
|
39
39
|
* @see https://www.biblioscape.com/rtf15_spec.htm RTF 1.5 Specification
|
|
40
40
|
* @see https://latex2rtf.sourceforge.net/RTF-Spec-1.2.pdf RTF 1.2 Specification
|
|
41
41
|
*/
|
|
42
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
42
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
43
43
|
/**
|
|
44
44
|
* Represents an RTF group (content enclosed in braces).
|
|
45
45
|
* Groups create formatting scopes and can contain other groups, control words, or text.
|
|
@@ -42,8 +42,8 @@
|
|
|
42
42
|
*/
|
|
43
43
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
44
44
|
exports.parseRtf = exports.SimpleRtfParser = void 0;
|
|
45
|
-
const
|
|
46
|
-
const
|
|
45
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
46
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
47
47
|
/**
|
|
48
48
|
* Lookup table mapping RTF control words to internal formats.
|
|
49
49
|
* Fully typed: if a key maps to an unsupported format, TypeScript throws an error.
|
|
@@ -88,13 +88,17 @@ const IMAGE_MIME_MAP = {
|
|
|
88
88
|
* ```
|
|
89
89
|
*/
|
|
90
90
|
class SimpleRtfParser {
|
|
91
|
+
/** Current position in the buffer */
|
|
92
|
+
index = 0;
|
|
93
|
+
/** The RTF content as a Buffer */
|
|
94
|
+
buffer;
|
|
95
|
+
/** Total length of the buffer */
|
|
96
|
+
length;
|
|
91
97
|
/**
|
|
92
98
|
* Creates a new RTF parser.
|
|
93
99
|
* @param buffer - The RTF file content as a Buffer
|
|
94
100
|
*/
|
|
95
101
|
constructor(buffer) {
|
|
96
|
-
/** Current position in the buffer */
|
|
97
|
-
this.index = 0;
|
|
98
102
|
this.buffer = buffer;
|
|
99
103
|
this.length = buffer.length;
|
|
100
104
|
}
|
|
@@ -1510,10 +1514,10 @@ const parseRtf = async (buffer, config) => {
|
|
|
1510
1514
|
// Passing base64 string directly would be interpreted as a file path,
|
|
1511
1515
|
// causing ENAMETOOLONG error for large images.
|
|
1512
1516
|
const imageBuffer = Buffer.from(attachment.data, 'base64');
|
|
1513
|
-
attachment.ocrText = (await (0,
|
|
1517
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
1514
1518
|
}
|
|
1515
1519
|
catch (e) {
|
|
1516
|
-
(0,
|
|
1520
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
1517
1521
|
}
|
|
1518
1522
|
}
|
|
1519
1523
|
}
|
|
@@ -58,7 +58,7 @@
|
|
|
58
58
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
|
|
59
59
|
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
|
|
60
60
|
*/
|
|
61
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
61
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
62
62
|
/**
|
|
63
63
|
* Parses a Word document (.docx) and extracts content, formatting, and metadata.
|
|
64
64
|
*
|