officeparser 5.2.1 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,778 @@
1
+ "use strict";
2
+ /**
3
+ * PowerPoint Presentation (PPTX) Parser
4
+ *
5
+ * **PPTX Format Overview:**
6
+ * PPTX is the default format for Microsoft PowerPoint since Office 2007, based on OOXML.
7
+ *
8
+ * **File Structure:**
9
+ * - `ppt/presentation.xml` - Presentation structure and slide list
10
+ * - `ppt/slides/slide1.xml` - Individual slide content
11
+ * - `ppt/notesSlides/notesSlide1.xml` - Speaker notes
12
+ * - `ppt/slideLayouts/*` - Slide layout definitions
13
+ * - `ppt/media/*` - Embedded images and media
14
+ *
15
+ * **Key Elements:**
16
+ * - `<p:sld>` - Slide
17
+ * - `<p:txBody>` - Text body containing paragraphs
18
+ * - `<a:p>` - Paragraph
19
+ * - `<a:r>` - Text run with formatting
20
+ * - `<a:t>` - Text content
21
+ *
22
+ * @module PowerPointParser
23
+ * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
24
+ */
25
+ Object.defineProperty(exports, "__esModule", { value: true });
26
+ exports.parsePowerPoint = void 0;
27
+ const xmldom_1 = require("@xmldom/xmldom");
28
+ const chartUtils_1 = require("../utils/chartUtils");
29
+ const errorUtils_1 = require("../utils/errorUtils");
30
+ const imageUtils_1 = require("../utils/imageUtils");
31
+ const ocrUtils_1 = require("../utils/ocrUtils");
32
+ const xmlUtils_1 = require("../utils/xmlUtils");
33
+ const zipUtils_1 = require("../utils/zipUtils");
34
+ /**
35
+ * Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
36
+ *
37
+ * @param buffer - The PPTX file as a Buffer
38
+ * @param config - Parser configuration
39
+ * @returns A promise resolving to the parsed AST
40
+ */
41
+ const parsePowerPoint = async (buffer, config) => {
42
+ const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
43
+ const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
44
+ const slideRelsRegex = /ppt\/slides\/_rels\/slide\d+\.xml\.rels/;
45
+ const slideNumberRegex = /lide(\d+)\.xml/;
46
+ const mediaFileRegex = /ppt\/media\/.*/;
47
+ const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
48
+ const corePropsFileRegex = /docProps\/core\.xml/;
49
+ const xmlSerializer = new xmldom_1.XMLSerializer();
50
+ const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
51
+ !!x.match(corePropsFileRegex) ||
52
+ !!x.match(slideRelsRegex) ||
53
+ (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
54
+ // Extract metadata
55
+ const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
56
+ const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
57
+ // Sort files
58
+ files.sort((a, b) => {
59
+ const aMatch = a.path.match(slideNumberRegex);
60
+ const bMatch = b.path.match(slideNumberRegex);
61
+ const aNum = aMatch ? parseInt(aMatch[1]) : 0;
62
+ const bNum = bMatch ? parseInt(bMatch[1]) : 0;
63
+ return aNum - bNum;
64
+ });
65
+ const content = [];
66
+ const rawContents = [];
67
+ const slideRelsMap = {};
68
+ let currentListId = 0;
69
+ let runningListIndex = 0;
70
+ let lastWasList = false;
71
+ let lastListType = null;
72
+ let lastListIndent = 0;
73
+ // per indent counters (for nested lists)
74
+ const levelCounters = {};
75
+ // Helper to parse a table node
76
+ const parseTable = (tblNode) => {
77
+ const rows = [];
78
+ const trNodes = (0, xmlUtils_1.getElementsByTagName)(tblNode, "a:tr");
79
+ for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
80
+ const trNode = trNodes[rIndex];
81
+ const cells = [];
82
+ const tcNodes = (0, xmlUtils_1.getElementsByTagName)(trNode, "a:tc");
83
+ for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
84
+ const tcNode = tcNodes[cIndex];
85
+ const cellChildren = [];
86
+ let cellText = '';
87
+ // Cells contain text bodies (txBody) which contain paragraphs
88
+ const txBody = (0, xmlUtils_1.getElementsByTagName)(tcNode, "a:txBody")[0];
89
+ if (txBody) {
90
+ const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
91
+ for (const p of paragraphs) {
92
+ // Reuse paragraph parsing logic if possible, or duplicate for now
93
+ // For simplicity, duplicating basic logic here as the main loop one is tied to shapes
94
+ const pNode = {
95
+ type: 'paragraph',
96
+ text: '',
97
+ children: [],
98
+ metadata: {}
99
+ };
100
+ if (config.includeRawContent) {
101
+ pNode.rawContent = p.toString();
102
+ }
103
+ const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
104
+ for (const r of runs) {
105
+ const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
106
+ if (t && t.childNodes[0]) {
107
+ const textContent = t.childNodes[0].nodeValue || '';
108
+ pNode.text += textContent;
109
+ const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
110
+ const formatting = {};
111
+ if (rPr) {
112
+ if (rPr.getAttribute("b") === "1")
113
+ formatting.bold = true;
114
+ if (rPr.getAttribute("i") === "1")
115
+ formatting.italic = true;
116
+ if (rPr.getAttribute("u") === "sng")
117
+ formatting.underline = true;
118
+ if (rPr.getAttribute("strike") === "sngStrike")
119
+ formatting.strikethrough = true;
120
+ const sz = rPr.getAttribute("sz");
121
+ if (sz)
122
+ formatting.size = (parseInt(sz) / 100).toString() + 'pt';
123
+ const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
124
+ if (solidFill) {
125
+ const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
126
+ if (srgbClr) {
127
+ const val = srgbClr.getAttribute("val");
128
+ if (val)
129
+ formatting.color = '#' + val;
130
+ }
131
+ }
132
+ const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
133
+ if (latin) {
134
+ const typeface = latin.getAttribute("typeface");
135
+ if (typeface)
136
+ formatting.font = typeface;
137
+ }
138
+ }
139
+ pNode.children?.push({
140
+ type: 'text',
141
+ text: textContent,
142
+ formatting: formatting
143
+ });
144
+ }
145
+ }
146
+ cellChildren.push(pNode);
147
+ cellText += pNode.text;
148
+ }
149
+ }
150
+ const cellNode = {
151
+ type: 'cell',
152
+ text: cellText,
153
+ children: cellChildren,
154
+ metadata: { row: rIndex, col: cIndex }
155
+ };
156
+ cells.push(cellNode);
157
+ }
158
+ const rowNode = {
159
+ type: 'row',
160
+ children: cells
161
+ };
162
+ rows.push(rowNode);
163
+ }
164
+ return {
165
+ type: 'table',
166
+ children: rows
167
+ };
168
+ };
169
+ /** Extract an AST node for p:pic */
170
+ const extractImageNode = (imageNode, slideNumber) => {
171
+ const blip = (0, xmlUtils_1.getElementsByTagName)(imageNode, "a:blip")[0];
172
+ if (!blip)
173
+ return null;
174
+ const rId = blip.getAttribute("r:embed");
175
+ if (!rId)
176
+ return null;
177
+ const rel = slideRelsMap[slideNumber]?.[rId];
178
+ if (!rel || rel.type !== "image")
179
+ return null;
180
+ const attachmentName = rel.target;
181
+ const nvPicPr = (0, xmlUtils_1.getElementsByTagName)(imageNode, "p:nvPicPr")[0];
182
+ const cNvPr = nvPicPr ? (0, xmlUtils_1.getElementsByTagName)(nvPicPr, "p:cNvPr")[0] : null;
183
+ const altText = cNvPr?.getAttribute("descr") || undefined;
184
+ return {
185
+ type: "image",
186
+ text: '',
187
+ metadata: {
188
+ attachmentName,
189
+ altText,
190
+ }
191
+ };
192
+ };
193
+ /**
194
+ * Extract an AST node for p:graphicFrame that contains a chart.
195
+ * This mirrors extractImageNode, but for charts instead of images.
196
+ *
197
+ * Steps:
198
+ * 1. Find <a:graphicData> inside the graphicFrame.
199
+ * 2. Check if uri is the chart namespace.
200
+ * 3. Extract <c:chart> and read r:id.
201
+ * 4. Resolve the relationship using slideRelsMap.
202
+ * 5. Produce a chart node with attachmentName (like images).
203
+ * 6. Chart text/data will be injected later in the pipeline.
204
+ *
205
+ * @param frameNode p:graphicFrame element
206
+ * @param slideNumber Current slide number for relationship resolution
207
+ * @returns A chart OfficeContentNode or null if not a chart.
208
+ */
209
+ const extractChartNode = (frameNode, slideNumber) => {
210
+ // Step 1: Find <a:graphicData>
211
+ const graphicData = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:graphicData")[0];
212
+ if (!graphicData) {
213
+ return null;
214
+ }
215
+ // Step 2: Verify chart namespace
216
+ // Must be: http://schemas.openxmlformats.org/drawingml/2006/chart
217
+ const uri = graphicData.getAttribute("uri");
218
+ const isChartGraphic = uri === "http://schemas.openxmlformats.org/drawingml/2006/chart";
219
+ if (!isChartGraphic) {
220
+ return null;
221
+ }
222
+ // Step 3: Find <c:chart>
223
+ const cChart = (0, xmlUtils_1.getElementsByTagName)(graphicData, "c:chart")[0];
224
+ if (!cChart) {
225
+ return null;
226
+ }
227
+ // Step 4: Extract r:id (relationship id)
228
+ const rId = cChart.getAttribute("r:id");
229
+ if (!rId) {
230
+ return null;
231
+ }
232
+ // Step 5: Resolve relationship target from slideRelsMap
233
+ const rel = slideRelsMap[slideNumber]?.[rId];
234
+ if (!rel || rel.type !== "chart") {
235
+ return null;
236
+ }
237
+ // rel.target will be something like "chart1.xml"
238
+ const attachmentName = rel.target;
239
+ // Step 6: Build AST node
240
+ const chartNode = {
241
+ type: "chart",
242
+ text: "",
243
+ metadata: {
244
+ attachmentName // name used to link to attachments & chartData
245
+ }
246
+ };
247
+ // Optional: include raw XML of the whole frame
248
+ if (config.includeRawContent) {
249
+ chartNode.rawContent = xmlSerializer.serializeToString(frameNode);
250
+ }
251
+ return chartNode;
252
+ };
253
+ /** Extract an AST node for p:graphicFrame */
254
+ const extractGraphicFrameNode = (frameNode, slideNumber) => {
255
+ const tbl = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:tbl")[0];
256
+ if (tbl) {
257
+ const tableNode = parseTable(tbl);
258
+ if (config.includeRawContent) {
259
+ tableNode.rawContent = xmlSerializer.serializeToString(frameNode);
260
+ }
261
+ if (tableNode.children && tableNode.children.length > 0) {
262
+ return tableNode;
263
+ }
264
+ }
265
+ if (frameNode.getElementsByTagName("c:chart").length > 0) {
266
+ const chartNode = extractChartNode(frameNode, slideNumber);
267
+ if (chartNode) {
268
+ return chartNode;
269
+ }
270
+ }
271
+ return null;
272
+ };
273
+ /** Extract text and hyperlinks from a p:sp shape. */
274
+ const extractShapeNodes = (spNode, slideNumber) => {
275
+ const nodes = [];
276
+ // Check for placeholder type (title, body, etc.)
277
+ const nvSpPr = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:nvSpPr")[0];
278
+ const nvPr = nvSpPr ? (0, xmlUtils_1.getElementsByTagName)(nvSpPr, "p:nvPr")[0] : null;
279
+ const ph = nvPr ? (0, xmlUtils_1.getElementsByTagName)(nvPr, "p:ph")[0] : null;
280
+ const type = ph ? ph.getAttribute("type") : "body";
281
+ const isTitle = type === "title" || type === "ctrTitle";
282
+ const txBody = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:txBody")[0];
283
+ if (txBody) {
284
+ const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
285
+ for (let i = 0; i < paragraphs.length; i++) {
286
+ const p = paragraphs[i];
287
+ const pNode = {
288
+ type: isTitle ? 'heading' : 'paragraph',
289
+ text: '',
290
+ children: [],
291
+ metadata: isTitle ? { level: 1 } : {}
292
+ };
293
+ // Paragraph Alignment and List Detection
294
+ const pPr = (0, xmlUtils_1.getElementsByTagName)(p, "a:pPr")[0];
295
+ let isList = false;
296
+ let listType = 'unordered';
297
+ let lvl = 0;
298
+ if (pPr) {
299
+ const lvlAttr = pPr.getAttribute("lvl");
300
+ if (lvlAttr)
301
+ lvl = parseInt(lvlAttr);
302
+ const buAutoNum = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buAutoNum")[0];
303
+ const buChar = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buChar")[0];
304
+ const buBlip = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buBlip")[0];
305
+ const buNode = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:bu")[0];
306
+ if (buAutoNum) {
307
+ isList = true;
308
+ listType = 'ordered';
309
+ }
310
+ else if (buChar || buBlip) {
311
+ isList = true;
312
+ listType = 'unordered';
313
+ }
314
+ else if (buNode) {
315
+ // inherited bullet from a list style
316
+ isList = true;
317
+ listType = 'unordered';
318
+ }
319
+ const algn = pPr.getAttribute("algn");
320
+ if (algn) {
321
+ const alignMap = {
322
+ 'l': 'left',
323
+ 'ctr': 'center',
324
+ 'r': 'right',
325
+ 'just': 'justify'
326
+ };
327
+ if (alignMap[algn]) {
328
+ pNode.metadata.alignment = alignMap[algn];
329
+ }
330
+ }
331
+ }
332
+ if (isList) {
333
+ pNode.type = 'list';
334
+ const ilvl = lvl;
335
+ // detect a new list when bullet type changes or previous was not a list
336
+ const newList = !lastWasList ||
337
+ listType !== lastListType;
338
+ if (newList) {
339
+ // new list → new ID
340
+ currentListId++;
341
+ // clear counters for nested levels
342
+ for (const k in levelCounters) {
343
+ delete levelCounters[k];
344
+ }
345
+ // start item index at 1
346
+ runningListIndex = 0;
347
+ levelCounters[ilvl] = 0;
348
+ }
349
+ else {
350
+ // same listId, but indentation may change
351
+ // if going deeper → start at 1 for that level
352
+ if (ilvl > lastListIndent) {
353
+ runningListIndex = 0;
354
+ levelCounters[ilvl] = 0;
355
+ }
356
+ // if going shallower → restore previous level counter + 1
357
+ else if (ilvl < lastListIndent) {
358
+ // remove deeper counters
359
+ for (const lvlKey in levelCounters) {
360
+ const lv = parseInt(lvlKey);
361
+ if (lv > ilvl)
362
+ delete levelCounters[lv];
363
+ }
364
+ // continue counter at this level
365
+ const prev = levelCounters[ilvl] || 0;
366
+ runningListIndex = prev + 1;
367
+ levelCounters[ilvl] = runningListIndex;
368
+ }
369
+ // same level → increment
370
+ else {
371
+ const prev = levelCounters[ilvl] || 0;
372
+ runningListIndex = prev + 1;
373
+ levelCounters[ilvl] = runningListIndex;
374
+ }
375
+ }
376
+ // update tracking state
377
+ lastWasList = true;
378
+ lastListType = listType;
379
+ lastListIndent = ilvl;
380
+ // metadata output
381
+ pNode.metadata = {
382
+ ...pNode.metadata,
383
+ listType,
384
+ indentation: ilvl,
385
+ listId: currentListId.toString(),
386
+ itemIndex: runningListIndex,
387
+ alignment: pNode.metadata?.alignment || 'left',
388
+ };
389
+ }
390
+ else {
391
+ lastWasList = false;
392
+ lastListType = null;
393
+ lastListIndent = 0;
394
+ }
395
+ if (isTitle) {
396
+ pNode.metadata = { ...pNode.metadata, level: 1 };
397
+ }
398
+ if (config.includeRawContent) {
399
+ pNode.rawContent = p.toString();
400
+ }
401
+ const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
402
+ for (let j = 0; j < runs.length; j++) {
403
+ const r = runs[j];
404
+ const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
405
+ if (t && t.childNodes[0]) {
406
+ const textContent = t.childNodes[0].nodeValue || '';
407
+ pNode.text += textContent;
408
+ const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
409
+ const formatting = {};
410
+ if (rPr) {
411
+ if (rPr.getAttribute("b") === "1")
412
+ formatting.bold = true;
413
+ if (rPr.getAttribute("i") === "1")
414
+ formatting.italic = true;
415
+ if (rPr.getAttribute("u") === "sng")
416
+ formatting.underline = true;
417
+ if (rPr.getAttribute("strike") === "sngStrike")
418
+ formatting.strikethrough = true;
419
+ const sz = rPr.getAttribute("sz");
420
+ if (sz)
421
+ formatting.size = (parseInt(sz) / 100).toString() + 'pt';
422
+ // Color extraction
423
+ const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
424
+ if (solidFill) {
425
+ const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
426
+ if (srgbClr) {
427
+ const val = srgbClr.getAttribute("val");
428
+ if (val)
429
+ formatting.color = '#' + val;
430
+ }
431
+ }
432
+ // Highlight extraction
433
+ const highlight = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:highlight")[0];
434
+ if (highlight) {
435
+ const srgbClr = (0, xmlUtils_1.getElementsByTagName)(highlight, "a:srgbClr")[0];
436
+ if (srgbClr) {
437
+ const val = srgbClr.getAttribute("val");
438
+ if (val)
439
+ formatting.backgroundColor = '#' + val;
440
+ }
441
+ }
442
+ // Font family
443
+ const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
444
+ if (latin) {
445
+ const typeface = latin.getAttribute("typeface");
446
+ if (typeface)
447
+ formatting.font = typeface;
448
+ }
449
+ // Subscript/Superscript
450
+ const baseline = rPr.getAttribute("baseline");
451
+ if (baseline) {
452
+ const baselineVal = parseInt(baseline);
453
+ if (baselineVal < 0)
454
+ formatting.subscript = true;
455
+ if (baselineVal > 0)
456
+ formatting.superscript = true;
457
+ }
458
+ }
459
+ const textNode = {
460
+ type: 'text',
461
+ text: textContent,
462
+ formatting: formatting
463
+ };
464
+ // Check for Hyperlinks
465
+ const hlinkClick = (0, xmlUtils_1.getElementsByTagName)(r, "a:hlinkClick")[0];
466
+ // Check if this run has a hyperlink click action
467
+ if (hlinkClick) {
468
+ // Relationship ID for the link
469
+ const rId = hlinkClick.getAttribute("r:id");
470
+ // Optional action attribute, often for internal jumps
471
+ const action = hlinkClick.getAttribute("action");
472
+ // Result placeholders
473
+ let link;
474
+ let linkType;
475
+ // Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
476
+ if (rId
477
+ && slideRelsMap[slideNumber]
478
+ && slideRelsMap[slideNumber][rId]
479
+ && slideRelsMap[slideNumber][rId].type === "hyperlink") {
480
+ // External URL
481
+ link = slideRelsMap[slideNumber][rId].target;
482
+ linkType = "external";
483
+ }
484
+ // Case 2: Relationship exists and is an internal slide reference
485
+ else if (rId
486
+ && slideRelsMap[slideNumber]
487
+ && slideRelsMap[slideNumber][rId]
488
+ && slideRelsMap[slideNumber][rId].type === "slide") {
489
+ // Example target: ppt/slides/slide3.xml
490
+ link = slideRelsMap[slideNumber][rId].target;
491
+ linkType = "internal";
492
+ }
493
+ // Case 3: action attribute like ppaction://hlinksldjump
494
+ else if (action) {
495
+ link = action;
496
+ linkType = "internal";
497
+ }
498
+ // Assign metadata only if a link was actually discovered
499
+ if (link) {
500
+ textNode.metadata = { link, linkType };
501
+ }
502
+ }
503
+ pNode.children?.push(textNode);
504
+ }
505
+ }
506
+ if (pNode.text) {
507
+ nodes.push(pNode);
508
+ }
509
+ }
510
+ }
511
+ return nodes;
512
+ };
513
+ /**
514
+ * Recursively traverses a PowerPoint shape tree (p:spTree),
515
+ * including grouped shapes (p:grpSp), and dispatches each element
516
+ * to the appropriate handler (shape, image, chart, etc.).
517
+ *
518
+ * This function preserves the visual order because it processes
519
+ * children in the order they appear in the XML. It also accumulates
520
+ * transforms so nested groups inherit positional transforms.
521
+ *
522
+ * @param treeNode The XML node representing <p:spTree>
523
+ * @param parentTransform Transform inherited from parent groups (defaults to identity)
524
+ */
525
+ function traverseSpTree(treeNode, slideNumber) {
526
+ const nodes = [];
527
+ // Process children in XML order (this preserves Z-order)
528
+ for (const child of Array.from(treeNode?.childNodes || [])) {
529
+ if (child.nodeType !== 1) {
530
+ continue;
531
+ }
532
+ const element = child;
533
+ const tag = element.tagName;
534
+ // Case 1: Normal shape
535
+ if (tag === "p:sp") {
536
+ nodes.push(...extractShapeNodes(element, slideNumber));
537
+ }
538
+ // Case 2: Inline picture
539
+ else if (tag === "p:pic") {
540
+ const imageNode = extractImageNode(element, slideNumber);
541
+ if (imageNode) {
542
+ nodes.push(imageNode);
543
+ }
544
+ }
545
+ // Case 3: Chart or other graphic frame
546
+ else if (tag === "p:graphicFrame") {
547
+ const tableNode = extractGraphicFrameNode(element, slideNumber);
548
+ if (tableNode) {
549
+ nodes.push(tableNode);
550
+ }
551
+ }
552
+ // Case 4: Grouped shape (recursive!)
553
+ else if (tag === "p:grpSp") {
554
+ // Extract the nested <p:spTree> inside the group
555
+ const nestedTree = element.getElementsByTagName("p:spTree")[0];
556
+ // Recurse into the nested tree
557
+ nodes.push(...traverseSpTree(nestedTree, slideNumber));
558
+ }
559
+ }
560
+ return nodes;
561
+ }
562
+ // First pass: Process relationships
563
+ for (const file of files) {
564
+ // Check whether this file is a slideX.xml.rels file
565
+ if (file.path.match(slideRelsRegex)) {
566
+ /**
567
+ * Builds a map of slide number to a map of relationship IDs containing:
568
+ * - type: The relationship category (image, hyperlink, chart, etc)
569
+ * - target: The fully normalized target path inside the PPTX zip
570
+ *
571
+ * Example structure:
572
+ * {
573
+ * 1: {
574
+ * "rId2": { type: "image", target: "ppt/media/image3.png" },
575
+ * "rId5": { type: "hyperlink", target: "https://example.com" }
576
+ * }
577
+ * }
578
+ *
579
+ * @param files All extracted PPTX ZIP files.
580
+ * @param slideRelsMap A map of slide number to relationship info.
581
+ */
582
+ // Extract slide number from path
583
+ const match = file.path.match(/slide(\d+)\.xml\.rels/);
584
+ if (match) {
585
+ // Convert matched number to integer
586
+ const slideNum = parseInt(match[1]);
587
+ // Prepare map for this slide
588
+ slideRelsMap[slideNum] = {};
589
+ // Parse the rels XML
590
+ const relsXml = (0, xmlUtils_1.parseXmlString)(file.content.toString());
591
+ // Get all Relationship nodes
592
+ const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
593
+ // Loop through each relationship node
594
+ for (let i = 0; i < relationships.length; i++) {
595
+ // Relationship ID, Example: "rId2"
596
+ const id = relationships[i].getAttribute("Id");
597
+ // Relationship Type, Example: "http://schemas.openxmlformats.org/officeDocument/2006/relationships/image"
598
+ const typeAttr = relationships[i].getAttribute("Type");
599
+ // Raw Target, may be relative or absolute
600
+ const targetRaw = relationships[i].getAttribute("Target");
601
+ // Only proceed if ID and Type exist
602
+ if (id && typeAttr && targetRaw) {
603
+ // Simplify Type to a short keyword (image, hyperlink, chart, etc)
604
+ // This is optional but very useful.
605
+ let simplifiedType = "other";
606
+ // Check image
607
+ if (typeAttr.includes("relationships/image")) {
608
+ simplifiedType = "image";
609
+ }
610
+ // Check hyperlink
611
+ else if (typeAttr.includes("relationships/hyperlink")) {
612
+ simplifiedType = "hyperlink";
613
+ }
614
+ // Check chart
615
+ else if (typeAttr.includes("relationships/chart")) {
616
+ simplifiedType = "chart";
617
+ }
618
+ // Check slide references
619
+ else if (typeAttr.includes("relationships/slide")) {
620
+ simplifiedType = "slide";
621
+ }
622
+ // Check notes
623
+ else if (typeAttr.includes("relationships/notesSlide")) {
624
+ simplifiedType = "notes";
625
+ }
626
+ // Now normalize the target only if it is a local file path.
627
+ // Hyperlinks are external and should not be normalized.
628
+ let normalizedTarget = targetRaw;
629
+ // Local paths never contain "http" or "https"
630
+ const isExternal = targetRaw.startsWith("http://") || targetRaw.startsWith("https://");
631
+ // If not external, normalize the target which is just the name of the item.
632
+ if (!isExternal) {
633
+ normalizedTarget = normalizedTarget.split('/').pop() || '';
634
+ }
635
+ // Finally store full relationship info
636
+ slideRelsMap[slideNum][id] = {
637
+ type: simplifiedType,
638
+ target: normalizedTarget
639
+ };
640
+ }
641
+ }
642
+ }
643
+ }
644
+ }
645
+ // Now for processing all the other files - slides and notes.
646
+ for (const file of files) {
647
+ if (file.path.match(mediaFileRegex))
648
+ continue;
649
+ if (file.path.match(chartFileRegex))
650
+ continue;
651
+ if (file.path.match(slideRelsRegex))
652
+ continue;
653
+ if (file.path.match(corePropsFileRegex))
654
+ continue;
655
+ const xmlContentString = file.content.toString();
656
+ const xml = (0, xmlUtils_1.parseXmlString)(xmlContentString);
657
+ if (config.includeRawContent) {
658
+ rawContents.push(xmlContentString);
659
+ }
660
+ const slideMatch = file.path.match(slideNumberRegex);
661
+ const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
662
+ const isNote = file.path.includes("notesSlide");
663
+ const slideNode = {
664
+ type: isNote ? 'note' : 'slide',
665
+ children: [],
666
+ metadata: {
667
+ slideNumber: slideNumber,
668
+ ...(isNote ? { noteId: `slide-note-${slideNumber}` } : {})
669
+ }
670
+ };
671
+ if (config.includeRawContent) {
672
+ slideNode.rawContent = file.content.toString();
673
+ }
674
+ /**
675
+ * Extract slide contents in correct document order by scanning p:spTree children.
676
+ * This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
677
+ */
678
+ const spTree = (0, xmlUtils_1.getElementsByTagName)(xml, "p:spTree")[0];
679
+ if (spTree) {
680
+ slideNode.children?.push(...traverseSpTree(spTree, slideNumber));
681
+ }
682
+ if (slideNode.children && slideNode.children.length > 0) {
683
+ content.push(slideNode);
684
+ }
685
+ }
686
+ const attachments = [];
687
+ const mediaFiles = files.filter(f => f.path.match(/ppt\/media\/.*/));
688
+ const chartFiles = files.filter(f => f.path.match(/ppt\/charts\/chart\d+\.xml/));
689
+ // First run to extract attachments and to assign ocr to image files.
690
+ if (config.extractAttachments) {
691
+ // Extract media files as attachments
692
+ for (const media of mediaFiles) {
693
+ const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
694
+ attachments.push(attachment);
695
+ if (config.ocr) {
696
+ if (attachment.mimeType.startsWith('image/')) {
697
+ try {
698
+ attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
699
+ }
700
+ catch (e) {
701
+ (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
702
+ }
703
+ }
704
+ }
705
+ }
706
+ // Extract chart files as attachments
707
+ for (const chart of chartFiles) {
708
+ const attachment = {
709
+ type: 'chart',
710
+ mimeType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
711
+ data: chart.content.toString('base64'),
712
+ name: chart.path.split('/').pop() || '',
713
+ extension: 'xml'
714
+ };
715
+ attachments.push(attachment);
716
+ // Extract text from chart XML
717
+ try {
718
+ const chartData = await (0, chartUtils_1.extractChartData)(chart.content);
719
+ // Assign chartData to attachment
720
+ attachment.chartData = chartData;
721
+ }
722
+ catch (e) {
723
+ (0, errorUtils_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
724
+ }
725
+ }
726
+ // Loop through nodes to find images and charts and link their text and chartData
727
+ const assignAttachmentData = (nodes) => {
728
+ for (const node of nodes) {
729
+ if ('attachmentName' in (node.metadata || {})) {
730
+ const meta = node.metadata;
731
+ const attachment = attachments.find(a => a.name === meta.attachmentName);
732
+ if (attachment) {
733
+ if (node.type === 'image') {
734
+ attachment.altText = meta.altText;
735
+ if (attachment.ocrText)
736
+ node.text = attachment.ocrText;
737
+ }
738
+ if (node.type === 'chart') {
739
+ node.text = attachment.chartData?.rawTexts.join(config.newlineDelimiter || '\n');
740
+ }
741
+ }
742
+ }
743
+ if (node.children) {
744
+ assignAttachmentData(node.children);
745
+ }
746
+ }
747
+ };
748
+ assignAttachmentData(content);
749
+ }
750
+ // Finally, if the notes are required to be at the end of the document, move them there.
751
+ if (!config.ignoreNotes && config.putNotesAtLast) {
752
+ content.sort((a, b) => {
753
+ const aIsNote = a.type === 'note' ? 1 : 0;
754
+ const bIsNote = b.type === 'note' ? 1 : 0;
755
+ return aIsNote - bIsNote;
756
+ });
757
+ }
758
+ return {
759
+ type: 'pptx',
760
+ metadata: metadata,
761
+ content: content,
762
+ attachments: attachments,
763
+ toText: () => content.map(c => {
764
+ // Recursive text extraction
765
+ const getText = (node) => {
766
+ let t = '';
767
+ if (node.children) {
768
+ t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
769
+ }
770
+ else
771
+ t += node.text || '';
772
+ return t;
773
+ };
774
+ return getText(c);
775
+ }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
776
+ };
777
+ };
778
+ exports.parsePowerPoint = parsePowerPoint;