officeparser 5.2.2 → 6.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +152 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +850 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -790
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,778 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* PowerPoint Presentation (PPTX) Parser
|
|
4
|
+
*
|
|
5
|
+
* **PPTX Format Overview:**
|
|
6
|
+
* PPTX is the default format for Microsoft PowerPoint since Office 2007, based on OOXML.
|
|
7
|
+
*
|
|
8
|
+
* **File Structure:**
|
|
9
|
+
* - `ppt/presentation.xml` - Presentation structure and slide list
|
|
10
|
+
* - `ppt/slides/slide1.xml` - Individual slide content
|
|
11
|
+
* - `ppt/notesSlides/notesSlide1.xml` - Speaker notes
|
|
12
|
+
* - `ppt/slideLayouts/*` - Slide layout definitions
|
|
13
|
+
* - `ppt/media/*` - Embedded images and media
|
|
14
|
+
*
|
|
15
|
+
* **Key Elements:**
|
|
16
|
+
* - `<p:sld>` - Slide
|
|
17
|
+
* - `<p:txBody>` - Text body containing paragraphs
|
|
18
|
+
* - `<a:p>` - Paragraph
|
|
19
|
+
* - `<a:r>` - Text run with formatting
|
|
20
|
+
* - `<a:t>` - Text content
|
|
21
|
+
*
|
|
22
|
+
* @module PowerPointParser
|
|
23
|
+
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
24
|
+
*/
|
|
25
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
|
+
exports.parsePowerPoint = void 0;
|
|
27
|
+
const xmldom_1 = require("@xmldom/xmldom");
|
|
28
|
+
const chartUtils_1 = require("../utils/chartUtils");
|
|
29
|
+
const errorUtils_1 = require("../utils/errorUtils");
|
|
30
|
+
const imageUtils_1 = require("../utils/imageUtils");
|
|
31
|
+
const ocrUtils_1 = require("../utils/ocrUtils");
|
|
32
|
+
const xmlUtils_1 = require("../utils/xmlUtils");
|
|
33
|
+
const zipUtils_1 = require("../utils/zipUtils");
|
|
34
|
+
/**
|
|
35
|
+
* Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
|
|
36
|
+
*
|
|
37
|
+
* @param buffer - The PPTX file as a Buffer
|
|
38
|
+
* @param config - Parser configuration
|
|
39
|
+
* @returns A promise resolving to the parsed AST
|
|
40
|
+
*/
|
|
41
|
+
const parsePowerPoint = async (buffer, config) => {
|
|
42
|
+
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
43
|
+
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
44
|
+
const slideRelsRegex = /ppt\/slides\/_rels\/slide\d+\.xml\.rels/;
|
|
45
|
+
const slideNumberRegex = /lide(\d+)\.xml/;
|
|
46
|
+
const mediaFileRegex = /ppt\/media\/.*/;
|
|
47
|
+
const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
|
|
48
|
+
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
49
|
+
const xmlSerializer = new xmldom_1.XMLSerializer();
|
|
50
|
+
const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
|
|
51
|
+
!!x.match(corePropsFileRegex) ||
|
|
52
|
+
!!x.match(slideRelsRegex) ||
|
|
53
|
+
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
|
|
54
|
+
// Extract metadata
|
|
55
|
+
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
56
|
+
const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
57
|
+
// Sort files
|
|
58
|
+
files.sort((a, b) => {
|
|
59
|
+
const aMatch = a.path.match(slideNumberRegex);
|
|
60
|
+
const bMatch = b.path.match(slideNumberRegex);
|
|
61
|
+
const aNum = aMatch ? parseInt(aMatch[1]) : 0;
|
|
62
|
+
const bNum = bMatch ? parseInt(bMatch[1]) : 0;
|
|
63
|
+
return aNum - bNum;
|
|
64
|
+
});
|
|
65
|
+
const content = [];
|
|
66
|
+
const rawContents = [];
|
|
67
|
+
const slideRelsMap = {};
|
|
68
|
+
let currentListId = 0;
|
|
69
|
+
let runningListIndex = 0;
|
|
70
|
+
let lastWasList = false;
|
|
71
|
+
let lastListType = null;
|
|
72
|
+
let lastListIndent = 0;
|
|
73
|
+
// per indent counters (for nested lists)
|
|
74
|
+
const levelCounters = {};
|
|
75
|
+
// Helper to parse a table node
|
|
76
|
+
const parseTable = (tblNode) => {
|
|
77
|
+
const rows = [];
|
|
78
|
+
const trNodes = (0, xmlUtils_1.getElementsByTagName)(tblNode, "a:tr");
|
|
79
|
+
for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
|
|
80
|
+
const trNode = trNodes[rIndex];
|
|
81
|
+
const cells = [];
|
|
82
|
+
const tcNodes = (0, xmlUtils_1.getElementsByTagName)(trNode, "a:tc");
|
|
83
|
+
for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
|
|
84
|
+
const tcNode = tcNodes[cIndex];
|
|
85
|
+
const cellChildren = [];
|
|
86
|
+
let cellText = '';
|
|
87
|
+
// Cells contain text bodies (txBody) which contain paragraphs
|
|
88
|
+
const txBody = (0, xmlUtils_1.getElementsByTagName)(tcNode, "a:txBody")[0];
|
|
89
|
+
if (txBody) {
|
|
90
|
+
const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
|
|
91
|
+
for (const p of paragraphs) {
|
|
92
|
+
// Reuse paragraph parsing logic if possible, or duplicate for now
|
|
93
|
+
// For simplicity, duplicating basic logic here as the main loop one is tied to shapes
|
|
94
|
+
const pNode = {
|
|
95
|
+
type: 'paragraph',
|
|
96
|
+
text: '',
|
|
97
|
+
children: [],
|
|
98
|
+
metadata: {}
|
|
99
|
+
};
|
|
100
|
+
if (config.includeRawContent) {
|
|
101
|
+
pNode.rawContent = p.toString();
|
|
102
|
+
}
|
|
103
|
+
const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
|
|
104
|
+
for (const r of runs) {
|
|
105
|
+
const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
|
|
106
|
+
if (t && t.childNodes[0]) {
|
|
107
|
+
const textContent = t.childNodes[0].nodeValue || '';
|
|
108
|
+
pNode.text += textContent;
|
|
109
|
+
const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
|
|
110
|
+
const formatting = {};
|
|
111
|
+
if (rPr) {
|
|
112
|
+
if (rPr.getAttribute("b") === "1")
|
|
113
|
+
formatting.bold = true;
|
|
114
|
+
if (rPr.getAttribute("i") === "1")
|
|
115
|
+
formatting.italic = true;
|
|
116
|
+
if (rPr.getAttribute("u") === "sng")
|
|
117
|
+
formatting.underline = true;
|
|
118
|
+
if (rPr.getAttribute("strike") === "sngStrike")
|
|
119
|
+
formatting.strikethrough = true;
|
|
120
|
+
const sz = rPr.getAttribute("sz");
|
|
121
|
+
if (sz)
|
|
122
|
+
formatting.size = (parseInt(sz) / 100).toString() + 'pt';
|
|
123
|
+
const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
|
|
124
|
+
if (solidFill) {
|
|
125
|
+
const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
|
|
126
|
+
if (srgbClr) {
|
|
127
|
+
const val = srgbClr.getAttribute("val");
|
|
128
|
+
if (val)
|
|
129
|
+
formatting.color = '#' + val;
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
|
|
133
|
+
if (latin) {
|
|
134
|
+
const typeface = latin.getAttribute("typeface");
|
|
135
|
+
if (typeface)
|
|
136
|
+
formatting.font = typeface;
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
pNode.children?.push({
|
|
140
|
+
type: 'text',
|
|
141
|
+
text: textContent,
|
|
142
|
+
formatting: formatting
|
|
143
|
+
});
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
cellChildren.push(pNode);
|
|
147
|
+
cellText += pNode.text;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
const cellNode = {
|
|
151
|
+
type: 'cell',
|
|
152
|
+
text: cellText,
|
|
153
|
+
children: cellChildren,
|
|
154
|
+
metadata: { row: rIndex, col: cIndex }
|
|
155
|
+
};
|
|
156
|
+
cells.push(cellNode);
|
|
157
|
+
}
|
|
158
|
+
const rowNode = {
|
|
159
|
+
type: 'row',
|
|
160
|
+
children: cells
|
|
161
|
+
};
|
|
162
|
+
rows.push(rowNode);
|
|
163
|
+
}
|
|
164
|
+
return {
|
|
165
|
+
type: 'table',
|
|
166
|
+
children: rows
|
|
167
|
+
};
|
|
168
|
+
};
|
|
169
|
+
/** Extract an AST node for p:pic */
|
|
170
|
+
const extractImageNode = (imageNode, slideNumber) => {
|
|
171
|
+
const blip = (0, xmlUtils_1.getElementsByTagName)(imageNode, "a:blip")[0];
|
|
172
|
+
if (!blip)
|
|
173
|
+
return null;
|
|
174
|
+
const rId = blip.getAttribute("r:embed");
|
|
175
|
+
if (!rId)
|
|
176
|
+
return null;
|
|
177
|
+
const rel = slideRelsMap[slideNumber]?.[rId];
|
|
178
|
+
if (!rel || rel.type !== "image")
|
|
179
|
+
return null;
|
|
180
|
+
const attachmentName = rel.target;
|
|
181
|
+
const nvPicPr = (0, xmlUtils_1.getElementsByTagName)(imageNode, "p:nvPicPr")[0];
|
|
182
|
+
const cNvPr = nvPicPr ? (0, xmlUtils_1.getElementsByTagName)(nvPicPr, "p:cNvPr")[0] : null;
|
|
183
|
+
const altText = cNvPr?.getAttribute("descr") || undefined;
|
|
184
|
+
return {
|
|
185
|
+
type: "image",
|
|
186
|
+
text: '',
|
|
187
|
+
metadata: {
|
|
188
|
+
attachmentName,
|
|
189
|
+
altText,
|
|
190
|
+
}
|
|
191
|
+
};
|
|
192
|
+
};
|
|
193
|
+
/**
|
|
194
|
+
* Extract an AST node for p:graphicFrame that contains a chart.
|
|
195
|
+
* This mirrors extractImageNode, but for charts instead of images.
|
|
196
|
+
*
|
|
197
|
+
* Steps:
|
|
198
|
+
* 1. Find <a:graphicData> inside the graphicFrame.
|
|
199
|
+
* 2. Check if uri is the chart namespace.
|
|
200
|
+
* 3. Extract <c:chart> and read r:id.
|
|
201
|
+
* 4. Resolve the relationship using slideRelsMap.
|
|
202
|
+
* 5. Produce a chart node with attachmentName (like images).
|
|
203
|
+
* 6. Chart text/data will be injected later in the pipeline.
|
|
204
|
+
*
|
|
205
|
+
* @param frameNode p:graphicFrame element
|
|
206
|
+
* @param slideNumber Current slide number for relationship resolution
|
|
207
|
+
* @returns A chart OfficeContentNode or null if not a chart.
|
|
208
|
+
*/
|
|
209
|
+
const extractChartNode = (frameNode, slideNumber) => {
|
|
210
|
+
// Step 1: Find <a:graphicData>
|
|
211
|
+
const graphicData = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:graphicData")[0];
|
|
212
|
+
if (!graphicData) {
|
|
213
|
+
return null;
|
|
214
|
+
}
|
|
215
|
+
// Step 2: Verify chart namespace
|
|
216
|
+
// Must be: http://schemas.openxmlformats.org/drawingml/2006/chart
|
|
217
|
+
const uri = graphicData.getAttribute("uri");
|
|
218
|
+
const isChartGraphic = uri === "http://schemas.openxmlformats.org/drawingml/2006/chart";
|
|
219
|
+
if (!isChartGraphic) {
|
|
220
|
+
return null;
|
|
221
|
+
}
|
|
222
|
+
// Step 3: Find <c:chart>
|
|
223
|
+
const cChart = (0, xmlUtils_1.getElementsByTagName)(graphicData, "c:chart")[0];
|
|
224
|
+
if (!cChart) {
|
|
225
|
+
return null;
|
|
226
|
+
}
|
|
227
|
+
// Step 4: Extract r:id (relationship id)
|
|
228
|
+
const rId = cChart.getAttribute("r:id");
|
|
229
|
+
if (!rId) {
|
|
230
|
+
return null;
|
|
231
|
+
}
|
|
232
|
+
// Step 5: Resolve relationship target from slideRelsMap
|
|
233
|
+
const rel = slideRelsMap[slideNumber]?.[rId];
|
|
234
|
+
if (!rel || rel.type !== "chart") {
|
|
235
|
+
return null;
|
|
236
|
+
}
|
|
237
|
+
// rel.target will be something like "chart1.xml"
|
|
238
|
+
const attachmentName = rel.target;
|
|
239
|
+
// Step 6: Build AST node
|
|
240
|
+
const chartNode = {
|
|
241
|
+
type: "chart",
|
|
242
|
+
text: "",
|
|
243
|
+
metadata: {
|
|
244
|
+
attachmentName // name used to link to attachments & chartData
|
|
245
|
+
}
|
|
246
|
+
};
|
|
247
|
+
// Optional: include raw XML of the whole frame
|
|
248
|
+
if (config.includeRawContent) {
|
|
249
|
+
chartNode.rawContent = xmlSerializer.serializeToString(frameNode);
|
|
250
|
+
}
|
|
251
|
+
return chartNode;
|
|
252
|
+
};
|
|
253
|
+
/** Extract an AST node for p:graphicFrame */
|
|
254
|
+
const extractGraphicFrameNode = (frameNode, slideNumber) => {
|
|
255
|
+
const tbl = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:tbl")[0];
|
|
256
|
+
if (tbl) {
|
|
257
|
+
const tableNode = parseTable(tbl);
|
|
258
|
+
if (config.includeRawContent) {
|
|
259
|
+
tableNode.rawContent = xmlSerializer.serializeToString(frameNode);
|
|
260
|
+
}
|
|
261
|
+
if (tableNode.children && tableNode.children.length > 0) {
|
|
262
|
+
return tableNode;
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
if (frameNode.getElementsByTagName("c:chart").length > 0) {
|
|
266
|
+
const chartNode = extractChartNode(frameNode, slideNumber);
|
|
267
|
+
if (chartNode) {
|
|
268
|
+
return chartNode;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
return null;
|
|
272
|
+
};
|
|
273
|
+
/** Extract text and hyperlinks from a p:sp shape. */
|
|
274
|
+
const extractShapeNodes = (spNode, slideNumber) => {
|
|
275
|
+
const nodes = [];
|
|
276
|
+
// Check for placeholder type (title, body, etc.)
|
|
277
|
+
const nvSpPr = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:nvSpPr")[0];
|
|
278
|
+
const nvPr = nvSpPr ? (0, xmlUtils_1.getElementsByTagName)(nvSpPr, "p:nvPr")[0] : null;
|
|
279
|
+
const ph = nvPr ? (0, xmlUtils_1.getElementsByTagName)(nvPr, "p:ph")[0] : null;
|
|
280
|
+
const type = ph ? ph.getAttribute("type") : "body";
|
|
281
|
+
const isTitle = type === "title" || type === "ctrTitle";
|
|
282
|
+
const txBody = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:txBody")[0];
|
|
283
|
+
if (txBody) {
|
|
284
|
+
const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
|
|
285
|
+
for (let i = 0; i < paragraphs.length; i++) {
|
|
286
|
+
const p = paragraphs[i];
|
|
287
|
+
const pNode = {
|
|
288
|
+
type: isTitle ? 'heading' : 'paragraph',
|
|
289
|
+
text: '',
|
|
290
|
+
children: [],
|
|
291
|
+
metadata: isTitle ? { level: 1 } : {}
|
|
292
|
+
};
|
|
293
|
+
// Paragraph Alignment and List Detection
|
|
294
|
+
const pPr = (0, xmlUtils_1.getElementsByTagName)(p, "a:pPr")[0];
|
|
295
|
+
let isList = false;
|
|
296
|
+
let listType = 'unordered';
|
|
297
|
+
let lvl = 0;
|
|
298
|
+
if (pPr) {
|
|
299
|
+
const lvlAttr = pPr.getAttribute("lvl");
|
|
300
|
+
if (lvlAttr)
|
|
301
|
+
lvl = parseInt(lvlAttr);
|
|
302
|
+
const buAutoNum = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buAutoNum")[0];
|
|
303
|
+
const buChar = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buChar")[0];
|
|
304
|
+
const buBlip = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buBlip")[0];
|
|
305
|
+
const buNode = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:bu")[0];
|
|
306
|
+
if (buAutoNum) {
|
|
307
|
+
isList = true;
|
|
308
|
+
listType = 'ordered';
|
|
309
|
+
}
|
|
310
|
+
else if (buChar || buBlip) {
|
|
311
|
+
isList = true;
|
|
312
|
+
listType = 'unordered';
|
|
313
|
+
}
|
|
314
|
+
else if (buNode) {
|
|
315
|
+
// inherited bullet from a list style
|
|
316
|
+
isList = true;
|
|
317
|
+
listType = 'unordered';
|
|
318
|
+
}
|
|
319
|
+
const algn = pPr.getAttribute("algn");
|
|
320
|
+
if (algn) {
|
|
321
|
+
const alignMap = {
|
|
322
|
+
'l': 'left',
|
|
323
|
+
'ctr': 'center',
|
|
324
|
+
'r': 'right',
|
|
325
|
+
'just': 'justify'
|
|
326
|
+
};
|
|
327
|
+
if (alignMap[algn]) {
|
|
328
|
+
pNode.metadata.alignment = alignMap[algn];
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
if (isList) {
|
|
333
|
+
pNode.type = 'list';
|
|
334
|
+
const ilvl = lvl;
|
|
335
|
+
// detect a new list when bullet type changes or previous was not a list
|
|
336
|
+
const newList = !lastWasList ||
|
|
337
|
+
listType !== lastListType;
|
|
338
|
+
if (newList) {
|
|
339
|
+
// new list → new ID
|
|
340
|
+
currentListId++;
|
|
341
|
+
// clear counters for nested levels
|
|
342
|
+
for (const k in levelCounters) {
|
|
343
|
+
delete levelCounters[k];
|
|
344
|
+
}
|
|
345
|
+
// start item index at 1
|
|
346
|
+
runningListIndex = 0;
|
|
347
|
+
levelCounters[ilvl] = 0;
|
|
348
|
+
}
|
|
349
|
+
else {
|
|
350
|
+
// same listId, but indentation may change
|
|
351
|
+
// if going deeper → start at 1 for that level
|
|
352
|
+
if (ilvl > lastListIndent) {
|
|
353
|
+
runningListIndex = 0;
|
|
354
|
+
levelCounters[ilvl] = 0;
|
|
355
|
+
}
|
|
356
|
+
// if going shallower → restore previous level counter + 1
|
|
357
|
+
else if (ilvl < lastListIndent) {
|
|
358
|
+
// remove deeper counters
|
|
359
|
+
for (const lvlKey in levelCounters) {
|
|
360
|
+
const lv = parseInt(lvlKey);
|
|
361
|
+
if (lv > ilvl)
|
|
362
|
+
delete levelCounters[lv];
|
|
363
|
+
}
|
|
364
|
+
// continue counter at this level
|
|
365
|
+
const prev = levelCounters[ilvl] || 0;
|
|
366
|
+
runningListIndex = prev + 1;
|
|
367
|
+
levelCounters[ilvl] = runningListIndex;
|
|
368
|
+
}
|
|
369
|
+
// same level → increment
|
|
370
|
+
else {
|
|
371
|
+
const prev = levelCounters[ilvl] || 0;
|
|
372
|
+
runningListIndex = prev + 1;
|
|
373
|
+
levelCounters[ilvl] = runningListIndex;
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
// update tracking state
|
|
377
|
+
lastWasList = true;
|
|
378
|
+
lastListType = listType;
|
|
379
|
+
lastListIndent = ilvl;
|
|
380
|
+
// metadata output
|
|
381
|
+
pNode.metadata = {
|
|
382
|
+
...pNode.metadata,
|
|
383
|
+
listType,
|
|
384
|
+
indentation: ilvl,
|
|
385
|
+
listId: currentListId.toString(),
|
|
386
|
+
itemIndex: runningListIndex,
|
|
387
|
+
alignment: pNode.metadata?.alignment || 'left',
|
|
388
|
+
};
|
|
389
|
+
}
|
|
390
|
+
else {
|
|
391
|
+
lastWasList = false;
|
|
392
|
+
lastListType = null;
|
|
393
|
+
lastListIndent = 0;
|
|
394
|
+
}
|
|
395
|
+
if (isTitle) {
|
|
396
|
+
pNode.metadata = { ...pNode.metadata, level: 1 };
|
|
397
|
+
}
|
|
398
|
+
if (config.includeRawContent) {
|
|
399
|
+
pNode.rawContent = p.toString();
|
|
400
|
+
}
|
|
401
|
+
const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
|
|
402
|
+
for (let j = 0; j < runs.length; j++) {
|
|
403
|
+
const r = runs[j];
|
|
404
|
+
const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
|
|
405
|
+
if (t && t.childNodes[0]) {
|
|
406
|
+
const textContent = t.childNodes[0].nodeValue || '';
|
|
407
|
+
pNode.text += textContent;
|
|
408
|
+
const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
|
|
409
|
+
const formatting = {};
|
|
410
|
+
if (rPr) {
|
|
411
|
+
if (rPr.getAttribute("b") === "1")
|
|
412
|
+
formatting.bold = true;
|
|
413
|
+
if (rPr.getAttribute("i") === "1")
|
|
414
|
+
formatting.italic = true;
|
|
415
|
+
if (rPr.getAttribute("u") === "sng")
|
|
416
|
+
formatting.underline = true;
|
|
417
|
+
if (rPr.getAttribute("strike") === "sngStrike")
|
|
418
|
+
formatting.strikethrough = true;
|
|
419
|
+
const sz = rPr.getAttribute("sz");
|
|
420
|
+
if (sz)
|
|
421
|
+
formatting.size = (parseInt(sz) / 100).toString() + 'pt';
|
|
422
|
+
// Color extraction
|
|
423
|
+
const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
|
|
424
|
+
if (solidFill) {
|
|
425
|
+
const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
|
|
426
|
+
if (srgbClr) {
|
|
427
|
+
const val = srgbClr.getAttribute("val");
|
|
428
|
+
if (val)
|
|
429
|
+
formatting.color = '#' + val;
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
// Highlight extraction
|
|
433
|
+
const highlight = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:highlight")[0];
|
|
434
|
+
if (highlight) {
|
|
435
|
+
const srgbClr = (0, xmlUtils_1.getElementsByTagName)(highlight, "a:srgbClr")[0];
|
|
436
|
+
if (srgbClr) {
|
|
437
|
+
const val = srgbClr.getAttribute("val");
|
|
438
|
+
if (val)
|
|
439
|
+
formatting.backgroundColor = '#' + val;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
// Font family
|
|
443
|
+
const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
|
|
444
|
+
if (latin) {
|
|
445
|
+
const typeface = latin.getAttribute("typeface");
|
|
446
|
+
if (typeface)
|
|
447
|
+
formatting.font = typeface;
|
|
448
|
+
}
|
|
449
|
+
// Subscript/Superscript
|
|
450
|
+
const baseline = rPr.getAttribute("baseline");
|
|
451
|
+
if (baseline) {
|
|
452
|
+
const baselineVal = parseInt(baseline);
|
|
453
|
+
if (baselineVal < 0)
|
|
454
|
+
formatting.subscript = true;
|
|
455
|
+
if (baselineVal > 0)
|
|
456
|
+
formatting.superscript = true;
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
const textNode = {
|
|
460
|
+
type: 'text',
|
|
461
|
+
text: textContent,
|
|
462
|
+
formatting: formatting
|
|
463
|
+
};
|
|
464
|
+
// Check for Hyperlinks
|
|
465
|
+
const hlinkClick = (0, xmlUtils_1.getElementsByTagName)(r, "a:hlinkClick")[0];
|
|
466
|
+
// Check if this run has a hyperlink click action
|
|
467
|
+
if (hlinkClick) {
|
|
468
|
+
// Relationship ID for the link
|
|
469
|
+
const rId = hlinkClick.getAttribute("r:id");
|
|
470
|
+
// Optional action attribute, often for internal jumps
|
|
471
|
+
const action = hlinkClick.getAttribute("action");
|
|
472
|
+
// Result placeholders
|
|
473
|
+
let link;
|
|
474
|
+
let linkType;
|
|
475
|
+
// Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
|
|
476
|
+
if (rId
|
|
477
|
+
&& slideRelsMap[slideNumber]
|
|
478
|
+
&& slideRelsMap[slideNumber][rId]
|
|
479
|
+
&& slideRelsMap[slideNumber][rId].type === "hyperlink") {
|
|
480
|
+
// External URL
|
|
481
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
482
|
+
linkType = "external";
|
|
483
|
+
}
|
|
484
|
+
// Case 2: Relationship exists and is an internal slide reference
|
|
485
|
+
else if (rId
|
|
486
|
+
&& slideRelsMap[slideNumber]
|
|
487
|
+
&& slideRelsMap[slideNumber][rId]
|
|
488
|
+
&& slideRelsMap[slideNumber][rId].type === "slide") {
|
|
489
|
+
// Example target: ppt/slides/slide3.xml
|
|
490
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
491
|
+
linkType = "internal";
|
|
492
|
+
}
|
|
493
|
+
// Case 3: action attribute like ppaction://hlinksldjump
|
|
494
|
+
else if (action) {
|
|
495
|
+
link = action;
|
|
496
|
+
linkType = "internal";
|
|
497
|
+
}
|
|
498
|
+
// Assign metadata only if a link was actually discovered
|
|
499
|
+
if (link) {
|
|
500
|
+
textNode.metadata = { link, linkType };
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
pNode.children?.push(textNode);
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
if (pNode.text) {
|
|
507
|
+
nodes.push(pNode);
|
|
508
|
+
}
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
return nodes;
|
|
512
|
+
};
|
|
513
|
+
/**
|
|
514
|
+
* Recursively traverses a PowerPoint shape tree (p:spTree),
|
|
515
|
+
* including grouped shapes (p:grpSp), and dispatches each element
|
|
516
|
+
* to the appropriate handler (shape, image, chart, etc.).
|
|
517
|
+
*
|
|
518
|
+
* This function preserves the visual order because it processes
|
|
519
|
+
* children in the order they appear in the XML. It also accumulates
|
|
520
|
+
* transforms so nested groups inherit positional transforms.
|
|
521
|
+
*
|
|
522
|
+
* @param treeNode The XML node representing <p:spTree>
|
|
523
|
+
* @param parentTransform Transform inherited from parent groups (defaults to identity)
|
|
524
|
+
*/
|
|
525
|
+
function traverseSpTree(treeNode, slideNumber) {
|
|
526
|
+
const nodes = [];
|
|
527
|
+
// Process children in XML order (this preserves Z-order)
|
|
528
|
+
for (const child of Array.from(treeNode?.childNodes || [])) {
|
|
529
|
+
if (child.nodeType !== 1) {
|
|
530
|
+
continue;
|
|
531
|
+
}
|
|
532
|
+
const element = child;
|
|
533
|
+
const tag = element.tagName;
|
|
534
|
+
// Case 1: Normal shape
|
|
535
|
+
if (tag === "p:sp") {
|
|
536
|
+
nodes.push(...extractShapeNodes(element, slideNumber));
|
|
537
|
+
}
|
|
538
|
+
// Case 2: Inline picture
|
|
539
|
+
else if (tag === "p:pic") {
|
|
540
|
+
const imageNode = extractImageNode(element, slideNumber);
|
|
541
|
+
if (imageNode) {
|
|
542
|
+
nodes.push(imageNode);
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
// Case 3: Chart or other graphic frame
|
|
546
|
+
else if (tag === "p:graphicFrame") {
|
|
547
|
+
const tableNode = extractGraphicFrameNode(element, slideNumber);
|
|
548
|
+
if (tableNode) {
|
|
549
|
+
nodes.push(tableNode);
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
// Case 4: Grouped shape (recursive!)
|
|
553
|
+
else if (tag === "p:grpSp") {
|
|
554
|
+
// Extract the nested <p:spTree> inside the group
|
|
555
|
+
const nestedTree = element.getElementsByTagName("p:spTree")[0];
|
|
556
|
+
// Recurse into the nested tree
|
|
557
|
+
nodes.push(...traverseSpTree(nestedTree, slideNumber));
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
return nodes;
|
|
561
|
+
}
|
|
562
|
+
// First pass: Process relationships
|
|
563
|
+
for (const file of files) {
|
|
564
|
+
// Check whether this file is a slideX.xml.rels file
|
|
565
|
+
if (file.path.match(slideRelsRegex)) {
|
|
566
|
+
/**
|
|
567
|
+
* Builds a map of slide number to a map of relationship IDs containing:
|
|
568
|
+
* - type: The relationship category (image, hyperlink, chart, etc)
|
|
569
|
+
* - target: The fully normalized target path inside the PPTX zip
|
|
570
|
+
*
|
|
571
|
+
* Example structure:
|
|
572
|
+
* {
|
|
573
|
+
* 1: {
|
|
574
|
+
* "rId2": { type: "image", target: "ppt/media/image3.png" },
|
|
575
|
+
* "rId5": { type: "hyperlink", target: "https://example.com" }
|
|
576
|
+
* }
|
|
577
|
+
* }
|
|
578
|
+
*
|
|
579
|
+
* @param files All extracted PPTX ZIP files.
|
|
580
|
+
* @param slideRelsMap A map of slide number to relationship info.
|
|
581
|
+
*/
|
|
582
|
+
// Extract slide number from path
|
|
583
|
+
const match = file.path.match(/slide(\d+)\.xml\.rels/);
|
|
584
|
+
if (match) {
|
|
585
|
+
// Convert matched number to integer
|
|
586
|
+
const slideNum = parseInt(match[1]);
|
|
587
|
+
// Prepare map for this slide
|
|
588
|
+
slideRelsMap[slideNum] = {};
|
|
589
|
+
// Parse the rels XML
|
|
590
|
+
const relsXml = (0, xmlUtils_1.parseXmlString)(file.content.toString());
|
|
591
|
+
// Get all Relationship nodes
|
|
592
|
+
const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
|
|
593
|
+
// Loop through each relationship node
|
|
594
|
+
for (let i = 0; i < relationships.length; i++) {
|
|
595
|
+
// Relationship ID, Example: "rId2"
|
|
596
|
+
const id = relationships[i].getAttribute("Id");
|
|
597
|
+
// Relationship Type, Example: "http://schemas.openxmlformats.org/officeDocument/2006/relationships/image"
|
|
598
|
+
const typeAttr = relationships[i].getAttribute("Type");
|
|
599
|
+
// Raw Target, may be relative or absolute
|
|
600
|
+
const targetRaw = relationships[i].getAttribute("Target");
|
|
601
|
+
// Only proceed if ID and Type exist
|
|
602
|
+
if (id && typeAttr && targetRaw) {
|
|
603
|
+
// Simplify Type to a short keyword (image, hyperlink, chart, etc)
|
|
604
|
+
// This is optional but very useful.
|
|
605
|
+
let simplifiedType = "other";
|
|
606
|
+
// Check image
|
|
607
|
+
if (typeAttr.includes("relationships/image")) {
|
|
608
|
+
simplifiedType = "image";
|
|
609
|
+
}
|
|
610
|
+
// Check hyperlink
|
|
611
|
+
else if (typeAttr.includes("relationships/hyperlink")) {
|
|
612
|
+
simplifiedType = "hyperlink";
|
|
613
|
+
}
|
|
614
|
+
// Check chart
|
|
615
|
+
else if (typeAttr.includes("relationships/chart")) {
|
|
616
|
+
simplifiedType = "chart";
|
|
617
|
+
}
|
|
618
|
+
// Check slide references
|
|
619
|
+
else if (typeAttr.includes("relationships/slide")) {
|
|
620
|
+
simplifiedType = "slide";
|
|
621
|
+
}
|
|
622
|
+
// Check notes
|
|
623
|
+
else if (typeAttr.includes("relationships/notesSlide")) {
|
|
624
|
+
simplifiedType = "notes";
|
|
625
|
+
}
|
|
626
|
+
// Now normalize the target only if it is a local file path.
|
|
627
|
+
// Hyperlinks are external and should not be normalized.
|
|
628
|
+
let normalizedTarget = targetRaw;
|
|
629
|
+
// Local paths never contain "http" or "https"
|
|
630
|
+
const isExternal = targetRaw.startsWith("http://") || targetRaw.startsWith("https://");
|
|
631
|
+
// If not external, normalize the target which is just the name of the item.
|
|
632
|
+
if (!isExternal) {
|
|
633
|
+
normalizedTarget = normalizedTarget.split('/').pop() || '';
|
|
634
|
+
}
|
|
635
|
+
// Finally store full relationship info
|
|
636
|
+
slideRelsMap[slideNum][id] = {
|
|
637
|
+
type: simplifiedType,
|
|
638
|
+
target: normalizedTarget
|
|
639
|
+
};
|
|
640
|
+
}
|
|
641
|
+
}
|
|
642
|
+
}
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
// Now for processing all the other files - slides and notes.
|
|
646
|
+
for (const file of files) {
|
|
647
|
+
if (file.path.match(mediaFileRegex))
|
|
648
|
+
continue;
|
|
649
|
+
if (file.path.match(chartFileRegex))
|
|
650
|
+
continue;
|
|
651
|
+
if (file.path.match(slideRelsRegex))
|
|
652
|
+
continue;
|
|
653
|
+
if (file.path.match(corePropsFileRegex))
|
|
654
|
+
continue;
|
|
655
|
+
const xmlContentString = file.content.toString();
|
|
656
|
+
const xml = (0, xmlUtils_1.parseXmlString)(xmlContentString);
|
|
657
|
+
if (config.includeRawContent) {
|
|
658
|
+
rawContents.push(xmlContentString);
|
|
659
|
+
}
|
|
660
|
+
const slideMatch = file.path.match(slideNumberRegex);
|
|
661
|
+
const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
|
|
662
|
+
const isNote = file.path.includes("notesSlide");
|
|
663
|
+
const slideNode = {
|
|
664
|
+
type: isNote ? 'note' : 'slide',
|
|
665
|
+
children: [],
|
|
666
|
+
metadata: {
|
|
667
|
+
slideNumber: slideNumber,
|
|
668
|
+
...(isNote ? { noteId: `slide-note-${slideNumber}` } : {})
|
|
669
|
+
}
|
|
670
|
+
};
|
|
671
|
+
if (config.includeRawContent) {
|
|
672
|
+
slideNode.rawContent = file.content.toString();
|
|
673
|
+
}
|
|
674
|
+
/**
|
|
675
|
+
* Extract slide contents in correct document order by scanning p:spTree children.
|
|
676
|
+
* This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
|
|
677
|
+
*/
|
|
678
|
+
const spTree = (0, xmlUtils_1.getElementsByTagName)(xml, "p:spTree")[0];
|
|
679
|
+
if (spTree) {
|
|
680
|
+
slideNode.children?.push(...traverseSpTree(spTree, slideNumber));
|
|
681
|
+
}
|
|
682
|
+
if (slideNode.children && slideNode.children.length > 0) {
|
|
683
|
+
content.push(slideNode);
|
|
684
|
+
}
|
|
685
|
+
}
|
|
686
|
+
const attachments = [];
|
|
687
|
+
const mediaFiles = files.filter(f => f.path.match(/ppt\/media\/.*/));
|
|
688
|
+
const chartFiles = files.filter(f => f.path.match(/ppt\/charts\/chart\d+\.xml/));
|
|
689
|
+
// First run to extract attachments and to assign ocr to image files.
|
|
690
|
+
if (config.extractAttachments) {
|
|
691
|
+
// Extract media files as attachments
|
|
692
|
+
for (const media of mediaFiles) {
|
|
693
|
+
const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
694
|
+
attachments.push(attachment);
|
|
695
|
+
if (config.ocr) {
|
|
696
|
+
if (attachment.mimeType.startsWith('image/')) {
|
|
697
|
+
try {
|
|
698
|
+
attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
|
|
699
|
+
}
|
|
700
|
+
catch (e) {
|
|
701
|
+
(0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
702
|
+
}
|
|
703
|
+
}
|
|
704
|
+
}
|
|
705
|
+
}
|
|
706
|
+
// Extract chart files as attachments
|
|
707
|
+
for (const chart of chartFiles) {
|
|
708
|
+
const attachment = {
|
|
709
|
+
type: 'chart',
|
|
710
|
+
mimeType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
|
|
711
|
+
data: chart.content.toString('base64'),
|
|
712
|
+
name: chart.path.split('/').pop() || '',
|
|
713
|
+
extension: 'xml'
|
|
714
|
+
};
|
|
715
|
+
attachments.push(attachment);
|
|
716
|
+
// Extract text from chart XML
|
|
717
|
+
try {
|
|
718
|
+
const chartData = await (0, chartUtils_1.extractChartData)(chart.content);
|
|
719
|
+
// Assign chartData to attachment
|
|
720
|
+
attachment.chartData = chartData;
|
|
721
|
+
}
|
|
722
|
+
catch (e) {
|
|
723
|
+
(0, errorUtils_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
// Loop through nodes to find images and charts and link their text and chartData
|
|
727
|
+
const assignAttachmentData = (nodes) => {
|
|
728
|
+
for (const node of nodes) {
|
|
729
|
+
if ('attachmentName' in (node.metadata || {})) {
|
|
730
|
+
const meta = node.metadata;
|
|
731
|
+
const attachment = attachments.find(a => a.name === meta.attachmentName);
|
|
732
|
+
if (attachment) {
|
|
733
|
+
if (node.type === 'image') {
|
|
734
|
+
attachment.altText = meta.altText;
|
|
735
|
+
if (attachment.ocrText)
|
|
736
|
+
node.text = attachment.ocrText;
|
|
737
|
+
}
|
|
738
|
+
if (node.type === 'chart') {
|
|
739
|
+
node.text = attachment.chartData?.rawTexts.join(config.newlineDelimiter || '\n');
|
|
740
|
+
}
|
|
741
|
+
}
|
|
742
|
+
}
|
|
743
|
+
if (node.children) {
|
|
744
|
+
assignAttachmentData(node.children);
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
};
|
|
748
|
+
assignAttachmentData(content);
|
|
749
|
+
}
|
|
750
|
+
// Finally, if the notes are required to be at the end of the document, move them there.
|
|
751
|
+
if (!config.ignoreNotes && config.putNotesAtLast) {
|
|
752
|
+
content.sort((a, b) => {
|
|
753
|
+
const aIsNote = a.type === 'note' ? 1 : 0;
|
|
754
|
+
const bIsNote = b.type === 'note' ? 1 : 0;
|
|
755
|
+
return aIsNote - bIsNote;
|
|
756
|
+
});
|
|
757
|
+
}
|
|
758
|
+
return {
|
|
759
|
+
type: 'pptx',
|
|
760
|
+
metadata: metadata,
|
|
761
|
+
content: content,
|
|
762
|
+
attachments: attachments,
|
|
763
|
+
toText: () => content.map(c => {
|
|
764
|
+
// Recursive text extraction
|
|
765
|
+
const getText = (node) => {
|
|
766
|
+
let t = '';
|
|
767
|
+
if (node.children) {
|
|
768
|
+
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
|
|
769
|
+
}
|
|
770
|
+
else
|
|
771
|
+
t += node.text || '';
|
|
772
|
+
return t;
|
|
773
|
+
};
|
|
774
|
+
return getText(c);
|
|
775
|
+
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
776
|
+
};
|
|
777
|
+
};
|
|
778
|
+
exports.parsePowerPoint = parsePowerPoint;
|