officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
|
@@ -24,13 +24,12 @@
|
|
|
24
24
|
*/
|
|
25
25
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
26
|
exports.parsePowerPoint = void 0;
|
|
27
|
-
const
|
|
28
|
-
const
|
|
29
|
-
const
|
|
30
|
-
const
|
|
31
|
-
const
|
|
32
|
-
const
|
|
33
|
-
const zipUtils_1 = require("../utils/zipUtils");
|
|
27
|
+
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
28
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
29
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
30
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
31
|
+
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
32
|
+
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
34
33
|
/**
|
|
35
34
|
* Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
|
|
36
35
|
*
|
|
@@ -46,14 +45,21 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
46
45
|
const mediaFileRegex = /ppt\/media\/.*/;
|
|
47
46
|
const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
|
|
48
47
|
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
49
|
-
const
|
|
50
|
-
const files = await (0,
|
|
48
|
+
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
49
|
+
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
|
|
51
50
|
!!x.match(corePropsFileRegex) ||
|
|
51
|
+
!!x.match(customPropsFileRegex) ||
|
|
52
52
|
!!x.match(slideRelsRegex) ||
|
|
53
53
|
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
|
|
54
54
|
// Extract metadata
|
|
55
55
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
56
|
-
const metadata = corePropsFile ? (0,
|
|
56
|
+
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
57
|
+
const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
|
|
58
|
+
if (customPropsFile) {
|
|
59
|
+
const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
|
|
60
|
+
if (Object.keys(customProperties).length > 0)
|
|
61
|
+
metadata.customProperties = customProperties;
|
|
62
|
+
}
|
|
57
63
|
// Sort files
|
|
58
64
|
files.sort((a, b) => {
|
|
59
65
|
const aMatch = a.path.match(slideNumberRegex);
|
|
@@ -73,21 +79,21 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
73
79
|
// per indent counters (for nested lists)
|
|
74
80
|
const levelCounters = {};
|
|
75
81
|
// Helper to parse a table node
|
|
76
|
-
const parseTable = (tblNode) => {
|
|
82
|
+
const parseTable = (tblNode, xmlContentString) => {
|
|
77
83
|
const rows = [];
|
|
78
|
-
const trNodes = (0,
|
|
84
|
+
const trNodes = (0, xmlUtils_js_1.getElementsByTagName)(tblNode, "a:tr");
|
|
79
85
|
for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
|
|
80
86
|
const trNode = trNodes[rIndex];
|
|
81
87
|
const cells = [];
|
|
82
|
-
const tcNodes = (0,
|
|
88
|
+
const tcNodes = (0, xmlUtils_js_1.getElementsByTagName)(trNode, "a:tc");
|
|
83
89
|
for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
|
|
84
90
|
const tcNode = tcNodes[cIndex];
|
|
85
91
|
const cellChildren = [];
|
|
86
92
|
let cellText = '';
|
|
87
93
|
// Cells contain text bodies (txBody) which contain paragraphs
|
|
88
|
-
const txBody = (0,
|
|
94
|
+
const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "a:txBody");
|
|
89
95
|
if (txBody) {
|
|
90
|
-
const paragraphs = (0,
|
|
96
|
+
const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
|
|
91
97
|
for (const p of paragraphs) {
|
|
92
98
|
// Reuse paragraph parsing logic if possible, or duplicate for now
|
|
93
99
|
// For simplicity, duplicating basic logic here as the main loop one is tied to shapes
|
|
@@ -98,15 +104,15 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
98
104
|
metadata: {}
|
|
99
105
|
};
|
|
100
106
|
if (config.includeRawContent) {
|
|
101
|
-
pNode.rawContent =
|
|
107
|
+
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
102
108
|
}
|
|
103
|
-
const runs = (0,
|
|
109
|
+
const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
|
|
104
110
|
for (const r of runs) {
|
|
105
|
-
const t = (0,
|
|
111
|
+
const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
|
|
106
112
|
if (t && t.childNodes[0]) {
|
|
107
113
|
const textContent = t.childNodes[0].nodeValue || '';
|
|
108
114
|
pNode.text += textContent;
|
|
109
|
-
const rPr = (0,
|
|
115
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
|
|
110
116
|
const formatting = {};
|
|
111
117
|
if (rPr) {
|
|
112
118
|
if (rPr.getAttribute("b") === "1")
|
|
@@ -120,16 +126,16 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
120
126
|
const sz = rPr.getAttribute("sz");
|
|
121
127
|
if (sz)
|
|
122
128
|
formatting.size = (parseInt(sz) / 100).toString() + 'pt';
|
|
123
|
-
const solidFill = (0,
|
|
129
|
+
const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
|
|
124
130
|
if (solidFill) {
|
|
125
|
-
const srgbClr = (0,
|
|
131
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
|
|
126
132
|
if (srgbClr) {
|
|
127
133
|
const val = srgbClr.getAttribute("val");
|
|
128
134
|
if (val)
|
|
129
135
|
formatting.color = '#' + val;
|
|
130
136
|
}
|
|
131
137
|
}
|
|
132
|
-
const latin = (0,
|
|
138
|
+
const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
|
|
133
139
|
if (latin) {
|
|
134
140
|
const typeface = latin.getAttribute("typeface");
|
|
135
141
|
if (typeface)
|
|
@@ -167,8 +173,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
167
173
|
};
|
|
168
174
|
};
|
|
169
175
|
/** Extract an AST node for p:pic */
|
|
170
|
-
const extractImageNode = (imageNode, slideNumber) => {
|
|
171
|
-
const blip = (0,
|
|
176
|
+
const extractImageNode = (imageNode, slideNumber, xmlContentString) => {
|
|
177
|
+
const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "a:blip");
|
|
172
178
|
if (!blip)
|
|
173
179
|
return null;
|
|
174
180
|
const rId = blip.getAttribute("r:embed");
|
|
@@ -178,8 +184,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
178
184
|
if (!rel || rel.type !== "image")
|
|
179
185
|
return null;
|
|
180
186
|
const attachmentName = rel.target;
|
|
181
|
-
const nvPicPr = (0,
|
|
182
|
-
const cNvPr = nvPicPr ? (0,
|
|
187
|
+
const nvPicPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "p:nvPicPr");
|
|
188
|
+
const cNvPr = nvPicPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPicPr, "p:cNvPr") : null;
|
|
183
189
|
const altText = cNvPr?.getAttribute("descr") || undefined;
|
|
184
190
|
return {
|
|
185
191
|
type: "image",
|
|
@@ -192,23 +198,11 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
192
198
|
};
|
|
193
199
|
/**
|
|
194
200
|
* Extract an AST node for p:graphicFrame that contains a chart.
|
|
195
|
-
*
|
|
196
|
-
*
|
|
197
|
-
* Steps:
|
|
198
|
-
* 1. Find <a:graphicData> inside the graphicFrame.
|
|
199
|
-
* 2. Check if uri is the chart namespace.
|
|
200
|
-
* 3. Extract <c:chart> and read r:id.
|
|
201
|
-
* 4. Resolve the relationship using slideRelsMap.
|
|
202
|
-
* 5. Produce a chart node with attachmentName (like images).
|
|
203
|
-
* 6. Chart text/data will be injected later in the pipeline.
|
|
204
|
-
*
|
|
205
|
-
* @param frameNode p:graphicFrame element
|
|
206
|
-
* @param slideNumber Current slide number for relationship resolution
|
|
207
|
-
* @returns A chart OfficeContentNode or null if not a chart.
|
|
201
|
+
* ... (comments omitted for brevity) ...
|
|
208
202
|
*/
|
|
209
|
-
const extractChartNode = (frameNode, slideNumber) => {
|
|
203
|
+
const extractChartNode = (frameNode, slideNumber, xmlContentString) => {
|
|
210
204
|
// Step 1: Find <a:graphicData>
|
|
211
|
-
const graphicData = (0,
|
|
205
|
+
const graphicData = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:graphicData");
|
|
212
206
|
if (!graphicData) {
|
|
213
207
|
return null;
|
|
214
208
|
}
|
|
@@ -220,7 +214,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
220
214
|
return null;
|
|
221
215
|
}
|
|
222
216
|
// Step 3: Find <c:chart>
|
|
223
|
-
const cChart = (0,
|
|
217
|
+
const cChart = (0, xmlUtils_js_1.getFirstElementByTagName)(graphicData, "c:chart");
|
|
224
218
|
if (!cChart) {
|
|
225
219
|
return null;
|
|
226
220
|
}
|
|
@@ -246,24 +240,24 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
246
240
|
};
|
|
247
241
|
// Optional: include raw XML of the whole frame
|
|
248
242
|
if (config.includeRawContent) {
|
|
249
|
-
chartNode.rawContent =
|
|
243
|
+
chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
|
|
250
244
|
}
|
|
251
245
|
return chartNode;
|
|
252
246
|
};
|
|
253
247
|
/** Extract an AST node for p:graphicFrame */
|
|
254
|
-
const extractGraphicFrameNode = (frameNode, slideNumber) => {
|
|
255
|
-
const tbl = (0,
|
|
248
|
+
const extractGraphicFrameNode = (frameNode, slideNumber, xmlContentString) => {
|
|
249
|
+
const tbl = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:tbl");
|
|
256
250
|
if (tbl) {
|
|
257
|
-
const tableNode = parseTable(tbl);
|
|
251
|
+
const tableNode = parseTable(tbl, xmlContentString);
|
|
258
252
|
if (config.includeRawContent) {
|
|
259
|
-
tableNode.rawContent =
|
|
253
|
+
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
|
|
260
254
|
}
|
|
261
255
|
if (tableNode.children && tableNode.children.length > 0) {
|
|
262
256
|
return tableNode;
|
|
263
257
|
}
|
|
264
258
|
}
|
|
265
259
|
if (frameNode.getElementsByTagName("c:chart").length > 0) {
|
|
266
|
-
const chartNode = extractChartNode(frameNode, slideNumber);
|
|
260
|
+
const chartNode = extractChartNode(frameNode, slideNumber, xmlContentString);
|
|
267
261
|
if (chartNode) {
|
|
268
262
|
return chartNode;
|
|
269
263
|
}
|
|
@@ -271,17 +265,17 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
271
265
|
return null;
|
|
272
266
|
};
|
|
273
267
|
/** Extract text and hyperlinks from a p:sp shape. */
|
|
274
|
-
const extractShapeNodes = (spNode, slideNumber) => {
|
|
268
|
+
const extractShapeNodes = (spNode, slideNumber, xmlContentString) => {
|
|
275
269
|
const nodes = [];
|
|
276
270
|
// Check for placeholder type (title, body, etc.)
|
|
277
|
-
const nvSpPr = (0,
|
|
278
|
-
const nvPr = nvSpPr ? (0,
|
|
279
|
-
const ph = nvPr ? (0,
|
|
271
|
+
const nvSpPr = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:nvSpPr");
|
|
272
|
+
const nvPr = nvSpPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvSpPr, "p:nvPr") : null;
|
|
273
|
+
const ph = nvPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPr, "p:ph") : null;
|
|
280
274
|
const type = ph ? ph.getAttribute("type") : "body";
|
|
281
275
|
const isTitle = type === "title" || type === "ctrTitle";
|
|
282
|
-
const txBody = (0,
|
|
276
|
+
const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:txBody");
|
|
283
277
|
if (txBody) {
|
|
284
|
-
const paragraphs = (0,
|
|
278
|
+
const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
|
|
285
279
|
for (let i = 0; i < paragraphs.length; i++) {
|
|
286
280
|
const p = paragraphs[i];
|
|
287
281
|
const pNode = {
|
|
@@ -291,7 +285,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
291
285
|
metadata: isTitle ? { level: 1 } : {}
|
|
292
286
|
};
|
|
293
287
|
// Paragraph Alignment and List Detection
|
|
294
|
-
const pPr = (0,
|
|
288
|
+
const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(p, "a:pPr");
|
|
295
289
|
let isList = false;
|
|
296
290
|
let listType = 'unordered';
|
|
297
291
|
let lvl = 0;
|
|
@@ -299,10 +293,10 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
299
293
|
const lvlAttr = pPr.getAttribute("lvl");
|
|
300
294
|
if (lvlAttr)
|
|
301
295
|
lvl = parseInt(lvlAttr);
|
|
302
|
-
const buAutoNum = (0,
|
|
303
|
-
const buChar = (0,
|
|
304
|
-
const buBlip = (0,
|
|
305
|
-
const buNode = (0,
|
|
296
|
+
const buAutoNum = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buAutoNum");
|
|
297
|
+
const buChar = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buChar");
|
|
298
|
+
const buBlip = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buBlip");
|
|
299
|
+
const buNode = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:bu");
|
|
306
300
|
if (buAutoNum) {
|
|
307
301
|
isList = true;
|
|
308
302
|
listType = 'ordered';
|
|
@@ -396,119 +390,131 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
396
390
|
pNode.metadata = { ...pNode.metadata, level: 1 };
|
|
397
391
|
}
|
|
398
392
|
if (config.includeRawContent) {
|
|
399
|
-
pNode.rawContent =
|
|
393
|
+
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
400
394
|
}
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
if (rPr
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
395
|
+
// Process all children of <a:p> in order (runs, breaks, fields)
|
|
396
|
+
const children = Array.from(p.childNodes);
|
|
397
|
+
let activeNode = pNode;
|
|
398
|
+
nodes.push(activeNode);
|
|
399
|
+
for (const childNode of children) {
|
|
400
|
+
if (!(0, xmlUtils_js_1.isElement)(childNode))
|
|
401
|
+
continue;
|
|
402
|
+
const element = childNode;
|
|
403
|
+
const tag = element.tagName;
|
|
404
|
+
if (tag === "a:r" || tag === "a:fld") {
|
|
405
|
+
const t = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:t");
|
|
406
|
+
if (t && t.childNodes[0]) {
|
|
407
|
+
const textContent = t.childNodes[0].nodeValue || "";
|
|
408
|
+
activeNode.text += textContent;
|
|
409
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:rPr");
|
|
410
|
+
const formatting = {};
|
|
411
|
+
if (rPr) {
|
|
412
|
+
if (rPr.getAttribute("b") === "1")
|
|
413
|
+
formatting.bold = true;
|
|
414
|
+
if (rPr.getAttribute("i") === "1")
|
|
415
|
+
formatting.italic = true;
|
|
416
|
+
if (rPr.getAttribute("u") === "sng")
|
|
417
|
+
formatting.underline = true;
|
|
418
|
+
if (rPr.getAttribute("strike") === "sngStrike")
|
|
419
|
+
formatting.strikethrough = true;
|
|
420
|
+
const sz = rPr.getAttribute("sz");
|
|
421
|
+
if (sz)
|
|
422
|
+
formatting.size = (parseInt(sz) / 100).toString() + "pt";
|
|
423
|
+
// Color extraction
|
|
424
|
+
const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
|
|
425
|
+
if (solidFill) {
|
|
426
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
|
|
427
|
+
if (srgbClr) {
|
|
428
|
+
const val = srgbClr.getAttribute("val");
|
|
429
|
+
if (val)
|
|
430
|
+
formatting.color = "#" + val;
|
|
431
|
+
}
|
|
430
432
|
}
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
433
|
+
// Highlight extraction
|
|
434
|
+
const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
|
|
435
|
+
if (highlight) {
|
|
436
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
|
|
437
|
+
if (srgbClr) {
|
|
438
|
+
const val = srgbClr.getAttribute("val");
|
|
439
|
+
if (val)
|
|
440
|
+
formatting.backgroundColor = "#" + val;
|
|
441
|
+
}
|
|
442
|
+
}
|
|
443
|
+
// Font family
|
|
444
|
+
const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
|
|
445
|
+
if (latin) {
|
|
446
|
+
const typeface = latin.getAttribute("typeface");
|
|
447
|
+
if (typeface)
|
|
448
|
+
formatting.font = typeface;
|
|
449
|
+
}
|
|
450
|
+
// Subscript/Superscript
|
|
451
|
+
const baseline = rPr.getAttribute("baseline");
|
|
452
|
+
if (baseline) {
|
|
453
|
+
const baselineVal = parseInt(baseline);
|
|
454
|
+
if (baselineVal < 0)
|
|
455
|
+
formatting.subscript = true;
|
|
456
|
+
if (baselineVal > 0)
|
|
457
|
+
formatting.superscript = true;
|
|
440
458
|
}
|
|
441
459
|
}
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
460
|
+
const textNode = {
|
|
461
|
+
type: 'text',
|
|
462
|
+
text: textContent,
|
|
463
|
+
formatting: formatting
|
|
464
|
+
};
|
|
465
|
+
// Check for Hyperlinks
|
|
466
|
+
const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:hlinkClick");
|
|
467
|
+
if (hlinkClick) {
|
|
468
|
+
const rId = hlinkClick.getAttribute("r:id");
|
|
469
|
+
const action = hlinkClick.getAttribute("action");
|
|
470
|
+
let link;
|
|
471
|
+
let linkType;
|
|
472
|
+
if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "hyperlink") {
|
|
473
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
474
|
+
linkType = "external";
|
|
475
|
+
}
|
|
476
|
+
else if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "slide") {
|
|
477
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
478
|
+
linkType = "internal";
|
|
479
|
+
}
|
|
480
|
+
else if (action) {
|
|
481
|
+
link = action;
|
|
482
|
+
linkType = "internal";
|
|
483
|
+
}
|
|
484
|
+
if (link) {
|
|
485
|
+
textNode.metadata = { link, linkType };
|
|
486
|
+
}
|
|
457
487
|
}
|
|
488
|
+
activeNode.children?.push(textNode);
|
|
458
489
|
}
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
let linkType;
|
|
475
|
-
// Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
|
|
476
|
-
if (rId
|
|
477
|
-
&& slideRelsMap[slideNumber]
|
|
478
|
-
&& slideRelsMap[slideNumber][rId]
|
|
479
|
-
&& slideRelsMap[slideNumber][rId].type === "hyperlink") {
|
|
480
|
-
// External URL
|
|
481
|
-
link = slideRelsMap[slideNumber][rId].target;
|
|
482
|
-
linkType = "external";
|
|
483
|
-
}
|
|
484
|
-
// Case 2: Relationship exists and is an internal slide reference
|
|
485
|
-
else if (rId
|
|
486
|
-
&& slideRelsMap[slideNumber]
|
|
487
|
-
&& slideRelsMap[slideNumber][rId]
|
|
488
|
-
&& slideRelsMap[slideNumber][rId].type === "slide") {
|
|
489
|
-
// Example target: ppt/slides/slide3.xml
|
|
490
|
-
link = slideRelsMap[slideNumber][rId].target;
|
|
491
|
-
linkType = "internal";
|
|
492
|
-
}
|
|
493
|
-
// Case 3: action attribute like ppaction://hlinksldjump
|
|
494
|
-
else if (action) {
|
|
495
|
-
link = action;
|
|
496
|
-
linkType = "internal";
|
|
497
|
-
}
|
|
498
|
-
// Assign metadata only if a link was actually discovered
|
|
499
|
-
if (link) {
|
|
500
|
-
textNode.metadata = { link, linkType };
|
|
490
|
+
}
|
|
491
|
+
else if (tag === "a:br") {
|
|
492
|
+
if (isList) {
|
|
493
|
+
// Split the list item on soft break into a paragraph node
|
|
494
|
+
activeNode = {
|
|
495
|
+
type: 'paragraph',
|
|
496
|
+
text: '',
|
|
497
|
+
children: [],
|
|
498
|
+
metadata: {
|
|
499
|
+
indentation: lvl,
|
|
500
|
+
alignment: pNode.metadata?.alignment || 'left'
|
|
501
|
+
}
|
|
502
|
+
};
|
|
503
|
+
if (config.includeRawContent) {
|
|
504
|
+
activeNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
501
505
|
}
|
|
506
|
+
nodes.push(activeNode);
|
|
507
|
+
}
|
|
508
|
+
else {
|
|
509
|
+
// In a normal paragraph, just add a newline
|
|
510
|
+
activeNode.text += "\n";
|
|
511
|
+
activeNode.children?.push({ type: 'text', text: "\n" });
|
|
502
512
|
}
|
|
503
|
-
pNode.children?.push(textNode);
|
|
504
513
|
}
|
|
505
514
|
}
|
|
506
|
-
if (pNode.text) {
|
|
507
|
-
nodes.push(pNode);
|
|
508
|
-
}
|
|
509
515
|
}
|
|
510
516
|
}
|
|
511
|
-
return nodes;
|
|
517
|
+
return nodes.filter(n => n.text?.trim() || (n.children && n.children.length > 0));
|
|
512
518
|
};
|
|
513
519
|
/**
|
|
514
520
|
* Recursively traverses a PowerPoint shape tree (p:spTree),
|
|
@@ -520,31 +526,32 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
520
526
|
* transforms so nested groups inherit positional transforms.
|
|
521
527
|
*
|
|
522
528
|
* @param treeNode The XML node representing <p:spTree>
|
|
523
|
-
* @param
|
|
529
|
+
* @param slideNumber Current slide number for relationship resolution
|
|
530
|
+
* @param xmlContentString The source XML string for raw content extraction
|
|
524
531
|
*/
|
|
525
|
-
function traverseSpTree(treeNode, slideNumber) {
|
|
532
|
+
function traverseSpTree(treeNode, slideNumber, xmlContentString) {
|
|
526
533
|
const nodes = [];
|
|
527
534
|
// Process children in XML order (this preserves Z-order)
|
|
528
535
|
for (const child of Array.from(treeNode?.childNodes || [])) {
|
|
529
|
-
if (
|
|
536
|
+
if (!(0, xmlUtils_js_1.isElement)(child)) {
|
|
530
537
|
continue;
|
|
531
538
|
}
|
|
532
539
|
const element = child;
|
|
533
540
|
const tag = element.tagName;
|
|
534
541
|
// Case 1: Normal shape
|
|
535
542
|
if (tag === "p:sp") {
|
|
536
|
-
nodes.push(...extractShapeNodes(element, slideNumber));
|
|
543
|
+
nodes.push(...extractShapeNodes(element, slideNumber, xmlContentString));
|
|
537
544
|
}
|
|
538
545
|
// Case 2: Inline picture
|
|
539
546
|
else if (tag === "p:pic") {
|
|
540
|
-
const imageNode = extractImageNode(element, slideNumber);
|
|
547
|
+
const imageNode = extractImageNode(element, slideNumber, xmlContentString);
|
|
541
548
|
if (imageNode) {
|
|
542
549
|
nodes.push(imageNode);
|
|
543
550
|
}
|
|
544
551
|
}
|
|
545
552
|
// Case 3: Chart or other graphic frame
|
|
546
553
|
else if (tag === "p:graphicFrame") {
|
|
547
|
-
const tableNode = extractGraphicFrameNode(element, slideNumber);
|
|
554
|
+
const tableNode = extractGraphicFrameNode(element, slideNumber, xmlContentString);
|
|
548
555
|
if (tableNode) {
|
|
549
556
|
nodes.push(tableNode);
|
|
550
557
|
}
|
|
@@ -552,9 +559,11 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
552
559
|
// Case 4: Grouped shape (recursive!)
|
|
553
560
|
else if (tag === "p:grpSp") {
|
|
554
561
|
// Extract the nested <p:spTree> inside the group
|
|
555
|
-
const nestedTree =
|
|
562
|
+
const nestedTree = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "p:spTree");
|
|
556
563
|
// Recurse into the nested tree
|
|
557
|
-
|
|
564
|
+
if (nestedTree) {
|
|
565
|
+
nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
|
|
566
|
+
}
|
|
558
567
|
}
|
|
559
568
|
}
|
|
560
569
|
return nodes;
|
|
@@ -587,9 +596,9 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
587
596
|
// Prepare map for this slide
|
|
588
597
|
slideRelsMap[slideNum] = {};
|
|
589
598
|
// Parse the rels XML
|
|
590
|
-
const relsXml = (0,
|
|
599
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(file.content.toString());
|
|
591
600
|
// Get all Relationship nodes
|
|
592
|
-
const relationships = (0,
|
|
601
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
593
602
|
// Loop through each relationship node
|
|
594
603
|
for (let i = 0; i < relationships.length; i++) {
|
|
595
604
|
// Relationship ID, Example: "rId2"
|
|
@@ -653,7 +662,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
653
662
|
if (file.path.match(corePropsFileRegex))
|
|
654
663
|
continue;
|
|
655
664
|
const xmlContentString = file.content.toString();
|
|
656
|
-
const xml = (0,
|
|
665
|
+
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
|
|
657
666
|
if (config.includeRawContent) {
|
|
658
667
|
rawContents.push(xmlContentString);
|
|
659
668
|
}
|
|
@@ -669,15 +678,15 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
669
678
|
}
|
|
670
679
|
};
|
|
671
680
|
if (config.includeRawContent) {
|
|
672
|
-
slideNode.rawContent =
|
|
681
|
+
slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(xml, xmlContentString, config);
|
|
673
682
|
}
|
|
674
683
|
/**
|
|
675
684
|
* Extract slide contents in correct document order by scanning p:spTree children.
|
|
676
685
|
* This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
|
|
677
686
|
*/
|
|
678
|
-
const spTree = (0,
|
|
687
|
+
const spTree = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "p:spTree");
|
|
679
688
|
if (spTree) {
|
|
680
|
-
slideNode.children?.push(...traverseSpTree(spTree, slideNumber));
|
|
689
|
+
slideNode.children?.push(...traverseSpTree(spTree, slideNumber, xmlContentString));
|
|
681
690
|
}
|
|
682
691
|
if (slideNode.children && slideNode.children.length > 0) {
|
|
683
692
|
content.push(slideNode);
|
|
@@ -690,15 +699,15 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
690
699
|
if (config.extractAttachments) {
|
|
691
700
|
// Extract media files as attachments
|
|
692
701
|
for (const media of mediaFiles) {
|
|
693
|
-
const attachment = (0,
|
|
702
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
694
703
|
attachments.push(attachment);
|
|
695
704
|
if (config.ocr) {
|
|
696
705
|
if (attachment.mimeType.startsWith('image/')) {
|
|
697
706
|
try {
|
|
698
|
-
attachment.ocrText = (await (0,
|
|
707
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
699
708
|
}
|
|
700
709
|
catch (e) {
|
|
701
|
-
(0,
|
|
710
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
702
711
|
}
|
|
703
712
|
}
|
|
704
713
|
}
|
|
@@ -715,12 +724,12 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
715
724
|
attachments.push(attachment);
|
|
716
725
|
// Extract text from chart XML
|
|
717
726
|
try {
|
|
718
|
-
const chartData = await (0,
|
|
727
|
+
const chartData = await (0, chartUtils_js_1.extractChartData)(chart.content);
|
|
719
728
|
// Assign chartData to attachment
|
|
720
729
|
attachment.chartData = chartData;
|
|
721
730
|
}
|
|
722
731
|
catch (e) {
|
|
723
|
-
(0,
|
|
732
|
+
(0, errorUtils_js_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
|
|
724
733
|
}
|
|
725
734
|
}
|
|
726
735
|
// Loop through nodes to find images and charts and link their text and chartData
|