officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
@@ -24,13 +24,12 @@
24
24
  */
25
25
  Object.defineProperty(exports, "__esModule", { value: true });
26
26
  exports.parsePowerPoint = void 0;
27
- const xmldom_1 = require("@xmldom/xmldom");
28
- const chartUtils_1 = require("../utils/chartUtils");
29
- const errorUtils_1 = require("../utils/errorUtils");
30
- const imageUtils_1 = require("../utils/imageUtils");
31
- const ocrUtils_1 = require("../utils/ocrUtils");
32
- const xmlUtils_1 = require("../utils/xmlUtils");
33
- const zipUtils_1 = require("../utils/zipUtils");
27
+ const chartUtils_js_1 = require("../utils/chartUtils.js");
28
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
29
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
30
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
31
+ const xmlUtils_js_1 = require("../utils/xmlUtils.js");
32
+ const zipUtils_js_1 = require("../utils/zipUtils.js");
34
33
  /**
35
34
  * Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
36
35
  *
@@ -46,14 +45,21 @@ const parsePowerPoint = async (buffer, config) => {
46
45
  const mediaFileRegex = /ppt\/media\/.*/;
47
46
  const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
48
47
  const corePropsFileRegex = /docProps\/core\.xml/;
49
- const xmlSerializer = new xmldom_1.XMLSerializer();
50
- const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
48
+ const customPropsFileRegex = /docProps\/custom\.xml/;
49
+ const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
51
50
  !!x.match(corePropsFileRegex) ||
51
+ !!x.match(customPropsFileRegex) ||
52
52
  !!x.match(slideRelsRegex) ||
53
53
  (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
54
54
  // Extract metadata
55
55
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
56
- const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
56
+ const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
57
+ const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
58
+ if (customPropsFile) {
59
+ const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
60
+ if (Object.keys(customProperties).length > 0)
61
+ metadata.customProperties = customProperties;
62
+ }
57
63
  // Sort files
58
64
  files.sort((a, b) => {
59
65
  const aMatch = a.path.match(slideNumberRegex);
@@ -73,21 +79,21 @@ const parsePowerPoint = async (buffer, config) => {
73
79
  // per indent counters (for nested lists)
74
80
  const levelCounters = {};
75
81
  // Helper to parse a table node
76
- const parseTable = (tblNode) => {
82
+ const parseTable = (tblNode, xmlContentString) => {
77
83
  const rows = [];
78
- const trNodes = (0, xmlUtils_1.getElementsByTagName)(tblNode, "a:tr");
84
+ const trNodes = (0, xmlUtils_js_1.getElementsByTagName)(tblNode, "a:tr");
79
85
  for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
80
86
  const trNode = trNodes[rIndex];
81
87
  const cells = [];
82
- const tcNodes = (0, xmlUtils_1.getElementsByTagName)(trNode, "a:tc");
88
+ const tcNodes = (0, xmlUtils_js_1.getElementsByTagName)(trNode, "a:tc");
83
89
  for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
84
90
  const tcNode = tcNodes[cIndex];
85
91
  const cellChildren = [];
86
92
  let cellText = '';
87
93
  // Cells contain text bodies (txBody) which contain paragraphs
88
- const txBody = (0, xmlUtils_1.getElementsByTagName)(tcNode, "a:txBody")[0];
94
+ const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "a:txBody");
89
95
  if (txBody) {
90
- const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
96
+ const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
91
97
  for (const p of paragraphs) {
92
98
  // Reuse paragraph parsing logic if possible, or duplicate for now
93
99
  // For simplicity, duplicating basic logic here as the main loop one is tied to shapes
@@ -98,15 +104,15 @@ const parsePowerPoint = async (buffer, config) => {
98
104
  metadata: {}
99
105
  };
100
106
  if (config.includeRawContent) {
101
- pNode.rawContent = p.toString();
107
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
102
108
  }
103
- const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
109
+ const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
104
110
  for (const r of runs) {
105
- const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
111
+ const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
106
112
  if (t && t.childNodes[0]) {
107
113
  const textContent = t.childNodes[0].nodeValue || '';
108
114
  pNode.text += textContent;
109
- const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
115
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
110
116
  const formatting = {};
111
117
  if (rPr) {
112
118
  if (rPr.getAttribute("b") === "1")
@@ -120,16 +126,16 @@ const parsePowerPoint = async (buffer, config) => {
120
126
  const sz = rPr.getAttribute("sz");
121
127
  if (sz)
122
128
  formatting.size = (parseInt(sz) / 100).toString() + 'pt';
123
- const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
129
+ const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
124
130
  if (solidFill) {
125
- const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
131
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
126
132
  if (srgbClr) {
127
133
  const val = srgbClr.getAttribute("val");
128
134
  if (val)
129
135
  formatting.color = '#' + val;
130
136
  }
131
137
  }
132
- const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
138
+ const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
133
139
  if (latin) {
134
140
  const typeface = latin.getAttribute("typeface");
135
141
  if (typeface)
@@ -167,8 +173,8 @@ const parsePowerPoint = async (buffer, config) => {
167
173
  };
168
174
  };
169
175
  /** Extract an AST node for p:pic */
170
- const extractImageNode = (imageNode, slideNumber) => {
171
- const blip = (0, xmlUtils_1.getElementsByTagName)(imageNode, "a:blip")[0];
176
+ const extractImageNode = (imageNode, slideNumber, xmlContentString) => {
177
+ const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "a:blip");
172
178
  if (!blip)
173
179
  return null;
174
180
  const rId = blip.getAttribute("r:embed");
@@ -178,8 +184,8 @@ const parsePowerPoint = async (buffer, config) => {
178
184
  if (!rel || rel.type !== "image")
179
185
  return null;
180
186
  const attachmentName = rel.target;
181
- const nvPicPr = (0, xmlUtils_1.getElementsByTagName)(imageNode, "p:nvPicPr")[0];
182
- const cNvPr = nvPicPr ? (0, xmlUtils_1.getElementsByTagName)(nvPicPr, "p:cNvPr")[0] : null;
187
+ const nvPicPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "p:nvPicPr");
188
+ const cNvPr = nvPicPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPicPr, "p:cNvPr") : null;
183
189
  const altText = cNvPr?.getAttribute("descr") || undefined;
184
190
  return {
185
191
  type: "image",
@@ -192,23 +198,11 @@ const parsePowerPoint = async (buffer, config) => {
192
198
  };
193
199
  /**
194
200
  * Extract an AST node for p:graphicFrame that contains a chart.
195
- * This mirrors extractImageNode, but for charts instead of images.
196
- *
197
- * Steps:
198
- * 1. Find <a:graphicData> inside the graphicFrame.
199
- * 2. Check if uri is the chart namespace.
200
- * 3. Extract <c:chart> and read r:id.
201
- * 4. Resolve the relationship using slideRelsMap.
202
- * 5. Produce a chart node with attachmentName (like images).
203
- * 6. Chart text/data will be injected later in the pipeline.
204
- *
205
- * @param frameNode p:graphicFrame element
206
- * @param slideNumber Current slide number for relationship resolution
207
- * @returns A chart OfficeContentNode or null if not a chart.
201
+ * ... (comments omitted for brevity) ...
208
202
  */
209
- const extractChartNode = (frameNode, slideNumber) => {
203
+ const extractChartNode = (frameNode, slideNumber, xmlContentString) => {
210
204
  // Step 1: Find <a:graphicData>
211
- const graphicData = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:graphicData")[0];
205
+ const graphicData = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:graphicData");
212
206
  if (!graphicData) {
213
207
  return null;
214
208
  }
@@ -220,7 +214,7 @@ const parsePowerPoint = async (buffer, config) => {
220
214
  return null;
221
215
  }
222
216
  // Step 3: Find <c:chart>
223
- const cChart = (0, xmlUtils_1.getElementsByTagName)(graphicData, "c:chart")[0];
217
+ const cChart = (0, xmlUtils_js_1.getFirstElementByTagName)(graphicData, "c:chart");
224
218
  if (!cChart) {
225
219
  return null;
226
220
  }
@@ -246,24 +240,24 @@ const parsePowerPoint = async (buffer, config) => {
246
240
  };
247
241
  // Optional: include raw XML of the whole frame
248
242
  if (config.includeRawContent) {
249
- chartNode.rawContent = xmlSerializer.serializeToString(frameNode);
243
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
250
244
  }
251
245
  return chartNode;
252
246
  };
253
247
  /** Extract an AST node for p:graphicFrame */
254
- const extractGraphicFrameNode = (frameNode, slideNumber) => {
255
- const tbl = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:tbl")[0];
248
+ const extractGraphicFrameNode = (frameNode, slideNumber, xmlContentString) => {
249
+ const tbl = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:tbl");
256
250
  if (tbl) {
257
- const tableNode = parseTable(tbl);
251
+ const tableNode = parseTable(tbl, xmlContentString);
258
252
  if (config.includeRawContent) {
259
- tableNode.rawContent = xmlSerializer.serializeToString(frameNode);
253
+ tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
260
254
  }
261
255
  if (tableNode.children && tableNode.children.length > 0) {
262
256
  return tableNode;
263
257
  }
264
258
  }
265
259
  if (frameNode.getElementsByTagName("c:chart").length > 0) {
266
- const chartNode = extractChartNode(frameNode, slideNumber);
260
+ const chartNode = extractChartNode(frameNode, slideNumber, xmlContentString);
267
261
  if (chartNode) {
268
262
  return chartNode;
269
263
  }
@@ -271,17 +265,17 @@ const parsePowerPoint = async (buffer, config) => {
271
265
  return null;
272
266
  };
273
267
  /** Extract text and hyperlinks from a p:sp shape. */
274
- const extractShapeNodes = (spNode, slideNumber) => {
268
+ const extractShapeNodes = (spNode, slideNumber, xmlContentString) => {
275
269
  const nodes = [];
276
270
  // Check for placeholder type (title, body, etc.)
277
- const nvSpPr = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:nvSpPr")[0];
278
- const nvPr = nvSpPr ? (0, xmlUtils_1.getElementsByTagName)(nvSpPr, "p:nvPr")[0] : null;
279
- const ph = nvPr ? (0, xmlUtils_1.getElementsByTagName)(nvPr, "p:ph")[0] : null;
271
+ const nvSpPr = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:nvSpPr");
272
+ const nvPr = nvSpPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvSpPr, "p:nvPr") : null;
273
+ const ph = nvPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPr, "p:ph") : null;
280
274
  const type = ph ? ph.getAttribute("type") : "body";
281
275
  const isTitle = type === "title" || type === "ctrTitle";
282
- const txBody = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:txBody")[0];
276
+ const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:txBody");
283
277
  if (txBody) {
284
- const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
278
+ const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
285
279
  for (let i = 0; i < paragraphs.length; i++) {
286
280
  const p = paragraphs[i];
287
281
  const pNode = {
@@ -291,7 +285,7 @@ const parsePowerPoint = async (buffer, config) => {
291
285
  metadata: isTitle ? { level: 1 } : {}
292
286
  };
293
287
  // Paragraph Alignment and List Detection
294
- const pPr = (0, xmlUtils_1.getElementsByTagName)(p, "a:pPr")[0];
288
+ const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(p, "a:pPr");
295
289
  let isList = false;
296
290
  let listType = 'unordered';
297
291
  let lvl = 0;
@@ -299,10 +293,10 @@ const parsePowerPoint = async (buffer, config) => {
299
293
  const lvlAttr = pPr.getAttribute("lvl");
300
294
  if (lvlAttr)
301
295
  lvl = parseInt(lvlAttr);
302
- const buAutoNum = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buAutoNum")[0];
303
- const buChar = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buChar")[0];
304
- const buBlip = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buBlip")[0];
305
- const buNode = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:bu")[0];
296
+ const buAutoNum = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buAutoNum");
297
+ const buChar = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buChar");
298
+ const buBlip = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buBlip");
299
+ const buNode = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:bu");
306
300
  if (buAutoNum) {
307
301
  isList = true;
308
302
  listType = 'ordered';
@@ -396,119 +390,131 @@ const parsePowerPoint = async (buffer, config) => {
396
390
  pNode.metadata = { ...pNode.metadata, level: 1 };
397
391
  }
398
392
  if (config.includeRawContent) {
399
- pNode.rawContent = p.toString();
393
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
400
394
  }
401
- const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
402
- for (let j = 0; j < runs.length; j++) {
403
- const r = runs[j];
404
- const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
405
- if (t && t.childNodes[0]) {
406
- const textContent = t.childNodes[0].nodeValue || '';
407
- pNode.text += textContent;
408
- const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
409
- const formatting = {};
410
- if (rPr) {
411
- if (rPr.getAttribute("b") === "1")
412
- formatting.bold = true;
413
- if (rPr.getAttribute("i") === "1")
414
- formatting.italic = true;
415
- if (rPr.getAttribute("u") === "sng")
416
- formatting.underline = true;
417
- if (rPr.getAttribute("strike") === "sngStrike")
418
- formatting.strikethrough = true;
419
- const sz = rPr.getAttribute("sz");
420
- if (sz)
421
- formatting.size = (parseInt(sz) / 100).toString() + 'pt';
422
- // Color extraction
423
- const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
424
- if (solidFill) {
425
- const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
426
- if (srgbClr) {
427
- const val = srgbClr.getAttribute("val");
428
- if (val)
429
- formatting.color = '#' + val;
395
+ // Process all children of <a:p> in order (runs, breaks, fields)
396
+ const children = Array.from(p.childNodes);
397
+ let activeNode = pNode;
398
+ nodes.push(activeNode);
399
+ for (const childNode of children) {
400
+ if (!(0, xmlUtils_js_1.isElement)(childNode))
401
+ continue;
402
+ const element = childNode;
403
+ const tag = element.tagName;
404
+ if (tag === "a:r" || tag === "a:fld") {
405
+ const t = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:t");
406
+ if (t && t.childNodes[0]) {
407
+ const textContent = t.childNodes[0].nodeValue || "";
408
+ activeNode.text += textContent;
409
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:rPr");
410
+ const formatting = {};
411
+ if (rPr) {
412
+ if (rPr.getAttribute("b") === "1")
413
+ formatting.bold = true;
414
+ if (rPr.getAttribute("i") === "1")
415
+ formatting.italic = true;
416
+ if (rPr.getAttribute("u") === "sng")
417
+ formatting.underline = true;
418
+ if (rPr.getAttribute("strike") === "sngStrike")
419
+ formatting.strikethrough = true;
420
+ const sz = rPr.getAttribute("sz");
421
+ if (sz)
422
+ formatting.size = (parseInt(sz) / 100).toString() + "pt";
423
+ // Color extraction
424
+ const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
425
+ if (solidFill) {
426
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
427
+ if (srgbClr) {
428
+ const val = srgbClr.getAttribute("val");
429
+ if (val)
430
+ formatting.color = "#" + val;
431
+ }
430
432
  }
431
- }
432
- // Highlight extraction
433
- const highlight = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:highlight")[0];
434
- if (highlight) {
435
- const srgbClr = (0, xmlUtils_1.getElementsByTagName)(highlight, "a:srgbClr")[0];
436
- if (srgbClr) {
437
- const val = srgbClr.getAttribute("val");
438
- if (val)
439
- formatting.backgroundColor = '#' + val;
433
+ // Highlight extraction
434
+ const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
435
+ if (highlight) {
436
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
437
+ if (srgbClr) {
438
+ const val = srgbClr.getAttribute("val");
439
+ if (val)
440
+ formatting.backgroundColor = "#" + val;
441
+ }
442
+ }
443
+ // Font family
444
+ const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
445
+ if (latin) {
446
+ const typeface = latin.getAttribute("typeface");
447
+ if (typeface)
448
+ formatting.font = typeface;
449
+ }
450
+ // Subscript/Superscript
451
+ const baseline = rPr.getAttribute("baseline");
452
+ if (baseline) {
453
+ const baselineVal = parseInt(baseline);
454
+ if (baselineVal < 0)
455
+ formatting.subscript = true;
456
+ if (baselineVal > 0)
457
+ formatting.superscript = true;
440
458
  }
441
459
  }
442
- // Font family
443
- const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
444
- if (latin) {
445
- const typeface = latin.getAttribute("typeface");
446
- if (typeface)
447
- formatting.font = typeface;
448
- }
449
- // Subscript/Superscript
450
- const baseline = rPr.getAttribute("baseline");
451
- if (baseline) {
452
- const baselineVal = parseInt(baseline);
453
- if (baselineVal < 0)
454
- formatting.subscript = true;
455
- if (baselineVal > 0)
456
- formatting.superscript = true;
460
+ const textNode = {
461
+ type: 'text',
462
+ text: textContent,
463
+ formatting: formatting
464
+ };
465
+ // Check for Hyperlinks
466
+ const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:hlinkClick");
467
+ if (hlinkClick) {
468
+ const rId = hlinkClick.getAttribute("r:id");
469
+ const action = hlinkClick.getAttribute("action");
470
+ let link;
471
+ let linkType;
472
+ if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "hyperlink") {
473
+ link = slideRelsMap[slideNumber][rId].target;
474
+ linkType = "external";
475
+ }
476
+ else if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "slide") {
477
+ link = slideRelsMap[slideNumber][rId].target;
478
+ linkType = "internal";
479
+ }
480
+ else if (action) {
481
+ link = action;
482
+ linkType = "internal";
483
+ }
484
+ if (link) {
485
+ textNode.metadata = { link, linkType };
486
+ }
457
487
  }
488
+ activeNode.children?.push(textNode);
458
489
  }
459
- const textNode = {
460
- type: 'text',
461
- text: textContent,
462
- formatting: formatting
463
- };
464
- // Check for Hyperlinks
465
- const hlinkClick = (0, xmlUtils_1.getElementsByTagName)(r, "a:hlinkClick")[0];
466
- // Check if this run has a hyperlink click action
467
- if (hlinkClick) {
468
- // Relationship ID for the link
469
- const rId = hlinkClick.getAttribute("r:id");
470
- // Optional action attribute, often for internal jumps
471
- const action = hlinkClick.getAttribute("action");
472
- // Result placeholders
473
- let link;
474
- let linkType;
475
- // Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
476
- if (rId
477
- && slideRelsMap[slideNumber]
478
- && slideRelsMap[slideNumber][rId]
479
- && slideRelsMap[slideNumber][rId].type === "hyperlink") {
480
- // External URL
481
- link = slideRelsMap[slideNumber][rId].target;
482
- linkType = "external";
483
- }
484
- // Case 2: Relationship exists and is an internal slide reference
485
- else if (rId
486
- && slideRelsMap[slideNumber]
487
- && slideRelsMap[slideNumber][rId]
488
- && slideRelsMap[slideNumber][rId].type === "slide") {
489
- // Example target: ppt/slides/slide3.xml
490
- link = slideRelsMap[slideNumber][rId].target;
491
- linkType = "internal";
492
- }
493
- // Case 3: action attribute like ppaction://hlinksldjump
494
- else if (action) {
495
- link = action;
496
- linkType = "internal";
497
- }
498
- // Assign metadata only if a link was actually discovered
499
- if (link) {
500
- textNode.metadata = { link, linkType };
490
+ }
491
+ else if (tag === "a:br") {
492
+ if (isList) {
493
+ // Split the list item on soft break into a paragraph node
494
+ activeNode = {
495
+ type: 'paragraph',
496
+ text: '',
497
+ children: [],
498
+ metadata: {
499
+ indentation: lvl,
500
+ alignment: pNode.metadata?.alignment || 'left'
501
+ }
502
+ };
503
+ if (config.includeRawContent) {
504
+ activeNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
501
505
  }
506
+ nodes.push(activeNode);
507
+ }
508
+ else {
509
+ // In a normal paragraph, just add a newline
510
+ activeNode.text += "\n";
511
+ activeNode.children?.push({ type: 'text', text: "\n" });
502
512
  }
503
- pNode.children?.push(textNode);
504
513
  }
505
514
  }
506
- if (pNode.text) {
507
- nodes.push(pNode);
508
- }
509
515
  }
510
516
  }
511
- return nodes;
517
+ return nodes.filter(n => n.text?.trim() || (n.children && n.children.length > 0));
512
518
  };
513
519
  /**
514
520
  * Recursively traverses a PowerPoint shape tree (p:spTree),
@@ -520,31 +526,32 @@ const parsePowerPoint = async (buffer, config) => {
520
526
  * transforms so nested groups inherit positional transforms.
521
527
  *
522
528
  * @param treeNode The XML node representing <p:spTree>
523
- * @param parentTransform Transform inherited from parent groups (defaults to identity)
529
+ * @param slideNumber Current slide number for relationship resolution
530
+ * @param xmlContentString The source XML string for raw content extraction
524
531
  */
525
- function traverseSpTree(treeNode, slideNumber) {
532
+ function traverseSpTree(treeNode, slideNumber, xmlContentString) {
526
533
  const nodes = [];
527
534
  // Process children in XML order (this preserves Z-order)
528
535
  for (const child of Array.from(treeNode?.childNodes || [])) {
529
- if (child.nodeType !== 1) {
536
+ if (!(0, xmlUtils_js_1.isElement)(child)) {
530
537
  continue;
531
538
  }
532
539
  const element = child;
533
540
  const tag = element.tagName;
534
541
  // Case 1: Normal shape
535
542
  if (tag === "p:sp") {
536
- nodes.push(...extractShapeNodes(element, slideNumber));
543
+ nodes.push(...extractShapeNodes(element, slideNumber, xmlContentString));
537
544
  }
538
545
  // Case 2: Inline picture
539
546
  else if (tag === "p:pic") {
540
- const imageNode = extractImageNode(element, slideNumber);
547
+ const imageNode = extractImageNode(element, slideNumber, xmlContentString);
541
548
  if (imageNode) {
542
549
  nodes.push(imageNode);
543
550
  }
544
551
  }
545
552
  // Case 3: Chart or other graphic frame
546
553
  else if (tag === "p:graphicFrame") {
547
- const tableNode = extractGraphicFrameNode(element, slideNumber);
554
+ const tableNode = extractGraphicFrameNode(element, slideNumber, xmlContentString);
548
555
  if (tableNode) {
549
556
  nodes.push(tableNode);
550
557
  }
@@ -552,9 +559,11 @@ const parsePowerPoint = async (buffer, config) => {
552
559
  // Case 4: Grouped shape (recursive!)
553
560
  else if (tag === "p:grpSp") {
554
561
  // Extract the nested <p:spTree> inside the group
555
- const nestedTree = element.getElementsByTagName("p:spTree")[0];
562
+ const nestedTree = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "p:spTree");
556
563
  // Recurse into the nested tree
557
- nodes.push(...traverseSpTree(nestedTree, slideNumber));
564
+ if (nestedTree) {
565
+ nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
566
+ }
558
567
  }
559
568
  }
560
569
  return nodes;
@@ -587,9 +596,9 @@ const parsePowerPoint = async (buffer, config) => {
587
596
  // Prepare map for this slide
588
597
  slideRelsMap[slideNum] = {};
589
598
  // Parse the rels XML
590
- const relsXml = (0, xmlUtils_1.parseXmlString)(file.content.toString());
599
+ const relsXml = (0, xmlUtils_js_1.parseXmlString)(file.content.toString());
591
600
  // Get all Relationship nodes
592
- const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
601
+ const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
593
602
  // Loop through each relationship node
594
603
  for (let i = 0; i < relationships.length; i++) {
595
604
  // Relationship ID, Example: "rId2"
@@ -653,7 +662,7 @@ const parsePowerPoint = async (buffer, config) => {
653
662
  if (file.path.match(corePropsFileRegex))
654
663
  continue;
655
664
  const xmlContentString = file.content.toString();
656
- const xml = (0, xmlUtils_1.parseXmlString)(xmlContentString);
665
+ const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
657
666
  if (config.includeRawContent) {
658
667
  rawContents.push(xmlContentString);
659
668
  }
@@ -669,15 +678,15 @@ const parsePowerPoint = async (buffer, config) => {
669
678
  }
670
679
  };
671
680
  if (config.includeRawContent) {
672
- slideNode.rawContent = file.content.toString();
681
+ slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(xml, xmlContentString, config);
673
682
  }
674
683
  /**
675
684
  * Extract slide contents in correct document order by scanning p:spTree children.
676
685
  * This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
677
686
  */
678
- const spTree = (0, xmlUtils_1.getElementsByTagName)(xml, "p:spTree")[0];
687
+ const spTree = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "p:spTree");
679
688
  if (spTree) {
680
- slideNode.children?.push(...traverseSpTree(spTree, slideNumber));
689
+ slideNode.children?.push(...traverseSpTree(spTree, slideNumber, xmlContentString));
681
690
  }
682
691
  if (slideNode.children && slideNode.children.length > 0) {
683
692
  content.push(slideNode);
@@ -690,15 +699,15 @@ const parsePowerPoint = async (buffer, config) => {
690
699
  if (config.extractAttachments) {
691
700
  // Extract media files as attachments
692
701
  for (const media of mediaFiles) {
693
- const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
702
+ const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
694
703
  attachments.push(attachment);
695
704
  if (config.ocr) {
696
705
  if (attachment.mimeType.startsWith('image/')) {
697
706
  try {
698
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
707
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
699
708
  }
700
709
  catch (e) {
701
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
710
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
702
711
  }
703
712
  }
704
713
  }
@@ -715,12 +724,12 @@ const parsePowerPoint = async (buffer, config) => {
715
724
  attachments.push(attachment);
716
725
  // Extract text from chart XML
717
726
  try {
718
- const chartData = await (0, chartUtils_1.extractChartData)(chart.content);
727
+ const chartData = await (0, chartUtils_js_1.extractChartData)(chart.content);
719
728
  // Assign chartData to attachment
720
729
  attachment.chartData = chartData;
721
730
  }
722
731
  catch (e) {
723
- (0, errorUtils_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
732
+ (0, errorUtils_js_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
724
733
  }
725
734
  }
726
735
  // Loop through nodes to find images and charts and link their text and chartData