officeparser 6.0.6 → 6.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +92 -13
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +43 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +116 -0
  6. package/dist/index.d.ts +3 -3
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +756 -0
  10. package/dist/officeparser.browser.iife.js +112 -0
  11. package/dist/officeparser.browser.mjs +111 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +71 -63
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +131 -114
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +85 -88
  20. package/dist/parsers/RtfParser.d.ts +1 -1
  21. package/dist/parsers/RtfParser.js +10 -6
  22. package/dist/parsers/WordParser.d.ts +1 -1
  23. package/dist/parsers/WordParser.js +109 -101
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +69 -1
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +39 -18
  39. package/dist/officeparser.browser.js +0 -153
  40. package/dist/officeparser.browser.js.map +0 -7
@@ -24,13 +24,12 @@
24
24
  */
25
25
  Object.defineProperty(exports, "__esModule", { value: true });
26
26
  exports.parsePowerPoint = void 0;
27
- const xmldom_1 = require("@xmldom/xmldom");
28
- const chartUtils_1 = require("../utils/chartUtils");
29
- const errorUtils_1 = require("../utils/errorUtils");
30
- const imageUtils_1 = require("../utils/imageUtils");
31
- const ocrUtils_1 = require("../utils/ocrUtils");
32
- const xmlUtils_1 = require("../utils/xmlUtils");
33
- const zipUtils_1 = require("../utils/zipUtils");
27
+ const chartUtils_js_1 = require("../utils/chartUtils.js");
28
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
29
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
30
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
31
+ const xmlUtils_js_1 = require("../utils/xmlUtils.js");
32
+ const zipUtils_js_1 = require("../utils/zipUtils.js");
34
33
  /**
35
34
  * Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
36
35
  *
@@ -46,14 +45,21 @@ const parsePowerPoint = async (buffer, config) => {
46
45
  const mediaFileRegex = /ppt\/media\/.*/;
47
46
  const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
48
47
  const corePropsFileRegex = /docProps\/core\.xml/;
49
- const xmlSerializer = new xmldom_1.XMLSerializer();
50
- const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
48
+ const customPropsFileRegex = /docProps\/custom\.xml/;
49
+ const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
51
50
  !!x.match(corePropsFileRegex) ||
51
+ !!x.match(customPropsFileRegex) ||
52
52
  !!x.match(slideRelsRegex) ||
53
53
  (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
54
54
  // Extract metadata
55
55
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
56
- const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
56
+ const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
57
+ const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
58
+ if (customPropsFile) {
59
+ const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
60
+ if (Object.keys(customProperties).length > 0)
61
+ metadata.customProperties = customProperties;
62
+ }
57
63
  // Sort files
58
64
  files.sort((a, b) => {
59
65
  const aMatch = a.path.match(slideNumberRegex);
@@ -73,21 +79,21 @@ const parsePowerPoint = async (buffer, config) => {
73
79
  // per indent counters (for nested lists)
74
80
  const levelCounters = {};
75
81
  // Helper to parse a table node
76
- const parseTable = (tblNode) => {
82
+ const parseTable = (tblNode, xmlContentString) => {
77
83
  const rows = [];
78
- const trNodes = (0, xmlUtils_1.getElementsByTagName)(tblNode, "a:tr");
84
+ const trNodes = (0, xmlUtils_js_1.getElementsByTagName)(tblNode, "a:tr");
79
85
  for (let rIndex = 0; rIndex < trNodes.length; rIndex++) {
80
86
  const trNode = trNodes[rIndex];
81
87
  const cells = [];
82
- const tcNodes = (0, xmlUtils_1.getElementsByTagName)(trNode, "a:tc");
88
+ const tcNodes = (0, xmlUtils_js_1.getElementsByTagName)(trNode, "a:tc");
83
89
  for (let cIndex = 0; cIndex < tcNodes.length; cIndex++) {
84
90
  const tcNode = tcNodes[cIndex];
85
91
  const cellChildren = [];
86
92
  let cellText = '';
87
93
  // Cells contain text bodies (txBody) which contain paragraphs
88
- const txBody = (0, xmlUtils_1.getElementsByTagName)(tcNode, "a:txBody")[0];
94
+ const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "a:txBody");
89
95
  if (txBody) {
90
- const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
96
+ const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
91
97
  for (const p of paragraphs) {
92
98
  // Reuse paragraph parsing logic if possible, or duplicate for now
93
99
  // For simplicity, duplicating basic logic here as the main loop one is tied to shapes
@@ -98,15 +104,15 @@ const parsePowerPoint = async (buffer, config) => {
98
104
  metadata: {}
99
105
  };
100
106
  if (config.includeRawContent) {
101
- pNode.rawContent = p.toString();
107
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
102
108
  }
103
- const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
109
+ const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
104
110
  for (const r of runs) {
105
- const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
111
+ const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
106
112
  if (t && t.childNodes[0]) {
107
113
  const textContent = t.childNodes[0].nodeValue || '';
108
114
  pNode.text += textContent;
109
- const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
115
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
110
116
  const formatting = {};
111
117
  if (rPr) {
112
118
  if (rPr.getAttribute("b") === "1")
@@ -120,16 +126,16 @@ const parsePowerPoint = async (buffer, config) => {
120
126
  const sz = rPr.getAttribute("sz");
121
127
  if (sz)
122
128
  formatting.size = (parseInt(sz) / 100).toString() + 'pt';
123
- const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
129
+ const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
124
130
  if (solidFill) {
125
- const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
131
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
126
132
  if (srgbClr) {
127
133
  const val = srgbClr.getAttribute("val");
128
134
  if (val)
129
135
  formatting.color = '#' + val;
130
136
  }
131
137
  }
132
- const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
138
+ const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
133
139
  if (latin) {
134
140
  const typeface = latin.getAttribute("typeface");
135
141
  if (typeface)
@@ -167,8 +173,8 @@ const parsePowerPoint = async (buffer, config) => {
167
173
  };
168
174
  };
169
175
  /** Extract an AST node for p:pic */
170
- const extractImageNode = (imageNode, slideNumber) => {
171
- const blip = (0, xmlUtils_1.getElementsByTagName)(imageNode, "a:blip")[0];
176
+ const extractImageNode = (imageNode, slideNumber, xmlContentString) => {
177
+ const blip = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "a:blip");
172
178
  if (!blip)
173
179
  return null;
174
180
  const rId = blip.getAttribute("r:embed");
@@ -178,8 +184,8 @@ const parsePowerPoint = async (buffer, config) => {
178
184
  if (!rel || rel.type !== "image")
179
185
  return null;
180
186
  const attachmentName = rel.target;
181
- const nvPicPr = (0, xmlUtils_1.getElementsByTagName)(imageNode, "p:nvPicPr")[0];
182
- const cNvPr = nvPicPr ? (0, xmlUtils_1.getElementsByTagName)(nvPicPr, "p:cNvPr")[0] : null;
187
+ const nvPicPr = (0, xmlUtils_js_1.getFirstElementByTagName)(imageNode, "p:nvPicPr");
188
+ const cNvPr = nvPicPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPicPr, "p:cNvPr") : null;
183
189
  const altText = cNvPr?.getAttribute("descr") || undefined;
184
190
  return {
185
191
  type: "image",
@@ -192,23 +198,11 @@ const parsePowerPoint = async (buffer, config) => {
192
198
  };
193
199
  /**
194
200
  * Extract an AST node for p:graphicFrame that contains a chart.
195
- * This mirrors extractImageNode, but for charts instead of images.
196
- *
197
- * Steps:
198
- * 1. Find <a:graphicData> inside the graphicFrame.
199
- * 2. Check if uri is the chart namespace.
200
- * 3. Extract <c:chart> and read r:id.
201
- * 4. Resolve the relationship using slideRelsMap.
202
- * 5. Produce a chart node with attachmentName (like images).
203
- * 6. Chart text/data will be injected later in the pipeline.
204
- *
205
- * @param frameNode p:graphicFrame element
206
- * @param slideNumber Current slide number for relationship resolution
207
- * @returns A chart OfficeContentNode or null if not a chart.
201
+ * ... (comments omitted for brevity) ...
208
202
  */
209
- const extractChartNode = (frameNode, slideNumber) => {
203
+ const extractChartNode = (frameNode, slideNumber, xmlContentString) => {
210
204
  // Step 1: Find <a:graphicData>
211
- const graphicData = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:graphicData")[0];
205
+ const graphicData = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:graphicData");
212
206
  if (!graphicData) {
213
207
  return null;
214
208
  }
@@ -220,7 +214,7 @@ const parsePowerPoint = async (buffer, config) => {
220
214
  return null;
221
215
  }
222
216
  // Step 3: Find <c:chart>
223
- const cChart = (0, xmlUtils_1.getElementsByTagName)(graphicData, "c:chart")[0];
217
+ const cChart = (0, xmlUtils_js_1.getFirstElementByTagName)(graphicData, "c:chart");
224
218
  if (!cChart) {
225
219
  return null;
226
220
  }
@@ -246,24 +240,24 @@ const parsePowerPoint = async (buffer, config) => {
246
240
  };
247
241
  // Optional: include raw XML of the whole frame
248
242
  if (config.includeRawContent) {
249
- chartNode.rawContent = xmlSerializer.serializeToString(frameNode);
243
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
250
244
  }
251
245
  return chartNode;
252
246
  };
253
247
  /** Extract an AST node for p:graphicFrame */
254
- const extractGraphicFrameNode = (frameNode, slideNumber) => {
255
- const tbl = (0, xmlUtils_1.getElementsByTagName)(frameNode, "a:tbl")[0];
248
+ const extractGraphicFrameNode = (frameNode, slideNumber, xmlContentString) => {
249
+ const tbl = (0, xmlUtils_js_1.getFirstElementByTagName)(frameNode, "a:tbl");
256
250
  if (tbl) {
257
- const tableNode = parseTable(tbl);
251
+ const tableNode = parseTable(tbl, xmlContentString);
258
252
  if (config.includeRawContent) {
259
- tableNode.rawContent = xmlSerializer.serializeToString(frameNode);
253
+ tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frameNode, xmlContentString, config);
260
254
  }
261
255
  if (tableNode.children && tableNode.children.length > 0) {
262
256
  return tableNode;
263
257
  }
264
258
  }
265
259
  if (frameNode.getElementsByTagName("c:chart").length > 0) {
266
- const chartNode = extractChartNode(frameNode, slideNumber);
260
+ const chartNode = extractChartNode(frameNode, slideNumber, xmlContentString);
267
261
  if (chartNode) {
268
262
  return chartNode;
269
263
  }
@@ -271,17 +265,17 @@ const parsePowerPoint = async (buffer, config) => {
271
265
  return null;
272
266
  };
273
267
  /** Extract text and hyperlinks from a p:sp shape. */
274
- const extractShapeNodes = (spNode, slideNumber) => {
268
+ const extractShapeNodes = (spNode, slideNumber, xmlContentString) => {
275
269
  const nodes = [];
276
270
  // Check for placeholder type (title, body, etc.)
277
- const nvSpPr = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:nvSpPr")[0];
278
- const nvPr = nvSpPr ? (0, xmlUtils_1.getElementsByTagName)(nvSpPr, "p:nvPr")[0] : null;
279
- const ph = nvPr ? (0, xmlUtils_1.getElementsByTagName)(nvPr, "p:ph")[0] : null;
271
+ const nvSpPr = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:nvSpPr");
272
+ const nvPr = nvSpPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvSpPr, "p:nvPr") : null;
273
+ const ph = nvPr ? (0, xmlUtils_js_1.getFirstElementByTagName)(nvPr, "p:ph") : null;
280
274
  const type = ph ? ph.getAttribute("type") : "body";
281
275
  const isTitle = type === "title" || type === "ctrTitle";
282
- const txBody = (0, xmlUtils_1.getElementsByTagName)(spNode, "p:txBody")[0];
276
+ const txBody = (0, xmlUtils_js_1.getFirstElementByTagName)(spNode, "p:txBody");
283
277
  if (txBody) {
284
- const paragraphs = (0, xmlUtils_1.getElementsByTagName)(txBody, "a:p");
278
+ const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
285
279
  for (let i = 0; i < paragraphs.length; i++) {
286
280
  const p = paragraphs[i];
287
281
  const pNode = {
@@ -291,7 +285,7 @@ const parsePowerPoint = async (buffer, config) => {
291
285
  metadata: isTitle ? { level: 1 } : {}
292
286
  };
293
287
  // Paragraph Alignment and List Detection
294
- const pPr = (0, xmlUtils_1.getElementsByTagName)(p, "a:pPr")[0];
288
+ const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(p, "a:pPr");
295
289
  let isList = false;
296
290
  let listType = 'unordered';
297
291
  let lvl = 0;
@@ -299,10 +293,10 @@ const parsePowerPoint = async (buffer, config) => {
299
293
  const lvlAttr = pPr.getAttribute("lvl");
300
294
  if (lvlAttr)
301
295
  lvl = parseInt(lvlAttr);
302
- const buAutoNum = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buAutoNum")[0];
303
- const buChar = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buChar")[0];
304
- const buBlip = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:buBlip")[0];
305
- const buNode = (0, xmlUtils_1.getElementsByTagName)(pPr, "a:bu")[0];
296
+ const buAutoNum = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buAutoNum");
297
+ const buChar = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buChar");
298
+ const buBlip = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:buBlip");
299
+ const buNode = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "a:bu");
306
300
  if (buAutoNum) {
307
301
  isList = true;
308
302
  listType = 'ordered';
@@ -396,16 +390,16 @@ const parsePowerPoint = async (buffer, config) => {
396
390
  pNode.metadata = { ...pNode.metadata, level: 1 };
397
391
  }
398
392
  if (config.includeRawContent) {
399
- pNode.rawContent = p.toString();
393
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
400
394
  }
401
- const runs = (0, xmlUtils_1.getElementsByTagName)(p, "a:r");
395
+ const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
402
396
  for (let j = 0; j < runs.length; j++) {
403
397
  const r = runs[j];
404
- const t = (0, xmlUtils_1.getElementsByTagName)(r, "a:t")[0];
398
+ const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
405
399
  if (t && t.childNodes[0]) {
406
400
  const textContent = t.childNodes[0].nodeValue || '';
407
401
  pNode.text += textContent;
408
- const rPr = (0, xmlUtils_1.getElementsByTagName)(r, "a:rPr")[0];
402
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
409
403
  const formatting = {};
410
404
  if (rPr) {
411
405
  if (rPr.getAttribute("b") === "1")
@@ -420,9 +414,9 @@ const parsePowerPoint = async (buffer, config) => {
420
414
  if (sz)
421
415
  formatting.size = (parseInt(sz) / 100).toString() + 'pt';
422
416
  // Color extraction
423
- const solidFill = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:solidFill")[0];
417
+ const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
424
418
  if (solidFill) {
425
- const srgbClr = (0, xmlUtils_1.getElementsByTagName)(solidFill, "a:srgbClr")[0];
419
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
426
420
  if (srgbClr) {
427
421
  const val = srgbClr.getAttribute("val");
428
422
  if (val)
@@ -430,9 +424,9 @@ const parsePowerPoint = async (buffer, config) => {
430
424
  }
431
425
  }
432
426
  // Highlight extraction
433
- const highlight = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:highlight")[0];
427
+ const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
434
428
  if (highlight) {
435
- const srgbClr = (0, xmlUtils_1.getElementsByTagName)(highlight, "a:srgbClr")[0];
429
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
436
430
  if (srgbClr) {
437
431
  const val = srgbClr.getAttribute("val");
438
432
  if (val)
@@ -440,7 +434,7 @@ const parsePowerPoint = async (buffer, config) => {
440
434
  }
441
435
  }
442
436
  // Font family
443
- const latin = (0, xmlUtils_1.getElementsByTagName)(rPr, "a:latin")[0];
437
+ const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
444
438
  if (latin) {
445
439
  const typeface = latin.getAttribute("typeface");
446
440
  if (typeface)
@@ -462,7 +456,7 @@ const parsePowerPoint = async (buffer, config) => {
462
456
  formatting: formatting
463
457
  };
464
458
  // Check for Hyperlinks
465
- const hlinkClick = (0, xmlUtils_1.getElementsByTagName)(r, "a:hlinkClick")[0];
459
+ const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:hlinkClick");
466
460
  // Check if this run has a hyperlink click action
467
461
  if (hlinkClick) {
468
462
  // Relationship ID for the link
@@ -520,31 +514,32 @@ const parsePowerPoint = async (buffer, config) => {
520
514
  * transforms so nested groups inherit positional transforms.
521
515
  *
522
516
  * @param treeNode The XML node representing <p:spTree>
523
- * @param parentTransform Transform inherited from parent groups (defaults to identity)
517
+ * @param slideNumber Current slide number for relationship resolution
518
+ * @param xmlContentString The source XML string for raw content extraction
524
519
  */
525
- function traverseSpTree(treeNode, slideNumber) {
520
+ function traverseSpTree(treeNode, slideNumber, xmlContentString) {
526
521
  const nodes = [];
527
522
  // Process children in XML order (this preserves Z-order)
528
523
  for (const child of Array.from(treeNode?.childNodes || [])) {
529
- if (child.nodeType !== 1) {
524
+ if (!(0, xmlUtils_js_1.isElement)(child)) {
530
525
  continue;
531
526
  }
532
527
  const element = child;
533
528
  const tag = element.tagName;
534
529
  // Case 1: Normal shape
535
530
  if (tag === "p:sp") {
536
- nodes.push(...extractShapeNodes(element, slideNumber));
531
+ nodes.push(...extractShapeNodes(element, slideNumber, xmlContentString));
537
532
  }
538
533
  // Case 2: Inline picture
539
534
  else if (tag === "p:pic") {
540
- const imageNode = extractImageNode(element, slideNumber);
535
+ const imageNode = extractImageNode(element, slideNumber, xmlContentString);
541
536
  if (imageNode) {
542
537
  nodes.push(imageNode);
543
538
  }
544
539
  }
545
540
  // Case 3: Chart or other graphic frame
546
541
  else if (tag === "p:graphicFrame") {
547
- const tableNode = extractGraphicFrameNode(element, slideNumber);
542
+ const tableNode = extractGraphicFrameNode(element, slideNumber, xmlContentString);
548
543
  if (tableNode) {
549
544
  nodes.push(tableNode);
550
545
  }
@@ -552,9 +547,11 @@ const parsePowerPoint = async (buffer, config) => {
552
547
  // Case 4: Grouped shape (recursive!)
553
548
  else if (tag === "p:grpSp") {
554
549
  // Extract the nested <p:spTree> inside the group
555
- const nestedTree = element.getElementsByTagName("p:spTree")[0];
550
+ const nestedTree = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "p:spTree");
556
551
  // Recurse into the nested tree
557
- nodes.push(...traverseSpTree(nestedTree, slideNumber));
552
+ if (nestedTree) {
553
+ nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
554
+ }
558
555
  }
559
556
  }
560
557
  return nodes;
@@ -587,9 +584,9 @@ const parsePowerPoint = async (buffer, config) => {
587
584
  // Prepare map for this slide
588
585
  slideRelsMap[slideNum] = {};
589
586
  // Parse the rels XML
590
- const relsXml = (0, xmlUtils_1.parseXmlString)(file.content.toString());
587
+ const relsXml = (0, xmlUtils_js_1.parseXmlString)(file.content.toString());
591
588
  // Get all Relationship nodes
592
- const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
589
+ const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
593
590
  // Loop through each relationship node
594
591
  for (let i = 0; i < relationships.length; i++) {
595
592
  // Relationship ID, Example: "rId2"
@@ -653,7 +650,7 @@ const parsePowerPoint = async (buffer, config) => {
653
650
  if (file.path.match(corePropsFileRegex))
654
651
  continue;
655
652
  const xmlContentString = file.content.toString();
656
- const xml = (0, xmlUtils_1.parseXmlString)(xmlContentString);
653
+ const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
657
654
  if (config.includeRawContent) {
658
655
  rawContents.push(xmlContentString);
659
656
  }
@@ -669,15 +666,15 @@ const parsePowerPoint = async (buffer, config) => {
669
666
  }
670
667
  };
671
668
  if (config.includeRawContent) {
672
- slideNode.rawContent = file.content.toString();
669
+ slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(xml, xmlContentString, config);
673
670
  }
674
671
  /**
675
672
  * Extract slide contents in correct document order by scanning p:spTree children.
676
673
  * This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
677
674
  */
678
- const spTree = (0, xmlUtils_1.getElementsByTagName)(xml, "p:spTree")[0];
675
+ const spTree = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "p:spTree");
679
676
  if (spTree) {
680
- slideNode.children?.push(...traverseSpTree(spTree, slideNumber));
677
+ slideNode.children?.push(...traverseSpTree(spTree, slideNumber, xmlContentString));
681
678
  }
682
679
  if (slideNode.children && slideNode.children.length > 0) {
683
680
  content.push(slideNode);
@@ -690,15 +687,15 @@ const parsePowerPoint = async (buffer, config) => {
690
687
  if (config.extractAttachments) {
691
688
  // Extract media files as attachments
692
689
  for (const media of mediaFiles) {
693
- const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
690
+ const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
694
691
  attachments.push(attachment);
695
692
  if (config.ocr) {
696
693
  if (attachment.mimeType.startsWith('image/')) {
697
694
  try {
698
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
695
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
699
696
  }
700
697
  catch (e) {
701
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
698
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
702
699
  }
703
700
  }
704
701
  }
@@ -715,12 +712,12 @@ const parsePowerPoint = async (buffer, config) => {
715
712
  attachments.push(attachment);
716
713
  // Extract text from chart XML
717
714
  try {
718
- const chartData = await (0, chartUtils_1.extractChartData)(chart.content);
715
+ const chartData = await (0, chartUtils_js_1.extractChartData)(chart.content);
719
716
  // Assign chartData to attachment
720
717
  attachment.chartData = chartData;
721
718
  }
722
719
  catch (e) {
723
- (0, errorUtils_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
720
+ (0, errorUtils_js_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
724
721
  }
725
722
  }
726
723
  // Loop through nodes to find images and charts and link their text and chartData
@@ -39,7 +39,7 @@
39
39
  * @see https://www.biblioscape.com/rtf15_spec.htm RTF 1.5 Specification
40
40
  * @see https://latex2rtf.sourceforge.net/RTF-Spec-1.2.pdf RTF 1.2 Specification
41
41
  */
42
- import { OfficeParserAST, OfficeParserConfig } from '../types';
42
+ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
43
43
  /**
44
44
  * Represents an RTF group (content enclosed in braces).
45
45
  * Groups create formatting scopes and can contain other groups, control words, or text.
@@ -42,8 +42,8 @@
42
42
  */
43
43
  Object.defineProperty(exports, "__esModule", { value: true });
44
44
  exports.parseRtf = exports.SimpleRtfParser = void 0;
45
- const errorUtils_1 = require("../utils/errorUtils");
46
- const ocrUtils_1 = require("../utils/ocrUtils");
45
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
46
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
47
47
  /**
48
48
  * Lookup table mapping RTF control words to internal formats.
49
49
  * Fully typed: if a key maps to an unsupported format, TypeScript throws an error.
@@ -88,13 +88,17 @@ const IMAGE_MIME_MAP = {
88
88
  * ```
89
89
  */
90
90
  class SimpleRtfParser {
91
+ /** Current position in the buffer */
92
+ index = 0;
93
+ /** The RTF content as a Buffer */
94
+ buffer;
95
+ /** Total length of the buffer */
96
+ length;
91
97
  /**
92
98
  * Creates a new RTF parser.
93
99
  * @param buffer - The RTF file content as a Buffer
94
100
  */
95
101
  constructor(buffer) {
96
- /** Current position in the buffer */
97
- this.index = 0;
98
102
  this.buffer = buffer;
99
103
  this.length = buffer.length;
100
104
  }
@@ -1510,10 +1514,10 @@ const parseRtf = async (buffer, config) => {
1510
1514
  // Passing base64 string directly would be interpreted as a file path,
1511
1515
  // causing ENAMETOOLONG error for large images.
1512
1516
  const imageBuffer = Buffer.from(attachment.data, 'base64');
1513
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(imageBuffer, config.ocrLanguage)).trim();
1517
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(imageBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1514
1518
  }
1515
1519
  catch (e) {
1516
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1520
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1517
1521
  }
1518
1522
  }
1519
1523
  }
@@ -58,7 +58,7 @@
58
58
  * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
59
59
  * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
60
60
  */
61
- import { OfficeParserAST, OfficeParserConfig } from '../types';
61
+ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
62
62
  /**
63
63
  * Parses a Word document (.docx) and extracts content, formatting, and metadata.
64
64
  *