officeparser 6.1.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +284 -86
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -28
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +107 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +878 -5
- package/dist/officeparser.browser.iife.js +703 -49
- package/dist/officeparser.browser.mjs +703 -49
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +237 -128
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +132 -123
- package/dist/parsers/RtfParser.d.ts +22 -2
- package/dist/parsers/RtfParser.js +1398 -1282
- package/dist/parsers/WordParser.d.ts +3 -2
- package/dist/parsers/WordParser.js +333 -115
- package/dist/sbom.cdx.json +103 -103
- package/dist/types.d.ts +833 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +28 -9
|
@@ -59,12 +59,15 @@
|
|
|
59
59
|
*/
|
|
60
60
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
61
61
|
exports.parsePdf = void 0;
|
|
62
|
+
const defaults_js_1 = require("../defaults.js");
|
|
63
|
+
const types_js_1 = require("../types.js");
|
|
64
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
65
|
+
const dateUtils_js_1 = require("../utils/dateUtils.js");
|
|
66
|
+
const envUtils_js_1 = require("../utils/envUtils.js");
|
|
62
67
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
63
68
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
64
|
-
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
65
69
|
const moduleLoader_js_1 = require("../utils/moduleLoader.js");
|
|
66
|
-
const
|
|
67
|
-
const envUtils_js_1 = require("../utils/envUtils.js");
|
|
70
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
68
71
|
/** Type guard for TextItem in PDF.js 5.x */
|
|
69
72
|
function isTextItem(item) {
|
|
70
73
|
return item && typeof item.str === 'string' && Array.isArray(item.transform) && item.transform.length >= 6;
|
|
@@ -265,29 +268,38 @@ function convertToRgbaBuffer(data, width, height, kind) {
|
|
|
265
268
|
const parsePdf = async (buffer, config) => {
|
|
266
269
|
const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
|
|
267
270
|
// Configure worker
|
|
268
|
-
|
|
269
|
-
|
|
271
|
+
const workerSrc = config.pdfWorkerSrc;
|
|
272
|
+
if (envUtils_js_1.isBrowser) {
|
|
273
|
+
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
270
274
|
}
|
|
271
275
|
else {
|
|
272
|
-
//
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
+
// Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
|
|
277
|
+
(0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
|
|
278
|
+
let resolved = false;
|
|
279
|
+
// If the user provided a custom path (not the default CDN one), use it.
|
|
280
|
+
// Otherwise, try to find it locally.
|
|
281
|
+
if (workerSrc !== defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.pdfWorkerSrc && workerSrc !== '') {
|
|
282
|
+
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
283
|
+
resolved = true;
|
|
276
284
|
}
|
|
277
285
|
else {
|
|
278
|
-
// Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
|
|
279
|
-
(0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
|
|
280
286
|
try {
|
|
281
287
|
// We use require.resolve to find the exact path of the installed package.
|
|
282
288
|
// @ts-ignore - 'require' is available in Node.js/CommonJS environment
|
|
283
|
-
const
|
|
284
|
-
|
|
289
|
+
const localWorkerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
|
|
290
|
+
// Use file:// URL for the worker source in Node.js to ensure compatibility with ESM-native PDF.js 5.x
|
|
291
|
+
// We use dynamic import for 'url' to avoid breaking browser bundles
|
|
292
|
+
const { pathToFileURL } = await import('url');
|
|
293
|
+
pdfjs.GlobalWorkerOptions.workerSrc = pathToFileURL(localWorkerPath).href;
|
|
294
|
+
resolved = true;
|
|
285
295
|
}
|
|
286
296
|
catch (e) {
|
|
287
|
-
|
|
288
|
-
console.warn("[PdfParser] Could not auto-resolve local worker path:", e);
|
|
297
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PDF_WORKER_FALLBACK, config, undefined, e);
|
|
289
298
|
}
|
|
290
299
|
}
|
|
300
|
+
if (!resolved) {
|
|
301
|
+
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
302
|
+
}
|
|
291
303
|
}
|
|
292
304
|
const uint8Array = new Uint8Array(buffer);
|
|
293
305
|
const loadingTask = pdfjs.getDocument({
|
|
@@ -302,7 +314,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
302
314
|
catch (e) {
|
|
303
315
|
const message = e instanceof Error ? e.message : String(e);
|
|
304
316
|
if (message.includes('workerSrc') || message.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
|
|
305
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
317
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
|
|
306
318
|
}
|
|
307
319
|
throw e;
|
|
308
320
|
}
|
|
@@ -382,8 +394,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
382
394
|
}
|
|
383
395
|
}
|
|
384
396
|
catch (e) {
|
|
385
|
-
|
|
386
|
-
console.error("Error extracting embedded attachments:", e);
|
|
397
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ATTACHMENT_EXTRACTION_FAILED, config, undefined, e);
|
|
387
398
|
}
|
|
388
399
|
// --- First Pass: Collect all items for font statistics ---
|
|
389
400
|
for (let i = 1; i <= numPages; i++) {
|
|
@@ -396,13 +407,13 @@ const parsePdf = async (buffer, config) => {
|
|
|
396
407
|
textContent = await page.getTextContent();
|
|
397
408
|
}
|
|
398
409
|
catch (e) {
|
|
399
|
-
|
|
400
|
-
console.warn(`[PdfParser] Error loading page ${i}:`, e);
|
|
410
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, i, e);
|
|
401
411
|
// Push empty items to maintain index alignment for second pass
|
|
402
412
|
allPageItems.push(pageItems);
|
|
403
413
|
continue;
|
|
404
414
|
}
|
|
405
415
|
const commonObjs = page.commonObjs;
|
|
416
|
+
const fontCache = new Map();
|
|
406
417
|
for (const item of textContent.items) {
|
|
407
418
|
// PDF.js 5.x: textContent.items can contain TextMarkedContent which lack
|
|
408
419
|
// 'str' and 'transform'. Skip these to avoid crashes and page skipping.
|
|
@@ -422,11 +433,15 @@ const parsePdf = async (buffer, config) => {
|
|
|
422
433
|
if (textItem.fontName && commonObjs) {
|
|
423
434
|
try {
|
|
424
435
|
if (commonObjs.has(textItem.fontName)) {
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
//
|
|
428
|
-
|
|
429
|
-
|
|
436
|
+
let fontData = fontCache.get(textItem.fontName);
|
|
437
|
+
if (!fontData) {
|
|
438
|
+
// Use callback-based get to ensure safe resolution
|
|
439
|
+
fontData = await new Promise((resolve) => {
|
|
440
|
+
// @ts-ignore - commonObjs.get is callback-based in legacy builds
|
|
441
|
+
commonObjs.get(textItem.fontName, (data) => resolve(data));
|
|
442
|
+
});
|
|
443
|
+
fontCache.set(textItem.fontName, fontData);
|
|
444
|
+
}
|
|
430
445
|
if (fontData?.name && typeof fontData.name === 'string') {
|
|
431
446
|
// Remove PDF subset prefix (6 uppercase letters + '+')
|
|
432
447
|
fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
|
|
@@ -485,9 +500,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
485
500
|
});
|
|
486
501
|
}
|
|
487
502
|
catch (e) {
|
|
488
|
-
|
|
489
|
-
console.error(`[PdfParser] Failed to load dependency ${dep}:`, e);
|
|
490
|
-
}
|
|
503
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.DEPENDENCY_LOAD_FAILED, config, dep, e);
|
|
491
504
|
}
|
|
492
505
|
}
|
|
493
506
|
}
|
|
@@ -507,7 +520,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
507
520
|
targetObjs.get(imgName, (data) => resolve(data));
|
|
508
521
|
});
|
|
509
522
|
// Browser-specific: Handle ImageBitmap if data is missing
|
|
510
|
-
if (
|
|
523
|
+
if (envUtils_js_1.isBrowser && !imgObj.data && imgObj.bitmap) {
|
|
511
524
|
try {
|
|
512
525
|
const canvas = document.createElement('canvas');
|
|
513
526
|
canvas.width = imgObj.width;
|
|
@@ -520,8 +533,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
520
533
|
}
|
|
521
534
|
}
|
|
522
535
|
catch (e) {
|
|
523
|
-
|
|
524
|
-
console.error(`[PdfParser] Failed to extract from ImageBitmap:`, e);
|
|
536
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_PROCESSING_FAILED, config, undefined, e);
|
|
525
537
|
}
|
|
526
538
|
}
|
|
527
539
|
if (imgObj?.data && imgObj.width > 0 && imgObj.height > 0) {
|
|
@@ -554,8 +566,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
554
566
|
}
|
|
555
567
|
}
|
|
556
568
|
catch (e) {
|
|
557
|
-
|
|
558
|
-
console.error(`Error extracting images from page ${i}:`, e);
|
|
569
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, `from page ${i}`, e);
|
|
559
570
|
}
|
|
560
571
|
}
|
|
561
572
|
allPageItems.push(pageItems);
|
|
@@ -570,8 +581,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
570
581
|
page = await pdfDocument.getPage(pageNum);
|
|
571
582
|
}
|
|
572
583
|
catch (e) {
|
|
573
|
-
|
|
574
|
-
console.warn(`[PdfParser] Error loading page ${pageNum} in second pass:`, e);
|
|
584
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, pageNum, e);
|
|
575
585
|
continue;
|
|
576
586
|
}
|
|
577
587
|
const pageItems = allPageItems[i];
|
|
@@ -594,8 +604,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
594
604
|
}
|
|
595
605
|
}
|
|
596
606
|
catch (e) {
|
|
597
|
-
|
|
598
|
-
console.error(`Error extracting annotations from page ${pageNum}:`, e);
|
|
607
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ANNOTATION_EXTRACTION_FAILED, config, pageNum, e);
|
|
599
608
|
}
|
|
600
609
|
// Sort items: Y descending (top to bottom), then X ascending (left to right)
|
|
601
610
|
pageItems.sort((a, b) => {
|
|
@@ -744,12 +753,11 @@ const parsePdf = async (buffer, config) => {
|
|
|
744
753
|
try {
|
|
745
754
|
// Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
|
|
746
755
|
if (item.width >= 10 && item.height >= 10) {
|
|
747
|
-
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, {
|
|
756
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { ...config.ocrConfig })).trim();
|
|
748
757
|
}
|
|
749
758
|
}
|
|
750
759
|
catch (e) {
|
|
751
|
-
|
|
752
|
-
console.error(`OCR failed for ${attachmentName}:`, e);
|
|
760
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachmentName, e);
|
|
753
761
|
}
|
|
754
762
|
}
|
|
755
763
|
attachments.push(attachment);
|
|
@@ -764,7 +772,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
764
772
|
});
|
|
765
773
|
}
|
|
766
774
|
catch (e) {
|
|
767
|
-
(0, errorUtils_js_1.logWarning)(
|
|
775
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, attachmentName, e);
|
|
768
776
|
}
|
|
769
777
|
}
|
|
770
778
|
}
|
|
@@ -777,17 +785,12 @@ const parsePdf = async (buffer, config) => {
|
|
|
777
785
|
content.push({
|
|
778
786
|
type: 'page',
|
|
779
787
|
children: pageContent,
|
|
780
|
-
text: pageContent.map(node => node.text).join(config.newlineDelimiter
|
|
788
|
+
text: pageContent.map(node => node.text).join(config.newlineDelimiter),
|
|
781
789
|
metadata: { pageNumber: pageNum }
|
|
782
790
|
});
|
|
783
791
|
}
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
metadata: metadata,
|
|
787
|
-
content: content,
|
|
788
|
-
attachments: attachments,
|
|
789
|
-
toText: () => content.map(c => c.text).join(config.newlineDelimiter ?? '\n\n')
|
|
790
|
-
};
|
|
792
|
+
const toTextSync = () => content.map(c => c.text).join(config.newlineDelimiter);
|
|
793
|
+
return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, toTextSync);
|
|
791
794
|
};
|
|
792
795
|
exports.parsePdf = parsePdf;
|
|
793
796
|
/**
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* @module PowerPointParser
|
|
22
22
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
23
23
|
*/
|
|
24
|
-
import {
|
|
24
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
25
25
|
/**
|
|
26
26
|
* Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
|
|
27
27
|
*
|
|
@@ -29,4 +29,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
|
29
29
|
* @param config - Parser configuration
|
|
30
30
|
* @returns A promise resolving to the parsed AST
|
|
31
31
|
*/
|
|
32
|
-
export declare const parsePowerPoint: (buffer: Buffer, config:
|
|
32
|
+
export declare const parsePowerPoint: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|
|
@@ -24,6 +24,8 @@
|
|
|
24
24
|
*/
|
|
25
25
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
26
|
exports.parsePowerPoint = void 0;
|
|
27
|
+
const types_js_1 = require("../types.js");
|
|
28
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
27
29
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
28
30
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
29
31
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
@@ -392,117 +394,129 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
392
394
|
if (config.includeRawContent) {
|
|
393
395
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
394
396
|
}
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
if (rPr
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
397
|
+
// Process all children of <a:p> in order (runs, breaks, fields)
|
|
398
|
+
const children = Array.from(p.childNodes);
|
|
399
|
+
let activeNode = pNode;
|
|
400
|
+
nodes.push(activeNode);
|
|
401
|
+
for (const childNode of children) {
|
|
402
|
+
if (!(0, xmlUtils_js_1.isElement)(childNode))
|
|
403
|
+
continue;
|
|
404
|
+
const element = childNode;
|
|
405
|
+
const tag = element.tagName;
|
|
406
|
+
if (tag === "a:r" || tag === "a:fld") {
|
|
407
|
+
const t = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:t");
|
|
408
|
+
if (t && t.childNodes[0]) {
|
|
409
|
+
const textContent = t.childNodes[0].nodeValue || "";
|
|
410
|
+
activeNode.text += textContent;
|
|
411
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:rPr");
|
|
412
|
+
const formatting = {};
|
|
413
|
+
if (rPr) {
|
|
414
|
+
if (rPr.getAttribute("b") === "1")
|
|
415
|
+
formatting.bold = true;
|
|
416
|
+
if (rPr.getAttribute("i") === "1")
|
|
417
|
+
formatting.italic = true;
|
|
418
|
+
if (rPr.getAttribute("u") === "sng")
|
|
419
|
+
formatting.underline = true;
|
|
420
|
+
if (rPr.getAttribute("strike") === "sngStrike")
|
|
421
|
+
formatting.strikethrough = true;
|
|
422
|
+
const sz = rPr.getAttribute("sz");
|
|
423
|
+
if (sz)
|
|
424
|
+
formatting.size = (parseInt(sz) / 100).toString() + "pt";
|
|
425
|
+
// Color extraction
|
|
426
|
+
const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
|
|
427
|
+
if (solidFill) {
|
|
428
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
|
|
429
|
+
if (srgbClr) {
|
|
430
|
+
const val = srgbClr.getAttribute("val");
|
|
431
|
+
if (val)
|
|
432
|
+
formatting.color = "#" + val;
|
|
433
|
+
}
|
|
424
434
|
}
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
435
|
+
// Highlight extraction
|
|
436
|
+
const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
|
|
437
|
+
if (highlight) {
|
|
438
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
|
|
439
|
+
if (srgbClr) {
|
|
440
|
+
const val = srgbClr.getAttribute("val");
|
|
441
|
+
if (val)
|
|
442
|
+
formatting.backgroundColor = "#" + val;
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
// Font family
|
|
446
|
+
const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
|
|
447
|
+
if (latin) {
|
|
448
|
+
const typeface = latin.getAttribute("typeface");
|
|
449
|
+
if (typeface)
|
|
450
|
+
formatting.font = typeface;
|
|
451
|
+
}
|
|
452
|
+
// Subscript/Superscript
|
|
453
|
+
const baseline = rPr.getAttribute("baseline");
|
|
454
|
+
if (baseline) {
|
|
455
|
+
const baselineVal = parseInt(baseline);
|
|
456
|
+
if (baselineVal < 0)
|
|
457
|
+
formatting.subscript = true;
|
|
458
|
+
if (baselineVal > 0)
|
|
459
|
+
formatting.superscript = true;
|
|
434
460
|
}
|
|
435
461
|
}
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
462
|
+
const textNode = {
|
|
463
|
+
type: 'text',
|
|
464
|
+
text: textContent,
|
|
465
|
+
formatting: formatting
|
|
466
|
+
};
|
|
467
|
+
// Check for Hyperlinks
|
|
468
|
+
const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:hlinkClick");
|
|
469
|
+
if (hlinkClick) {
|
|
470
|
+
const rId = hlinkClick.getAttribute("r:id");
|
|
471
|
+
const action = hlinkClick.getAttribute("action");
|
|
472
|
+
let link;
|
|
473
|
+
let linkType;
|
|
474
|
+
if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "hyperlink") {
|
|
475
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
476
|
+
linkType = "external";
|
|
477
|
+
}
|
|
478
|
+
else if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "slide") {
|
|
479
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
480
|
+
linkType = "internal";
|
|
481
|
+
}
|
|
482
|
+
else if (action) {
|
|
483
|
+
link = action;
|
|
484
|
+
linkType = "internal";
|
|
485
|
+
}
|
|
486
|
+
if (link) {
|
|
487
|
+
textNode.metadata = { link, linkType };
|
|
488
|
+
}
|
|
451
489
|
}
|
|
490
|
+
activeNode.children?.push(textNode);
|
|
452
491
|
}
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
let linkType;
|
|
469
|
-
// Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
|
|
470
|
-
if (rId
|
|
471
|
-
&& slideRelsMap[slideNumber]
|
|
472
|
-
&& slideRelsMap[slideNumber][rId]
|
|
473
|
-
&& slideRelsMap[slideNumber][rId].type === "hyperlink") {
|
|
474
|
-
// External URL
|
|
475
|
-
link = slideRelsMap[slideNumber][rId].target;
|
|
476
|
-
linkType = "external";
|
|
477
|
-
}
|
|
478
|
-
// Case 2: Relationship exists and is an internal slide reference
|
|
479
|
-
else if (rId
|
|
480
|
-
&& slideRelsMap[slideNumber]
|
|
481
|
-
&& slideRelsMap[slideNumber][rId]
|
|
482
|
-
&& slideRelsMap[slideNumber][rId].type === "slide") {
|
|
483
|
-
// Example target: ppt/slides/slide3.xml
|
|
484
|
-
link = slideRelsMap[slideNumber][rId].target;
|
|
485
|
-
linkType = "internal";
|
|
486
|
-
}
|
|
487
|
-
// Case 3: action attribute like ppaction://hlinksldjump
|
|
488
|
-
else if (action) {
|
|
489
|
-
link = action;
|
|
490
|
-
linkType = "internal";
|
|
491
|
-
}
|
|
492
|
-
// Assign metadata only if a link was actually discovered
|
|
493
|
-
if (link) {
|
|
494
|
-
textNode.metadata = { link, linkType };
|
|
492
|
+
}
|
|
493
|
+
else if (tag === "a:br") {
|
|
494
|
+
if (isList) {
|
|
495
|
+
// Split the list item on soft break into a paragraph node
|
|
496
|
+
activeNode = {
|
|
497
|
+
type: 'paragraph',
|
|
498
|
+
text: '',
|
|
499
|
+
children: [],
|
|
500
|
+
metadata: {
|
|
501
|
+
indentation: lvl,
|
|
502
|
+
alignment: pNode.metadata?.alignment || 'left'
|
|
503
|
+
}
|
|
504
|
+
};
|
|
505
|
+
if (config.includeRawContent) {
|
|
506
|
+
activeNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
495
507
|
}
|
|
508
|
+
nodes.push(activeNode);
|
|
509
|
+
}
|
|
510
|
+
else {
|
|
511
|
+
// In a normal paragraph, just add a newline
|
|
512
|
+
activeNode.text += "\n";
|
|
513
|
+
activeNode.children?.push({ type: 'text', text: "\n" });
|
|
496
514
|
}
|
|
497
|
-
pNode.children?.push(textNode);
|
|
498
515
|
}
|
|
499
516
|
}
|
|
500
|
-
if (pNode.text) {
|
|
501
|
-
nodes.push(pNode);
|
|
502
|
-
}
|
|
503
517
|
}
|
|
504
518
|
}
|
|
505
|
-
return nodes;
|
|
519
|
+
return nodes.filter(n => n.text?.trim() || (n.children && n.children.length > 0));
|
|
506
520
|
};
|
|
507
521
|
/**
|
|
508
522
|
* Recursively traverses a PowerPoint shape tree (p:spTree),
|
|
@@ -692,10 +706,10 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
692
706
|
if (config.ocr) {
|
|
693
707
|
if (attachment.mimeType.startsWith('image/')) {
|
|
694
708
|
try {
|
|
695
|
-
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, {
|
|
709
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
|
|
696
710
|
}
|
|
697
711
|
catch (e) {
|
|
698
|
-
(0, errorUtils_js_1.logWarning)(
|
|
712
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
|
|
699
713
|
}
|
|
700
714
|
}
|
|
701
715
|
}
|
|
@@ -717,7 +731,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
717
731
|
attachment.chartData = chartData;
|
|
718
732
|
}
|
|
719
733
|
catch (e) {
|
|
720
|
-
(0, errorUtils_js_1.logWarning)(
|
|
734
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.CHART_DATA_EXTRACTION_FAILED, config, chart.path, e);
|
|
721
735
|
}
|
|
722
736
|
}
|
|
723
737
|
// Loop through nodes to find images and charts and link their text and chartData
|
|
@@ -733,7 +747,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
733
747
|
node.text = attachment.ocrText;
|
|
734
748
|
}
|
|
735
749
|
if (node.type === 'chart') {
|
|
736
|
-
node.text = attachment.chartData?.rawTexts.join(config.newlineDelimiter
|
|
750
|
+
node.text = attachment.chartData?.rawTexts.join(config.newlineDelimiter);
|
|
737
751
|
}
|
|
738
752
|
}
|
|
739
753
|
}
|
|
@@ -752,24 +766,19 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
752
766
|
return aIsNote - bIsNote;
|
|
753
767
|
});
|
|
754
768
|
}
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
return t;
|
|
770
|
-
};
|
|
771
|
-
return getText(c);
|
|
772
|
-
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
773
|
-
};
|
|
769
|
+
const toTextSync = () => content.map(c => {
|
|
770
|
+
// Recursive text extraction
|
|
771
|
+
const getText = (node) => {
|
|
772
|
+
let t = '';
|
|
773
|
+
if (node.children) {
|
|
774
|
+
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter);
|
|
775
|
+
}
|
|
776
|
+
else
|
|
777
|
+
t += node.text || '';
|
|
778
|
+
return t;
|
|
779
|
+
};
|
|
780
|
+
return getText(c);
|
|
781
|
+
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
782
|
+
return (0, astUtils_js_1.createAST)('pptx', metadata, content, attachments, config, toTextSync);
|
|
774
783
|
};
|
|
775
784
|
exports.parsePowerPoint = parsePowerPoint;
|
|
@@ -39,7 +39,7 @@
|
|
|
39
39
|
* @see https://www.biblioscape.com/rtf15_spec.htm RTF 1.5 Specification
|
|
40
40
|
* @see https://latex2rtf.sourceforge.net/RTF-Spec-1.2.pdf RTF 1.2 Specification
|
|
41
41
|
*/
|
|
42
|
-
import {
|
|
42
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
43
43
|
/**
|
|
44
44
|
* Represents an RTF group (content enclosed in braces).
|
|
45
45
|
* Groups create formatting scopes and can contain other groups, control words, or text.
|
|
@@ -130,6 +130,12 @@ export declare class SimpleRtfParser {
|
|
|
130
130
|
private index;
|
|
131
131
|
/** The RTF content as a Buffer */
|
|
132
132
|
private buffer;
|
|
133
|
+
/** Current code page for character decoding (default is Windows-1252) */
|
|
134
|
+
private codePage;
|
|
135
|
+
/** Cached TextDecoders for different code pages */
|
|
136
|
+
private decoders;
|
|
137
|
+
/** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
|
|
138
|
+
private pendingBytes;
|
|
133
139
|
/** Total length of the buffer */
|
|
134
140
|
private length;
|
|
135
141
|
/**
|
|
@@ -140,6 +146,20 @@ export declare class SimpleRtfParser {
|
|
|
140
146
|
parse(): RtfGroup;
|
|
141
147
|
private parseControl;
|
|
142
148
|
private parseText;
|
|
149
|
+
/**
|
|
150
|
+
* Flushes the pending bytes buffer as a text node to the current group.
|
|
151
|
+
* @param group The group to append the text node to
|
|
152
|
+
*/
|
|
153
|
+
private flushPendingText;
|
|
154
|
+
/**
|
|
155
|
+
* Decodes a byte array using a "UTF-8 first" strategy.
|
|
156
|
+
* If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
|
|
157
|
+
* Otherwise, falls back to the specified code page.
|
|
158
|
+
* @param bytes The bytes to decode
|
|
159
|
+
* @param codePage The RTF code page ID
|
|
160
|
+
* @returns The decoded string
|
|
161
|
+
*/
|
|
162
|
+
private decodeBytes;
|
|
143
163
|
}
|
|
144
164
|
/**
|
|
145
165
|
* Parses an RTF file and returns the AST.
|
|
@@ -164,4 +184,4 @@ export declare class SimpleRtfParser {
|
|
|
164
184
|
* @param config The parser configuration.
|
|
165
185
|
* @returns The parsed AST.
|
|
166
186
|
*/
|
|
167
|
-
export declare const parseRtf: (buffer: Buffer, config:
|
|
187
|
+
export declare const parseRtf: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|