officeparser 5.2.1 → 6.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +165 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +847 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -776
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* OCR (Optical Character Recognition) Utilities
|
|
4
|
+
*
|
|
5
|
+
* This module provides functions for extracting text from images using Tesseract.js.
|
|
6
|
+
* Used when `config.ocr` is enabled to extract text from embedded images in documents.
|
|
7
|
+
*
|
|
8
|
+
* @module ocrUtils
|
|
9
|
+
*/
|
|
10
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
11
|
+
exports.performOcr = void 0;
|
|
12
|
+
const tesseract_js_1 = require("tesseract.js");
|
|
13
|
+
/**
|
|
14
|
+
* Performs Optical Character Recognition (OCR) on an image to extract text.
|
|
15
|
+
*
|
|
16
|
+
* Uses Tesseract.js to recognize text in the provided image buffer.
|
|
17
|
+
* This is useful for extracting text from screenshots, scanned documents,
|
|
18
|
+
* charts with labels, or any image containing text.
|
|
19
|
+
*
|
|
20
|
+
* The function creates a new Tesseract worker, processes the image,
|
|
21
|
+
* and properly terminates the worker to free resources.
|
|
22
|
+
*
|
|
23
|
+
* @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
|
|
24
|
+
* @param language - The language code for OCR (default: 'eng' for English).
|
|
25
|
+
* Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
|
|
26
|
+
* Multiple languages can be combined with '+': 'eng+fra'
|
|
27
|
+
* @returns A promise that resolves to the recognized text as a string
|
|
28
|
+
* @throws {Error} If the image cannot be processed or Tesseract initialization fails
|
|
29
|
+
*
|
|
30
|
+
* @example
|
|
31
|
+
* ```typescript
|
|
32
|
+
* // Extract text from an English image
|
|
33
|
+
* const text = await performOcr(imageBuffer, 'eng');
|
|
34
|
+
* console.log(text); // "Annual Revenue: $1.2M"
|
|
35
|
+
*
|
|
36
|
+
* // Extract text from a multilingual image
|
|
37
|
+
* const text = await performOcr(imageBuffer, 'eng+spa');
|
|
38
|
+
* ```
|
|
39
|
+
*
|
|
40
|
+
* @see https://github.com/naptha/tesseract.js for supported languages and options
|
|
41
|
+
*/
|
|
42
|
+
const performOcr = async (image, language = 'eng') => {
|
|
43
|
+
// Step 1: Create a Tesseract worker with the specified language
|
|
44
|
+
// We pass 1 for OEM (LSTM) and a silent logger to suppress console output
|
|
45
|
+
const worker = await (0, tesseract_js_1.createWorker)(language, 1, {
|
|
46
|
+
logger: () => { }
|
|
47
|
+
});
|
|
48
|
+
// Step 2: Prepare image data
|
|
49
|
+
let inputImage = image;
|
|
50
|
+
// In browser environment, convert Buffer to Blob for better compatibility
|
|
51
|
+
// @ts-ignore
|
|
52
|
+
if (typeof window !== 'undefined' && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
|
|
53
|
+
inputImage = new Blob([image], { type: 'image/bmp' });
|
|
54
|
+
}
|
|
55
|
+
// Step 3: Perform OCR
|
|
56
|
+
const ret = await worker.recognize(inputImage);
|
|
57
|
+
// Step 4: Terminate worker
|
|
58
|
+
await worker.terminate();
|
|
59
|
+
return ret.data.text;
|
|
60
|
+
};
|
|
61
|
+
exports.performOcr = performOcr;
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* XML Parsing Utilities
|
|
3
|
+
*
|
|
4
|
+
* Provides helper functions for parsing and navigating XML documents.
|
|
5
|
+
* Used extensively by OOXML parsers (DOCX, XLSX, PPTX) and OpenOffice parsers (ODT, ODP, ODS).
|
|
6
|
+
*
|
|
7
|
+
* OOXML (Office Open XML) is an XML-based format used by Microsoft Office.
|
|
8
|
+
* Documents are ZIP archives containing multiple XML files describing structure, content, and formatting.
|
|
9
|
+
*
|
|
10
|
+
* @module xmlUtils
|
|
11
|
+
*/
|
|
12
|
+
import { OfficeMetadata } from '../types';
|
|
13
|
+
/**
|
|
14
|
+
* Parses an XML string into a DOM Document object.
|
|
15
|
+
*
|
|
16
|
+
* Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
|
|
17
|
+
* This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
|
|
18
|
+
*
|
|
19
|
+
* @param xml - The XML content as a string
|
|
20
|
+
* @returns A Document object that can be queried using standard DOM methods
|
|
21
|
+
* @example
|
|
22
|
+
* ```typescript
|
|
23
|
+
* const xmlString = '<root><item>Hello</item></root>';
|
|
24
|
+
* const doc = parseXmlString(xmlString);
|
|
25
|
+
* const items = doc.getElementsByTagName('item');
|
|
26
|
+
* console.log(items[0].textContent); // "Hello"
|
|
27
|
+
* ```
|
|
28
|
+
*/
|
|
29
|
+
export declare const parseXmlString: (xml: string) => Document;
|
|
30
|
+
/**
|
|
31
|
+
* Gets all elements with a specific tag name and returns them as an array.
|
|
32
|
+
*
|
|
33
|
+
* This is a convenience wrapper around the DOM API's getElementsByTagName method
|
|
34
|
+
* that converts the HTMLCollection/NodeList to a proper JavaScript array for easier manipulation.
|
|
35
|
+
*
|
|
36
|
+
* @param element - The element or document to search within
|
|
37
|
+
* @param tagName - The tag name to search for (e.g., 'w:t', 'w:p', 'item')
|
|
38
|
+
* @returns An array of matching elements (empty array if none found)
|
|
39
|
+
* @example
|
|
40
|
+
* ```typescript
|
|
41
|
+
* const paragraphs = getElementsByTagName(doc, 'w:p');
|
|
42
|
+
* paragraphs.forEach(p => console.log(p.textContent));
|
|
43
|
+
* ```
|
|
44
|
+
*/
|
|
45
|
+
export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
|
|
46
|
+
/**
|
|
47
|
+
* Gets direct child elements with a specific tag name.
|
|
48
|
+
* Unlike getElementsByTagName, this does not search recursively.
|
|
49
|
+
*
|
|
50
|
+
* @param parent - The parent element
|
|
51
|
+
* @param tagName - The tag name to search for
|
|
52
|
+
* @returns An array of matching direct child elements
|
|
53
|
+
*/
|
|
54
|
+
export declare const getDirectChildren: (parent: Element, tagName: string) => Element[];
|
|
55
|
+
/**
|
|
56
|
+
* Parses OOXML document metadata from the docProps/core.xml file.
|
|
57
|
+
*
|
|
58
|
+
* OOXML documents (DOCX, XLSX, PPTX) store metadata in a standard location:
|
|
59
|
+
* `docProps/core.xml` within the ZIP archive.
|
|
60
|
+
*
|
|
61
|
+
* This file follows the Dublin Core metadata standard with OOXML-specific extensions.
|
|
62
|
+
* Common metadata elements:
|
|
63
|
+
* - dc:title - Document title
|
|
64
|
+
* - dc:creator - Original author
|
|
65
|
+
* - cp:lastModifiedBy - User who last modified the document
|
|
66
|
+
* - dcterms:created - Creation timestamp
|
|
67
|
+
* - dcterms:modified - Last modification timestamp
|
|
68
|
+
*
|
|
69
|
+
* @param xmlContent - The raw XML content string from docProps/core.xml
|
|
70
|
+
* @returns An OfficeMetadata object with extracted properties (empty object if parsing fails)
|
|
71
|
+
* @example
|
|
72
|
+
* ```typescript
|
|
73
|
+
* const coreXml = files.find(f => f.path === 'docProps/core.xml').content.toString();
|
|
74
|
+
* const metadata = parseOfficeMetadata(coreXml);
|
|
75
|
+
*
|
|
76
|
+
* console.log(metadata.author); // "John Smith"
|
|
77
|
+
* console.log(metadata.title); // "Annual Report"
|
|
78
|
+
* console.log(metadata.created); // Date object
|
|
79
|
+
* ```
|
|
80
|
+
*
|
|
81
|
+
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
|
|
82
|
+
*/
|
|
83
|
+
export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata;
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* XML Parsing Utilities
|
|
4
|
+
*
|
|
5
|
+
* Provides helper functions for parsing and navigating XML documents.
|
|
6
|
+
* Used extensively by OOXML parsers (DOCX, XLSX, PPTX) and OpenOffice parsers (ODT, ODP, ODS).
|
|
7
|
+
*
|
|
8
|
+
* OOXML (Office Open XML) is an XML-based format used by Microsoft Office.
|
|
9
|
+
* Documents are ZIP archives containing multiple XML files describing structure, content, and formatting.
|
|
10
|
+
*
|
|
11
|
+
* @module xmlUtils
|
|
12
|
+
*/
|
|
13
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
|
+
exports.parseOfficeMetadata = exports.getDirectChildren = exports.getElementsByTagName = exports.parseXmlString = void 0;
|
|
15
|
+
const xmldom_1 = require("@xmldom/xmldom");
|
|
16
|
+
/**
|
|
17
|
+
* Parses an XML string into a DOM Document object.
|
|
18
|
+
*
|
|
19
|
+
* Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
|
|
20
|
+
* This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
|
|
21
|
+
*
|
|
22
|
+
* @param xml - The XML content as a string
|
|
23
|
+
* @returns A Document object that can be queried using standard DOM methods
|
|
24
|
+
* @example
|
|
25
|
+
* ```typescript
|
|
26
|
+
* const xmlString = '<root><item>Hello</item></root>';
|
|
27
|
+
* const doc = parseXmlString(xmlString);
|
|
28
|
+
* const items = doc.getElementsByTagName('item');
|
|
29
|
+
* console.log(items[0].textContent); // "Hello"
|
|
30
|
+
* ```
|
|
31
|
+
*/
|
|
32
|
+
const parseXmlString = (xml) => {
|
|
33
|
+
const parser = new xmldom_1.DOMParser();
|
|
34
|
+
return parser.parseFromString(xml, "text/xml");
|
|
35
|
+
};
|
|
36
|
+
exports.parseXmlString = parseXmlString;
|
|
37
|
+
/**
|
|
38
|
+
* Gets all elements with a specific tag name and returns them as an array.
|
|
39
|
+
*
|
|
40
|
+
* This is a convenience wrapper around the DOM API's getElementsByTagName method
|
|
41
|
+
* that converts the HTMLCollection/NodeList to a proper JavaScript array for easier manipulation.
|
|
42
|
+
*
|
|
43
|
+
* @param element - The element or document to search within
|
|
44
|
+
* @param tagName - The tag name to search for (e.g., 'w:t', 'w:p', 'item')
|
|
45
|
+
* @returns An array of matching elements (empty array if none found)
|
|
46
|
+
* @example
|
|
47
|
+
* ```typescript
|
|
48
|
+
* const paragraphs = getElementsByTagName(doc, 'w:p');
|
|
49
|
+
* paragraphs.forEach(p => console.log(p.textContent));
|
|
50
|
+
* ```
|
|
51
|
+
*/
|
|
52
|
+
const getElementsByTagName = (element, tagName) => {
|
|
53
|
+
return Array.from(element.getElementsByTagName(tagName));
|
|
54
|
+
};
|
|
55
|
+
exports.getElementsByTagName = getElementsByTagName;
|
|
56
|
+
/**
|
|
57
|
+
* Gets direct child elements with a specific tag name.
|
|
58
|
+
* Unlike getElementsByTagName, this does not search recursively.
|
|
59
|
+
*
|
|
60
|
+
* @param parent - The parent element
|
|
61
|
+
* @param tagName - The tag name to search for
|
|
62
|
+
* @returns An array of matching direct child elements
|
|
63
|
+
*/
|
|
64
|
+
const getDirectChildren = (parent, tagName) => {
|
|
65
|
+
const result = [];
|
|
66
|
+
if (!parent.childNodes)
|
|
67
|
+
return result;
|
|
68
|
+
for (let i = 0; i < parent.childNodes.length; i++) {
|
|
69
|
+
const child = parent.childNodes[i];
|
|
70
|
+
if (child.nodeType === 1 && child.tagName === tagName) { // 1 = ELEMENT_NODE
|
|
71
|
+
result.push(child);
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
return result;
|
|
75
|
+
};
|
|
76
|
+
exports.getDirectChildren = getDirectChildren;
|
|
77
|
+
/**
|
|
78
|
+
* Parses OOXML document metadata from the docProps/core.xml file.
|
|
79
|
+
*
|
|
80
|
+
* OOXML documents (DOCX, XLSX, PPTX) store metadata in a standard location:
|
|
81
|
+
* `docProps/core.xml` within the ZIP archive.
|
|
82
|
+
*
|
|
83
|
+
* This file follows the Dublin Core metadata standard with OOXML-specific extensions.
|
|
84
|
+
* Common metadata elements:
|
|
85
|
+
* - dc:title - Document title
|
|
86
|
+
* - dc:creator - Original author
|
|
87
|
+
* - cp:lastModifiedBy - User who last modified the document
|
|
88
|
+
* - dcterms:created - Creation timestamp
|
|
89
|
+
* - dcterms:modified - Last modification timestamp
|
|
90
|
+
*
|
|
91
|
+
* @param xmlContent - The raw XML content string from docProps/core.xml
|
|
92
|
+
* @returns An OfficeMetadata object with extracted properties (empty object if parsing fails)
|
|
93
|
+
* @example
|
|
94
|
+
* ```typescript
|
|
95
|
+
* const coreXml = files.find(f => f.path === 'docProps/core.xml').content.toString();
|
|
96
|
+
* const metadata = parseOfficeMetadata(coreXml);
|
|
97
|
+
*
|
|
98
|
+
* console.log(metadata.author); // "John Smith"
|
|
99
|
+
* console.log(metadata.title); // "Annual Report"
|
|
100
|
+
* console.log(metadata.created); // Date object
|
|
101
|
+
* ```
|
|
102
|
+
*
|
|
103
|
+
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
|
|
104
|
+
*/
|
|
105
|
+
const parseOfficeMetadata = (xmlContent) => {
|
|
106
|
+
// Step 1: Parse the XML content into a DOM document
|
|
107
|
+
const xml = (0, exports.parseXmlString)(xmlContent);
|
|
108
|
+
const metadata = {};
|
|
109
|
+
// Check for OOXML Core Properties
|
|
110
|
+
const coreProperties = (0, exports.getElementsByTagName)(xml, "cp:coreProperties")[0];
|
|
111
|
+
if (coreProperties) {
|
|
112
|
+
// Step 3: Extract title (Dublin Core element)
|
|
113
|
+
const title = (0, exports.getElementsByTagName)(coreProperties, "dc:title")[0];
|
|
114
|
+
if (title && title.textContent)
|
|
115
|
+
metadata.title = title.textContent;
|
|
116
|
+
// Step 4: Extract author/creator (Dublin Core element)
|
|
117
|
+
const author = (0, exports.getElementsByTagName)(coreProperties, "dc:creator")[0];
|
|
118
|
+
if (author && author.textContent)
|
|
119
|
+
metadata.author = author.textContent;
|
|
120
|
+
// Step 5: Extract last modifier (OOXML Core Properties element)
|
|
121
|
+
const lastModifiedBy = (0, exports.getElementsByTagName)(coreProperties, "cp:lastModifiedBy")[0];
|
|
122
|
+
if (lastModifiedBy && lastModifiedBy.textContent)
|
|
123
|
+
metadata.lastModifiedBy = lastModifiedBy.textContent;
|
|
124
|
+
// Step 6: Extract creation date (Dublin Core Terms element)
|
|
125
|
+
const created = (0, exports.getElementsByTagName)(coreProperties, "dcterms:created")[0];
|
|
126
|
+
if (created && created.textContent)
|
|
127
|
+
metadata.created = new Date(created.textContent);
|
|
128
|
+
// Step 7: Extract last modification date (Dublin Core Terms element)
|
|
129
|
+
const modified = (0, exports.getElementsByTagName)(coreProperties, "dcterms:modified")[0];
|
|
130
|
+
if (modified && modified.textContent)
|
|
131
|
+
metadata.modified = new Date(modified.textContent);
|
|
132
|
+
return metadata;
|
|
133
|
+
}
|
|
134
|
+
// Check for ODF Meta
|
|
135
|
+
const officeMeta = (0, exports.getElementsByTagName)(xml, "office:meta")[0];
|
|
136
|
+
if (officeMeta) {
|
|
137
|
+
const title = (0, exports.getElementsByTagName)(officeMeta, "dc:title")[0];
|
|
138
|
+
if (title && title.textContent)
|
|
139
|
+
metadata.title = title.textContent;
|
|
140
|
+
const author = (0, exports.getElementsByTagName)(officeMeta, "dc:creator")[0];
|
|
141
|
+
if (author && author.textContent)
|
|
142
|
+
metadata.author = author.textContent;
|
|
143
|
+
const description = (0, exports.getElementsByTagName)(officeMeta, "dc:description")[0];
|
|
144
|
+
if (description && description.textContent)
|
|
145
|
+
metadata.description = description.textContent;
|
|
146
|
+
const subject = (0, exports.getElementsByTagName)(officeMeta, "dc:subject")[0];
|
|
147
|
+
if (subject && subject.textContent)
|
|
148
|
+
metadata.subject = subject.textContent;
|
|
149
|
+
const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
|
|
150
|
+
if (created && created.textContent)
|
|
151
|
+
metadata.created = new Date(created.textContent);
|
|
152
|
+
const modified = (0, exports.getElementsByTagName)(officeMeta, "dc:date")[0];
|
|
153
|
+
if (modified && modified.textContent)
|
|
154
|
+
metadata.modified = new Date(modified.textContent);
|
|
155
|
+
}
|
|
156
|
+
return metadata;
|
|
157
|
+
};
|
|
158
|
+
exports.parseOfficeMetadata = parseOfficeMetadata;
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ZIP Archive Extraction Utilities
|
|
3
|
+
*
|
|
4
|
+
* Provides functions for extracting files from ZIP archives.
|
|
5
|
+
* Essential for parsing OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) files,
|
|
6
|
+
* which are all ZIP archives containing XML and media files.
|
|
7
|
+
*
|
|
8
|
+
* Office File Structure:
|
|
9
|
+
* - DOCX: ZIP containing word/document.xml, word/styles.xml, word/media/*, etc.
|
|
10
|
+
* - XLSX: ZIP containing xl/workbook.xml, xl/worksheets/sheet1.xml, etc.
|
|
11
|
+
* - PPTX: ZIP containing ppt/slides/slide1.xml, ppt/media/*, etc.
|
|
12
|
+
* - ODF: Similar structure with content.xml, styles.xml, etc.
|
|
13
|
+
*
|
|
14
|
+
* @module zipUtils
|
|
15
|
+
*/
|
|
16
|
+
/// <reference types="node" />
|
|
17
|
+
/**
|
|
18
|
+
* Represents a file extracted from a ZIP archive.
|
|
19
|
+
* Contains the file's path within the archive and its content as a Buffer.
|
|
20
|
+
*/
|
|
21
|
+
interface ZipFileContent {
|
|
22
|
+
/**
|
|
23
|
+
* The relative path of the file within the ZIP archive.
|
|
24
|
+
* @example "word/document.xml", "xl/worksheets/sheet1.xml", "ppt/slides/slide1.xml"
|
|
25
|
+
*/
|
|
26
|
+
path: string;
|
|
27
|
+
/**
|
|
28
|
+
* The file content as a Node.js Buffer.
|
|
29
|
+
* Can be converted to string for XML files or used directly for binary files (images, etc.).
|
|
30
|
+
* @example Buffer containing XML text or binary image data
|
|
31
|
+
*/
|
|
32
|
+
content: Buffer;
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* Extracts files from a ZIP archive with optional filtering.
|
|
36
|
+
*
|
|
37
|
+
* This function:
|
|
38
|
+
* 1. Opens the ZIP archive from a Buffer
|
|
39
|
+
* 2. Iterates through all entries in the archive
|
|
40
|
+
* 3. Applies a filter function to determine which files to extract
|
|
41
|
+
* 4. Extracts matching files and returns them as an array
|
|
42
|
+
*
|
|
43
|
+
* Uses lazy entry reading for better memory efficiency with large archives.
|
|
44
|
+
* Files are extracted asynchronously and collected into an array.
|
|
45
|
+
*
|
|
46
|
+
* @param zipInput - The ZIP file as a Node.js Buffer
|
|
47
|
+
* @param filterFn - A predicate function to determine which files to extract.
|
|
48
|
+
* Receives the filename and returns true to extract, false to skip.
|
|
49
|
+
* @returns A promise resolving to an array of extracted files
|
|
50
|
+
* @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
|
|
51
|
+
*
|
|
52
|
+
* @example
|
|
53
|
+
* ```typescript
|
|
54
|
+
* // Extract only XML files from a DOCX
|
|
55
|
+
* const files = await extractFiles(docxBuffer, (fileName) => fileName.endsWith('.xml'));
|
|
56
|
+
*
|
|
57
|
+
* // Extract document.xml specifically
|
|
58
|
+
* const files = await extractFiles(docxBuffer, (fileName) =>
|
|
59
|
+
* fileName === 'word/document.xml'
|
|
60
|
+
* );
|
|
61
|
+
*
|
|
62
|
+
* // Extract all files
|
|
63
|
+
* const allFiles = await extractFiles(zipBuffer, () => true);
|
|
64
|
+
*
|
|
65
|
+
* // Extract everything except media files
|
|
66
|
+
* const files = await extractFiles(zipBuffer, (fileName) =>
|
|
67
|
+
* !fileName.startsWith('word/media/')
|
|
68
|
+
* );
|
|
69
|
+
* ```
|
|
70
|
+
*
|
|
71
|
+
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
72
|
+
*/
|
|
73
|
+
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean) => Promise<ZipFileContent[]>;
|
|
74
|
+
export {};
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* ZIP Archive Extraction Utilities
|
|
4
|
+
*
|
|
5
|
+
* Provides functions for extracting files from ZIP archives.
|
|
6
|
+
* Essential for parsing OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) files,
|
|
7
|
+
* which are all ZIP archives containing XML and media files.
|
|
8
|
+
*
|
|
9
|
+
* Office File Structure:
|
|
10
|
+
* - DOCX: ZIP containing word/document.xml, word/styles.xml, word/media/*, etc.
|
|
11
|
+
* - XLSX: ZIP containing xl/workbook.xml, xl/worksheets/sheet1.xml, etc.
|
|
12
|
+
* - PPTX: ZIP containing ppt/slides/slide1.xml, ppt/media/*, etc.
|
|
13
|
+
* - ODF: Similar structure with content.xml, styles.xml, etc.
|
|
14
|
+
*
|
|
15
|
+
* @module zipUtils
|
|
16
|
+
*/
|
|
17
|
+
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
18
|
+
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
19
|
+
};
|
|
20
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
21
|
+
exports.extractFiles = void 0;
|
|
22
|
+
const yauzl_1 = __importDefault(require("yauzl"));
|
|
23
|
+
const concat_stream_1 = __importDefault(require("concat-stream"));
|
|
24
|
+
/**
|
|
25
|
+
* Extracts files from a ZIP archive with optional filtering.
|
|
26
|
+
*
|
|
27
|
+
* This function:
|
|
28
|
+
* 1. Opens the ZIP archive from a Buffer
|
|
29
|
+
* 2. Iterates through all entries in the archive
|
|
30
|
+
* 3. Applies a filter function to determine which files to extract
|
|
31
|
+
* 4. Extracts matching files and returns them as an array
|
|
32
|
+
*
|
|
33
|
+
* Uses lazy entry reading for better memory efficiency with large archives.
|
|
34
|
+
* Files are extracted asynchronously and collected into an array.
|
|
35
|
+
*
|
|
36
|
+
* @param zipInput - The ZIP file as a Node.js Buffer
|
|
37
|
+
* @param filterFn - A predicate function to determine which files to extract.
|
|
38
|
+
* Receives the filename and returns true to extract, false to skip.
|
|
39
|
+
* @returns A promise resolving to an array of extracted files
|
|
40
|
+
* @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
|
|
41
|
+
*
|
|
42
|
+
* @example
|
|
43
|
+
* ```typescript
|
|
44
|
+
* // Extract only XML files from a DOCX
|
|
45
|
+
* const files = await extractFiles(docxBuffer, (fileName) => fileName.endsWith('.xml'));
|
|
46
|
+
*
|
|
47
|
+
* // Extract document.xml specifically
|
|
48
|
+
* const files = await extractFiles(docxBuffer, (fileName) =>
|
|
49
|
+
* fileName === 'word/document.xml'
|
|
50
|
+
* );
|
|
51
|
+
*
|
|
52
|
+
* // Extract all files
|
|
53
|
+
* const allFiles = await extractFiles(zipBuffer, () => true);
|
|
54
|
+
*
|
|
55
|
+
* // Extract everything except media files
|
|
56
|
+
* const files = await extractFiles(zipBuffer, (fileName) =>
|
|
57
|
+
* !fileName.startsWith('word/media/')
|
|
58
|
+
* );
|
|
59
|
+
* ```
|
|
60
|
+
*
|
|
61
|
+
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
62
|
+
*/
|
|
63
|
+
const extractFiles = (zipInput, filterFn) => {
|
|
64
|
+
return new Promise((resolve, reject) => {
|
|
65
|
+
// Step 1: Open the ZIP archive from the buffer
|
|
66
|
+
// lazyEntries: true means we manually control when to read each entry (better memory usage)
|
|
67
|
+
yauzl_1.default.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
|
|
68
|
+
if (err)
|
|
69
|
+
return reject(err);
|
|
70
|
+
if (!zipfile)
|
|
71
|
+
return reject(new Error("Failed to open zip file"));
|
|
72
|
+
// Array to collect all extracted files
|
|
73
|
+
const extractedFiles = [];
|
|
74
|
+
// Step 2: Start reading the first entry
|
|
75
|
+
// This triggers the 'entry' event
|
|
76
|
+
zipfile.readEntry();
|
|
77
|
+
// Step 3: Handle each entry (file or directory) in the ZIP
|
|
78
|
+
zipfile.on('entry', (entry) => {
|
|
79
|
+
// Step 3a: Check if this file should be extracted using the filter function
|
|
80
|
+
if (filterFn(entry.fileName)) {
|
|
81
|
+
// Step 3b: Open a read stream for this entry
|
|
82
|
+
zipfile.openReadStream(entry, (err, readStream) => {
|
|
83
|
+
if (err)
|
|
84
|
+
return reject(err);
|
|
85
|
+
if (!readStream)
|
|
86
|
+
return reject(new Error("Failed to open read stream"));
|
|
87
|
+
// Step 3c: Pipe the stream through concat to collect all data into a single Buffer
|
|
88
|
+
// This is necessary because streams deliver data in chunks
|
|
89
|
+
readStream.pipe((0, concat_stream_1.default)((data) => {
|
|
90
|
+
// Step 3d: Add the extracted file to our results
|
|
91
|
+
extractedFiles.push({
|
|
92
|
+
path: entry.fileName,
|
|
93
|
+
content: data
|
|
94
|
+
});
|
|
95
|
+
// Step 3e: Continue to the next entry
|
|
96
|
+
zipfile.readEntry();
|
|
97
|
+
}));
|
|
98
|
+
});
|
|
99
|
+
}
|
|
100
|
+
else {
|
|
101
|
+
// Step 3f: Skip this entry and move to the next one
|
|
102
|
+
zipfile.readEntry();
|
|
103
|
+
}
|
|
104
|
+
});
|
|
105
|
+
// Step 4: All entries have been processed
|
|
106
|
+
zipfile.on('end', () => resolve(extractedFiles));
|
|
107
|
+
// Step 5: Handle any errors during extraction
|
|
108
|
+
zipfile.on('error', reject);
|
|
109
|
+
});
|
|
110
|
+
});
|
|
111
|
+
};
|
|
112
|
+
exports.extractFiles = extractFiles;
|
package/package.json
CHANGED
|
@@ -1,22 +1,32 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "
|
|
4
|
-
"description": "A Node.js
|
|
5
|
-
"main": "
|
|
3
|
+
"version": "6.0.1",
|
|
4
|
+
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf) into structured AST with rich metadata, formatting, and attachment support.",
|
|
5
|
+
"main": "dist/index.js",
|
|
6
|
+
"types": "dist/index.d.ts",
|
|
7
|
+
"sideEffects": false,
|
|
8
|
+
"engines": {
|
|
9
|
+
"node": ">=18.0.0"
|
|
10
|
+
},
|
|
6
11
|
"files": [
|
|
7
|
-
"
|
|
8
|
-
"typings/officeParser.d.ts",
|
|
9
|
-
"pdfjs-dist-build/*"
|
|
12
|
+
"dist"
|
|
10
13
|
],
|
|
11
|
-
"types": "typings/officeParser.d.ts",
|
|
12
14
|
"scripts": {
|
|
13
|
-
"
|
|
15
|
+
"build": "tsc && node build_browser.js && npm run sync:docs",
|
|
16
|
+
"build:node": "tsc",
|
|
17
|
+
"build:browser": "node build_browser.js && npm run sync:docs",
|
|
18
|
+
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.js docs/dist/",
|
|
19
|
+
"test": "npm run test:clean && npm run build && npx tsx test/testOfficeParser.ts",
|
|
20
|
+
"test:clean": "rm -rf test/results",
|
|
21
|
+
"clean": "rm -rf dist && npm run test:clean",
|
|
22
|
+
"prepublishOnly": "npm run build",
|
|
23
|
+
"prepare": "husky"
|
|
14
24
|
},
|
|
15
25
|
"repository": {
|
|
16
26
|
"type": "git",
|
|
17
27
|
"url": "git+https://github.com/harshankur/officeParser.git"
|
|
18
28
|
},
|
|
19
|
-
"bin": "
|
|
29
|
+
"bin": "dist/index.js",
|
|
20
30
|
"keywords": [
|
|
21
31
|
"office",
|
|
22
32
|
"docx",
|
|
@@ -26,15 +36,28 @@
|
|
|
26
36
|
"odp",
|
|
27
37
|
"ods",
|
|
28
38
|
"pdf",
|
|
39
|
+
"rtf",
|
|
29
40
|
"parser",
|
|
30
|
-
"text",
|
|
31
|
-
"
|
|
32
|
-
"document",
|
|
41
|
+
"text extraction",
|
|
42
|
+
"document parser",
|
|
33
43
|
"word",
|
|
34
44
|
"excel",
|
|
35
|
-
"worksheet",
|
|
36
45
|
"powerpoint",
|
|
37
|
-
"
|
|
46
|
+
"spreadsheet",
|
|
47
|
+
"presentation",
|
|
48
|
+
"slides",
|
|
49
|
+
"ast",
|
|
50
|
+
"ocr",
|
|
51
|
+
"typescript",
|
|
52
|
+
"browser",
|
|
53
|
+
"metadata",
|
|
54
|
+
"formatting",
|
|
55
|
+
"attachments",
|
|
56
|
+
"tesseract",
|
|
57
|
+
"pdf.js",
|
|
58
|
+
"structured-data",
|
|
59
|
+
"openoffice",
|
|
60
|
+
"libreoffice"
|
|
38
61
|
],
|
|
39
62
|
"author": "Harsh Ankur",
|
|
40
63
|
"license": "MIT",
|
|
@@ -46,8 +69,8 @@
|
|
|
46
69
|
"@xmldom/xmldom": "^0.8.10",
|
|
47
70
|
"concat-stream": "^2.0.0",
|
|
48
71
|
"file-type": "^16.5.4",
|
|
49
|
-
"
|
|
50
|
-
"
|
|
72
|
+
"pdfjs-dist": "5.4.530",
|
|
73
|
+
"tesseract.js": "^6.0.0",
|
|
51
74
|
"yauzl": "^3.1.3"
|
|
52
75
|
},
|
|
53
76
|
"devDependencies": {
|
|
@@ -55,6 +78,10 @@
|
|
|
55
78
|
"@types/node": "^18.16.1",
|
|
56
79
|
"@types/xmldom": "^0.1.33",
|
|
57
80
|
"@types/yauzl": "^2.10.3",
|
|
81
|
+
"esbuild": "^0.27.0",
|
|
82
|
+
"esbuild-plugin-polyfill-node": "^0.3.0",
|
|
83
|
+
"husky": "^9.1.7",
|
|
84
|
+
"tsx": "^4.21.0",
|
|
58
85
|
"typescript": "^5.0.3"
|
|
59
86
|
}
|
|
60
|
-
}
|
|
87
|
+
}
|