officeparser 5.2.1 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,61 @@
1
+ "use strict";
2
+ /**
3
+ * OCR (Optical Character Recognition) Utilities
4
+ *
5
+ * This module provides functions for extracting text from images using Tesseract.js.
6
+ * Used when `config.ocr` is enabled to extract text from embedded images in documents.
7
+ *
8
+ * @module ocrUtils
9
+ */
10
+ Object.defineProperty(exports, "__esModule", { value: true });
11
+ exports.performOcr = void 0;
12
+ const tesseract_js_1 = require("tesseract.js");
13
+ /**
14
+ * Performs Optical Character Recognition (OCR) on an image to extract text.
15
+ *
16
+ * Uses Tesseract.js to recognize text in the provided image buffer.
17
+ * This is useful for extracting text from screenshots, scanned documents,
18
+ * charts with labels, or any image containing text.
19
+ *
20
+ * The function creates a new Tesseract worker, processes the image,
21
+ * and properly terminates the worker to free resources.
22
+ *
23
+ * @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
24
+ * @param language - The language code for OCR (default: 'eng' for English).
25
+ * Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
26
+ * Multiple languages can be combined with '+': 'eng+fra'
27
+ * @returns A promise that resolves to the recognized text as a string
28
+ * @throws {Error} If the image cannot be processed or Tesseract initialization fails
29
+ *
30
+ * @example
31
+ * ```typescript
32
+ * // Extract text from an English image
33
+ * const text = await performOcr(imageBuffer, 'eng');
34
+ * console.log(text); // "Annual Revenue: $1.2M"
35
+ *
36
+ * // Extract text from a multilingual image
37
+ * const text = await performOcr(imageBuffer, 'eng+spa');
38
+ * ```
39
+ *
40
+ * @see https://github.com/naptha/tesseract.js for supported languages and options
41
+ */
42
+ const performOcr = async (image, language = 'eng') => {
43
+ // Step 1: Create a Tesseract worker with the specified language
44
+ // We pass 1 for OEM (LSTM) and a silent logger to suppress console output
45
+ const worker = await (0, tesseract_js_1.createWorker)(language, 1, {
46
+ logger: () => { }
47
+ });
48
+ // Step 2: Prepare image data
49
+ let inputImage = image;
50
+ // In browser environment, convert Buffer to Blob for better compatibility
51
+ // @ts-ignore
52
+ if (typeof window !== 'undefined' && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
53
+ inputImage = new Blob([image], { type: 'image/bmp' });
54
+ }
55
+ // Step 3: Perform OCR
56
+ const ret = await worker.recognize(inputImage);
57
+ // Step 4: Terminate worker
58
+ await worker.terminate();
59
+ return ret.data.text;
60
+ };
61
+ exports.performOcr = performOcr;
@@ -0,0 +1,83 @@
1
+ /**
2
+ * XML Parsing Utilities
3
+ *
4
+ * Provides helper functions for parsing and navigating XML documents.
5
+ * Used extensively by OOXML parsers (DOCX, XLSX, PPTX) and OpenOffice parsers (ODT, ODP, ODS).
6
+ *
7
+ * OOXML (Office Open XML) is an XML-based format used by Microsoft Office.
8
+ * Documents are ZIP archives containing multiple XML files describing structure, content, and formatting.
9
+ *
10
+ * @module xmlUtils
11
+ */
12
+ import { OfficeMetadata } from '../types';
13
+ /**
14
+ * Parses an XML string into a DOM Document object.
15
+ *
16
+ * Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
17
+ * This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
18
+ *
19
+ * @param xml - The XML content as a string
20
+ * @returns A Document object that can be queried using standard DOM methods
21
+ * @example
22
+ * ```typescript
23
+ * const xmlString = '<root><item>Hello</item></root>';
24
+ * const doc = parseXmlString(xmlString);
25
+ * const items = doc.getElementsByTagName('item');
26
+ * console.log(items[0].textContent); // "Hello"
27
+ * ```
28
+ */
29
+ export declare const parseXmlString: (xml: string) => Document;
30
+ /**
31
+ * Gets all elements with a specific tag name and returns them as an array.
32
+ *
33
+ * This is a convenience wrapper around the DOM API's getElementsByTagName method
34
+ * that converts the HTMLCollection/NodeList to a proper JavaScript array for easier manipulation.
35
+ *
36
+ * @param element - The element or document to search within
37
+ * @param tagName - The tag name to search for (e.g., 'w:t', 'w:p', 'item')
38
+ * @returns An array of matching elements (empty array if none found)
39
+ * @example
40
+ * ```typescript
41
+ * const paragraphs = getElementsByTagName(doc, 'w:p');
42
+ * paragraphs.forEach(p => console.log(p.textContent));
43
+ * ```
44
+ */
45
+ export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
46
+ /**
47
+ * Gets direct child elements with a specific tag name.
48
+ * Unlike getElementsByTagName, this does not search recursively.
49
+ *
50
+ * @param parent - The parent element
51
+ * @param tagName - The tag name to search for
52
+ * @returns An array of matching direct child elements
53
+ */
54
+ export declare const getDirectChildren: (parent: Element, tagName: string) => Element[];
55
+ /**
56
+ * Parses OOXML document metadata from the docProps/core.xml file.
57
+ *
58
+ * OOXML documents (DOCX, XLSX, PPTX) store metadata in a standard location:
59
+ * `docProps/core.xml` within the ZIP archive.
60
+ *
61
+ * This file follows the Dublin Core metadata standard with OOXML-specific extensions.
62
+ * Common metadata elements:
63
+ * - dc:title - Document title
64
+ * - dc:creator - Original author
65
+ * - cp:lastModifiedBy - User who last modified the document
66
+ * - dcterms:created - Creation timestamp
67
+ * - dcterms:modified - Last modification timestamp
68
+ *
69
+ * @param xmlContent - The raw XML content string from docProps/core.xml
70
+ * @returns An OfficeMetadata object with extracted properties (empty object if parsing fails)
71
+ * @example
72
+ * ```typescript
73
+ * const coreXml = files.find(f => f.path === 'docProps/core.xml').content.toString();
74
+ * const metadata = parseOfficeMetadata(coreXml);
75
+ *
76
+ * console.log(metadata.author); // "John Smith"
77
+ * console.log(metadata.title); // "Annual Report"
78
+ * console.log(metadata.created); // Date object
79
+ * ```
80
+ *
81
+ * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
82
+ */
83
+ export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata;
@@ -0,0 +1,158 @@
1
+ "use strict";
2
+ /**
3
+ * XML Parsing Utilities
4
+ *
5
+ * Provides helper functions for parsing and navigating XML documents.
6
+ * Used extensively by OOXML parsers (DOCX, XLSX, PPTX) and OpenOffice parsers (ODT, ODP, ODS).
7
+ *
8
+ * OOXML (Office Open XML) is an XML-based format used by Microsoft Office.
9
+ * Documents are ZIP archives containing multiple XML files describing structure, content, and formatting.
10
+ *
11
+ * @module xmlUtils
12
+ */
13
+ Object.defineProperty(exports, "__esModule", { value: true });
14
+ exports.parseOfficeMetadata = exports.getDirectChildren = exports.getElementsByTagName = exports.parseXmlString = void 0;
15
+ const xmldom_1 = require("@xmldom/xmldom");
16
+ /**
17
+ * Parses an XML string into a DOM Document object.
18
+ *
19
+ * Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
20
+ * This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
21
+ *
22
+ * @param xml - The XML content as a string
23
+ * @returns A Document object that can be queried using standard DOM methods
24
+ * @example
25
+ * ```typescript
26
+ * const xmlString = '<root><item>Hello</item></root>';
27
+ * const doc = parseXmlString(xmlString);
28
+ * const items = doc.getElementsByTagName('item');
29
+ * console.log(items[0].textContent); // "Hello"
30
+ * ```
31
+ */
32
+ const parseXmlString = (xml) => {
33
+ const parser = new xmldom_1.DOMParser();
34
+ return parser.parseFromString(xml, "text/xml");
35
+ };
36
+ exports.parseXmlString = parseXmlString;
37
+ /**
38
+ * Gets all elements with a specific tag name and returns them as an array.
39
+ *
40
+ * This is a convenience wrapper around the DOM API's getElementsByTagName method
41
+ * that converts the HTMLCollection/NodeList to a proper JavaScript array for easier manipulation.
42
+ *
43
+ * @param element - The element or document to search within
44
+ * @param tagName - The tag name to search for (e.g., 'w:t', 'w:p', 'item')
45
+ * @returns An array of matching elements (empty array if none found)
46
+ * @example
47
+ * ```typescript
48
+ * const paragraphs = getElementsByTagName(doc, 'w:p');
49
+ * paragraphs.forEach(p => console.log(p.textContent));
50
+ * ```
51
+ */
52
+ const getElementsByTagName = (element, tagName) => {
53
+ return Array.from(element.getElementsByTagName(tagName));
54
+ };
55
+ exports.getElementsByTagName = getElementsByTagName;
56
+ /**
57
+ * Gets direct child elements with a specific tag name.
58
+ * Unlike getElementsByTagName, this does not search recursively.
59
+ *
60
+ * @param parent - The parent element
61
+ * @param tagName - The tag name to search for
62
+ * @returns An array of matching direct child elements
63
+ */
64
+ const getDirectChildren = (parent, tagName) => {
65
+ const result = [];
66
+ if (!parent.childNodes)
67
+ return result;
68
+ for (let i = 0; i < parent.childNodes.length; i++) {
69
+ const child = parent.childNodes[i];
70
+ if (child.nodeType === 1 && child.tagName === tagName) { // 1 = ELEMENT_NODE
71
+ result.push(child);
72
+ }
73
+ }
74
+ return result;
75
+ };
76
+ exports.getDirectChildren = getDirectChildren;
77
+ /**
78
+ * Parses OOXML document metadata from the docProps/core.xml file.
79
+ *
80
+ * OOXML documents (DOCX, XLSX, PPTX) store metadata in a standard location:
81
+ * `docProps/core.xml` within the ZIP archive.
82
+ *
83
+ * This file follows the Dublin Core metadata standard with OOXML-specific extensions.
84
+ * Common metadata elements:
85
+ * - dc:title - Document title
86
+ * - dc:creator - Original author
87
+ * - cp:lastModifiedBy - User who last modified the document
88
+ * - dcterms:created - Creation timestamp
89
+ * - dcterms:modified - Last modification timestamp
90
+ *
91
+ * @param xmlContent - The raw XML content string from docProps/core.xml
92
+ * @returns An OfficeMetadata object with extracted properties (empty object if parsing fails)
93
+ * @example
94
+ * ```typescript
95
+ * const coreXml = files.find(f => f.path === 'docProps/core.xml').content.toString();
96
+ * const metadata = parseOfficeMetadata(coreXml);
97
+ *
98
+ * console.log(metadata.author); // "John Smith"
99
+ * console.log(metadata.title); // "Annual Report"
100
+ * console.log(metadata.created); // Date object
101
+ * ```
102
+ *
103
+ * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
104
+ */
105
+ const parseOfficeMetadata = (xmlContent) => {
106
+ // Step 1: Parse the XML content into a DOM document
107
+ const xml = (0, exports.parseXmlString)(xmlContent);
108
+ const metadata = {};
109
+ // Check for OOXML Core Properties
110
+ const coreProperties = (0, exports.getElementsByTagName)(xml, "cp:coreProperties")[0];
111
+ if (coreProperties) {
112
+ // Step 3: Extract title (Dublin Core element)
113
+ const title = (0, exports.getElementsByTagName)(coreProperties, "dc:title")[0];
114
+ if (title && title.textContent)
115
+ metadata.title = title.textContent;
116
+ // Step 4: Extract author/creator (Dublin Core element)
117
+ const author = (0, exports.getElementsByTagName)(coreProperties, "dc:creator")[0];
118
+ if (author && author.textContent)
119
+ metadata.author = author.textContent;
120
+ // Step 5: Extract last modifier (OOXML Core Properties element)
121
+ const lastModifiedBy = (0, exports.getElementsByTagName)(coreProperties, "cp:lastModifiedBy")[0];
122
+ if (lastModifiedBy && lastModifiedBy.textContent)
123
+ metadata.lastModifiedBy = lastModifiedBy.textContent;
124
+ // Step 6: Extract creation date (Dublin Core Terms element)
125
+ const created = (0, exports.getElementsByTagName)(coreProperties, "dcterms:created")[0];
126
+ if (created && created.textContent)
127
+ metadata.created = new Date(created.textContent);
128
+ // Step 7: Extract last modification date (Dublin Core Terms element)
129
+ const modified = (0, exports.getElementsByTagName)(coreProperties, "dcterms:modified")[0];
130
+ if (modified && modified.textContent)
131
+ metadata.modified = new Date(modified.textContent);
132
+ return metadata;
133
+ }
134
+ // Check for ODF Meta
135
+ const officeMeta = (0, exports.getElementsByTagName)(xml, "office:meta")[0];
136
+ if (officeMeta) {
137
+ const title = (0, exports.getElementsByTagName)(officeMeta, "dc:title")[0];
138
+ if (title && title.textContent)
139
+ metadata.title = title.textContent;
140
+ const author = (0, exports.getElementsByTagName)(officeMeta, "dc:creator")[0];
141
+ if (author && author.textContent)
142
+ metadata.author = author.textContent;
143
+ const description = (0, exports.getElementsByTagName)(officeMeta, "dc:description")[0];
144
+ if (description && description.textContent)
145
+ metadata.description = description.textContent;
146
+ const subject = (0, exports.getElementsByTagName)(officeMeta, "dc:subject")[0];
147
+ if (subject && subject.textContent)
148
+ metadata.subject = subject.textContent;
149
+ const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
150
+ if (created && created.textContent)
151
+ metadata.created = new Date(created.textContent);
152
+ const modified = (0, exports.getElementsByTagName)(officeMeta, "dc:date")[0];
153
+ if (modified && modified.textContent)
154
+ metadata.modified = new Date(modified.textContent);
155
+ }
156
+ return metadata;
157
+ };
158
+ exports.parseOfficeMetadata = parseOfficeMetadata;
@@ -0,0 +1,74 @@
1
+ /**
2
+ * ZIP Archive Extraction Utilities
3
+ *
4
+ * Provides functions for extracting files from ZIP archives.
5
+ * Essential for parsing OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) files,
6
+ * which are all ZIP archives containing XML and media files.
7
+ *
8
+ * Office File Structure:
9
+ * - DOCX: ZIP containing word/document.xml, word/styles.xml, word/media/*, etc.
10
+ * - XLSX: ZIP containing xl/workbook.xml, xl/worksheets/sheet1.xml, etc.
11
+ * - PPTX: ZIP containing ppt/slides/slide1.xml, ppt/media/*, etc.
12
+ * - ODF: Similar structure with content.xml, styles.xml, etc.
13
+ *
14
+ * @module zipUtils
15
+ */
16
+ /// <reference types="node" />
17
+ /**
18
+ * Represents a file extracted from a ZIP archive.
19
+ * Contains the file's path within the archive and its content as a Buffer.
20
+ */
21
+ interface ZipFileContent {
22
+ /**
23
+ * The relative path of the file within the ZIP archive.
24
+ * @example "word/document.xml", "xl/worksheets/sheet1.xml", "ppt/slides/slide1.xml"
25
+ */
26
+ path: string;
27
+ /**
28
+ * The file content as a Node.js Buffer.
29
+ * Can be converted to string for XML files or used directly for binary files (images, etc.).
30
+ * @example Buffer containing XML text or binary image data
31
+ */
32
+ content: Buffer;
33
+ }
34
+ /**
35
+ * Extracts files from a ZIP archive with optional filtering.
36
+ *
37
+ * This function:
38
+ * 1. Opens the ZIP archive from a Buffer
39
+ * 2. Iterates through all entries in the archive
40
+ * 3. Applies a filter function to determine which files to extract
41
+ * 4. Extracts matching files and returns them as an array
42
+ *
43
+ * Uses lazy entry reading for better memory efficiency with large archives.
44
+ * Files are extracted asynchronously and collected into an array.
45
+ *
46
+ * @param zipInput - The ZIP file as a Node.js Buffer
47
+ * @param filterFn - A predicate function to determine which files to extract.
48
+ * Receives the filename and returns true to extract, false to skip.
49
+ * @returns A promise resolving to an array of extracted files
50
+ * @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
51
+ *
52
+ * @example
53
+ * ```typescript
54
+ * // Extract only XML files from a DOCX
55
+ * const files = await extractFiles(docxBuffer, (fileName) => fileName.endsWith('.xml'));
56
+ *
57
+ * // Extract document.xml specifically
58
+ * const files = await extractFiles(docxBuffer, (fileName) =>
59
+ * fileName === 'word/document.xml'
60
+ * );
61
+ *
62
+ * // Extract all files
63
+ * const allFiles = await extractFiles(zipBuffer, () => true);
64
+ *
65
+ * // Extract everything except media files
66
+ * const files = await extractFiles(zipBuffer, (fileName) =>
67
+ * !fileName.startsWith('word/media/')
68
+ * );
69
+ * ```
70
+ *
71
+ * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
72
+ */
73
+ export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean) => Promise<ZipFileContent[]>;
74
+ export {};
@@ -0,0 +1,112 @@
1
+ "use strict";
2
+ /**
3
+ * ZIP Archive Extraction Utilities
4
+ *
5
+ * Provides functions for extracting files from ZIP archives.
6
+ * Essential for parsing OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) files,
7
+ * which are all ZIP archives containing XML and media files.
8
+ *
9
+ * Office File Structure:
10
+ * - DOCX: ZIP containing word/document.xml, word/styles.xml, word/media/*, etc.
11
+ * - XLSX: ZIP containing xl/workbook.xml, xl/worksheets/sheet1.xml, etc.
12
+ * - PPTX: ZIP containing ppt/slides/slide1.xml, ppt/media/*, etc.
13
+ * - ODF: Similar structure with content.xml, styles.xml, etc.
14
+ *
15
+ * @module zipUtils
16
+ */
17
+ var __importDefault = (this && this.__importDefault) || function (mod) {
18
+ return (mod && mod.__esModule) ? mod : { "default": mod };
19
+ };
20
+ Object.defineProperty(exports, "__esModule", { value: true });
21
+ exports.extractFiles = void 0;
22
+ const yauzl_1 = __importDefault(require("yauzl"));
23
+ const concat_stream_1 = __importDefault(require("concat-stream"));
24
+ /**
25
+ * Extracts files from a ZIP archive with optional filtering.
26
+ *
27
+ * This function:
28
+ * 1. Opens the ZIP archive from a Buffer
29
+ * 2. Iterates through all entries in the archive
30
+ * 3. Applies a filter function to determine which files to extract
31
+ * 4. Extracts matching files and returns them as an array
32
+ *
33
+ * Uses lazy entry reading for better memory efficiency with large archives.
34
+ * Files are extracted asynchronously and collected into an array.
35
+ *
36
+ * @param zipInput - The ZIP file as a Node.js Buffer
37
+ * @param filterFn - A predicate function to determine which files to extract.
38
+ * Receives the filename and returns true to extract, false to skip.
39
+ * @returns A promise resolving to an array of extracted files
40
+ * @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
41
+ *
42
+ * @example
43
+ * ```typescript
44
+ * // Extract only XML files from a DOCX
45
+ * const files = await extractFiles(docxBuffer, (fileName) => fileName.endsWith('.xml'));
46
+ *
47
+ * // Extract document.xml specifically
48
+ * const files = await extractFiles(docxBuffer, (fileName) =>
49
+ * fileName === 'word/document.xml'
50
+ * );
51
+ *
52
+ * // Extract all files
53
+ * const allFiles = await extractFiles(zipBuffer, () => true);
54
+ *
55
+ * // Extract everything except media files
56
+ * const files = await extractFiles(zipBuffer, (fileName) =>
57
+ * !fileName.startsWith('word/media/')
58
+ * );
59
+ * ```
60
+ *
61
+ * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
62
+ */
63
+ const extractFiles = (zipInput, filterFn) => {
64
+ return new Promise((resolve, reject) => {
65
+ // Step 1: Open the ZIP archive from the buffer
66
+ // lazyEntries: true means we manually control when to read each entry (better memory usage)
67
+ yauzl_1.default.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
68
+ if (err)
69
+ return reject(err);
70
+ if (!zipfile)
71
+ return reject(new Error("Failed to open zip file"));
72
+ // Array to collect all extracted files
73
+ const extractedFiles = [];
74
+ // Step 2: Start reading the first entry
75
+ // This triggers the 'entry' event
76
+ zipfile.readEntry();
77
+ // Step 3: Handle each entry (file or directory) in the ZIP
78
+ zipfile.on('entry', (entry) => {
79
+ // Step 3a: Check if this file should be extracted using the filter function
80
+ if (filterFn(entry.fileName)) {
81
+ // Step 3b: Open a read stream for this entry
82
+ zipfile.openReadStream(entry, (err, readStream) => {
83
+ if (err)
84
+ return reject(err);
85
+ if (!readStream)
86
+ return reject(new Error("Failed to open read stream"));
87
+ // Step 3c: Pipe the stream through concat to collect all data into a single Buffer
88
+ // This is necessary because streams deliver data in chunks
89
+ readStream.pipe((0, concat_stream_1.default)((data) => {
90
+ // Step 3d: Add the extracted file to our results
91
+ extractedFiles.push({
92
+ path: entry.fileName,
93
+ content: data
94
+ });
95
+ // Step 3e: Continue to the next entry
96
+ zipfile.readEntry();
97
+ }));
98
+ });
99
+ }
100
+ else {
101
+ // Step 3f: Skip this entry and move to the next one
102
+ zipfile.readEntry();
103
+ }
104
+ });
105
+ // Step 4: All entries have been processed
106
+ zipfile.on('end', () => resolve(extractedFiles));
107
+ // Step 5: Handle any errors during extraction
108
+ zipfile.on('error', reject);
109
+ });
110
+ });
111
+ };
112
+ exports.extractFiles = extractFiles;
package/package.json CHANGED
@@ -1,22 +1,32 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "5.2.1",
4
- "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
- "main": "officeParser.js",
3
+ "version": "6.0.1",
4
+ "description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf) into structured AST with rich metadata, formatting, and attachment support.",
5
+ "main": "dist/index.js",
6
+ "types": "dist/index.d.ts",
7
+ "sideEffects": false,
8
+ "engines": {
9
+ "node": ">=18.0.0"
10
+ },
6
11
  "files": [
7
- "officeParser.js",
8
- "typings/officeParser.d.ts",
9
- "pdfjs-dist-build/*"
12
+ "dist"
10
13
  ],
11
- "types": "typings/officeParser.d.ts",
12
14
  "scripts": {
13
- "test": "node test/testOfficeParser.js"
15
+ "build": "tsc && node build_browser.js && npm run sync:docs",
16
+ "build:node": "tsc",
17
+ "build:browser": "node build_browser.js && npm run sync:docs",
18
+ "sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.js docs/dist/",
19
+ "test": "npm run test:clean && npm run build && npx tsx test/testOfficeParser.ts",
20
+ "test:clean": "rm -rf test/results",
21
+ "clean": "rm -rf dist && npm run test:clean",
22
+ "prepublishOnly": "npm run build",
23
+ "prepare": "husky"
14
24
  },
15
25
  "repository": {
16
26
  "type": "git",
17
27
  "url": "git+https://github.com/harshankur/officeParser.git"
18
28
  },
19
- "bin": "officeParser.js",
29
+ "bin": "dist/index.js",
20
30
  "keywords": [
21
31
  "office",
22
32
  "docx",
@@ -26,15 +36,28 @@
26
36
  "odp",
27
37
  "ods",
28
38
  "pdf",
39
+ "rtf",
29
40
  "parser",
30
- "text",
31
- "extract text",
32
- "document",
41
+ "text extraction",
42
+ "document parser",
33
43
  "word",
34
44
  "excel",
35
- "worksheet",
36
45
  "powerpoint",
37
- "slides"
46
+ "spreadsheet",
47
+ "presentation",
48
+ "slides",
49
+ "ast",
50
+ "ocr",
51
+ "typescript",
52
+ "browser",
53
+ "metadata",
54
+ "formatting",
55
+ "attachments",
56
+ "tesseract",
57
+ "pdf.js",
58
+ "structured-data",
59
+ "openoffice",
60
+ "libreoffice"
38
61
  ],
39
62
  "author": "Harsh Ankur",
40
63
  "license": "MIT",
@@ -46,8 +69,8 @@
46
69
  "@xmldom/xmldom": "^0.8.10",
47
70
  "concat-stream": "^2.0.0",
48
71
  "file-type": "^16.5.4",
49
- "node-ensure": "^0.0.0",
50
- "pdfjs-dist": "^5.3.31",
72
+ "pdfjs-dist": "5.4.530",
73
+ "tesseract.js": "^6.0.0",
51
74
  "yauzl": "^3.1.3"
52
75
  },
53
76
  "devDependencies": {
@@ -55,6 +78,10 @@
55
78
  "@types/node": "^18.16.1",
56
79
  "@types/xmldom": "^0.1.33",
57
80
  "@types/yauzl": "^2.10.3",
81
+ "esbuild": "^0.27.0",
82
+ "esbuild-plugin-polyfill-node": "^0.3.0",
83
+ "husky": "^9.1.7",
84
+ "tsx": "^4.21.0",
58
85
  "typescript": "^5.0.3"
59
86
  }
60
- }
87
+ }