officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/dist/utils/ocrUtils.js
CHANGED
|
@@ -5,11 +5,176 @@
|
|
|
5
5
|
* This module provides functions for extracting text from images using Tesseract.js.
|
|
6
6
|
* Used when `config.ocr` is enabled to extract text from embedded images in documents.
|
|
7
7
|
*
|
|
8
|
+
* Includes a worker pool via OcrSchedulerManager to improve performance when
|
|
9
|
+
* processing multiple images.
|
|
10
|
+
*
|
|
8
11
|
* @module ocrUtils
|
|
9
12
|
*/
|
|
10
13
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
11
|
-
exports.performOcr = void 0;
|
|
12
|
-
|
|
14
|
+
exports.terminateOcr = exports.performOcr = void 0;
|
|
15
|
+
/**
|
|
16
|
+
* Manages a pool of Tesseract workers with "Smart Affinity".
|
|
17
|
+
*
|
|
18
|
+
* Instead of a simple scheduler, this manager allows workers to persist with
|
|
19
|
+
* a specific language affinity. If a new language is requested and the pool
|
|
20
|
+
* is at capacity, it re-initializes the Least Recently Used (LRU) idle worker
|
|
21
|
+
* rather than resetting the entire pool.
|
|
22
|
+
*
|
|
23
|
+
* Implements lazy loading of tesseract.js to ensure no background processes
|
|
24
|
+
* are spawned unless OCR is explicitly used.
|
|
25
|
+
*/
|
|
26
|
+
class OcrSchedulerManager {
|
|
27
|
+
static instance;
|
|
28
|
+
pool = [];
|
|
29
|
+
queue = [];
|
|
30
|
+
MAX_WORKERS = 4;
|
|
31
|
+
idleTimeout = 10000; // 10s default
|
|
32
|
+
timeoutId = null;
|
|
33
|
+
constructor() { }
|
|
34
|
+
/**
|
|
35
|
+
* Returns the singleton instance of the manager.
|
|
36
|
+
*/
|
|
37
|
+
static getInstance() {
|
|
38
|
+
if (!OcrSchedulerManager.instance) {
|
|
39
|
+
OcrSchedulerManager.instance = new OcrSchedulerManager();
|
|
40
|
+
}
|
|
41
|
+
return OcrSchedulerManager.instance;
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Checks if the singleton instance has been initialized.
|
|
45
|
+
*/
|
|
46
|
+
static hasInstance() {
|
|
47
|
+
return !!OcrSchedulerManager.instance;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Resets the inactivity timer. If the timer reaches its duration,
|
|
51
|
+
* all workers are terminated automatically.
|
|
52
|
+
*/
|
|
53
|
+
resetIdleTimer() {
|
|
54
|
+
if (this.timeoutId) {
|
|
55
|
+
clearTimeout(this.timeoutId);
|
|
56
|
+
}
|
|
57
|
+
if (this.idleTimeout > 0) {
|
|
58
|
+
this.timeoutId = setTimeout(async () => {
|
|
59
|
+
await this.terminate();
|
|
60
|
+
}, this.idleTimeout);
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Performs OCR on an image using the smart worker pool.
|
|
65
|
+
*
|
|
66
|
+
* @param image - Image data (Buffer, string path, or Blob)
|
|
67
|
+
* @param config - OCR configuration (language, custom paths)
|
|
68
|
+
* @returns Recognized text
|
|
69
|
+
*/
|
|
70
|
+
async recognize(image, config) {
|
|
71
|
+
return new Promise((resolve, reject) => {
|
|
72
|
+
// Update idle timeout if provided
|
|
73
|
+
if (config?.autoTerminateTimeout !== undefined) {
|
|
74
|
+
this.idleTimeout = config.autoTerminateTimeout;
|
|
75
|
+
}
|
|
76
|
+
// Reset the inactivity timer every time a new job is requested
|
|
77
|
+
this.resetIdleTimer();
|
|
78
|
+
// Add job to queue and trigger processing
|
|
79
|
+
this.queue.push({ image, config: config || {}, resolve, reject });
|
|
80
|
+
this.processQueue();
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* Attempts to process the next job in the queue using an available worker.
|
|
85
|
+
*/
|
|
86
|
+
async processQueue() {
|
|
87
|
+
if (this.queue.length === 0)
|
|
88
|
+
return;
|
|
89
|
+
const nextJob = this.queue[0];
|
|
90
|
+
const requestedLanguage = nextJob.config.language || 'eng';
|
|
91
|
+
// 1. Find an idle worker with the EXACT language affinity
|
|
92
|
+
let managed = this.pool.find(mw => !mw.isBusy && mw.language === requestedLanguage);
|
|
93
|
+
// 2. If not found and we have room, create a new worker
|
|
94
|
+
if (!managed && this.pool.length < this.MAX_WORKERS) {
|
|
95
|
+
try {
|
|
96
|
+
const { createWorker } = await import('tesseract.js');
|
|
97
|
+
const options = { logger: () => { } };
|
|
98
|
+
if (nextJob.config.workerPath)
|
|
99
|
+
options.workerPath = nextJob.config.workerPath;
|
|
100
|
+
if (nextJob.config.corePath)
|
|
101
|
+
options.corePath = nextJob.config.corePath;
|
|
102
|
+
if (nextJob.config.langPath)
|
|
103
|
+
options.langPath = nextJob.config.langPath;
|
|
104
|
+
const worker = await createWorker(requestedLanguage, 1, options);
|
|
105
|
+
managed = {
|
|
106
|
+
worker,
|
|
107
|
+
language: requestedLanguage,
|
|
108
|
+
lastUsed: Date.now(),
|
|
109
|
+
isBusy: false
|
|
110
|
+
};
|
|
111
|
+
this.pool.push(managed);
|
|
112
|
+
}
|
|
113
|
+
catch (err) {
|
|
114
|
+
const job = this.queue.shift();
|
|
115
|
+
job?.reject(err);
|
|
116
|
+
this.processQueue(); // Try next job
|
|
117
|
+
return;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
// 3. If still not found and we are at capacity, find the LRU idle worker and re-initialize it
|
|
121
|
+
if (!managed) {
|
|
122
|
+
const idleWorkers = this.pool.filter(mw => !mw.isBusy);
|
|
123
|
+
if (idleWorkers.length > 0) {
|
|
124
|
+
// Find Least Recently Used idle worker
|
|
125
|
+
managed = idleWorkers.reduce((prev, curr) => (prev.lastUsed < curr.lastUsed ? prev : curr));
|
|
126
|
+
try {
|
|
127
|
+
// Smart Re-initialization (v5 API)
|
|
128
|
+
await managed.worker.reinitialize(requestedLanguage);
|
|
129
|
+
managed.language = requestedLanguage;
|
|
130
|
+
}
|
|
131
|
+
catch (err) {
|
|
132
|
+
// If reinitialization fails, we might need to recreate it, but for simplicity
|
|
133
|
+
// we'll just fail this job and try another worker next time.
|
|
134
|
+
const job = this.queue.shift();
|
|
135
|
+
job?.reject(err);
|
|
136
|
+
this.processQueue();
|
|
137
|
+
return;
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
// 4. If we have a worker ready, execute the job
|
|
142
|
+
if (managed) {
|
|
143
|
+
const job = this.queue.shift();
|
|
144
|
+
if (!job)
|
|
145
|
+
return;
|
|
146
|
+
managed.isBusy = true;
|
|
147
|
+
managed.lastUsed = Date.now();
|
|
148
|
+
try {
|
|
149
|
+
const { data: { text } } = await managed.worker.recognize(job.image);
|
|
150
|
+
job.resolve(text);
|
|
151
|
+
}
|
|
152
|
+
catch (err) {
|
|
153
|
+
job.reject(err);
|
|
154
|
+
}
|
|
155
|
+
finally {
|
|
156
|
+
managed.isBusy = false;
|
|
157
|
+
managed.lastUsed = Date.now();
|
|
158
|
+
// Check if there are more jobs waiting
|
|
159
|
+
this.processQueue();
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
// If no worker is available (all busy), the job stays in the queue
|
|
163
|
+
// and will be picked up when a worker finishes.
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* Terminates all workers in the pool and resets the state.
|
|
167
|
+
*/
|
|
168
|
+
async terminate() {
|
|
169
|
+
if (this.timeoutId) {
|
|
170
|
+
clearTimeout(this.timeoutId);
|
|
171
|
+
this.timeoutId = null;
|
|
172
|
+
}
|
|
173
|
+
const workersToTerminate = this.pool.map(mw => mw.worker.terminate());
|
|
174
|
+
await Promise.all(workersToTerminate);
|
|
175
|
+
this.pool = [];
|
|
176
|
+
}
|
|
177
|
+
}
|
|
13
178
|
/**
|
|
14
179
|
* Performs Optical Character Recognition (OCR) on an image to extract text.
|
|
15
180
|
*
|
|
@@ -17,45 +182,41 @@ const tesseract_js_1 = require("tesseract.js");
|
|
|
17
182
|
* This is useful for extracting text from screenshots, scanned documents,
|
|
18
183
|
* charts with labels, or any image containing text.
|
|
19
184
|
*
|
|
20
|
-
*
|
|
21
|
-
* and properly terminates the worker to free resources.
|
|
185
|
+
* This function uses a shared worker pool to minimize initialization overhead.
|
|
22
186
|
*
|
|
23
|
-
* @param
|
|
24
|
-
* @param
|
|
25
|
-
* Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
|
|
26
|
-
* Multiple languages can be combined with '+': 'eng+fra'
|
|
187
|
+
* @param image - The image data as a Buffer, file path, or Blob
|
|
188
|
+
* @param config - Optional configuration for language and custom worker paths
|
|
27
189
|
* @returns A promise that resolves to the recognized text as a string
|
|
28
190
|
* @throws {Error} If the image cannot be processed or Tesseract initialization fails
|
|
29
191
|
*
|
|
30
192
|
* @example
|
|
31
193
|
* ```typescript
|
|
32
194
|
* // Extract text from an English image
|
|
33
|
-
* const text = await performOcr(imageBuffer, 'eng');
|
|
34
|
-
* console.log(text); // "Annual Revenue: $1.2M"
|
|
35
|
-
*
|
|
36
|
-
* // Extract text from a multilingual image
|
|
37
|
-
* const text = await performOcr(imageBuffer, 'eng+spa');
|
|
195
|
+
* const text = await performOcr(imageBuffer, { language: 'eng' });
|
|
38
196
|
* ```
|
|
39
197
|
*
|
|
40
198
|
* @see https://github.com/naptha/tesseract.js for supported languages and options
|
|
41
199
|
*/
|
|
42
|
-
const performOcr = async (image,
|
|
43
|
-
//
|
|
44
|
-
// We pass 1 for OEM (LSTM) and a silent logger to suppress console output
|
|
45
|
-
const worker = await (0, tesseract_js_1.createWorker)(language, 1, {
|
|
46
|
-
logger: () => { }
|
|
47
|
-
});
|
|
48
|
-
// Step 2: Prepare image data
|
|
200
|
+
const performOcr = async (image, config) => {
|
|
201
|
+
// Prepare image data
|
|
49
202
|
let inputImage = image;
|
|
50
203
|
// In browser environment, convert Buffer to Blob for better compatibility
|
|
51
204
|
// @ts-ignore
|
|
52
205
|
if (typeof window !== 'undefined' && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
|
|
53
206
|
inputImage = new Blob([image], { type: 'image/bmp' });
|
|
54
207
|
}
|
|
55
|
-
|
|
56
|
-
const ret = await worker.recognize(inputImage);
|
|
57
|
-
// Step 4: Terminate worker
|
|
58
|
-
await worker.terminate();
|
|
59
|
-
return ret.data.text;
|
|
208
|
+
return await OcrSchedulerManager.getInstance().recognize(inputImage, config);
|
|
60
209
|
};
|
|
61
210
|
exports.performOcr = performOcr;
|
|
211
|
+
/**
|
|
212
|
+
* Terminates all OCR workers and cleans up resources.
|
|
213
|
+
*
|
|
214
|
+
* Should be called when the application is shutting down or OCR is no longer needed
|
|
215
|
+
* to prevent memory leaks and dangling worker processes.
|
|
216
|
+
*/
|
|
217
|
+
const terminateOcr = async () => {
|
|
218
|
+
if (OcrSchedulerManager.hasInstance()) {
|
|
219
|
+
await OcrSchedulerManager.getInstance().terminate();
|
|
220
|
+
}
|
|
221
|
+
};
|
|
222
|
+
exports.terminateOcr = terminateOcr;
|
package/dist/utils/xmlUtils.d.ts
CHANGED
|
@@ -10,23 +10,22 @@
|
|
|
10
10
|
* @module xmlUtils
|
|
11
11
|
*/
|
|
12
12
|
import { OfficeMetadata } from '../types';
|
|
13
|
+
/**
|
|
14
|
+
* Type guard for Element nodes.
|
|
15
|
+
*/
|
|
16
|
+
export declare const isElement: (node: Node) => node is Element;
|
|
13
17
|
/**
|
|
14
18
|
* Parses an XML string into a DOM Document object.
|
|
15
19
|
*
|
|
16
20
|
* Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
|
|
17
|
-
* This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
|
|
18
21
|
*
|
|
19
22
|
* @param xml - The XML content as a string
|
|
23
|
+
* @param options - Optional parser settings (e.g., enable locators for source mapping)
|
|
20
24
|
* @returns A Document object that can be queried using standard DOM methods
|
|
21
|
-
* @example
|
|
22
|
-
* ```typescript
|
|
23
|
-
* const xmlString = '<root><item>Hello</item></root>';
|
|
24
|
-
* const doc = parseXmlString(xmlString);
|
|
25
|
-
* const items = doc.getElementsByTagName('item');
|
|
26
|
-
* console.log(items[0].textContent); // "Hello"
|
|
27
|
-
* ```
|
|
28
25
|
*/
|
|
29
|
-
export declare const parseXmlString: (xml: string
|
|
26
|
+
export declare const parseXmlString: (xml: string, options?: {
|
|
27
|
+
locator?: boolean;
|
|
28
|
+
}) => Document;
|
|
30
29
|
/**
|
|
31
30
|
* Gets all elements with a specific tag name and returns them as an array.
|
|
32
31
|
*
|
|
@@ -43,6 +42,54 @@ export declare const parseXmlString: (xml: string) => Document;
|
|
|
43
42
|
* ```
|
|
44
43
|
*/
|
|
45
44
|
export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
|
|
45
|
+
/**
|
|
46
|
+
* Serializes a DOM Node (Document, Element, etc.) back into an XML string.
|
|
47
|
+
* This is cross-platform and works in both Node.js and Browser environments.
|
|
48
|
+
*
|
|
49
|
+
* @param node - The DOM node to serialize
|
|
50
|
+
* @param options - Serialization options
|
|
51
|
+
* @returns The XML string representation
|
|
52
|
+
*/
|
|
53
|
+
export declare const serializeXml: (node: Node, options?: {
|
|
54
|
+
preserveWhitespace?: boolean;
|
|
55
|
+
}) => string;
|
|
56
|
+
/**
|
|
57
|
+
* Attempts to extract the original raw substring from the source XML for a given node.
|
|
58
|
+
* Requires the document to have been parsed with { locator: true }.
|
|
59
|
+
*
|
|
60
|
+
* @param node - The DOM node to extract source for
|
|
61
|
+
* @param sourceXml - The original XML source string
|
|
62
|
+
* @returns The raw XML substring, or undefined if it cannot be reliably determined
|
|
63
|
+
*/
|
|
64
|
+
export declare const getSourceSubstring: (node: any, sourceXml: string) => string | undefined;
|
|
65
|
+
/**
|
|
66
|
+
* High-level helper to get raw content for a node based on OfficeParserConfig.
|
|
67
|
+
*
|
|
68
|
+
* @param node - The DOM node
|
|
69
|
+
* @param sourceXml - The original source XML string
|
|
70
|
+
* @param config - The parser configuration
|
|
71
|
+
* @returns The raw content string (serialized or original)
|
|
72
|
+
*/
|
|
73
|
+
export declare const getRawContent: (node: Node, sourceXml: string, config: {
|
|
74
|
+
serializeRawContent?: boolean;
|
|
75
|
+
preserveXmlWhitespace?: boolean;
|
|
76
|
+
}) => string;
|
|
77
|
+
/**
|
|
78
|
+
* Gets the first element with the specified tag name within a parent element.
|
|
79
|
+
*
|
|
80
|
+
* @param parent - The parent element or document to search within
|
|
81
|
+
* @param tagName - The tag name to search for
|
|
82
|
+
* @returns The first matching element, or undefined if none found
|
|
83
|
+
*/
|
|
84
|
+
export declare const getFirstElementByTagName: (parent: Element | Document, tagName: string) => Element | undefined;
|
|
85
|
+
/**
|
|
86
|
+
* Gets the value of an attribute from an element.
|
|
87
|
+
*
|
|
88
|
+
* @param element - The element to get the attribute from
|
|
89
|
+
* @param attrName - The name of the attribute
|
|
90
|
+
* @returns The attribute value or undefined if not set
|
|
91
|
+
*/
|
|
92
|
+
export declare const getAttribute: (element: Element, attrName: string) => string | undefined;
|
|
46
93
|
/**
|
|
47
94
|
* Gets direct child elements with a specific tag name.
|
|
48
95
|
* Unlike getElementsByTagName, this does not search recursively.
|
|
@@ -81,3 +128,27 @@ export declare const getDirectChildren: (parent: Element, tagName: string) => El
|
|
|
81
128
|
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
|
|
82
129
|
*/
|
|
83
130
|
export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata;
|
|
131
|
+
/**
|
|
132
|
+
* Parses OOXML custom document properties from `docProps/custom.xml`.
|
|
133
|
+
*
|
|
134
|
+
* Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
|
|
135
|
+
* (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
|
|
136
|
+
*
|
|
137
|
+
* Property values are typed using the `vt:` namespace (docPropsVTypes):
|
|
138
|
+
* - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
|
|
139
|
+
* - `vt:bool` → boolean
|
|
140
|
+
* - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
|
|
141
|
+
* - `vt:filetime` / `vt:date` → Date
|
|
142
|
+
*
|
|
143
|
+
* @param xmlContent - Raw XML string from `docProps/custom.xml`
|
|
144
|
+
* @returns A record of property name → typed value (empty object if none found)
|
|
145
|
+
* @example
|
|
146
|
+
* ```typescript
|
|
147
|
+
* const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
|
|
148
|
+
* const props = parseOOXMLCustomProperties(customXml);
|
|
149
|
+
* console.log(props['Department']); // "Engineering"
|
|
150
|
+
* console.log(props['Priority']); // 1 (number)
|
|
151
|
+
* console.log(props['Reviewed']); // true (boolean)
|
|
152
|
+
* ```
|
|
153
|
+
*/
|
|
154
|
+
export declare const parseOOXMLCustomProperties: (xmlContent: string) => Record<string, string | number | boolean | Date>;
|
package/dist/utils/xmlUtils.js
CHANGED
|
@@ -11,27 +11,32 @@
|
|
|
11
11
|
* @module xmlUtils
|
|
12
12
|
*/
|
|
13
13
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
|
-
exports.parseOfficeMetadata = exports.getDirectChildren = exports.getElementsByTagName = exports.parseXmlString = void 0;
|
|
14
|
+
exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
|
|
15
15
|
const xmldom_1 = require("@xmldom/xmldom");
|
|
16
|
+
const dateUtils_js_1 = require("./dateUtils.js");
|
|
17
|
+
/**
|
|
18
|
+
* Type guard for Element nodes.
|
|
19
|
+
*/
|
|
20
|
+
const isElement = (node) => {
|
|
21
|
+
return node.nodeType === 1;
|
|
22
|
+
};
|
|
23
|
+
exports.isElement = isElement;
|
|
16
24
|
/**
|
|
17
25
|
* Parses an XML string into a DOM Document object.
|
|
18
26
|
*
|
|
19
27
|
* Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
|
|
20
|
-
* This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
|
|
21
28
|
*
|
|
22
29
|
* @param xml - The XML content as a string
|
|
30
|
+
* @param options - Optional parser settings (e.g., enable locators for source mapping)
|
|
23
31
|
* @returns A Document object that can be queried using standard DOM methods
|
|
24
|
-
* @example
|
|
25
|
-
* ```typescript
|
|
26
|
-
* const xmlString = '<root><item>Hello</item></root>';
|
|
27
|
-
* const doc = parseXmlString(xmlString);
|
|
28
|
-
* const items = doc.getElementsByTagName('item');
|
|
29
|
-
* console.log(items[0].textContent); // "Hello"
|
|
30
|
-
* ```
|
|
31
32
|
*/
|
|
32
|
-
const parseXmlString = (xml) => {
|
|
33
|
-
const parser = new xmldom_1.DOMParser();
|
|
34
|
-
|
|
33
|
+
const parseXmlString = (xml, options = {}) => {
|
|
34
|
+
const parser = new xmldom_1.DOMParser(options);
|
|
35
|
+
// @xmldom/xmldom 0.9.x is strict: a UTF-8 BOM (U+FEFF) prepended to the
|
|
36
|
+
// XML string causes a fatalError because the XML declaration is no longer
|
|
37
|
+
// at position 0. Strip it before parsing.
|
|
38
|
+
const sanitized = xml.charCodeAt(0) === 0xFEFF ? xml.slice(1) : xml.trim();
|
|
39
|
+
return parser.parseFromString(sanitized, "text/xml");
|
|
35
40
|
};
|
|
36
41
|
exports.parseXmlString = parseXmlString;
|
|
37
42
|
/**
|
|
@@ -50,9 +55,115 @@ exports.parseXmlString = parseXmlString;
|
|
|
50
55
|
* ```
|
|
51
56
|
*/
|
|
52
57
|
const getElementsByTagName = (element, tagName) => {
|
|
53
|
-
|
|
58
|
+
const results = Array.from(element.getElementsByTagName(tagName));
|
|
59
|
+
// Resilience: If prefixed tag (e.g., 'dc:title') not found, try local name (e.g., 'title')
|
|
60
|
+
if (results.length === 0 && tagName.includes(':')) {
|
|
61
|
+
const localName = tagName.split(':').pop();
|
|
62
|
+
return Array.from(element.getElementsByTagName(localName));
|
|
63
|
+
}
|
|
64
|
+
return results;
|
|
54
65
|
};
|
|
55
66
|
exports.getElementsByTagName = getElementsByTagName;
|
|
67
|
+
/**
|
|
68
|
+
* Serializes a DOM Node (Document, Element, etc.) back into an XML string.
|
|
69
|
+
* This is cross-platform and works in both Node.js and Browser environments.
|
|
70
|
+
*
|
|
71
|
+
* @param node - The DOM node to serialize
|
|
72
|
+
* @param options - Serialization options
|
|
73
|
+
* @returns The XML string representation
|
|
74
|
+
*/
|
|
75
|
+
const serializeXml = (node, options = {}) => {
|
|
76
|
+
// Note: xmldom's XMLSerializer doesn't natively support a 'pretty' or 'preserve'
|
|
77
|
+
// flag in a way that matches all user expectations, but it defaults to
|
|
78
|
+
// preserving structure. Formatting (indentation) is usually handled by the
|
|
79
|
+
// parser's initial whitespace handling.
|
|
80
|
+
// @ts-ignore - xmldom's Node is compatible with the global Node interface
|
|
81
|
+
return new xmldom_1.XMLSerializer().serializeToString(node);
|
|
82
|
+
};
|
|
83
|
+
exports.serializeXml = serializeXml;
|
|
84
|
+
/**
|
|
85
|
+
* Attempts to extract the original raw substring from the source XML for a given node.
|
|
86
|
+
* Requires the document to have been parsed with { locator: true }.
|
|
87
|
+
*
|
|
88
|
+
* @param node - The DOM node to extract source for
|
|
89
|
+
* @param sourceXml - The original XML source string
|
|
90
|
+
* @returns The raw XML substring, or undefined if it cannot be reliably determined
|
|
91
|
+
*/
|
|
92
|
+
const getSourceSubstring = (node, sourceXml) => {
|
|
93
|
+
if (!node || typeof node.lineNumber !== 'number' || typeof node.columnNumber !== 'number') {
|
|
94
|
+
return undefined;
|
|
95
|
+
}
|
|
96
|
+
// Convert line/column to absolute index
|
|
97
|
+
const lines = sourceXml.split('\n');
|
|
98
|
+
let startIdx = 0;
|
|
99
|
+
for (let i = 0; i < node.lineNumber - 1; i++) {
|
|
100
|
+
startIdx += lines[i].length + 1; // +1 for newline
|
|
101
|
+
}
|
|
102
|
+
startIdx += node.columnNumber - 1;
|
|
103
|
+
// To find the end of the node, we look for the closing tag.
|
|
104
|
+
// This is a heuristic approach that works well for simple structured nodes (p, tbl, etc.)
|
|
105
|
+
// but might be complex for overlapping namespaces or malformed XML.
|
|
106
|
+
if ((0, exports.isElement)(node)) {
|
|
107
|
+
const tagName = node.tagName;
|
|
108
|
+
const closingTag = `</${tagName}>`;
|
|
109
|
+
const endIdx = sourceXml.indexOf(closingTag, startIdx);
|
|
110
|
+
if (endIdx !== -1) {
|
|
111
|
+
return sourceXml.substring(startIdx, endIdx + closingTag.length);
|
|
112
|
+
}
|
|
113
|
+
// Self-closing tag handling (e.g., <w:p/>)
|
|
114
|
+
const selfClosingEnd = sourceXml.indexOf('/>', startIdx);
|
|
115
|
+
const nextOpenTag = sourceXml.indexOf('<', startIdx + 1);
|
|
116
|
+
if (selfClosingEnd !== -1 && (nextOpenTag === -1 || selfClosingEnd < nextOpenTag)) {
|
|
117
|
+
return sourceXml.substring(startIdx, selfClosingEnd + 2);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
return undefined;
|
|
121
|
+
};
|
|
122
|
+
exports.getSourceSubstring = getSourceSubstring;
|
|
123
|
+
/**
|
|
124
|
+
* High-level helper to get raw content for a node based on OfficeParserConfig.
|
|
125
|
+
*
|
|
126
|
+
* @param node - The DOM node
|
|
127
|
+
* @param sourceXml - The original source XML string
|
|
128
|
+
* @param config - The parser configuration
|
|
129
|
+
* @returns The raw content string (serialized or original)
|
|
130
|
+
*/
|
|
131
|
+
const getRawContent = (node, sourceXml, config) => {
|
|
132
|
+
if (config.serializeRawContent === false) {
|
|
133
|
+
const original = (0, exports.getSourceSubstring)(node, sourceXml);
|
|
134
|
+
if (original)
|
|
135
|
+
return original;
|
|
136
|
+
}
|
|
137
|
+
return (0, exports.serializeXml)(node, { preserveWhitespace: config.preserveXmlWhitespace });
|
|
138
|
+
};
|
|
139
|
+
exports.getRawContent = getRawContent;
|
|
140
|
+
/**
|
|
141
|
+
* Gets the first element with the specified tag name within a parent element.
|
|
142
|
+
*
|
|
143
|
+
* @param parent - The parent element or document to search within
|
|
144
|
+
* @param tagName - The tag name to search for
|
|
145
|
+
* @returns The first matching element, or undefined if none found
|
|
146
|
+
*/
|
|
147
|
+
const getFirstElementByTagName = (parent, tagName) => {
|
|
148
|
+
const elements = parent.getElementsByTagName(tagName);
|
|
149
|
+
if (elements && elements.length > 0) {
|
|
150
|
+
return elements[0];
|
|
151
|
+
}
|
|
152
|
+
return undefined;
|
|
153
|
+
};
|
|
154
|
+
exports.getFirstElementByTagName = getFirstElementByTagName;
|
|
155
|
+
/**
|
|
156
|
+
* Gets the value of an attribute from an element.
|
|
157
|
+
*
|
|
158
|
+
* @param element - The element to get the attribute from
|
|
159
|
+
* @param attrName - The name of the attribute
|
|
160
|
+
* @returns The attribute value or undefined if not set
|
|
161
|
+
*/
|
|
162
|
+
const getAttribute = (element, attrName) => {
|
|
163
|
+
const attr = element.getAttribute(attrName);
|
|
164
|
+
return attr !== null ? attr : undefined;
|
|
165
|
+
};
|
|
166
|
+
exports.getAttribute = getAttribute;
|
|
56
167
|
/**
|
|
57
168
|
* Gets direct child elements with a specific tag name.
|
|
58
169
|
* Unlike getElementsByTagName, this does not search recursively.
|
|
@@ -67,7 +178,7 @@ const getDirectChildren = (parent, tagName) => {
|
|
|
67
178
|
return result;
|
|
68
179
|
for (let i = 0; i < parent.childNodes.length; i++) {
|
|
69
180
|
const child = parent.childNodes[i];
|
|
70
|
-
if (child
|
|
181
|
+
if ((0, exports.isElement)(child) && child.tagName === tagName) {
|
|
71
182
|
result.push(child);
|
|
72
183
|
}
|
|
73
184
|
}
|
|
@@ -124,11 +235,18 @@ const parseOfficeMetadata = (xmlContent) => {
|
|
|
124
235
|
// Step 6: Extract creation date (Dublin Core Terms element)
|
|
125
236
|
const created = (0, exports.getElementsByTagName)(coreProperties, "dcterms:created")[0];
|
|
126
237
|
if (created && created.textContent)
|
|
127
|
-
metadata.created =
|
|
238
|
+
metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
|
|
128
239
|
// Step 7: Extract last modification date (Dublin Core Terms element)
|
|
129
240
|
const modified = (0, exports.getElementsByTagName)(coreProperties, "dcterms:modified")[0];
|
|
130
241
|
if (modified && modified.textContent)
|
|
131
|
-
metadata.modified =
|
|
242
|
+
metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
|
|
243
|
+
// Step 8: Extract description and subject (Dublin Core elements)
|
|
244
|
+
const description = (0, exports.getElementsByTagName)(coreProperties, "dc:description")[0];
|
|
245
|
+
if (description && description.textContent)
|
|
246
|
+
metadata.description = description.textContent;
|
|
247
|
+
const subject = (0, exports.getElementsByTagName)(coreProperties, "dc:subject")[0];
|
|
248
|
+
if (subject && subject.textContent)
|
|
249
|
+
metadata.subject = subject.textContent;
|
|
132
250
|
return metadata;
|
|
133
251
|
}
|
|
134
252
|
// Check for ODF Meta
|
|
@@ -148,11 +266,111 @@ const parseOfficeMetadata = (xmlContent) => {
|
|
|
148
266
|
metadata.subject = subject.textContent;
|
|
149
267
|
const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
|
|
150
268
|
if (created && created.textContent)
|
|
151
|
-
metadata.created =
|
|
269
|
+
metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
|
|
152
270
|
const modified = (0, exports.getElementsByTagName)(officeMeta, "dc:date")[0];
|
|
153
271
|
if (modified && modified.textContent)
|
|
154
|
-
metadata.modified =
|
|
272
|
+
metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
|
|
273
|
+
// Extract user-defined custom properties (meta:user-defined)
|
|
274
|
+
const userDefined = (0, exports.getElementsByTagName)(officeMeta, "meta:user-defined");
|
|
275
|
+
if (userDefined.length > 0) {
|
|
276
|
+
const customProperties = {};
|
|
277
|
+
for (const el of userDefined) {
|
|
278
|
+
const name = el.getAttribute("meta:name");
|
|
279
|
+
if (!name || !el.textContent)
|
|
280
|
+
continue;
|
|
281
|
+
const valueType = el.getAttribute("meta:value-type") || "string";
|
|
282
|
+
const raw = el.textContent;
|
|
283
|
+
if (valueType === "boolean") {
|
|
284
|
+
customProperties[name] = raw.toLowerCase() === "true";
|
|
285
|
+
}
|
|
286
|
+
else if (valueType === "float") {
|
|
287
|
+
const num = Number(raw);
|
|
288
|
+
if (!isNaN(num))
|
|
289
|
+
customProperties[name] = num;
|
|
290
|
+
}
|
|
291
|
+
else if (valueType === "date" || valueType === "time") {
|
|
292
|
+
const date = (0, dateUtils_js_1.parseOfficeDate)(raw);
|
|
293
|
+
if (date)
|
|
294
|
+
customProperties[name] = date;
|
|
295
|
+
else
|
|
296
|
+
customProperties[name] = raw;
|
|
297
|
+
}
|
|
298
|
+
else {
|
|
299
|
+
customProperties[name] = raw;
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
if (Object.keys(customProperties).length > 0) {
|
|
303
|
+
metadata.customProperties = customProperties;
|
|
304
|
+
}
|
|
305
|
+
}
|
|
155
306
|
}
|
|
156
307
|
return metadata;
|
|
157
308
|
};
|
|
158
309
|
exports.parseOfficeMetadata = parseOfficeMetadata;
|
|
310
|
+
/**
|
|
311
|
+
* Parses OOXML custom document properties from `docProps/custom.xml`.
|
|
312
|
+
*
|
|
313
|
+
* Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
|
|
314
|
+
* (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
|
|
315
|
+
*
|
|
316
|
+
* Property values are typed using the `vt:` namespace (docPropsVTypes):
|
|
317
|
+
* - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
|
|
318
|
+
* - `vt:bool` → boolean
|
|
319
|
+
* - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
|
|
320
|
+
* - `vt:filetime` / `vt:date` → Date
|
|
321
|
+
*
|
|
322
|
+
* @param xmlContent - Raw XML string from `docProps/custom.xml`
|
|
323
|
+
* @returns A record of property name → typed value (empty object if none found)
|
|
324
|
+
* @example
|
|
325
|
+
* ```typescript
|
|
326
|
+
* const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
|
|
327
|
+
* const props = parseOOXMLCustomProperties(customXml);
|
|
328
|
+
* console.log(props['Department']); // "Engineering"
|
|
329
|
+
* console.log(props['Priority']); // 1 (number)
|
|
330
|
+
* console.log(props['Reviewed']); // true (boolean)
|
|
331
|
+
* ```
|
|
332
|
+
*/
|
|
333
|
+
const parseOOXMLCustomProperties = (xmlContent) => {
|
|
334
|
+
const xml = (0, exports.parseXmlString)(xmlContent);
|
|
335
|
+
const result = {};
|
|
336
|
+
const properties = (0, exports.getElementsByTagName)(xml, "property");
|
|
337
|
+
for (const prop of properties) {
|
|
338
|
+
const name = prop.getAttribute("name");
|
|
339
|
+
if (!name)
|
|
340
|
+
continue;
|
|
341
|
+
// The value is the first child element (typed using vt: namespace)
|
|
342
|
+
for (let i = 0; i < prop.childNodes.length; i++) {
|
|
343
|
+
const child = prop.childNodes[i];
|
|
344
|
+
if (child.nodeType !== 1)
|
|
345
|
+
continue; // skip non-elements
|
|
346
|
+
const el = child;
|
|
347
|
+
const tag = el.tagName || '';
|
|
348
|
+
const text = el.textContent || '';
|
|
349
|
+
if (/vt:lpwstr|vt:lpstr|vt:bstr/.test(tag)) {
|
|
350
|
+
result[name] = text;
|
|
351
|
+
}
|
|
352
|
+
else if (/vt:bool/.test(tag)) {
|
|
353
|
+
result[name] = text.toLowerCase() === 'true';
|
|
354
|
+
}
|
|
355
|
+
else if (/vt:(i[1248]|ui[1248]|int|uint|r4|r8|decimal)/.test(tag)) {
|
|
356
|
+
const num = Number(text);
|
|
357
|
+
if (!isNaN(num))
|
|
358
|
+
result[name] = num;
|
|
359
|
+
}
|
|
360
|
+
else if (/vt:filetime|vt:date/.test(tag)) {
|
|
361
|
+
const date = (0, dateUtils_js_1.parseOfficeDate)(text);
|
|
362
|
+
if (date)
|
|
363
|
+
result[name] = date;
|
|
364
|
+
else
|
|
365
|
+
result[name] = text;
|
|
366
|
+
}
|
|
367
|
+
else if (text) {
|
|
368
|
+
// Fallback: store as string for any other vt: type
|
|
369
|
+
result[name] = text;
|
|
370
|
+
}
|
|
371
|
+
break; // only one value element per property
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
return result;
|
|
375
|
+
};
|
|
376
|
+
exports.parseOOXMLCustomProperties = parseOOXMLCustomProperties;
|