officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
@@ -5,11 +5,176 @@
5
5
  * This module provides functions for extracting text from images using Tesseract.js.
6
6
  * Used when `config.ocr` is enabled to extract text from embedded images in documents.
7
7
  *
8
+ * Includes a worker pool via OcrSchedulerManager to improve performance when
9
+ * processing multiple images.
10
+ *
8
11
  * @module ocrUtils
9
12
  */
10
13
  Object.defineProperty(exports, "__esModule", { value: true });
11
- exports.performOcr = void 0;
12
- const tesseract_js_1 = require("tesseract.js");
14
+ exports.terminateOcr = exports.performOcr = void 0;
15
+ /**
16
+ * Manages a pool of Tesseract workers with "Smart Affinity".
17
+ *
18
+ * Instead of a simple scheduler, this manager allows workers to persist with
19
+ * a specific language affinity. If a new language is requested and the pool
20
+ * is at capacity, it re-initializes the Least Recently Used (LRU) idle worker
21
+ * rather than resetting the entire pool.
22
+ *
23
+ * Implements lazy loading of tesseract.js to ensure no background processes
24
+ * are spawned unless OCR is explicitly used.
25
+ */
26
+ class OcrSchedulerManager {
27
+ static instance;
28
+ pool = [];
29
+ queue = [];
30
+ MAX_WORKERS = 4;
31
+ idleTimeout = 10000; // 10s default
32
+ timeoutId = null;
33
+ constructor() { }
34
+ /**
35
+ * Returns the singleton instance of the manager.
36
+ */
37
+ static getInstance() {
38
+ if (!OcrSchedulerManager.instance) {
39
+ OcrSchedulerManager.instance = new OcrSchedulerManager();
40
+ }
41
+ return OcrSchedulerManager.instance;
42
+ }
43
+ /**
44
+ * Checks if the singleton instance has been initialized.
45
+ */
46
+ static hasInstance() {
47
+ return !!OcrSchedulerManager.instance;
48
+ }
49
+ /**
50
+ * Resets the inactivity timer. If the timer reaches its duration,
51
+ * all workers are terminated automatically.
52
+ */
53
+ resetIdleTimer() {
54
+ if (this.timeoutId) {
55
+ clearTimeout(this.timeoutId);
56
+ }
57
+ if (this.idleTimeout > 0) {
58
+ this.timeoutId = setTimeout(async () => {
59
+ await this.terminate();
60
+ }, this.idleTimeout);
61
+ }
62
+ }
63
+ /**
64
+ * Performs OCR on an image using the smart worker pool.
65
+ *
66
+ * @param image - Image data (Buffer, string path, or Blob)
67
+ * @param config - OCR configuration (language, custom paths)
68
+ * @returns Recognized text
69
+ */
70
+ async recognize(image, config) {
71
+ return new Promise((resolve, reject) => {
72
+ // Update idle timeout if provided
73
+ if (config?.autoTerminateTimeout !== undefined) {
74
+ this.idleTimeout = config.autoTerminateTimeout;
75
+ }
76
+ // Reset the inactivity timer every time a new job is requested
77
+ this.resetIdleTimer();
78
+ // Add job to queue and trigger processing
79
+ this.queue.push({ image, config: config || {}, resolve, reject });
80
+ this.processQueue();
81
+ });
82
+ }
83
+ /**
84
+ * Attempts to process the next job in the queue using an available worker.
85
+ */
86
+ async processQueue() {
87
+ if (this.queue.length === 0)
88
+ return;
89
+ const nextJob = this.queue[0];
90
+ const requestedLanguage = nextJob.config.language || 'eng';
91
+ // 1. Find an idle worker with the EXACT language affinity
92
+ let managed = this.pool.find(mw => !mw.isBusy && mw.language === requestedLanguage);
93
+ // 2. If not found and we have room, create a new worker
94
+ if (!managed && this.pool.length < this.MAX_WORKERS) {
95
+ try {
96
+ const { createWorker } = await import('tesseract.js');
97
+ const options = { logger: () => { } };
98
+ if (nextJob.config.workerPath)
99
+ options.workerPath = nextJob.config.workerPath;
100
+ if (nextJob.config.corePath)
101
+ options.corePath = nextJob.config.corePath;
102
+ if (nextJob.config.langPath)
103
+ options.langPath = nextJob.config.langPath;
104
+ const worker = await createWorker(requestedLanguage, 1, options);
105
+ managed = {
106
+ worker,
107
+ language: requestedLanguage,
108
+ lastUsed: Date.now(),
109
+ isBusy: false
110
+ };
111
+ this.pool.push(managed);
112
+ }
113
+ catch (err) {
114
+ const job = this.queue.shift();
115
+ job?.reject(err);
116
+ this.processQueue(); // Try next job
117
+ return;
118
+ }
119
+ }
120
+ // 3. If still not found and we are at capacity, find the LRU idle worker and re-initialize it
121
+ if (!managed) {
122
+ const idleWorkers = this.pool.filter(mw => !mw.isBusy);
123
+ if (idleWorkers.length > 0) {
124
+ // Find Least Recently Used idle worker
125
+ managed = idleWorkers.reduce((prev, curr) => (prev.lastUsed < curr.lastUsed ? prev : curr));
126
+ try {
127
+ // Smart Re-initialization (v5 API)
128
+ await managed.worker.reinitialize(requestedLanguage);
129
+ managed.language = requestedLanguage;
130
+ }
131
+ catch (err) {
132
+ // If reinitialization fails, we might need to recreate it, but for simplicity
133
+ // we'll just fail this job and try another worker next time.
134
+ const job = this.queue.shift();
135
+ job?.reject(err);
136
+ this.processQueue();
137
+ return;
138
+ }
139
+ }
140
+ }
141
+ // 4. If we have a worker ready, execute the job
142
+ if (managed) {
143
+ const job = this.queue.shift();
144
+ if (!job)
145
+ return;
146
+ managed.isBusy = true;
147
+ managed.lastUsed = Date.now();
148
+ try {
149
+ const { data: { text } } = await managed.worker.recognize(job.image);
150
+ job.resolve(text);
151
+ }
152
+ catch (err) {
153
+ job.reject(err);
154
+ }
155
+ finally {
156
+ managed.isBusy = false;
157
+ managed.lastUsed = Date.now();
158
+ // Check if there are more jobs waiting
159
+ this.processQueue();
160
+ }
161
+ }
162
+ // If no worker is available (all busy), the job stays in the queue
163
+ // and will be picked up when a worker finishes.
164
+ }
165
+ /**
166
+ * Terminates all workers in the pool and resets the state.
167
+ */
168
+ async terminate() {
169
+ if (this.timeoutId) {
170
+ clearTimeout(this.timeoutId);
171
+ this.timeoutId = null;
172
+ }
173
+ const workersToTerminate = this.pool.map(mw => mw.worker.terminate());
174
+ await Promise.all(workersToTerminate);
175
+ this.pool = [];
176
+ }
177
+ }
13
178
  /**
14
179
  * Performs Optical Character Recognition (OCR) on an image to extract text.
15
180
  *
@@ -17,45 +182,41 @@ const tesseract_js_1 = require("tesseract.js");
17
182
  * This is useful for extracting text from screenshots, scanned documents,
18
183
  * charts with labels, or any image containing text.
19
184
  *
20
- * The function creates a new Tesseract worker, processes the image,
21
- * and properly terminates the worker to free resources.
185
+ * This function uses a shared worker pool to minimize initialization overhead.
22
186
  *
23
- * @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
24
- * @param language - The language code for OCR (default: 'eng' for English).
25
- * Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
26
- * Multiple languages can be combined with '+': 'eng+fra'
187
+ * @param image - The image data as a Buffer, file path, or Blob
188
+ * @param config - Optional configuration for language and custom worker paths
27
189
  * @returns A promise that resolves to the recognized text as a string
28
190
  * @throws {Error} If the image cannot be processed or Tesseract initialization fails
29
191
  *
30
192
  * @example
31
193
  * ```typescript
32
194
  * // Extract text from an English image
33
- * const text = await performOcr(imageBuffer, 'eng');
34
- * console.log(text); // "Annual Revenue: $1.2M"
35
- *
36
- * // Extract text from a multilingual image
37
- * const text = await performOcr(imageBuffer, 'eng+spa');
195
+ * const text = await performOcr(imageBuffer, { language: 'eng' });
38
196
  * ```
39
197
  *
40
198
  * @see https://github.com/naptha/tesseract.js for supported languages and options
41
199
  */
42
- const performOcr = async (image, language = 'eng') => {
43
- // Step 1: Create a Tesseract worker with the specified language
44
- // We pass 1 for OEM (LSTM) and a silent logger to suppress console output
45
- const worker = await (0, tesseract_js_1.createWorker)(language, 1, {
46
- logger: () => { }
47
- });
48
- // Step 2: Prepare image data
200
+ const performOcr = async (image, config) => {
201
+ // Prepare image data
49
202
  let inputImage = image;
50
203
  // In browser environment, convert Buffer to Blob for better compatibility
51
204
  // @ts-ignore
52
205
  if (typeof window !== 'undefined' && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
53
206
  inputImage = new Blob([image], { type: 'image/bmp' });
54
207
  }
55
- // Step 3: Perform OCR
56
- const ret = await worker.recognize(inputImage);
57
- // Step 4: Terminate worker
58
- await worker.terminate();
59
- return ret.data.text;
208
+ return await OcrSchedulerManager.getInstance().recognize(inputImage, config);
60
209
  };
61
210
  exports.performOcr = performOcr;
211
+ /**
212
+ * Terminates all OCR workers and cleans up resources.
213
+ *
214
+ * Should be called when the application is shutting down or OCR is no longer needed
215
+ * to prevent memory leaks and dangling worker processes.
216
+ */
217
+ const terminateOcr = async () => {
218
+ if (OcrSchedulerManager.hasInstance()) {
219
+ await OcrSchedulerManager.getInstance().terminate();
220
+ }
221
+ };
222
+ exports.terminateOcr = terminateOcr;
@@ -10,23 +10,22 @@
10
10
  * @module xmlUtils
11
11
  */
12
12
  import { OfficeMetadata } from '../types';
13
+ /**
14
+ * Type guard for Element nodes.
15
+ */
16
+ export declare const isElement: (node: Node) => node is Element;
13
17
  /**
14
18
  * Parses an XML string into a DOM Document object.
15
19
  *
16
20
  * Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
17
- * This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
18
21
  *
19
22
  * @param xml - The XML content as a string
23
+ * @param options - Optional parser settings (e.g., enable locators for source mapping)
20
24
  * @returns A Document object that can be queried using standard DOM methods
21
- * @example
22
- * ```typescript
23
- * const xmlString = '<root><item>Hello</item></root>';
24
- * const doc = parseXmlString(xmlString);
25
- * const items = doc.getElementsByTagName('item');
26
- * console.log(items[0].textContent); // "Hello"
27
- * ```
28
25
  */
29
- export declare const parseXmlString: (xml: string) => Document;
26
+ export declare const parseXmlString: (xml: string, options?: {
27
+ locator?: boolean;
28
+ }) => Document;
30
29
  /**
31
30
  * Gets all elements with a specific tag name and returns them as an array.
32
31
  *
@@ -43,6 +42,54 @@ export declare const parseXmlString: (xml: string) => Document;
43
42
  * ```
44
43
  */
45
44
  export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
45
+ /**
46
+ * Serializes a DOM Node (Document, Element, etc.) back into an XML string.
47
+ * This is cross-platform and works in both Node.js and Browser environments.
48
+ *
49
+ * @param node - The DOM node to serialize
50
+ * @param options - Serialization options
51
+ * @returns The XML string representation
52
+ */
53
+ export declare const serializeXml: (node: Node, options?: {
54
+ preserveWhitespace?: boolean;
55
+ }) => string;
56
+ /**
57
+ * Attempts to extract the original raw substring from the source XML for a given node.
58
+ * Requires the document to have been parsed with { locator: true }.
59
+ *
60
+ * @param node - The DOM node to extract source for
61
+ * @param sourceXml - The original XML source string
62
+ * @returns The raw XML substring, or undefined if it cannot be reliably determined
63
+ */
64
+ export declare const getSourceSubstring: (node: any, sourceXml: string) => string | undefined;
65
+ /**
66
+ * High-level helper to get raw content for a node based on OfficeParserConfig.
67
+ *
68
+ * @param node - The DOM node
69
+ * @param sourceXml - The original source XML string
70
+ * @param config - The parser configuration
71
+ * @returns The raw content string (serialized or original)
72
+ */
73
+ export declare const getRawContent: (node: Node, sourceXml: string, config: {
74
+ serializeRawContent?: boolean;
75
+ preserveXmlWhitespace?: boolean;
76
+ }) => string;
77
+ /**
78
+ * Gets the first element with the specified tag name within a parent element.
79
+ *
80
+ * @param parent - The parent element or document to search within
81
+ * @param tagName - The tag name to search for
82
+ * @returns The first matching element, or undefined if none found
83
+ */
84
+ export declare const getFirstElementByTagName: (parent: Element | Document, tagName: string) => Element | undefined;
85
+ /**
86
+ * Gets the value of an attribute from an element.
87
+ *
88
+ * @param element - The element to get the attribute from
89
+ * @param attrName - The name of the attribute
90
+ * @returns The attribute value or undefined if not set
91
+ */
92
+ export declare const getAttribute: (element: Element, attrName: string) => string | undefined;
46
93
  /**
47
94
  * Gets direct child elements with a specific tag name.
48
95
  * Unlike getElementsByTagName, this does not search recursively.
@@ -81,3 +128,27 @@ export declare const getDirectChildren: (parent: Element, tagName: string) => El
81
128
  * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
82
129
  */
83
130
  export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata;
131
+ /**
132
+ * Parses OOXML custom document properties from `docProps/custom.xml`.
133
+ *
134
+ * Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
135
+ * (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
136
+ *
137
+ * Property values are typed using the `vt:` namespace (docPropsVTypes):
138
+ * - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
139
+ * - `vt:bool` → boolean
140
+ * - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
141
+ * - `vt:filetime` / `vt:date` → Date
142
+ *
143
+ * @param xmlContent - Raw XML string from `docProps/custom.xml`
144
+ * @returns A record of property name → typed value (empty object if none found)
145
+ * @example
146
+ * ```typescript
147
+ * const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
148
+ * const props = parseOOXMLCustomProperties(customXml);
149
+ * console.log(props['Department']); // "Engineering"
150
+ * console.log(props['Priority']); // 1 (number)
151
+ * console.log(props['Reviewed']); // true (boolean)
152
+ * ```
153
+ */
154
+ export declare const parseOOXMLCustomProperties: (xmlContent: string) => Record<string, string | number | boolean | Date>;
@@ -11,27 +11,32 @@
11
11
  * @module xmlUtils
12
12
  */
13
13
  Object.defineProperty(exports, "__esModule", { value: true });
14
- exports.parseOfficeMetadata = exports.getDirectChildren = exports.getElementsByTagName = exports.parseXmlString = void 0;
14
+ exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
15
15
  const xmldom_1 = require("@xmldom/xmldom");
16
+ const dateUtils_js_1 = require("./dateUtils.js");
17
+ /**
18
+ * Type guard for Element nodes.
19
+ */
20
+ const isElement = (node) => {
21
+ return node.nodeType === 1;
22
+ };
23
+ exports.isElement = isElement;
16
24
  /**
17
25
  * Parses an XML string into a DOM Document object.
18
26
  *
19
27
  * Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
20
- * This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
21
28
  *
22
29
  * @param xml - The XML content as a string
30
+ * @param options - Optional parser settings (e.g., enable locators for source mapping)
23
31
  * @returns A Document object that can be queried using standard DOM methods
24
- * @example
25
- * ```typescript
26
- * const xmlString = '<root><item>Hello</item></root>';
27
- * const doc = parseXmlString(xmlString);
28
- * const items = doc.getElementsByTagName('item');
29
- * console.log(items[0].textContent); // "Hello"
30
- * ```
31
32
  */
32
- const parseXmlString = (xml) => {
33
- const parser = new xmldom_1.DOMParser();
34
- return parser.parseFromString(xml, "text/xml");
33
+ const parseXmlString = (xml, options = {}) => {
34
+ const parser = new xmldom_1.DOMParser(options);
35
+ // @xmldom/xmldom 0.9.x is strict: a UTF-8 BOM (U+FEFF) prepended to the
36
+ // XML string causes a fatalError because the XML declaration is no longer
37
+ // at position 0. Strip it before parsing.
38
+ const sanitized = xml.charCodeAt(0) === 0xFEFF ? xml.slice(1) : xml.trim();
39
+ return parser.parseFromString(sanitized, "text/xml");
35
40
  };
36
41
  exports.parseXmlString = parseXmlString;
37
42
  /**
@@ -50,9 +55,115 @@ exports.parseXmlString = parseXmlString;
50
55
  * ```
51
56
  */
52
57
  const getElementsByTagName = (element, tagName) => {
53
- return Array.from(element.getElementsByTagName(tagName));
58
+ const results = Array.from(element.getElementsByTagName(tagName));
59
+ // Resilience: If prefixed tag (e.g., 'dc:title') not found, try local name (e.g., 'title')
60
+ if (results.length === 0 && tagName.includes(':')) {
61
+ const localName = tagName.split(':').pop();
62
+ return Array.from(element.getElementsByTagName(localName));
63
+ }
64
+ return results;
54
65
  };
55
66
  exports.getElementsByTagName = getElementsByTagName;
67
+ /**
68
+ * Serializes a DOM Node (Document, Element, etc.) back into an XML string.
69
+ * This is cross-platform and works in both Node.js and Browser environments.
70
+ *
71
+ * @param node - The DOM node to serialize
72
+ * @param options - Serialization options
73
+ * @returns The XML string representation
74
+ */
75
+ const serializeXml = (node, options = {}) => {
76
+ // Note: xmldom's XMLSerializer doesn't natively support a 'pretty' or 'preserve'
77
+ // flag in a way that matches all user expectations, but it defaults to
78
+ // preserving structure. Formatting (indentation) is usually handled by the
79
+ // parser's initial whitespace handling.
80
+ // @ts-ignore - xmldom's Node is compatible with the global Node interface
81
+ return new xmldom_1.XMLSerializer().serializeToString(node);
82
+ };
83
+ exports.serializeXml = serializeXml;
84
+ /**
85
+ * Attempts to extract the original raw substring from the source XML for a given node.
86
+ * Requires the document to have been parsed with { locator: true }.
87
+ *
88
+ * @param node - The DOM node to extract source for
89
+ * @param sourceXml - The original XML source string
90
+ * @returns The raw XML substring, or undefined if it cannot be reliably determined
91
+ */
92
+ const getSourceSubstring = (node, sourceXml) => {
93
+ if (!node || typeof node.lineNumber !== 'number' || typeof node.columnNumber !== 'number') {
94
+ return undefined;
95
+ }
96
+ // Convert line/column to absolute index
97
+ const lines = sourceXml.split('\n');
98
+ let startIdx = 0;
99
+ for (let i = 0; i < node.lineNumber - 1; i++) {
100
+ startIdx += lines[i].length + 1; // +1 for newline
101
+ }
102
+ startIdx += node.columnNumber - 1;
103
+ // To find the end of the node, we look for the closing tag.
104
+ // This is a heuristic approach that works well for simple structured nodes (p, tbl, etc.)
105
+ // but might be complex for overlapping namespaces or malformed XML.
106
+ if ((0, exports.isElement)(node)) {
107
+ const tagName = node.tagName;
108
+ const closingTag = `</${tagName}>`;
109
+ const endIdx = sourceXml.indexOf(closingTag, startIdx);
110
+ if (endIdx !== -1) {
111
+ return sourceXml.substring(startIdx, endIdx + closingTag.length);
112
+ }
113
+ // Self-closing tag handling (e.g., <w:p/>)
114
+ const selfClosingEnd = sourceXml.indexOf('/>', startIdx);
115
+ const nextOpenTag = sourceXml.indexOf('<', startIdx + 1);
116
+ if (selfClosingEnd !== -1 && (nextOpenTag === -1 || selfClosingEnd < nextOpenTag)) {
117
+ return sourceXml.substring(startIdx, selfClosingEnd + 2);
118
+ }
119
+ }
120
+ return undefined;
121
+ };
122
+ exports.getSourceSubstring = getSourceSubstring;
123
+ /**
124
+ * High-level helper to get raw content for a node based on OfficeParserConfig.
125
+ *
126
+ * @param node - The DOM node
127
+ * @param sourceXml - The original source XML string
128
+ * @param config - The parser configuration
129
+ * @returns The raw content string (serialized or original)
130
+ */
131
+ const getRawContent = (node, sourceXml, config) => {
132
+ if (config.serializeRawContent === false) {
133
+ const original = (0, exports.getSourceSubstring)(node, sourceXml);
134
+ if (original)
135
+ return original;
136
+ }
137
+ return (0, exports.serializeXml)(node, { preserveWhitespace: config.preserveXmlWhitespace });
138
+ };
139
+ exports.getRawContent = getRawContent;
140
+ /**
141
+ * Gets the first element with the specified tag name within a parent element.
142
+ *
143
+ * @param parent - The parent element or document to search within
144
+ * @param tagName - The tag name to search for
145
+ * @returns The first matching element, or undefined if none found
146
+ */
147
+ const getFirstElementByTagName = (parent, tagName) => {
148
+ const elements = parent.getElementsByTagName(tagName);
149
+ if (elements && elements.length > 0) {
150
+ return elements[0];
151
+ }
152
+ return undefined;
153
+ };
154
+ exports.getFirstElementByTagName = getFirstElementByTagName;
155
+ /**
156
+ * Gets the value of an attribute from an element.
157
+ *
158
+ * @param element - The element to get the attribute from
159
+ * @param attrName - The name of the attribute
160
+ * @returns The attribute value or undefined if not set
161
+ */
162
+ const getAttribute = (element, attrName) => {
163
+ const attr = element.getAttribute(attrName);
164
+ return attr !== null ? attr : undefined;
165
+ };
166
+ exports.getAttribute = getAttribute;
56
167
  /**
57
168
  * Gets direct child elements with a specific tag name.
58
169
  * Unlike getElementsByTagName, this does not search recursively.
@@ -67,7 +178,7 @@ const getDirectChildren = (parent, tagName) => {
67
178
  return result;
68
179
  for (let i = 0; i < parent.childNodes.length; i++) {
69
180
  const child = parent.childNodes[i];
70
- if (child.nodeType === 1 && child.tagName === tagName) { // 1 = ELEMENT_NODE
181
+ if ((0, exports.isElement)(child) && child.tagName === tagName) {
71
182
  result.push(child);
72
183
  }
73
184
  }
@@ -124,11 +235,18 @@ const parseOfficeMetadata = (xmlContent) => {
124
235
  // Step 6: Extract creation date (Dublin Core Terms element)
125
236
  const created = (0, exports.getElementsByTagName)(coreProperties, "dcterms:created")[0];
126
237
  if (created && created.textContent)
127
- metadata.created = new Date(created.textContent);
238
+ metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
128
239
  // Step 7: Extract last modification date (Dublin Core Terms element)
129
240
  const modified = (0, exports.getElementsByTagName)(coreProperties, "dcterms:modified")[0];
130
241
  if (modified && modified.textContent)
131
- metadata.modified = new Date(modified.textContent);
242
+ metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
243
+ // Step 8: Extract description and subject (Dublin Core elements)
244
+ const description = (0, exports.getElementsByTagName)(coreProperties, "dc:description")[0];
245
+ if (description && description.textContent)
246
+ metadata.description = description.textContent;
247
+ const subject = (0, exports.getElementsByTagName)(coreProperties, "dc:subject")[0];
248
+ if (subject && subject.textContent)
249
+ metadata.subject = subject.textContent;
132
250
  return metadata;
133
251
  }
134
252
  // Check for ODF Meta
@@ -148,11 +266,111 @@ const parseOfficeMetadata = (xmlContent) => {
148
266
  metadata.subject = subject.textContent;
149
267
  const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
150
268
  if (created && created.textContent)
151
- metadata.created = new Date(created.textContent);
269
+ metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
152
270
  const modified = (0, exports.getElementsByTagName)(officeMeta, "dc:date")[0];
153
271
  if (modified && modified.textContent)
154
- metadata.modified = new Date(modified.textContent);
272
+ metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
273
+ // Extract user-defined custom properties (meta:user-defined)
274
+ const userDefined = (0, exports.getElementsByTagName)(officeMeta, "meta:user-defined");
275
+ if (userDefined.length > 0) {
276
+ const customProperties = {};
277
+ for (const el of userDefined) {
278
+ const name = el.getAttribute("meta:name");
279
+ if (!name || !el.textContent)
280
+ continue;
281
+ const valueType = el.getAttribute("meta:value-type") || "string";
282
+ const raw = el.textContent;
283
+ if (valueType === "boolean") {
284
+ customProperties[name] = raw.toLowerCase() === "true";
285
+ }
286
+ else if (valueType === "float") {
287
+ const num = Number(raw);
288
+ if (!isNaN(num))
289
+ customProperties[name] = num;
290
+ }
291
+ else if (valueType === "date" || valueType === "time") {
292
+ const date = (0, dateUtils_js_1.parseOfficeDate)(raw);
293
+ if (date)
294
+ customProperties[name] = date;
295
+ else
296
+ customProperties[name] = raw;
297
+ }
298
+ else {
299
+ customProperties[name] = raw;
300
+ }
301
+ }
302
+ if (Object.keys(customProperties).length > 0) {
303
+ metadata.customProperties = customProperties;
304
+ }
305
+ }
155
306
  }
156
307
  return metadata;
157
308
  };
158
309
  exports.parseOfficeMetadata = parseOfficeMetadata;
310
+ /**
311
+ * Parses OOXML custom document properties from `docProps/custom.xml`.
312
+ *
313
+ * Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
314
+ * (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
315
+ *
316
+ * Property values are typed using the `vt:` namespace (docPropsVTypes):
317
+ * - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
318
+ * - `vt:bool` → boolean
319
+ * - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
320
+ * - `vt:filetime` / `vt:date` → Date
321
+ *
322
+ * @param xmlContent - Raw XML string from `docProps/custom.xml`
323
+ * @returns A record of property name → typed value (empty object if none found)
324
+ * @example
325
+ * ```typescript
326
+ * const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
327
+ * const props = parseOOXMLCustomProperties(customXml);
328
+ * console.log(props['Department']); // "Engineering"
329
+ * console.log(props['Priority']); // 1 (number)
330
+ * console.log(props['Reviewed']); // true (boolean)
331
+ * ```
332
+ */
333
+ const parseOOXMLCustomProperties = (xmlContent) => {
334
+ const xml = (0, exports.parseXmlString)(xmlContent);
335
+ const result = {};
336
+ const properties = (0, exports.getElementsByTagName)(xml, "property");
337
+ for (const prop of properties) {
338
+ const name = prop.getAttribute("name");
339
+ if (!name)
340
+ continue;
341
+ // The value is the first child element (typed using vt: namespace)
342
+ for (let i = 0; i < prop.childNodes.length; i++) {
343
+ const child = prop.childNodes[i];
344
+ if (child.nodeType !== 1)
345
+ continue; // skip non-elements
346
+ const el = child;
347
+ const tag = el.tagName || '';
348
+ const text = el.textContent || '';
349
+ if (/vt:lpwstr|vt:lpstr|vt:bstr/.test(tag)) {
350
+ result[name] = text;
351
+ }
352
+ else if (/vt:bool/.test(tag)) {
353
+ result[name] = text.toLowerCase() === 'true';
354
+ }
355
+ else if (/vt:(i[1248]|ui[1248]|int|uint|r4|r8|decimal)/.test(tag)) {
356
+ const num = Number(text);
357
+ if (!isNaN(num))
358
+ result[name] = num;
359
+ }
360
+ else if (/vt:filetime|vt:date/.test(tag)) {
361
+ const date = (0, dateUtils_js_1.parseOfficeDate)(text);
362
+ if (date)
363
+ result[name] = date;
364
+ else
365
+ result[name] = text;
366
+ }
367
+ else if (text) {
368
+ // Fallback: store as string for any other vt: type
369
+ result[name] = text;
370
+ }
371
+ break; // only one value element per property
372
+ }
373
+ }
374
+ return result;
375
+ };
376
+ exports.parseOOXMLCustomProperties = parseOOXMLCustomProperties;