officeparser 6.0.7 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -13
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +43 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +116 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +79 -1
- package/dist/officeparser.browser.iife.js +112 -0
- package/dist/officeparser.browser.mjs +111 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +71 -63
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +131 -114
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +85 -88
- package/dist/parsers/RtfParser.d.ts +1 -1
- package/dist/parsers/RtfParser.js +10 -6
- package/dist/parsers/WordParser.d.ts +1 -1
- package/dist/parsers/WordParser.js +109 -101
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +69 -1
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/dist/utils/xmlUtils.d.ts
CHANGED
|
@@ -10,23 +10,22 @@
|
|
|
10
10
|
* @module xmlUtils
|
|
11
11
|
*/
|
|
12
12
|
import { OfficeMetadata } from '../types';
|
|
13
|
+
/**
|
|
14
|
+
* Type guard for Element nodes.
|
|
15
|
+
*/
|
|
16
|
+
export declare const isElement: (node: Node) => node is Element;
|
|
13
17
|
/**
|
|
14
18
|
* Parses an XML string into a DOM Document object.
|
|
15
19
|
*
|
|
16
20
|
* Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
|
|
17
|
-
* This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
|
|
18
21
|
*
|
|
19
22
|
* @param xml - The XML content as a string
|
|
23
|
+
* @param options - Optional parser settings (e.g., enable locators for source mapping)
|
|
20
24
|
* @returns A Document object that can be queried using standard DOM methods
|
|
21
|
-
* @example
|
|
22
|
-
* ```typescript
|
|
23
|
-
* const xmlString = '<root><item>Hello</item></root>';
|
|
24
|
-
* const doc = parseXmlString(xmlString);
|
|
25
|
-
* const items = doc.getElementsByTagName('item');
|
|
26
|
-
* console.log(items[0].textContent); // "Hello"
|
|
27
|
-
* ```
|
|
28
25
|
*/
|
|
29
|
-
export declare const parseXmlString: (xml: string
|
|
26
|
+
export declare const parseXmlString: (xml: string, options?: {
|
|
27
|
+
locator?: boolean;
|
|
28
|
+
}) => Document;
|
|
30
29
|
/**
|
|
31
30
|
* Gets all elements with a specific tag name and returns them as an array.
|
|
32
31
|
*
|
|
@@ -43,6 +42,54 @@ export declare const parseXmlString: (xml: string) => Document;
|
|
|
43
42
|
* ```
|
|
44
43
|
*/
|
|
45
44
|
export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
|
|
45
|
+
/**
|
|
46
|
+
* Serializes a DOM Node (Document, Element, etc.) back into an XML string.
|
|
47
|
+
* This is cross-platform and works in both Node.js and Browser environments.
|
|
48
|
+
*
|
|
49
|
+
* @param node - The DOM node to serialize
|
|
50
|
+
* @param options - Serialization options
|
|
51
|
+
* @returns The XML string representation
|
|
52
|
+
*/
|
|
53
|
+
export declare const serializeXml: (node: Node, options?: {
|
|
54
|
+
preserveWhitespace?: boolean;
|
|
55
|
+
}) => string;
|
|
56
|
+
/**
|
|
57
|
+
* Attempts to extract the original raw substring from the source XML for a given node.
|
|
58
|
+
* Requires the document to have been parsed with { locator: true }.
|
|
59
|
+
*
|
|
60
|
+
* @param node - The DOM node to extract source for
|
|
61
|
+
* @param sourceXml - The original XML source string
|
|
62
|
+
* @returns The raw XML substring, or undefined if it cannot be reliably determined
|
|
63
|
+
*/
|
|
64
|
+
export declare const getSourceSubstring: (node: any, sourceXml: string) => string | undefined;
|
|
65
|
+
/**
|
|
66
|
+
* High-level helper to get raw content for a node based on OfficeParserConfig.
|
|
67
|
+
*
|
|
68
|
+
* @param node - The DOM node
|
|
69
|
+
* @param sourceXml - The original source XML string
|
|
70
|
+
* @param config - The parser configuration
|
|
71
|
+
* @returns The raw content string (serialized or original)
|
|
72
|
+
*/
|
|
73
|
+
export declare const getRawContent: (node: Node, sourceXml: string, config: {
|
|
74
|
+
serializeRawContent?: boolean;
|
|
75
|
+
preserveXmlWhitespace?: boolean;
|
|
76
|
+
}) => string;
|
|
77
|
+
/**
|
|
78
|
+
* Gets the first element with the specified tag name within a parent element.
|
|
79
|
+
*
|
|
80
|
+
* @param parent - The parent element or document to search within
|
|
81
|
+
* @param tagName - The tag name to search for
|
|
82
|
+
* @returns The first matching element, or undefined if none found
|
|
83
|
+
*/
|
|
84
|
+
export declare const getFirstElementByTagName: (parent: Element | Document, tagName: string) => Element | undefined;
|
|
85
|
+
/**
|
|
86
|
+
* Gets the value of an attribute from an element.
|
|
87
|
+
*
|
|
88
|
+
* @param element - The element to get the attribute from
|
|
89
|
+
* @param attrName - The name of the attribute
|
|
90
|
+
* @returns The attribute value or undefined if not set
|
|
91
|
+
*/
|
|
92
|
+
export declare const getAttribute: (element: Element, attrName: string) => string | undefined;
|
|
46
93
|
/**
|
|
47
94
|
* Gets direct child elements with a specific tag name.
|
|
48
95
|
* Unlike getElementsByTagName, this does not search recursively.
|
|
@@ -81,3 +128,27 @@ export declare const getDirectChildren: (parent: Element, tagName: string) => El
|
|
|
81
128
|
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
|
|
82
129
|
*/
|
|
83
130
|
export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata;
|
|
131
|
+
/**
|
|
132
|
+
* Parses OOXML custom document properties from `docProps/custom.xml`.
|
|
133
|
+
*
|
|
134
|
+
* Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
|
|
135
|
+
* (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
|
|
136
|
+
*
|
|
137
|
+
* Property values are typed using the `vt:` namespace (docPropsVTypes):
|
|
138
|
+
* - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
|
|
139
|
+
* - `vt:bool` → boolean
|
|
140
|
+
* - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
|
|
141
|
+
* - `vt:filetime` / `vt:date` → Date
|
|
142
|
+
*
|
|
143
|
+
* @param xmlContent - Raw XML string from `docProps/custom.xml`
|
|
144
|
+
* @returns A record of property name → typed value (empty object if none found)
|
|
145
|
+
* @example
|
|
146
|
+
* ```typescript
|
|
147
|
+
* const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
|
|
148
|
+
* const props = parseOOXMLCustomProperties(customXml);
|
|
149
|
+
* console.log(props['Department']); // "Engineering"
|
|
150
|
+
* console.log(props['Priority']); // 1 (number)
|
|
151
|
+
* console.log(props['Reviewed']); // true (boolean)
|
|
152
|
+
* ```
|
|
153
|
+
*/
|
|
154
|
+
export declare const parseOOXMLCustomProperties: (xmlContent: string) => Record<string, string | number | boolean | Date>;
|
package/dist/utils/xmlUtils.js
CHANGED
|
@@ -11,27 +11,32 @@
|
|
|
11
11
|
* @module xmlUtils
|
|
12
12
|
*/
|
|
13
13
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
|
-
exports.parseOfficeMetadata = exports.getDirectChildren = exports.getElementsByTagName = exports.parseXmlString = void 0;
|
|
14
|
+
exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
|
|
15
15
|
const xmldom_1 = require("@xmldom/xmldom");
|
|
16
|
+
const dateUtils_js_1 = require("./dateUtils.js");
|
|
17
|
+
/**
|
|
18
|
+
* Type guard for Element nodes.
|
|
19
|
+
*/
|
|
20
|
+
const isElement = (node) => {
|
|
21
|
+
return node.nodeType === 1;
|
|
22
|
+
};
|
|
23
|
+
exports.isElement = isElement;
|
|
16
24
|
/**
|
|
17
25
|
* Parses an XML string into a DOM Document object.
|
|
18
26
|
*
|
|
19
27
|
* Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
|
|
20
|
-
* This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
|
|
21
28
|
*
|
|
22
29
|
* @param xml - The XML content as a string
|
|
30
|
+
* @param options - Optional parser settings (e.g., enable locators for source mapping)
|
|
23
31
|
* @returns A Document object that can be queried using standard DOM methods
|
|
24
|
-
* @example
|
|
25
|
-
* ```typescript
|
|
26
|
-
* const xmlString = '<root><item>Hello</item></root>';
|
|
27
|
-
* const doc = parseXmlString(xmlString);
|
|
28
|
-
* const items = doc.getElementsByTagName('item');
|
|
29
|
-
* console.log(items[0].textContent); // "Hello"
|
|
30
|
-
* ```
|
|
31
32
|
*/
|
|
32
|
-
const parseXmlString = (xml) => {
|
|
33
|
-
const parser = new xmldom_1.DOMParser();
|
|
34
|
-
|
|
33
|
+
const parseXmlString = (xml, options = {}) => {
|
|
34
|
+
const parser = new xmldom_1.DOMParser(options);
|
|
35
|
+
// @xmldom/xmldom 0.9.x is strict: a UTF-8 BOM (U+FEFF) prepended to the
|
|
36
|
+
// XML string causes a fatalError because the XML declaration is no longer
|
|
37
|
+
// at position 0. Strip it before parsing.
|
|
38
|
+
const sanitized = xml.charCodeAt(0) === 0xFEFF ? xml.slice(1) : xml.trim();
|
|
39
|
+
return parser.parseFromString(sanitized, "text/xml");
|
|
35
40
|
};
|
|
36
41
|
exports.parseXmlString = parseXmlString;
|
|
37
42
|
/**
|
|
@@ -50,9 +55,115 @@ exports.parseXmlString = parseXmlString;
|
|
|
50
55
|
* ```
|
|
51
56
|
*/
|
|
52
57
|
const getElementsByTagName = (element, tagName) => {
|
|
53
|
-
|
|
58
|
+
const results = Array.from(element.getElementsByTagName(tagName));
|
|
59
|
+
// Resilience: If prefixed tag (e.g., 'dc:title') not found, try local name (e.g., 'title')
|
|
60
|
+
if (results.length === 0 && tagName.includes(':')) {
|
|
61
|
+
const localName = tagName.split(':').pop();
|
|
62
|
+
return Array.from(element.getElementsByTagName(localName));
|
|
63
|
+
}
|
|
64
|
+
return results;
|
|
54
65
|
};
|
|
55
66
|
exports.getElementsByTagName = getElementsByTagName;
|
|
67
|
+
/**
|
|
68
|
+
* Serializes a DOM Node (Document, Element, etc.) back into an XML string.
|
|
69
|
+
* This is cross-platform and works in both Node.js and Browser environments.
|
|
70
|
+
*
|
|
71
|
+
* @param node - The DOM node to serialize
|
|
72
|
+
* @param options - Serialization options
|
|
73
|
+
* @returns The XML string representation
|
|
74
|
+
*/
|
|
75
|
+
const serializeXml = (node, options = {}) => {
|
|
76
|
+
// Note: xmldom's XMLSerializer doesn't natively support a 'pretty' or 'preserve'
|
|
77
|
+
// flag in a way that matches all user expectations, but it defaults to
|
|
78
|
+
// preserving structure. Formatting (indentation) is usually handled by the
|
|
79
|
+
// parser's initial whitespace handling.
|
|
80
|
+
// @ts-ignore - xmldom's Node is compatible with the global Node interface
|
|
81
|
+
return new xmldom_1.XMLSerializer().serializeToString(node);
|
|
82
|
+
};
|
|
83
|
+
exports.serializeXml = serializeXml;
|
|
84
|
+
/**
|
|
85
|
+
* Attempts to extract the original raw substring from the source XML for a given node.
|
|
86
|
+
* Requires the document to have been parsed with { locator: true }.
|
|
87
|
+
*
|
|
88
|
+
* @param node - The DOM node to extract source for
|
|
89
|
+
* @param sourceXml - The original XML source string
|
|
90
|
+
* @returns The raw XML substring, or undefined if it cannot be reliably determined
|
|
91
|
+
*/
|
|
92
|
+
const getSourceSubstring = (node, sourceXml) => {
|
|
93
|
+
if (!node || typeof node.lineNumber !== 'number' || typeof node.columnNumber !== 'number') {
|
|
94
|
+
return undefined;
|
|
95
|
+
}
|
|
96
|
+
// Convert line/column to absolute index
|
|
97
|
+
const lines = sourceXml.split('\n');
|
|
98
|
+
let startIdx = 0;
|
|
99
|
+
for (let i = 0; i < node.lineNumber - 1; i++) {
|
|
100
|
+
startIdx += lines[i].length + 1; // +1 for newline
|
|
101
|
+
}
|
|
102
|
+
startIdx += node.columnNumber - 1;
|
|
103
|
+
// To find the end of the node, we look for the closing tag.
|
|
104
|
+
// This is a heuristic approach that works well for simple structured nodes (p, tbl, etc.)
|
|
105
|
+
// but might be complex for overlapping namespaces or malformed XML.
|
|
106
|
+
if ((0, exports.isElement)(node)) {
|
|
107
|
+
const tagName = node.tagName;
|
|
108
|
+
const closingTag = `</${tagName}>`;
|
|
109
|
+
const endIdx = sourceXml.indexOf(closingTag, startIdx);
|
|
110
|
+
if (endIdx !== -1) {
|
|
111
|
+
return sourceXml.substring(startIdx, endIdx + closingTag.length);
|
|
112
|
+
}
|
|
113
|
+
// Self-closing tag handling (e.g., <w:p/>)
|
|
114
|
+
const selfClosingEnd = sourceXml.indexOf('/>', startIdx);
|
|
115
|
+
const nextOpenTag = sourceXml.indexOf('<', startIdx + 1);
|
|
116
|
+
if (selfClosingEnd !== -1 && (nextOpenTag === -1 || selfClosingEnd < nextOpenTag)) {
|
|
117
|
+
return sourceXml.substring(startIdx, selfClosingEnd + 2);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
return undefined;
|
|
121
|
+
};
|
|
122
|
+
exports.getSourceSubstring = getSourceSubstring;
|
|
123
|
+
/**
|
|
124
|
+
* High-level helper to get raw content for a node based on OfficeParserConfig.
|
|
125
|
+
*
|
|
126
|
+
* @param node - The DOM node
|
|
127
|
+
* @param sourceXml - The original source XML string
|
|
128
|
+
* @param config - The parser configuration
|
|
129
|
+
* @returns The raw content string (serialized or original)
|
|
130
|
+
*/
|
|
131
|
+
const getRawContent = (node, sourceXml, config) => {
|
|
132
|
+
if (config.serializeRawContent === false) {
|
|
133
|
+
const original = (0, exports.getSourceSubstring)(node, sourceXml);
|
|
134
|
+
if (original)
|
|
135
|
+
return original;
|
|
136
|
+
}
|
|
137
|
+
return (0, exports.serializeXml)(node, { preserveWhitespace: config.preserveXmlWhitespace });
|
|
138
|
+
};
|
|
139
|
+
exports.getRawContent = getRawContent;
|
|
140
|
+
/**
|
|
141
|
+
* Gets the first element with the specified tag name within a parent element.
|
|
142
|
+
*
|
|
143
|
+
* @param parent - The parent element or document to search within
|
|
144
|
+
* @param tagName - The tag name to search for
|
|
145
|
+
* @returns The first matching element, or undefined if none found
|
|
146
|
+
*/
|
|
147
|
+
const getFirstElementByTagName = (parent, tagName) => {
|
|
148
|
+
const elements = parent.getElementsByTagName(tagName);
|
|
149
|
+
if (elements && elements.length > 0) {
|
|
150
|
+
return elements[0];
|
|
151
|
+
}
|
|
152
|
+
return undefined;
|
|
153
|
+
};
|
|
154
|
+
exports.getFirstElementByTagName = getFirstElementByTagName;
|
|
155
|
+
/**
|
|
156
|
+
* Gets the value of an attribute from an element.
|
|
157
|
+
*
|
|
158
|
+
* @param element - The element to get the attribute from
|
|
159
|
+
* @param attrName - The name of the attribute
|
|
160
|
+
* @returns The attribute value or undefined if not set
|
|
161
|
+
*/
|
|
162
|
+
const getAttribute = (element, attrName) => {
|
|
163
|
+
const attr = element.getAttribute(attrName);
|
|
164
|
+
return attr !== null ? attr : undefined;
|
|
165
|
+
};
|
|
166
|
+
exports.getAttribute = getAttribute;
|
|
56
167
|
/**
|
|
57
168
|
* Gets direct child elements with a specific tag name.
|
|
58
169
|
* Unlike getElementsByTagName, this does not search recursively.
|
|
@@ -67,7 +178,7 @@ const getDirectChildren = (parent, tagName) => {
|
|
|
67
178
|
return result;
|
|
68
179
|
for (let i = 0; i < parent.childNodes.length; i++) {
|
|
69
180
|
const child = parent.childNodes[i];
|
|
70
|
-
if (child
|
|
181
|
+
if ((0, exports.isElement)(child) && child.tagName === tagName) {
|
|
71
182
|
result.push(child);
|
|
72
183
|
}
|
|
73
184
|
}
|
|
@@ -124,11 +235,18 @@ const parseOfficeMetadata = (xmlContent) => {
|
|
|
124
235
|
// Step 6: Extract creation date (Dublin Core Terms element)
|
|
125
236
|
const created = (0, exports.getElementsByTagName)(coreProperties, "dcterms:created")[0];
|
|
126
237
|
if (created && created.textContent)
|
|
127
|
-
metadata.created =
|
|
238
|
+
metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
|
|
128
239
|
// Step 7: Extract last modification date (Dublin Core Terms element)
|
|
129
240
|
const modified = (0, exports.getElementsByTagName)(coreProperties, "dcterms:modified")[0];
|
|
130
241
|
if (modified && modified.textContent)
|
|
131
|
-
metadata.modified =
|
|
242
|
+
metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
|
|
243
|
+
// Step 8: Extract description and subject (Dublin Core elements)
|
|
244
|
+
const description = (0, exports.getElementsByTagName)(coreProperties, "dc:description")[0];
|
|
245
|
+
if (description && description.textContent)
|
|
246
|
+
metadata.description = description.textContent;
|
|
247
|
+
const subject = (0, exports.getElementsByTagName)(coreProperties, "dc:subject")[0];
|
|
248
|
+
if (subject && subject.textContent)
|
|
249
|
+
metadata.subject = subject.textContent;
|
|
132
250
|
return metadata;
|
|
133
251
|
}
|
|
134
252
|
// Check for ODF Meta
|
|
@@ -148,11 +266,111 @@ const parseOfficeMetadata = (xmlContent) => {
|
|
|
148
266
|
metadata.subject = subject.textContent;
|
|
149
267
|
const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
|
|
150
268
|
if (created && created.textContent)
|
|
151
|
-
metadata.created =
|
|
269
|
+
metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
|
|
152
270
|
const modified = (0, exports.getElementsByTagName)(officeMeta, "dc:date")[0];
|
|
153
271
|
if (modified && modified.textContent)
|
|
154
|
-
metadata.modified =
|
|
272
|
+
metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
|
|
273
|
+
// Extract user-defined custom properties (meta:user-defined)
|
|
274
|
+
const userDefined = (0, exports.getElementsByTagName)(officeMeta, "meta:user-defined");
|
|
275
|
+
if (userDefined.length > 0) {
|
|
276
|
+
const customProperties = {};
|
|
277
|
+
for (const el of userDefined) {
|
|
278
|
+
const name = el.getAttribute("meta:name");
|
|
279
|
+
if (!name || !el.textContent)
|
|
280
|
+
continue;
|
|
281
|
+
const valueType = el.getAttribute("meta:value-type") || "string";
|
|
282
|
+
const raw = el.textContent;
|
|
283
|
+
if (valueType === "boolean") {
|
|
284
|
+
customProperties[name] = raw.toLowerCase() === "true";
|
|
285
|
+
}
|
|
286
|
+
else if (valueType === "float") {
|
|
287
|
+
const num = Number(raw);
|
|
288
|
+
if (!isNaN(num))
|
|
289
|
+
customProperties[name] = num;
|
|
290
|
+
}
|
|
291
|
+
else if (valueType === "date" || valueType === "time") {
|
|
292
|
+
const date = (0, dateUtils_js_1.parseOfficeDate)(raw);
|
|
293
|
+
if (date)
|
|
294
|
+
customProperties[name] = date;
|
|
295
|
+
else
|
|
296
|
+
customProperties[name] = raw;
|
|
297
|
+
}
|
|
298
|
+
else {
|
|
299
|
+
customProperties[name] = raw;
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
if (Object.keys(customProperties).length > 0) {
|
|
303
|
+
metadata.customProperties = customProperties;
|
|
304
|
+
}
|
|
305
|
+
}
|
|
155
306
|
}
|
|
156
307
|
return metadata;
|
|
157
308
|
};
|
|
158
309
|
exports.parseOfficeMetadata = parseOfficeMetadata;
|
|
310
|
+
/**
|
|
311
|
+
* Parses OOXML custom document properties from `docProps/custom.xml`.
|
|
312
|
+
*
|
|
313
|
+
* Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
|
|
314
|
+
* (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
|
|
315
|
+
*
|
|
316
|
+
* Property values are typed using the `vt:` namespace (docPropsVTypes):
|
|
317
|
+
* - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
|
|
318
|
+
* - `vt:bool` → boolean
|
|
319
|
+
* - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
|
|
320
|
+
* - `vt:filetime` / `vt:date` → Date
|
|
321
|
+
*
|
|
322
|
+
* @param xmlContent - Raw XML string from `docProps/custom.xml`
|
|
323
|
+
* @returns A record of property name → typed value (empty object if none found)
|
|
324
|
+
* @example
|
|
325
|
+
* ```typescript
|
|
326
|
+
* const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
|
|
327
|
+
* const props = parseOOXMLCustomProperties(customXml);
|
|
328
|
+
* console.log(props['Department']); // "Engineering"
|
|
329
|
+
* console.log(props['Priority']); // 1 (number)
|
|
330
|
+
* console.log(props['Reviewed']); // true (boolean)
|
|
331
|
+
* ```
|
|
332
|
+
*/
|
|
333
|
+
const parseOOXMLCustomProperties = (xmlContent) => {
|
|
334
|
+
const xml = (0, exports.parseXmlString)(xmlContent);
|
|
335
|
+
const result = {};
|
|
336
|
+
const properties = (0, exports.getElementsByTagName)(xml, "property");
|
|
337
|
+
for (const prop of properties) {
|
|
338
|
+
const name = prop.getAttribute("name");
|
|
339
|
+
if (!name)
|
|
340
|
+
continue;
|
|
341
|
+
// The value is the first child element (typed using vt: namespace)
|
|
342
|
+
for (let i = 0; i < prop.childNodes.length; i++) {
|
|
343
|
+
const child = prop.childNodes[i];
|
|
344
|
+
if (child.nodeType !== 1)
|
|
345
|
+
continue; // skip non-elements
|
|
346
|
+
const el = child;
|
|
347
|
+
const tag = el.tagName || '';
|
|
348
|
+
const text = el.textContent || '';
|
|
349
|
+
if (/vt:lpwstr|vt:lpstr|vt:bstr/.test(tag)) {
|
|
350
|
+
result[name] = text;
|
|
351
|
+
}
|
|
352
|
+
else if (/vt:bool/.test(tag)) {
|
|
353
|
+
result[name] = text.toLowerCase() === 'true';
|
|
354
|
+
}
|
|
355
|
+
else if (/vt:(i[1248]|ui[1248]|int|uint|r4|r8|decimal)/.test(tag)) {
|
|
356
|
+
const num = Number(text);
|
|
357
|
+
if (!isNaN(num))
|
|
358
|
+
result[name] = num;
|
|
359
|
+
}
|
|
360
|
+
else if (/vt:filetime|vt:date/.test(tag)) {
|
|
361
|
+
const date = (0, dateUtils_js_1.parseOfficeDate)(text);
|
|
362
|
+
if (date)
|
|
363
|
+
result[name] = date;
|
|
364
|
+
else
|
|
365
|
+
result[name] = text;
|
|
366
|
+
}
|
|
367
|
+
else if (text) {
|
|
368
|
+
// Fallback: store as string for any other vt: type
|
|
369
|
+
result[name] = text;
|
|
370
|
+
}
|
|
371
|
+
break; // only one value element per property
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
return result;
|
|
375
|
+
};
|
|
376
|
+
exports.parseOOXMLCustomProperties = parseOOXMLCustomProperties;
|
package/dist/utils/zipUtils.js
CHANGED
|
@@ -14,13 +14,9 @@
|
|
|
14
14
|
*
|
|
15
15
|
* @module zipUtils
|
|
16
16
|
*/
|
|
17
|
-
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
18
|
-
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
19
|
-
};
|
|
20
17
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
21
18
|
exports.extractFiles = void 0;
|
|
22
|
-
const
|
|
23
|
-
const concat_stream_1 = __importDefault(require("concat-stream"));
|
|
19
|
+
const fflate_1 = require("fflate");
|
|
24
20
|
/**
|
|
25
21
|
* Extracts files from a ZIP archive with optional filtering.
|
|
26
22
|
*
|
|
@@ -62,50 +58,13 @@ const concat_stream_1 = __importDefault(require("concat-stream"));
|
|
|
62
58
|
*/
|
|
63
59
|
const extractFiles = (zipInput, filterFn) => {
|
|
64
60
|
return new Promise((resolve, reject) => {
|
|
65
|
-
|
|
66
|
-
// lazyEntries: true means we manually control when to read each entry (better memory usage)
|
|
67
|
-
yauzl_1.default.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
|
|
61
|
+
(0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), { filter: (file) => filterFn(file.name) }, (err, decompressed) => {
|
|
68
62
|
if (err)
|
|
69
63
|
return reject(err);
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
// Step 2: Start reading the first entry
|
|
75
|
-
// This triggers the 'entry' event
|
|
76
|
-
zipfile.readEntry();
|
|
77
|
-
// Step 3: Handle each entry (file or directory) in the ZIP
|
|
78
|
-
zipfile.on('entry', (entry) => {
|
|
79
|
-
// Step 3a: Check if this file should be extracted using the filter function
|
|
80
|
-
if (filterFn(entry.fileName)) {
|
|
81
|
-
// Step 3b: Open a read stream for this entry
|
|
82
|
-
zipfile.openReadStream(entry, (err, readStream) => {
|
|
83
|
-
if (err)
|
|
84
|
-
return reject(err);
|
|
85
|
-
if (!readStream)
|
|
86
|
-
return reject(new Error("Failed to open read stream"));
|
|
87
|
-
// Step 3c: Pipe the stream through concat to collect all data into a single Buffer
|
|
88
|
-
// This is necessary because streams deliver data in chunks
|
|
89
|
-
readStream.pipe((0, concat_stream_1.default)((data) => {
|
|
90
|
-
// Step 3d: Add the extracted file to our results
|
|
91
|
-
extractedFiles.push({
|
|
92
|
-
path: entry.fileName,
|
|
93
|
-
content: data
|
|
94
|
-
});
|
|
95
|
-
// Step 3e: Continue to the next entry
|
|
96
|
-
zipfile.readEntry();
|
|
97
|
-
}));
|
|
98
|
-
});
|
|
99
|
-
}
|
|
100
|
-
else {
|
|
101
|
-
// Step 3f: Skip this entry and move to the next one
|
|
102
|
-
zipfile.readEntry();
|
|
103
|
-
}
|
|
104
|
-
});
|
|
105
|
-
// Step 4: All entries have been processed
|
|
106
|
-
zipfile.on('end', () => resolve(extractedFiles));
|
|
107
|
-
// Step 5: Handle any errors during extraction
|
|
108
|
-
zipfile.on('error', reject);
|
|
64
|
+
resolve(Object.entries(decompressed).map(([path, data]) => ({
|
|
65
|
+
path,
|
|
66
|
+
content: Buffer.from(data)
|
|
67
|
+
})));
|
|
109
68
|
});
|
|
110
69
|
});
|
|
111
70
|
};
|
package/package.json
CHANGED
|
@@ -1,9 +1,20 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "6.0
|
|
3
|
+
"version": "6.1.0",
|
|
4
4
|
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf) into structured AST with rich metadata, formatting, and attachment support.",
|
|
5
|
+
"funding": "https://github.com/sponsors/harshankur",
|
|
5
6
|
"main": "dist/index.js",
|
|
7
|
+
"module": "dist/index.mjs",
|
|
6
8
|
"types": "dist/index.d.ts",
|
|
9
|
+
"browser": "./dist/officeparser.browser.mjs",
|
|
10
|
+
"exports": {
|
|
11
|
+
".": {
|
|
12
|
+
"types": "./dist/index.d.ts",
|
|
13
|
+
"browser": "./dist/officeparser.browser.mjs",
|
|
14
|
+
"import": "./dist/index.mjs",
|
|
15
|
+
"require": "./dist/index.js"
|
|
16
|
+
}
|
|
17
|
+
},
|
|
7
18
|
"sideEffects": false,
|
|
8
19
|
"engines": {
|
|
9
20
|
"node": ">=18.0.0"
|
|
@@ -12,15 +23,21 @@
|
|
|
12
23
|
"dist"
|
|
13
24
|
],
|
|
14
25
|
"scripts": {
|
|
15
|
-
"build": "npm run sync:versions && npm run build:node && npm run build:browser:types && npm run build:browser",
|
|
26
|
+
"build": "npm run sync:versions && npm run build:node && npm run build:esm-wrapper && npm run build:browser:types && npm run build:browser",
|
|
16
27
|
"build:node": "tsc",
|
|
28
|
+
"build:esm-wrapper": "node scripts/generate-esm-wrapper.js",
|
|
17
29
|
"build:browser:types": "dts-bundle-generator --no-check -o dist/officeparser.browser.d.ts src/index.ts",
|
|
18
30
|
"build:browser": "node build_browser.js && npm run sync:docs",
|
|
19
31
|
"sync:versions": "node scripts/sync-pdfjs-versions.js",
|
|
20
|
-
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.js docs/dist/",
|
|
21
|
-
"test": "npm run test:clean && npm run build &&
|
|
32
|
+
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/",
|
|
33
|
+
"test": "npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser",
|
|
34
|
+
"test:baseline": "npm run test baseline",
|
|
35
|
+
"test:parser": "npx tsx test/testOfficeParser.ts",
|
|
36
|
+
"test:artifacts": "npx tsx test/testShippingArtifacts.ts",
|
|
37
|
+
"test:license": "npm run sbom && node scripts/validate-licenses.js",
|
|
22
38
|
"test:clean": "rm -rf test/results",
|
|
23
39
|
"clean": "rm -rf dist && npm run test:clean",
|
|
40
|
+
"sbom": "npx --yes @cyclonedx/cyclonedx-npm --output-format json --output-file dist/sbom.cdx.json --omit dev",
|
|
24
41
|
"prepublishOnly": "npm run build",
|
|
25
42
|
"prepare": "husky"
|
|
26
43
|
},
|
|
@@ -29,7 +46,7 @@
|
|
|
29
46
|
"url": "git+https://github.com/harshankur/officeParser.git"
|
|
30
47
|
},
|
|
31
48
|
"bin": {
|
|
32
|
-
"officeparser": "dist/
|
|
49
|
+
"officeparser": "dist/cli.js"
|
|
33
50
|
},
|
|
34
51
|
"publishConfig": {
|
|
35
52
|
"access": "public",
|
|
@@ -74,22 +91,20 @@
|
|
|
74
91
|
},
|
|
75
92
|
"homepage": "https://officeparser.harshankur.com",
|
|
76
93
|
"dependencies": {
|
|
77
|
-
"@xmldom/xmldom": "^0.
|
|
78
|
-
"
|
|
79
|
-
"file-type": "^
|
|
80
|
-
"pdfjs-dist": "
|
|
81
|
-
"tesseract.js": "^7.0.0"
|
|
82
|
-
"yauzl": "^3.2.1"
|
|
94
|
+
"@xmldom/xmldom": "^0.9.9",
|
|
95
|
+
"fflate": "^0.8.2",
|
|
96
|
+
"file-type": "^22.0.1",
|
|
97
|
+
"pdfjs-dist": "5.6.205",
|
|
98
|
+
"tesseract.js": "^7.0.0"
|
|
83
99
|
},
|
|
84
100
|
"devDependencies": {
|
|
85
|
-
"@types/
|
|
86
|
-
"
|
|
87
|
-
"@types/xmldom": "^0.1.34",
|
|
88
|
-
"@types/yauzl": "^2.10.3",
|
|
101
|
+
"@types/node": "^25.5.0",
|
|
102
|
+
"buffer": "^6.0.3",
|
|
89
103
|
"dts-bundle-generator": "^9.5.1",
|
|
90
104
|
"esbuild": "^0.27.4",
|
|
91
|
-
"esbuild-
|
|
105
|
+
"esbuild-plugins-node-modules-polyfill": "^1.8.1",
|
|
92
106
|
"husky": "^9.1.7",
|
|
107
|
+
"process": "^0.11.10",
|
|
93
108
|
"tsx": "^4.21.0",
|
|
94
109
|
"typescript": "^6.0.2"
|
|
95
110
|
}
|