officeparser 6.0.6 → 6.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +92 -13
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +43 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +116 -0
  6. package/dist/index.d.ts +3 -3
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +756 -0
  10. package/dist/officeparser.browser.iife.js +112 -0
  11. package/dist/officeparser.browser.mjs +111 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +71 -63
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +131 -114
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +85 -88
  20. package/dist/parsers/RtfParser.d.ts +1 -1
  21. package/dist/parsers/RtfParser.js +10 -6
  22. package/dist/parsers/WordParser.d.ts +1 -1
  23. package/dist/parsers/WordParser.js +109 -101
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +69 -1
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +39 -18
  39. package/dist/officeparser.browser.js +0 -153
  40. package/dist/officeparser.browser.js.map +0 -7
@@ -10,23 +10,22 @@
10
10
  * @module xmlUtils
11
11
  */
12
12
  import { OfficeMetadata } from '../types';
13
+ /**
14
+ * Type guard for Element nodes.
15
+ */
16
+ export declare const isElement: (node: Node) => node is Element;
13
17
  /**
14
18
  * Parses an XML string into a DOM Document object.
15
19
  *
16
20
  * Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
17
- * This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
18
21
  *
19
22
  * @param xml - The XML content as a string
23
+ * @param options - Optional parser settings (e.g., enable locators for source mapping)
20
24
  * @returns A Document object that can be queried using standard DOM methods
21
- * @example
22
- * ```typescript
23
- * const xmlString = '<root><item>Hello</item></root>';
24
- * const doc = parseXmlString(xmlString);
25
- * const items = doc.getElementsByTagName('item');
26
- * console.log(items[0].textContent); // "Hello"
27
- * ```
28
25
  */
29
- export declare const parseXmlString: (xml: string) => Document;
26
+ export declare const parseXmlString: (xml: string, options?: {
27
+ locator?: boolean;
28
+ }) => Document;
30
29
  /**
31
30
  * Gets all elements with a specific tag name and returns them as an array.
32
31
  *
@@ -43,6 +42,54 @@ export declare const parseXmlString: (xml: string) => Document;
43
42
  * ```
44
43
  */
45
44
  export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
45
+ /**
46
+ * Serializes a DOM Node (Document, Element, etc.) back into an XML string.
47
+ * This is cross-platform and works in both Node.js and Browser environments.
48
+ *
49
+ * @param node - The DOM node to serialize
50
+ * @param options - Serialization options
51
+ * @returns The XML string representation
52
+ */
53
+ export declare const serializeXml: (node: Node, options?: {
54
+ preserveWhitespace?: boolean;
55
+ }) => string;
56
+ /**
57
+ * Attempts to extract the original raw substring from the source XML for a given node.
58
+ * Requires the document to have been parsed with { locator: true }.
59
+ *
60
+ * @param node - The DOM node to extract source for
61
+ * @param sourceXml - The original XML source string
62
+ * @returns The raw XML substring, or undefined if it cannot be reliably determined
63
+ */
64
+ export declare const getSourceSubstring: (node: any, sourceXml: string) => string | undefined;
65
+ /**
66
+ * High-level helper to get raw content for a node based on OfficeParserConfig.
67
+ *
68
+ * @param node - The DOM node
69
+ * @param sourceXml - The original source XML string
70
+ * @param config - The parser configuration
71
+ * @returns The raw content string (serialized or original)
72
+ */
73
+ export declare const getRawContent: (node: Node, sourceXml: string, config: {
74
+ serializeRawContent?: boolean;
75
+ preserveXmlWhitespace?: boolean;
76
+ }) => string;
77
+ /**
78
+ * Gets the first element with the specified tag name within a parent element.
79
+ *
80
+ * @param parent - The parent element or document to search within
81
+ * @param tagName - The tag name to search for
82
+ * @returns The first matching element, or undefined if none found
83
+ */
84
+ export declare const getFirstElementByTagName: (parent: Element | Document, tagName: string) => Element | undefined;
85
+ /**
86
+ * Gets the value of an attribute from an element.
87
+ *
88
+ * @param element - The element to get the attribute from
89
+ * @param attrName - The name of the attribute
90
+ * @returns The attribute value or undefined if not set
91
+ */
92
+ export declare const getAttribute: (element: Element, attrName: string) => string | undefined;
46
93
  /**
47
94
  * Gets direct child elements with a specific tag name.
48
95
  * Unlike getElementsByTagName, this does not search recursively.
@@ -81,3 +128,27 @@ export declare const getDirectChildren: (parent: Element, tagName: string) => El
81
128
  * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-oe376/6c085e39-c695-4f83-91e8-3f277bb4e111
82
129
  */
83
130
  export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata;
131
+ /**
132
+ * Parses OOXML custom document properties from `docProps/custom.xml`.
133
+ *
134
+ * Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
135
+ * (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
136
+ *
137
+ * Property values are typed using the `vt:` namespace (docPropsVTypes):
138
+ * - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
139
+ * - `vt:bool` → boolean
140
+ * - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
141
+ * - `vt:filetime` / `vt:date` → Date
142
+ *
143
+ * @param xmlContent - Raw XML string from `docProps/custom.xml`
144
+ * @returns A record of property name → typed value (empty object if none found)
145
+ * @example
146
+ * ```typescript
147
+ * const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
148
+ * const props = parseOOXMLCustomProperties(customXml);
149
+ * console.log(props['Department']); // "Engineering"
150
+ * console.log(props['Priority']); // 1 (number)
151
+ * console.log(props['Reviewed']); // true (boolean)
152
+ * ```
153
+ */
154
+ export declare const parseOOXMLCustomProperties: (xmlContent: string) => Record<string, string | number | boolean | Date>;
@@ -11,27 +11,32 @@
11
11
  * @module xmlUtils
12
12
  */
13
13
  Object.defineProperty(exports, "__esModule", { value: true });
14
- exports.parseOfficeMetadata = exports.getDirectChildren = exports.getElementsByTagName = exports.parseXmlString = void 0;
14
+ exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
15
15
  const xmldom_1 = require("@xmldom/xmldom");
16
+ const dateUtils_js_1 = require("./dateUtils.js");
17
+ /**
18
+ * Type guard for Element nodes.
19
+ */
20
+ const isElement = (node) => {
21
+ return node.nodeType === 1;
22
+ };
23
+ exports.isElement = isElement;
16
24
  /**
17
25
  * Parses an XML string into a DOM Document object.
18
26
  *
19
27
  * Uses the @xmldom/xmldom library to parse XML strings in a Node.js environment.
20
- * This is necessary because Node.js doesn't have a built-in DOM parser like browsers do.
21
28
  *
22
29
  * @param xml - The XML content as a string
30
+ * @param options - Optional parser settings (e.g., enable locators for source mapping)
23
31
  * @returns A Document object that can be queried using standard DOM methods
24
- * @example
25
- * ```typescript
26
- * const xmlString = '<root><item>Hello</item></root>';
27
- * const doc = parseXmlString(xmlString);
28
- * const items = doc.getElementsByTagName('item');
29
- * console.log(items[0].textContent); // "Hello"
30
- * ```
31
32
  */
32
- const parseXmlString = (xml) => {
33
- const parser = new xmldom_1.DOMParser();
34
- return parser.parseFromString(xml, "text/xml");
33
+ const parseXmlString = (xml, options = {}) => {
34
+ const parser = new xmldom_1.DOMParser(options);
35
+ // @xmldom/xmldom 0.9.x is strict: a UTF-8 BOM (U+FEFF) prepended to the
36
+ // XML string causes a fatalError because the XML declaration is no longer
37
+ // at position 0. Strip it before parsing.
38
+ const sanitized = xml.charCodeAt(0) === 0xFEFF ? xml.slice(1) : xml.trim();
39
+ return parser.parseFromString(sanitized, "text/xml");
35
40
  };
36
41
  exports.parseXmlString = parseXmlString;
37
42
  /**
@@ -50,9 +55,115 @@ exports.parseXmlString = parseXmlString;
50
55
  * ```
51
56
  */
52
57
  const getElementsByTagName = (element, tagName) => {
53
- return Array.from(element.getElementsByTagName(tagName));
58
+ const results = Array.from(element.getElementsByTagName(tagName));
59
+ // Resilience: If prefixed tag (e.g., 'dc:title') not found, try local name (e.g., 'title')
60
+ if (results.length === 0 && tagName.includes(':')) {
61
+ const localName = tagName.split(':').pop();
62
+ return Array.from(element.getElementsByTagName(localName));
63
+ }
64
+ return results;
54
65
  };
55
66
  exports.getElementsByTagName = getElementsByTagName;
67
+ /**
68
+ * Serializes a DOM Node (Document, Element, etc.) back into an XML string.
69
+ * This is cross-platform and works in both Node.js and Browser environments.
70
+ *
71
+ * @param node - The DOM node to serialize
72
+ * @param options - Serialization options
73
+ * @returns The XML string representation
74
+ */
75
+ const serializeXml = (node, options = {}) => {
76
+ // Note: xmldom's XMLSerializer doesn't natively support a 'pretty' or 'preserve'
77
+ // flag in a way that matches all user expectations, but it defaults to
78
+ // preserving structure. Formatting (indentation) is usually handled by the
79
+ // parser's initial whitespace handling.
80
+ // @ts-ignore - xmldom's Node is compatible with the global Node interface
81
+ return new xmldom_1.XMLSerializer().serializeToString(node);
82
+ };
83
+ exports.serializeXml = serializeXml;
84
+ /**
85
+ * Attempts to extract the original raw substring from the source XML for a given node.
86
+ * Requires the document to have been parsed with { locator: true }.
87
+ *
88
+ * @param node - The DOM node to extract source for
89
+ * @param sourceXml - The original XML source string
90
+ * @returns The raw XML substring, or undefined if it cannot be reliably determined
91
+ */
92
+ const getSourceSubstring = (node, sourceXml) => {
93
+ if (!node || typeof node.lineNumber !== 'number' || typeof node.columnNumber !== 'number') {
94
+ return undefined;
95
+ }
96
+ // Convert line/column to absolute index
97
+ const lines = sourceXml.split('\n');
98
+ let startIdx = 0;
99
+ for (let i = 0; i < node.lineNumber - 1; i++) {
100
+ startIdx += lines[i].length + 1; // +1 for newline
101
+ }
102
+ startIdx += node.columnNumber - 1;
103
+ // To find the end of the node, we look for the closing tag.
104
+ // This is a heuristic approach that works well for simple structured nodes (p, tbl, etc.)
105
+ // but might be complex for overlapping namespaces or malformed XML.
106
+ if ((0, exports.isElement)(node)) {
107
+ const tagName = node.tagName;
108
+ const closingTag = `</${tagName}>`;
109
+ const endIdx = sourceXml.indexOf(closingTag, startIdx);
110
+ if (endIdx !== -1) {
111
+ return sourceXml.substring(startIdx, endIdx + closingTag.length);
112
+ }
113
+ // Self-closing tag handling (e.g., <w:p/>)
114
+ const selfClosingEnd = sourceXml.indexOf('/>', startIdx);
115
+ const nextOpenTag = sourceXml.indexOf('<', startIdx + 1);
116
+ if (selfClosingEnd !== -1 && (nextOpenTag === -1 || selfClosingEnd < nextOpenTag)) {
117
+ return sourceXml.substring(startIdx, selfClosingEnd + 2);
118
+ }
119
+ }
120
+ return undefined;
121
+ };
122
+ exports.getSourceSubstring = getSourceSubstring;
123
+ /**
124
+ * High-level helper to get raw content for a node based on OfficeParserConfig.
125
+ *
126
+ * @param node - The DOM node
127
+ * @param sourceXml - The original source XML string
128
+ * @param config - The parser configuration
129
+ * @returns The raw content string (serialized or original)
130
+ */
131
+ const getRawContent = (node, sourceXml, config) => {
132
+ if (config.serializeRawContent === false) {
133
+ const original = (0, exports.getSourceSubstring)(node, sourceXml);
134
+ if (original)
135
+ return original;
136
+ }
137
+ return (0, exports.serializeXml)(node, { preserveWhitespace: config.preserveXmlWhitespace });
138
+ };
139
+ exports.getRawContent = getRawContent;
140
+ /**
141
+ * Gets the first element with the specified tag name within a parent element.
142
+ *
143
+ * @param parent - The parent element or document to search within
144
+ * @param tagName - The tag name to search for
145
+ * @returns The first matching element, or undefined if none found
146
+ */
147
+ const getFirstElementByTagName = (parent, tagName) => {
148
+ const elements = parent.getElementsByTagName(tagName);
149
+ if (elements && elements.length > 0) {
150
+ return elements[0];
151
+ }
152
+ return undefined;
153
+ };
154
+ exports.getFirstElementByTagName = getFirstElementByTagName;
155
+ /**
156
+ * Gets the value of an attribute from an element.
157
+ *
158
+ * @param element - The element to get the attribute from
159
+ * @param attrName - The name of the attribute
160
+ * @returns The attribute value or undefined if not set
161
+ */
162
+ const getAttribute = (element, attrName) => {
163
+ const attr = element.getAttribute(attrName);
164
+ return attr !== null ? attr : undefined;
165
+ };
166
+ exports.getAttribute = getAttribute;
56
167
  /**
57
168
  * Gets direct child elements with a specific tag name.
58
169
  * Unlike getElementsByTagName, this does not search recursively.
@@ -67,7 +178,7 @@ const getDirectChildren = (parent, tagName) => {
67
178
  return result;
68
179
  for (let i = 0; i < parent.childNodes.length; i++) {
69
180
  const child = parent.childNodes[i];
70
- if (child.nodeType === 1 && child.tagName === tagName) { // 1 = ELEMENT_NODE
181
+ if ((0, exports.isElement)(child) && child.tagName === tagName) {
71
182
  result.push(child);
72
183
  }
73
184
  }
@@ -124,11 +235,18 @@ const parseOfficeMetadata = (xmlContent) => {
124
235
  // Step 6: Extract creation date (Dublin Core Terms element)
125
236
  const created = (0, exports.getElementsByTagName)(coreProperties, "dcterms:created")[0];
126
237
  if (created && created.textContent)
127
- metadata.created = new Date(created.textContent);
238
+ metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
128
239
  // Step 7: Extract last modification date (Dublin Core Terms element)
129
240
  const modified = (0, exports.getElementsByTagName)(coreProperties, "dcterms:modified")[0];
130
241
  if (modified && modified.textContent)
131
- metadata.modified = new Date(modified.textContent);
242
+ metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
243
+ // Step 8: Extract description and subject (Dublin Core elements)
244
+ const description = (0, exports.getElementsByTagName)(coreProperties, "dc:description")[0];
245
+ if (description && description.textContent)
246
+ metadata.description = description.textContent;
247
+ const subject = (0, exports.getElementsByTagName)(coreProperties, "dc:subject")[0];
248
+ if (subject && subject.textContent)
249
+ metadata.subject = subject.textContent;
132
250
  return metadata;
133
251
  }
134
252
  // Check for ODF Meta
@@ -148,11 +266,111 @@ const parseOfficeMetadata = (xmlContent) => {
148
266
  metadata.subject = subject.textContent;
149
267
  const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
150
268
  if (created && created.textContent)
151
- metadata.created = new Date(created.textContent);
269
+ metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
152
270
  const modified = (0, exports.getElementsByTagName)(officeMeta, "dc:date")[0];
153
271
  if (modified && modified.textContent)
154
- metadata.modified = new Date(modified.textContent);
272
+ metadata.modified = (0, dateUtils_js_1.parseOfficeDate)(modified.textContent);
273
+ // Extract user-defined custom properties (meta:user-defined)
274
+ const userDefined = (0, exports.getElementsByTagName)(officeMeta, "meta:user-defined");
275
+ if (userDefined.length > 0) {
276
+ const customProperties = {};
277
+ for (const el of userDefined) {
278
+ const name = el.getAttribute("meta:name");
279
+ if (!name || !el.textContent)
280
+ continue;
281
+ const valueType = el.getAttribute("meta:value-type") || "string";
282
+ const raw = el.textContent;
283
+ if (valueType === "boolean") {
284
+ customProperties[name] = raw.toLowerCase() === "true";
285
+ }
286
+ else if (valueType === "float") {
287
+ const num = Number(raw);
288
+ if (!isNaN(num))
289
+ customProperties[name] = num;
290
+ }
291
+ else if (valueType === "date" || valueType === "time") {
292
+ const date = (0, dateUtils_js_1.parseOfficeDate)(raw);
293
+ if (date)
294
+ customProperties[name] = date;
295
+ else
296
+ customProperties[name] = raw;
297
+ }
298
+ else {
299
+ customProperties[name] = raw;
300
+ }
301
+ }
302
+ if (Object.keys(customProperties).length > 0) {
303
+ metadata.customProperties = customProperties;
304
+ }
305
+ }
155
306
  }
156
307
  return metadata;
157
308
  };
158
309
  exports.parseOfficeMetadata = parseOfficeMetadata;
310
+ /**
311
+ * Parses OOXML custom document properties from `docProps/custom.xml`.
312
+ *
313
+ * Custom properties are user-defined key/value pairs that authors can attach to OOXML documents
314
+ * (DOCX, XLSX, PPTX). They are stored in `docProps/custom.xml` inside the ZIP archive.
315
+ *
316
+ * Property values are typed using the `vt:` namespace (docPropsVTypes):
317
+ * - `vt:lpwstr` / `vt:lpstr` / `vt:bstr` → string
318
+ * - `vt:bool` → boolean
319
+ * - `vt:i1`..`vt:i8`, `vt:int`, `vt:r4`, `vt:r8`, `vt:decimal` → number
320
+ * - `vt:filetime` / `vt:date` → Date
321
+ *
322
+ * @param xmlContent - Raw XML string from `docProps/custom.xml`
323
+ * @returns A record of property name → typed value (empty object if none found)
324
+ * @example
325
+ * ```typescript
326
+ * const customXml = files.find(f => f.path === 'docProps/custom.xml').content.toString();
327
+ * const props = parseOOXMLCustomProperties(customXml);
328
+ * console.log(props['Department']); // "Engineering"
329
+ * console.log(props['Priority']); // 1 (number)
330
+ * console.log(props['Reviewed']); // true (boolean)
331
+ * ```
332
+ */
333
+ const parseOOXMLCustomProperties = (xmlContent) => {
334
+ const xml = (0, exports.parseXmlString)(xmlContent);
335
+ const result = {};
336
+ const properties = (0, exports.getElementsByTagName)(xml, "property");
337
+ for (const prop of properties) {
338
+ const name = prop.getAttribute("name");
339
+ if (!name)
340
+ continue;
341
+ // The value is the first child element (typed using vt: namespace)
342
+ for (let i = 0; i < prop.childNodes.length; i++) {
343
+ const child = prop.childNodes[i];
344
+ if (child.nodeType !== 1)
345
+ continue; // skip non-elements
346
+ const el = child;
347
+ const tag = el.tagName || '';
348
+ const text = el.textContent || '';
349
+ if (/vt:lpwstr|vt:lpstr|vt:bstr/.test(tag)) {
350
+ result[name] = text;
351
+ }
352
+ else if (/vt:bool/.test(tag)) {
353
+ result[name] = text.toLowerCase() === 'true';
354
+ }
355
+ else if (/vt:(i[1248]|ui[1248]|int|uint|r4|r8|decimal)/.test(tag)) {
356
+ const num = Number(text);
357
+ if (!isNaN(num))
358
+ result[name] = num;
359
+ }
360
+ else if (/vt:filetime|vt:date/.test(tag)) {
361
+ const date = (0, dateUtils_js_1.parseOfficeDate)(text);
362
+ if (date)
363
+ result[name] = date;
364
+ else
365
+ result[name] = text;
366
+ }
367
+ else if (text) {
368
+ // Fallback: store as string for any other vt: type
369
+ result[name] = text;
370
+ }
371
+ break; // only one value element per property
372
+ }
373
+ }
374
+ return result;
375
+ };
376
+ exports.parseOOXMLCustomProperties = parseOOXMLCustomProperties;
@@ -14,13 +14,9 @@
14
14
  *
15
15
  * @module zipUtils
16
16
  */
17
- var __importDefault = (this && this.__importDefault) || function (mod) {
18
- return (mod && mod.__esModule) ? mod : { "default": mod };
19
- };
20
17
  Object.defineProperty(exports, "__esModule", { value: true });
21
18
  exports.extractFiles = void 0;
22
- const yauzl_1 = __importDefault(require("yauzl"));
23
- const concat_stream_1 = __importDefault(require("concat-stream"));
19
+ const fflate_1 = require("fflate");
24
20
  /**
25
21
  * Extracts files from a ZIP archive with optional filtering.
26
22
  *
@@ -62,50 +58,13 @@ const concat_stream_1 = __importDefault(require("concat-stream"));
62
58
  */
63
59
  const extractFiles = (zipInput, filterFn) => {
64
60
  return new Promise((resolve, reject) => {
65
- // Step 1: Open the ZIP archive from the buffer
66
- // lazyEntries: true means we manually control when to read each entry (better memory usage)
67
- yauzl_1.default.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
61
+ (0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), { filter: (file) => filterFn(file.name) }, (err, decompressed) => {
68
62
  if (err)
69
63
  return reject(err);
70
- if (!zipfile)
71
- return reject(new Error("Failed to open zip file"));
72
- // Array to collect all extracted files
73
- const extractedFiles = [];
74
- // Step 2: Start reading the first entry
75
- // This triggers the 'entry' event
76
- zipfile.readEntry();
77
- // Step 3: Handle each entry (file or directory) in the ZIP
78
- zipfile.on('entry', (entry) => {
79
- // Step 3a: Check if this file should be extracted using the filter function
80
- if (filterFn(entry.fileName)) {
81
- // Step 3b: Open a read stream for this entry
82
- zipfile.openReadStream(entry, (err, readStream) => {
83
- if (err)
84
- return reject(err);
85
- if (!readStream)
86
- return reject(new Error("Failed to open read stream"));
87
- // Step 3c: Pipe the stream through concat to collect all data into a single Buffer
88
- // This is necessary because streams deliver data in chunks
89
- readStream.pipe((0, concat_stream_1.default)((data) => {
90
- // Step 3d: Add the extracted file to our results
91
- extractedFiles.push({
92
- path: entry.fileName,
93
- content: data
94
- });
95
- // Step 3e: Continue to the next entry
96
- zipfile.readEntry();
97
- }));
98
- });
99
- }
100
- else {
101
- // Step 3f: Skip this entry and move to the next one
102
- zipfile.readEntry();
103
- }
104
- });
105
- // Step 4: All entries have been processed
106
- zipfile.on('end', () => resolve(extractedFiles));
107
- // Step 5: Handle any errors during extraction
108
- zipfile.on('error', reject);
64
+ resolve(Object.entries(decompressed).map(([path, data]) => ({
65
+ path,
66
+ content: Buffer.from(data)
67
+ })));
109
68
  });
110
69
  });
111
70
  };
package/package.json CHANGED
@@ -1,9 +1,20 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "6.0.6",
3
+ "version": "6.1.0",
4
4
  "description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf) into structured AST with rich metadata, formatting, and attachment support.",
5
+ "funding": "https://github.com/sponsors/harshankur",
5
6
  "main": "dist/index.js",
7
+ "module": "dist/index.mjs",
6
8
  "types": "dist/index.d.ts",
9
+ "browser": "./dist/officeparser.browser.mjs",
10
+ "exports": {
11
+ ".": {
12
+ "types": "./dist/index.d.ts",
13
+ "browser": "./dist/officeparser.browser.mjs",
14
+ "import": "./dist/index.mjs",
15
+ "require": "./dist/index.js"
16
+ }
17
+ },
7
18
  "sideEffects": false,
8
19
  "engines": {
9
20
  "node": ">=18.0.0"
@@ -12,14 +23,21 @@
12
23
  "dist"
13
24
  ],
14
25
  "scripts": {
15
- "build": "npm run sync:versions && npm run build:node && npm run build:browser",
26
+ "build": "npm run sync:versions && npm run build:node && npm run build:esm-wrapper && npm run build:browser:types && npm run build:browser",
16
27
  "build:node": "tsc",
28
+ "build:esm-wrapper": "node scripts/generate-esm-wrapper.js",
29
+ "build:browser:types": "dts-bundle-generator --no-check -o dist/officeparser.browser.d.ts src/index.ts",
17
30
  "build:browser": "node build_browser.js && npm run sync:docs",
18
31
  "sync:versions": "node scripts/sync-pdfjs-versions.js",
19
- "sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.js docs/dist/",
20
- "test": "npm run test:clean && npm run build && npx tsx test/testOfficeParser.ts",
32
+ "sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/",
33
+ "test": "npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser",
34
+ "test:baseline": "npm run test baseline",
35
+ "test:parser": "npx tsx test/testOfficeParser.ts",
36
+ "test:artifacts": "npx tsx test/testShippingArtifacts.ts",
37
+ "test:license": "npm run sbom && node scripts/validate-licenses.js",
21
38
  "test:clean": "rm -rf test/results",
22
39
  "clean": "rm -rf dist && npm run test:clean",
40
+ "sbom": "npx --yes @cyclonedx/cyclonedx-npm --output-format json --output-file dist/sbom.cdx.json --omit dev",
23
41
  "prepublishOnly": "npm run build",
24
42
  "prepare": "husky"
25
43
  },
@@ -28,7 +46,11 @@
28
46
  "url": "git+https://github.com/harshankur/officeParser.git"
29
47
  },
30
48
  "bin": {
31
- "officeparser": "dist/index.js"
49
+ "officeparser": "dist/cli.js"
50
+ },
51
+ "publishConfig": {
52
+ "access": "public",
53
+ "provenance": true
32
54
  },
33
55
  "keywords": [
34
56
  "office",
@@ -67,24 +89,23 @@
67
89
  "bugs": {
68
90
  "url": "https://github.com/harshankur/officeParser/issues"
69
91
  },
70
- "homepage": "https://harshankur.github.io/officeParser/",
92
+ "homepage": "https://officeparser.harshankur.com",
71
93
  "dependencies": {
72
- "@xmldom/xmldom": "^0.8.11",
73
- "concat-stream": "^2.0.0",
74
- "file-type": "^19.6.0",
75
- "pdfjs-dist": "^5.5.207",
76
- "tesseract.js": "^7.0.0",
77
- "yauzl": "^3.2.1"
94
+ "@xmldom/xmldom": "^0.9.9",
95
+ "fflate": "^0.8.2",
96
+ "file-type": "^22.0.1",
97
+ "pdfjs-dist": "5.6.205",
98
+ "tesseract.js": "^7.0.0"
78
99
  },
79
100
  "devDependencies": {
80
- "@types/concat-stream": "^2.0.3",
81
- "@types/node": "^22.13.10",
82
- "@types/xmldom": "^0.1.34",
83
- "@types/yauzl": "^2.10.3",
101
+ "@types/node": "^25.5.0",
102
+ "buffer": "^6.0.3",
103
+ "dts-bundle-generator": "^9.5.1",
84
104
  "esbuild": "^0.27.4",
85
- "esbuild-plugin-polyfill-node": "^0.3.0",
105
+ "esbuild-plugins-node-modules-polyfill": "^1.8.1",
86
106
  "husky": "^9.1.7",
107
+ "process": "^0.11.10",
87
108
  "tsx": "^4.21.0",
88
- "typescript": "^5.9.3"
109
+ "typescript": "^6.0.2"
89
110
  }
90
111
  }