officeparser 7.1.0 → 7.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +152 -56
  2. package/dist/OfficeGenerator.d.ts +6 -2
  3. package/dist/OfficeGenerator.js +30 -9
  4. package/dist/OfficeParser.d.ts +1 -1
  5. package/dist/OfficeParser.js +1 -1
  6. package/dist/cli.d.ts +18 -12
  7. package/dist/cli.js +255 -81
  8. package/dist/defaults.js +12 -1
  9. package/dist/generators/BaseGenerator.d.ts +4 -3
  10. package/dist/generators/BaseGenerator.js +13 -1
  11. package/dist/generators/ChunkingGenerator.js +32 -5
  12. package/dist/generators/CsvGenerator.d.ts +1 -1
  13. package/dist/generators/HtmlGenerator.d.ts +2 -1
  14. package/dist/generators/HtmlGenerator.js +481 -42
  15. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  16. package/dist/generators/MarkdownGenerator.js +35 -2
  17. package/dist/generators/PdfGenerator.d.ts +1 -1
  18. package/dist/generators/PdfGenerator.js +0 -6
  19. package/dist/generators/RtfGenerator.d.ts +2 -1
  20. package/dist/generators/RtfGenerator.js +49 -6
  21. package/dist/generators/TextGenerator.d.ts +1 -1
  22. package/dist/generators/TextGenerator.js +6 -0
  23. package/dist/officeparser.browser.d.ts +267 -54
  24. package/dist/officeparser.browser.iife.js +599 -187
  25. package/dist/officeparser.browser.mjs +599 -187
  26. package/dist/parsers/CsvParser.js +1 -1
  27. package/dist/parsers/ExcelParser.js +63 -19
  28. package/dist/parsers/HtmlParser.js +10 -1
  29. package/dist/parsers/MarkdownParser.js +13 -10
  30. package/dist/parsers/OpenOfficeParser.js +57 -34
  31. package/dist/parsers/PdfParser.js +28 -3
  32. package/dist/parsers/PowerPointParser.js +164 -40
  33. package/dist/parsers/RtfParser.js +28 -24
  34. package/dist/parsers/WordParser.js +154 -11
  35. package/dist/sbom.cdx.json +100 -100
  36. package/dist/types.d.ts +268 -53
  37. package/dist/types.js +4 -0
  38. package/dist/utils/astUtils.d.ts +2 -2
  39. package/dist/utils/astUtils.js +2 -1
  40. package/dist/utils/configUtils.d.ts +5 -0
  41. package/dist/utils/configUtils.js +55 -1
  42. package/dist/utils/errorUtils.js +3 -1
  43. package/dist/utils/moduleLoader.js +55 -11
  44. package/dist/utils/xmlUtils.d.ts +9 -0
  45. package/dist/utils/xmlUtils.js +53 -1
  46. package/package.json +6 -3
@@ -19,7 +19,7 @@ async function loadNodeEsmModule(specifier) {
19
19
  // This is especially important in Node 18 for sub-paths of packages.
20
20
  try {
21
21
  const { pathToFileURL } = await import('url');
22
- // @ts-ignore - require.resolve is available in Node.js
22
+ // @ts-ignore - require.resolve is available in Node.js CJS context
23
23
  const absolutePath = require.resolve(specifier);
24
24
  const fileUrl = pathToFileURL(absolutePath).href;
25
25
  return import(fileUrl);
@@ -29,6 +29,16 @@ async function loadNodeEsmModule(specifier) {
29
29
  return import(specifier);
30
30
  }
31
31
  }
32
+ /**
33
+ * Returns true if require.resolve is available in this runtime context.
34
+ * It is NOT available in native ESM (e.g. the .mjs wrapper) or browser environments.
35
+ * Checking this guards against a ReferenceError that would silently trigger
36
+ * the bundled fallback path for non-bundled ESM consumers.
37
+ */
38
+ function isRequireAvailable() {
39
+ // @ts-ignore - require may not exist in ESM context
40
+ return typeof require !== 'undefined' && typeof require.resolve === 'function';
41
+ }
32
42
  /**
33
43
  * Specialized loader for file-type
34
44
  */
@@ -36,10 +46,19 @@ async function loadFileType() {
36
46
  if (!envUtils_js_1.isBrowser) {
37
47
  // Ensure environment polyfills for Node.js 18 support
38
48
  (0, envUtils_js_1.ensureEnvPolyfills)();
39
- // Node.js path: Use dynamic import wrapper for CJS compatibility
40
- return loadNodeEsmModule('file-type');
49
+ if (isRequireAvailable()) {
50
+ // Check if node_modules is available at runtime.
51
+ // @ts-ignore - require.resolve is available in Node.js CJS context
52
+ require.resolve(String('file-type'));
53
+ // node_modules present: use the path-resolved loader for Node 18 ESM compatibility
54
+ return loadNodeEsmModule('file-type');
55
+ }
41
56
  }
42
- // Browser path: standard dynamic import() is handled by bundlers (e.g. esbuild/Vite)
57
+ // Covers three cases with one return:
58
+ // 1. Node.js standalone/bundled (SEA or bundled CJS): bundler inlines the module at build time.
59
+ // 2. Node.js native ESM (no require available): runtime resolves the bare specifier.
60
+ // 3. Browser: bundler (esbuild/Vite) handles the static-looking dynamic import().
61
+ // Note: bypasses the Node 18 sub-path ESM fix in loadNodeEsmModule, but bundlers handle it.
43
62
  return import('file-type');
44
63
  }
45
64
  /**
@@ -49,14 +68,39 @@ async function loadPdfJs() {
49
68
  if (!envUtils_js_1.isBrowser) {
50
69
  // Ensure environment polyfills for Node.js 18 support
51
70
  (0, envUtils_js_1.ensureEnvPolyfills)();
52
- // Node.js environment: require legacy build for stability with ESM-only main
53
- try {
54
- return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
55
- }
56
- catch {
57
- return await loadNodeEsmModule('pdfjs-dist');
71
+ if (isRequireAvailable()) {
72
+ // @ts-ignore - require.resolve is available in Node.js CJS context
73
+ const pkgExists = (() => { try {
74
+ require.resolve(String('pdfjs-dist'));
75
+ return true;
76
+ }
77
+ catch {
78
+ return false;
79
+ } })();
80
+ if (pkgExists) {
81
+ // node_modules present: try the legacy build path first for stability with
82
+ // ESM-only main, then fall back to the package root.
83
+ // Errors from these loaders are NOT swallowed into the bundled-fallback path.
84
+ try {
85
+ return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
86
+ }
87
+ catch {
88
+ return await loadNodeEsmModule('pdfjs-dist');
89
+ }
90
+ }
58
91
  }
92
+ // Standalone/bundled fallback for Node.js (SEA, bundled CJS without node_modules, or native ESM context):
93
+ // Load both PDF.js and its worker, and register the worker on globalThis to enable the offline fake-worker.
94
+ // @ts-ignore - Ignore type check for local .mjs files in node_modules/packaging
95
+ const [pdfjs, pdfjsWorker] = await Promise.all([
96
+ // @ts-ignore - mjs imports may not have types
97
+ import('pdfjs-dist/legacy/build/pdf.mjs'),
98
+ // @ts-ignore - mjs imports may not have types
99
+ import('pdfjs-dist/legacy/build/pdf.worker.mjs')
100
+ ]);
101
+ globalThis.pdfjsWorker = pdfjsWorker;
102
+ return pdfjs;
59
103
  }
60
- // Browser environment: esbuild handles standard static-looking dynamic import()
104
+ // Browser environment: bundler (esbuild/Vite) handles the standard dynamic import()
61
105
  return import('pdfjs-dist');
62
106
  }
@@ -144,6 +144,15 @@ export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata
144
144
  * ```
145
145
  */
146
146
  export declare const parseOOXMLCustomProperties: (xmlContent: string) => Record<string, string | number | boolean | Date>;
147
+ /**
148
+ * Parses OOXML application properties from `docProps/app.xml`.
149
+ *
150
+ * Application properties contain document statistics and application settings.
151
+ *
152
+ * @param xmlContent - Raw XML string from `docProps/app.xml`
153
+ * @returns A record of property name -> typed value
154
+ */
155
+ export declare const parseOOXMLAppProperties: (xmlContent: string) => Record<string, string | number | boolean>;
147
156
  /**
148
157
  * Decodes XML entities (standard named entities, decimal, and hexadecimal entities) in a string.
149
158
  * Useful when parsing content with regular expressions instead of a full DOM parser.
@@ -11,7 +11,7 @@
11
11
  * @module xmlUtils
12
12
  */
13
13
  Object.defineProperty(exports, "__esModule", { value: true });
14
- exports.decodeXmlEntities = exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
14
+ exports.decodeXmlEntities = exports.parseOOXMLAppProperties = exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
15
15
  const xmldom_1 = require("@xmldom/xmldom");
16
16
  const dateUtils_js_1 = require("./dateUtils.js");
17
17
  /**
@@ -221,6 +221,13 @@ const parseOfficeMetadata = (xmlContent) => {
221
221
  // Check for OOXML Core Properties
222
222
  const coreProperties = (0, exports.getElementsByTagName)(xml, "cp:coreProperties")[0];
223
223
  if (coreProperties) {
224
+ metadata.nativeProperties = {};
225
+ for (let i = 0; i < coreProperties.childNodes.length; i++) {
226
+ const child = coreProperties.childNodes[i];
227
+ if ((0, exports.isElement)(child)) {
228
+ metadata.nativeProperties[child.tagName] = child.textContent;
229
+ }
230
+ }
224
231
  // Step 3: Extract title (Dublin Core element)
225
232
  const title = (0, exports.getElementsByTagName)(coreProperties, "dc:title")[0];
226
233
  if (title && title.textContent)
@@ -248,11 +255,21 @@ const parseOfficeMetadata = (xmlContent) => {
248
255
  const subject = (0, exports.getElementsByTagName)(coreProperties, "dc:subject")[0];
249
256
  if (subject && subject.textContent)
250
257
  metadata.subject = subject.textContent;
258
+ const keywords = (0, exports.getElementsByTagName)(coreProperties, "cp:keywords")[0];
259
+ if (keywords && keywords.textContent)
260
+ metadata.keywords = keywords.textContent;
251
261
  return metadata;
252
262
  }
253
263
  // Check for ODF Meta
254
264
  const officeMeta = (0, exports.getElementsByTagName)(xml, "office:meta")[0];
255
265
  if (officeMeta) {
266
+ metadata.nativeProperties = {};
267
+ for (let i = 0; i < officeMeta.childNodes.length; i++) {
268
+ const child = officeMeta.childNodes[i];
269
+ if ((0, exports.isElement)(child)) {
270
+ metadata.nativeProperties[child.tagName] = child.textContent;
271
+ }
272
+ }
256
273
  const title = (0, exports.getElementsByTagName)(officeMeta, "dc:title")[0];
257
274
  if (title && title.textContent)
258
275
  metadata.title = title.textContent;
@@ -265,6 +282,10 @@ const parseOfficeMetadata = (xmlContent) => {
265
282
  const subject = (0, exports.getElementsByTagName)(officeMeta, "dc:subject")[0];
266
283
  if (subject && subject.textContent)
267
284
  metadata.subject = subject.textContent;
285
+ const keywordElements = (0, exports.getElementsByTagName)(officeMeta, "meta:keyword");
286
+ if (keywordElements.length > 0) {
287
+ metadata.keywords = keywordElements.map(k => k.textContent).filter(Boolean).join(', ');
288
+ }
268
289
  const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
269
290
  if (created && created.textContent)
270
291
  metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
@@ -375,6 +396,37 @@ const parseOOXMLCustomProperties = (xmlContent) => {
375
396
  return result;
376
397
  };
377
398
  exports.parseOOXMLCustomProperties = parseOOXMLCustomProperties;
399
+ /**
400
+ * Parses OOXML application properties from `docProps/app.xml`.
401
+ *
402
+ * Application properties contain document statistics and application settings.
403
+ *
404
+ * @param xmlContent - Raw XML string from `docProps/app.xml`
405
+ * @returns A record of property name -> typed value
406
+ */
407
+ const parseOOXMLAppProperties = (xmlContent) => {
408
+ const xml = (0, exports.parseXmlString)(xmlContent);
409
+ const result = {};
410
+ const appProperties = (0, exports.getElementsByTagName)(xml, "Properties")[0];
411
+ if (appProperties) {
412
+ for (let i = 0; i < appProperties.childNodes.length; i++) {
413
+ const child = appProperties.childNodes[i];
414
+ if ((0, exports.isElement)(child)) {
415
+ const text = child.textContent || '';
416
+ if (text.toLowerCase() === 'true')
417
+ result[child.tagName] = true;
418
+ else if (text.toLowerCase() === 'false')
419
+ result[child.tagName] = false;
420
+ else if (!isNaN(Number(text)) && text.trim() !== '')
421
+ result[child.tagName] = Number(text);
422
+ else
423
+ result[child.tagName] = text;
424
+ }
425
+ }
426
+ }
427
+ return result;
428
+ };
429
+ exports.parseOOXMLAppProperties = parseOOXMLAppProperties;
378
430
  /**
379
431
  * Decodes XML entities (standard named entities, decimal, and hexadecimal entities) in a string.
380
432
  * Useful when parsing content with regular expressions instead of a full DOM parser.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "7.1.0",
3
+ "version": "7.2.1",
4
4
  "description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, and RAG-focused chunks.",
5
5
  "funding": "https://github.com/sponsors/harshankur",
6
6
  "main": "dist/index.js",
@@ -31,16 +31,18 @@
31
31
  "sync:versions": "node scripts/sync-pdfjs-versions.js",
32
32
  "sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/ && mkdir -p docs/test/files && cp test/files/* docs/test/files/",
33
33
  "lint": "eslint src",
34
- "test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator",
34
+ "test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator && npm run test:cli",
35
35
  "test:baseline": "npm run test:parser:baseline && npm run test:generator:baseline",
36
36
  "test:parser": "npx tsx test/parser/testOfficeParser.ts",
37
37
  "test:parser:baseline": "npx tsx test/parser/testOfficeParser.ts baseline",
38
38
  "test:generator": "npx tsx test/generator/testOfficeGenerator.ts",
39
39
  "test:generator:baseline": "npx tsx test/generator/testOfficeGenerator.ts baseline",
40
40
  "test:artifacts": "npx tsx test/testShippingArtifacts.ts",
41
+ "test:cli": "npx tsx test/cli/testCli.ts",
41
42
  "test:visualizer": "node test/testVisualizer.js",
43
+ "test:integration": "node test/testIntegration.js",
42
44
  "test:license": "npm run sbom && node scripts/validate-licenses.js",
43
- "test:clean": "rm -rf test/results test/generator/results test/generator/output test/parser/results test/parser/output",
45
+ "test:clean": "rm -rf test/results test/generator/results test/generator/output test/parser/results test/parser/output test/cli/results",
44
46
  "clean": "rm -rf dist && npm run test:clean",
45
47
  "sbom": "npx --yes @cyclonedx/cyclonedx-npm --output-format json --output-file dist/sbom.cdx.json --omit dev",
46
48
  "prepublishOnly": "npm run build",
@@ -123,6 +125,7 @@
123
125
  "esbuild-plugins-node-modules-polyfill": "^1.8.1",
124
126
  "eslint": "^10.3.0",
125
127
  "husky": "^9.1.7",
128
+ "postject": "^1.0.0-alpha.6",
126
129
  "process": "^0.11.10",
127
130
  "puppeteer": "^22.15.0",
128
131
  "tsx": "^4.21.0",