officeparser 7.0.3 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +27 -1
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +31 -4
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +51 -10
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +377 -53
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +6 -1
- package/dist/parsers/ExcelParser.js +69 -21
- package/dist/parsers/HtmlParser.js +15 -1
- package/dist/parsers/MarkdownParser.js +18 -10
- package/dist/parsers/OpenOfficeParser.js +61 -34
- package/dist/parsers/PdfParser.js +26 -1
- package/dist/parsers/PowerPointParser.js +168 -40
- package/dist/parsers/RtfParser.js +30 -24
- package/dist/parsers/WordParser.js +158 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +383 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +69 -2
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +39 -3
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +17 -0
- package/dist/utils/xmlUtils.js +85 -1
- package/package.json +3 -2
package/dist/utils/xmlUtils.d.ts
CHANGED
|
@@ -144,3 +144,20 @@ export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata
|
|
|
144
144
|
* ```
|
|
145
145
|
*/
|
|
146
146
|
export declare const parseOOXMLCustomProperties: (xmlContent: string) => Record<string, string | number | boolean | Date>;
|
|
147
|
+
/**
|
|
148
|
+
* Parses OOXML application properties from `docProps/app.xml`.
|
|
149
|
+
*
|
|
150
|
+
* Application properties contain document statistics and application settings.
|
|
151
|
+
*
|
|
152
|
+
* @param xmlContent - Raw XML string from `docProps/app.xml`
|
|
153
|
+
* @returns A record of property name -> typed value
|
|
154
|
+
*/
|
|
155
|
+
export declare const parseOOXMLAppProperties: (xmlContent: string) => Record<string, string | number | boolean>;
|
|
156
|
+
/**
|
|
157
|
+
* Decodes XML entities (standard named entities, decimal, and hexadecimal entities) in a string.
|
|
158
|
+
* Useful when parsing content with regular expressions instead of a full DOM parser.
|
|
159
|
+
*
|
|
160
|
+
* @param text - The XML-encoded string
|
|
161
|
+
* @returns The decoded string
|
|
162
|
+
*/
|
|
163
|
+
export declare const decodeXmlEntities: (text: string) => string;
|
package/dist/utils/xmlUtils.js
CHANGED
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
* @module xmlUtils
|
|
12
12
|
*/
|
|
13
13
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
|
-
exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
|
|
14
|
+
exports.decodeXmlEntities = exports.parseOOXMLAppProperties = exports.parseOOXMLCustomProperties = exports.parseOfficeMetadata = exports.getDirectChildren = exports.getAttribute = exports.getFirstElementByTagName = exports.getRawContent = exports.getSourceSubstring = exports.serializeXml = exports.getElementsByTagName = exports.parseXmlString = exports.isElement = void 0;
|
|
15
15
|
const xmldom_1 = require("@xmldom/xmldom");
|
|
16
16
|
const dateUtils_js_1 = require("./dateUtils.js");
|
|
17
17
|
/**
|
|
@@ -221,6 +221,13 @@ const parseOfficeMetadata = (xmlContent) => {
|
|
|
221
221
|
// Check for OOXML Core Properties
|
|
222
222
|
const coreProperties = (0, exports.getElementsByTagName)(xml, "cp:coreProperties")[0];
|
|
223
223
|
if (coreProperties) {
|
|
224
|
+
metadata.nativeProperties = {};
|
|
225
|
+
for (let i = 0; i < coreProperties.childNodes.length; i++) {
|
|
226
|
+
const child = coreProperties.childNodes[i];
|
|
227
|
+
if ((0, exports.isElement)(child)) {
|
|
228
|
+
metadata.nativeProperties[child.tagName] = child.textContent;
|
|
229
|
+
}
|
|
230
|
+
}
|
|
224
231
|
// Step 3: Extract title (Dublin Core element)
|
|
225
232
|
const title = (0, exports.getElementsByTagName)(coreProperties, "dc:title")[0];
|
|
226
233
|
if (title && title.textContent)
|
|
@@ -248,11 +255,21 @@ const parseOfficeMetadata = (xmlContent) => {
|
|
|
248
255
|
const subject = (0, exports.getElementsByTagName)(coreProperties, "dc:subject")[0];
|
|
249
256
|
if (subject && subject.textContent)
|
|
250
257
|
metadata.subject = subject.textContent;
|
|
258
|
+
const keywords = (0, exports.getElementsByTagName)(coreProperties, "cp:keywords")[0];
|
|
259
|
+
if (keywords && keywords.textContent)
|
|
260
|
+
metadata.keywords = keywords.textContent;
|
|
251
261
|
return metadata;
|
|
252
262
|
}
|
|
253
263
|
// Check for ODF Meta
|
|
254
264
|
const officeMeta = (0, exports.getElementsByTagName)(xml, "office:meta")[0];
|
|
255
265
|
if (officeMeta) {
|
|
266
|
+
metadata.nativeProperties = {};
|
|
267
|
+
for (let i = 0; i < officeMeta.childNodes.length; i++) {
|
|
268
|
+
const child = officeMeta.childNodes[i];
|
|
269
|
+
if ((0, exports.isElement)(child)) {
|
|
270
|
+
metadata.nativeProperties[child.tagName] = child.textContent;
|
|
271
|
+
}
|
|
272
|
+
}
|
|
256
273
|
const title = (0, exports.getElementsByTagName)(officeMeta, "dc:title")[0];
|
|
257
274
|
if (title && title.textContent)
|
|
258
275
|
metadata.title = title.textContent;
|
|
@@ -265,6 +282,10 @@ const parseOfficeMetadata = (xmlContent) => {
|
|
|
265
282
|
const subject = (0, exports.getElementsByTagName)(officeMeta, "dc:subject")[0];
|
|
266
283
|
if (subject && subject.textContent)
|
|
267
284
|
metadata.subject = subject.textContent;
|
|
285
|
+
const keywordElements = (0, exports.getElementsByTagName)(officeMeta, "meta:keyword");
|
|
286
|
+
if (keywordElements.length > 0) {
|
|
287
|
+
metadata.keywords = keywordElements.map(k => k.textContent).filter(Boolean).join(', ');
|
|
288
|
+
}
|
|
268
289
|
const created = (0, exports.getElementsByTagName)(officeMeta, "meta:creation-date")[0];
|
|
269
290
|
if (created && created.textContent)
|
|
270
291
|
metadata.created = (0, dateUtils_js_1.parseOfficeDate)(created.textContent);
|
|
@@ -375,3 +396,66 @@ const parseOOXMLCustomProperties = (xmlContent) => {
|
|
|
375
396
|
return result;
|
|
376
397
|
};
|
|
377
398
|
exports.parseOOXMLCustomProperties = parseOOXMLCustomProperties;
|
|
399
|
+
/**
|
|
400
|
+
* Parses OOXML application properties from `docProps/app.xml`.
|
|
401
|
+
*
|
|
402
|
+
* Application properties contain document statistics and application settings.
|
|
403
|
+
*
|
|
404
|
+
* @param xmlContent - Raw XML string from `docProps/app.xml`
|
|
405
|
+
* @returns A record of property name -> typed value
|
|
406
|
+
*/
|
|
407
|
+
const parseOOXMLAppProperties = (xmlContent) => {
|
|
408
|
+
const xml = (0, exports.parseXmlString)(xmlContent);
|
|
409
|
+
const result = {};
|
|
410
|
+
const appProperties = (0, exports.getElementsByTagName)(xml, "Properties")[0];
|
|
411
|
+
if (appProperties) {
|
|
412
|
+
for (let i = 0; i < appProperties.childNodes.length; i++) {
|
|
413
|
+
const child = appProperties.childNodes[i];
|
|
414
|
+
if ((0, exports.isElement)(child)) {
|
|
415
|
+
const text = child.textContent || '';
|
|
416
|
+
if (text.toLowerCase() === 'true')
|
|
417
|
+
result[child.tagName] = true;
|
|
418
|
+
else if (text.toLowerCase() === 'false')
|
|
419
|
+
result[child.tagName] = false;
|
|
420
|
+
else if (!isNaN(Number(text)) && text.trim() !== '')
|
|
421
|
+
result[child.tagName] = Number(text);
|
|
422
|
+
else
|
|
423
|
+
result[child.tagName] = text;
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
return result;
|
|
428
|
+
};
|
|
429
|
+
exports.parseOOXMLAppProperties = parseOOXMLAppProperties;
|
|
430
|
+
/**
|
|
431
|
+
* Decodes XML entities (standard named entities, decimal, and hexadecimal entities) in a string.
|
|
432
|
+
* Useful when parsing content with regular expressions instead of a full DOM parser.
|
|
433
|
+
*
|
|
434
|
+
* @param text - The XML-encoded string
|
|
435
|
+
* @returns The decoded string
|
|
436
|
+
*/
|
|
437
|
+
const decodeXmlEntities = (text) => {
|
|
438
|
+
return text.replace(/&([^;]+);/g, (match, entity) => {
|
|
439
|
+
if (entity.startsWith('#')) {
|
|
440
|
+
if (entity[1] === 'x' || entity[1] === 'X') {
|
|
441
|
+
const hex = entity.slice(2);
|
|
442
|
+
const code = parseInt(hex, 16);
|
|
443
|
+
return !isNaN(code) ? String.fromCodePoint(code) : match;
|
|
444
|
+
}
|
|
445
|
+
else {
|
|
446
|
+
const dec = entity.slice(1);
|
|
447
|
+
const code = parseInt(dec, 10);
|
|
448
|
+
return !isNaN(code) ? String.fromCodePoint(code) : match;
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
switch (entity) {
|
|
452
|
+
case 'amp': return '&';
|
|
453
|
+
case 'lt': return '<';
|
|
454
|
+
case 'gt': return '>';
|
|
455
|
+
case 'quot': return '"';
|
|
456
|
+
case 'apos': return "'";
|
|
457
|
+
default: return match;
|
|
458
|
+
}
|
|
459
|
+
});
|
|
460
|
+
};
|
|
461
|
+
exports.decodeXmlEntities = decodeXmlEntities;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "7.0
|
|
3
|
+
"version": "7.2.0",
|
|
4
4
|
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, and RAG-focused chunks.",
|
|
5
5
|
"funding": "https://github.com/sponsors/harshankur",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -29,7 +29,7 @@
|
|
|
29
29
|
"build:browser:types": "dts-bundle-generator --no-check -o dist/officeparser.browser.d.ts src/index.ts",
|
|
30
30
|
"build:browser": "node build_browser.js && npm run sync:docs",
|
|
31
31
|
"sync:versions": "node scripts/sync-pdfjs-versions.js",
|
|
32
|
-
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/",
|
|
32
|
+
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/ && mkdir -p docs/test/files && cp test/files/* docs/test/files/",
|
|
33
33
|
"lint": "eslint src",
|
|
34
34
|
"test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator",
|
|
35
35
|
"test:baseline": "npm run test:parser:baseline && npm run test:generator:baseline",
|
|
@@ -38,6 +38,7 @@
|
|
|
38
38
|
"test:generator": "npx tsx test/generator/testOfficeGenerator.ts",
|
|
39
39
|
"test:generator:baseline": "npx tsx test/generator/testOfficeGenerator.ts baseline",
|
|
40
40
|
"test:artifacts": "npx tsx test/testShippingArtifacts.ts",
|
|
41
|
+
"test:visualizer": "node test/testVisualizer.js",
|
|
41
42
|
"test:license": "npm run sbom && node scripts/validate-licenses.js",
|
|
42
43
|
"test:clean": "rm -rf test/results test/generator/results test/generator/output test/parser/results test/parser/output",
|
|
43
44
|
"clean": "rm -rf dist && npm run test:clean",
|