officeparser 7.0.0 → 7.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +710 -539
- package/dist/OfficeConverter.d.ts +4 -3
- package/dist/OfficeConverter.js +4 -3
- package/dist/OfficeParser.d.ts +1 -1
- package/dist/OfficeParser.js +27 -10
- package/dist/officeparser.browser.d.ts +6 -3
- package/dist/officeparser.browser.iife.js +28 -28
- package/dist/officeparser.browser.mjs +28 -28
- package/dist/sbom.cdx.json +98 -98
- package/dist/types.d.ts +2 -0
- package/dist/types.js +2 -0
- package/dist/utils/envUtils.d.ts +8 -3
- package/dist/utils/envUtils.js +113 -84
- package/dist/utils/errorUtils.js +2 -1
- package/dist/utils/moduleLoader.js +4 -2
- package/package.json +2 -2
|
@@ -34,10 +34,11 @@ export declare class OfficeConverter {
|
|
|
34
34
|
* // Convert Word to Markdown with a single call
|
|
35
35
|
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
36
|
*
|
|
37
|
-
* // Convert PDF to HTML
|
|
37
|
+
* // Convert PDF to HTML (Note: OCR is disabled in this one-step API)
|
|
38
38
|
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
-
*
|
|
40
|
-
*
|
|
39
|
+
* generatorConfig: {
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* }
|
|
41
42
|
* });
|
|
42
43
|
* ```
|
|
43
44
|
*/
|
package/dist/OfficeConverter.js
CHANGED
|
@@ -34,10 +34,11 @@ class OfficeConverter {
|
|
|
34
34
|
* // Convert Word to Markdown with a single call
|
|
35
35
|
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
36
|
*
|
|
37
|
-
* // Convert PDF to HTML
|
|
37
|
+
* // Convert PDF to HTML (Note: OCR is disabled in this one-step API)
|
|
38
38
|
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
-
*
|
|
40
|
-
*
|
|
39
|
+
* generatorConfig: {
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* }
|
|
41
42
|
* });
|
|
42
43
|
* ```
|
|
43
44
|
*/
|
package/dist/OfficeParser.d.ts
CHANGED
package/dist/OfficeParser.js
CHANGED
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
* const ast = await OfficeParser.parseOffice(buffer);
|
|
32
32
|
*
|
|
33
33
|
* // Get plain text
|
|
34
|
-
* console.log(ast.
|
|
34
|
+
* console.log((await ast.to('text')).value);
|
|
35
35
|
* ```
|
|
36
36
|
*
|
|
37
37
|
* @module OfficeParser
|
|
@@ -158,12 +158,13 @@ class OfficeParser {
|
|
|
158
158
|
else {
|
|
159
159
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
|
|
160
160
|
}
|
|
161
|
-
//
|
|
162
|
-
//
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
161
|
+
// Attempt to detect file type from buffer only if extension is unknown.
|
|
162
|
+
// This matches v6 behavior and prevents crashes in older Node environments
|
|
163
|
+
// where file-type 22.x might be incompatible.
|
|
164
|
+
if (buffer.length > 0 && !ext) {
|
|
165
|
+
try {
|
|
166
|
+
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
167
|
+
const type = await fileTypeFromBuffer(buffer);
|
|
167
168
|
if (type) {
|
|
168
169
|
ext = type.ext;
|
|
169
170
|
}
|
|
@@ -173,9 +174,25 @@ class OfficeParser {
|
|
|
173
174
|
// lack magic bytes. We'll let the switch default handle it.
|
|
174
175
|
}
|
|
175
176
|
}
|
|
176
|
-
|
|
177
|
-
//
|
|
178
|
-
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.
|
|
177
|
+
catch (error) {
|
|
178
|
+
// Log warning but don't crash; the switch below will handle unsupported/missing ext
|
|
179
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
else if (buffer.length > 0 && ext) {
|
|
183
|
+
// If extension is known, we can optionally verify it, but we wrap it
|
|
184
|
+
// in a try-catch to avoid breaking Node 18 if file-type fails to load.
|
|
185
|
+
try {
|
|
186
|
+
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
187
|
+
const type = await fileTypeFromBuffer(buffer);
|
|
188
|
+
if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
|
|
189
|
+
// Mismatch found between authoritative extension and detected content
|
|
190
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
catch (error) {
|
|
194
|
+
// Log warning so user knows verification could not be performed
|
|
195
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
|
|
179
196
|
}
|
|
180
197
|
}
|
|
181
198
|
if (!ext) {
|
|
@@ -63,6 +63,8 @@ export declare enum OfficeWarningType {
|
|
|
63
63
|
SHEET_RANGE_NOT_FOUND = "SHEET_RANGE_NOT_FOUND",
|
|
64
64
|
/** Buffer content type does not match the provided or expected file extension */
|
|
65
65
|
BUFFER_TYPE_MISMATCH = "BUFFER_TYPE_MISMATCH",
|
|
66
|
+
/** Failed to detect file type from buffer due to library error or incompatibility */
|
|
67
|
+
FILE_TYPE_DETECTION_FAILED = "FILE_TYPE_DETECTION_FAILED",
|
|
66
68
|
/** No chunks were generated for the document given the current strategy */
|
|
67
69
|
EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
|
|
68
70
|
/** A node was skipped because it only contained whitespace */
|
|
@@ -1608,10 +1610,11 @@ export declare class OfficeConverter {
|
|
|
1608
1610
|
* // Convert Word to Markdown with a single call
|
|
1609
1611
|
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
1610
1612
|
*
|
|
1611
|
-
* // Convert PDF to HTML
|
|
1613
|
+
* // Convert PDF to HTML (Note: OCR is disabled in this one-step API)
|
|
1612
1614
|
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
1613
|
-
*
|
|
1614
|
-
*
|
|
1615
|
+
* generatorConfig: {
|
|
1616
|
+
* includeImages: true
|
|
1617
|
+
* }
|
|
1615
1618
|
* });
|
|
1616
1619
|
* ```
|
|
1617
1620
|
*/
|