officeparser 7.0.0 → 7.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -34,10 +34,11 @@ export declare class OfficeConverter {
34
34
  * // Convert Word to Markdown with a single call
35
35
  * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
36
  *
37
- * // Convert PDF to HTML with OCR enabled for images
37
+ * // Convert PDF to HTML (Note: OCR is disabled in this one-step API)
38
38
  * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
- * ocr: true,
40
- * includeImages: true
39
+ * generatorConfig: {
40
+ * includeImages: true
41
+ * }
41
42
  * });
42
43
  * ```
43
44
  */
@@ -34,10 +34,11 @@ class OfficeConverter {
34
34
  * // Convert Word to Markdown with a single call
35
35
  * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
36
  *
37
- * // Convert PDF to HTML with OCR enabled for images
37
+ * // Convert PDF to HTML (Note: OCR is disabled in this one-step API)
38
38
  * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
- * ocr: true,
40
- * includeImages: true
39
+ * generatorConfig: {
40
+ * includeImages: true
41
+ * }
41
42
  * });
42
43
  * ```
43
44
  */
@@ -30,7 +30,7 @@
30
30
  * const ast = await OfficeParser.parseOffice(buffer);
31
31
  *
32
32
  * // Get plain text
33
- * console.log(ast.toText());
33
+ * console.log((await ast.to('text')).value);
34
34
  * ```
35
35
  *
36
36
  * @module OfficeParser
@@ -31,7 +31,7 @@
31
31
  * const ast = await OfficeParser.parseOffice(buffer);
32
32
  *
33
33
  * // Get plain text
34
- * console.log(ast.toText());
34
+ * console.log((await ast.to('text')).value);
35
35
  * ```
36
36
  *
37
37
  * @module OfficeParser
@@ -158,12 +158,13 @@ class OfficeParser {
158
158
  else {
159
159
  throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
160
160
  }
161
- // Always attempt to detect file type from buffer if it exists,
162
- // but respect the authoritative 'ext' if it was already set.
163
- if (buffer.length > 0) {
164
- const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
165
- const type = await fileTypeFromBuffer(buffer);
166
- if (!ext) {
161
+ // Attempt to detect file type from buffer only if extension is unknown.
162
+ // This matches v6 behavior and prevents crashes in older Node environments
163
+ // where file-type 22.x might be incompatible.
164
+ if (buffer.length > 0 && !ext) {
165
+ try {
166
+ const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
167
+ const type = await fileTypeFromBuffer(buffer);
167
168
  if (type) {
168
169
  ext = type.ext;
169
170
  }
@@ -173,9 +174,25 @@ class OfficeParser {
173
174
  // lack magic bytes. We'll let the switch default handle it.
174
175
  }
175
176
  }
176
- else if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
177
- // Mismatch found between authoritative extension and detected content
178
- (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
177
+ catch (error) {
178
+ // Log warning but don't crash; the switch below will handle unsupported/missing ext
179
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
180
+ }
181
+ }
182
+ else if (buffer.length > 0 && ext) {
183
+ // If extension is known, we can optionally verify it, but we wrap it
184
+ // in a try-catch to avoid breaking Node 18 if file-type fails to load.
185
+ try {
186
+ const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
187
+ const type = await fileTypeFromBuffer(buffer);
188
+ if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
189
+ // Mismatch found between authoritative extension and detected content
190
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
191
+ }
192
+ }
193
+ catch (error) {
194
+ // Log warning so user knows verification could not be performed
195
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
179
196
  }
180
197
  }
181
198
  if (!ext) {
@@ -63,6 +63,8 @@ export declare enum OfficeWarningType {
63
63
  SHEET_RANGE_NOT_FOUND = "SHEET_RANGE_NOT_FOUND",
64
64
  /** Buffer content type does not match the provided or expected file extension */
65
65
  BUFFER_TYPE_MISMATCH = "BUFFER_TYPE_MISMATCH",
66
+ /** Failed to detect file type from buffer due to library error or incompatibility */
67
+ FILE_TYPE_DETECTION_FAILED = "FILE_TYPE_DETECTION_FAILED",
66
68
  /** No chunks were generated for the document given the current strategy */
67
69
  EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
68
70
  /** A node was skipped because it only contained whitespace */
@@ -1608,10 +1610,11 @@ export declare class OfficeConverter {
1608
1610
  * // Convert Word to Markdown with a single call
1609
1611
  * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
1610
1612
  *
1611
- * // Convert PDF to HTML with OCR enabled for images
1613
+ * // Convert PDF to HTML (Note: OCR is disabled in this one-step API)
1612
1614
  * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
1613
- * ocr: true,
1614
- * includeImages: true
1615
+ * generatorConfig: {
1616
+ * includeImages: true
1617
+ * }
1615
1618
  * });
1616
1619
  * ```
1617
1620
  */