officeparser 7.0.3 → 7.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +152 -18
  2. package/dist/OfficeGenerator.d.ts +1 -1
  3. package/dist/OfficeGenerator.js +16 -7
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +4 -0
  6. package/dist/cli.js +12 -3
  7. package/dist/defaults.js +27 -1
  8. package/dist/generators/BaseGenerator.d.ts +3 -3
  9. package/dist/generators/ChunkingGenerator.js +31 -4
  10. package/dist/generators/CsvGenerator.d.ts +1 -1
  11. package/dist/generators/HtmlGenerator.d.ts +2 -1
  12. package/dist/generators/HtmlGenerator.js +462 -40
  13. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  14. package/dist/generators/MarkdownGenerator.js +3 -1
  15. package/dist/generators/PdfGenerator.d.ts +1 -1
  16. package/dist/generators/PdfGenerator.js +51 -10
  17. package/dist/generators/RtfGenerator.d.ts +2 -1
  18. package/dist/generators/RtfGenerator.js +43 -6
  19. package/dist/generators/TextGenerator.d.ts +1 -1
  20. package/dist/officeparser.browser.d.ts +377 -53
  21. package/dist/officeparser.browser.iife.js +380 -93
  22. package/dist/officeparser.browser.mjs +380 -93
  23. package/dist/parsers/CsvParser.js +6 -1
  24. package/dist/parsers/ExcelParser.js +69 -21
  25. package/dist/parsers/HtmlParser.js +15 -1
  26. package/dist/parsers/MarkdownParser.js +18 -10
  27. package/dist/parsers/OpenOfficeParser.js +61 -34
  28. package/dist/parsers/PdfParser.js +26 -1
  29. package/dist/parsers/PowerPointParser.js +168 -40
  30. package/dist/parsers/RtfParser.js +30 -24
  31. package/dist/parsers/WordParser.js +158 -11
  32. package/dist/sbom.cdx.json +100 -100
  33. package/dist/types.d.ts +383 -53
  34. package/dist/types.js +4 -0
  35. package/dist/utils/astUtils.d.ts +2 -2
  36. package/dist/utils/astUtils.js +2 -1
  37. package/dist/utils/configUtils.d.ts +5 -0
  38. package/dist/utils/configUtils.js +69 -2
  39. package/dist/utils/errorUtils.d.ts +20 -0
  40. package/dist/utils/errorUtils.js +39 -3
  41. package/dist/utils/moduleLoader.js +3 -3
  42. package/dist/utils/ocrUtils.js +271 -66
  43. package/dist/utils/xmlUtils.d.ts +17 -0
  44. package/dist/utils/xmlUtils.js +85 -1
  45. package/package.json +3 -2
package/README.md CHANGED
@@ -100,7 +100,7 @@ npx officeparser document.pdf --format=chunks
100
100
  |------|--------|---------|-------------|
101
101
  | `--format` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
102
102
  | `--output` | path | — | Write output to a file |
103
- | `--toText` | `true\|false` | `false` | **Deprecated.** Use `--format=text` |
103
+ | ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--format=text` |
104
104
  | `--ignoreNotes` | `true\|false` | `false` | Ignore speaker notes (PPTX/ODP) |
105
105
  | `--putNotesAtLast` | `true\|false` | `false` | Collect notes at end of output |
106
106
  | `--newlineDelimiter` | string | `\n` | Delimiter between lines |
@@ -108,7 +108,7 @@ npx officeparser document.pdf --format=chunks
108
108
  | `--ocr` | `true\|false` | `false` | Enable OCR for images |
109
109
  | `--includeRawContent` | `true\|false` | `false` | Include raw XML/RTF in nodes |
110
110
  | `--includeBreakNodes` | `true\|false` | `false` | Include break nodes (DOCX only) |
111
- | `--outputErrorToConsole` | `true\|false` | `false` | **Deprecated.** Use `onWarning` callback |
111
+ | ~~`--outputErrorToConsole`~~ | `true\|false` | `false` | **Deprecated.** Use `onWarning` callback |
112
112
  | `--verbose` | `true\|false` | `false` | Show full error stack traces |
113
113
 
114
114
  ---
@@ -178,6 +178,60 @@ const ast = await officeParser.parseOffice(buffer);
178
178
  > const ast = await officeParser.parseOffice(markdownBuffer, { fileType: 'md' });
179
179
  > ```
180
180
 
181
+ ### Cancellation with AbortSignal
182
+
183
+ You can pass a standard `AbortSignal` (e.g. from an `AbortController`) to cancel an active parse operation. This is especially useful for setting request-level timeouts or canceling long-running parses (like large PDFs with OCR).
184
+
185
+ ```js
186
+ const controller = new AbortController();
187
+
188
+ // Cancel parsing if it takes longer than 5 seconds
189
+ setTimeout(() => controller.abort(), 5000);
190
+
191
+ try {
192
+ const ast = await officeParser.parseOffice('large_scanned_file.pdf', {
193
+ abortSignal: controller.signal,
194
+ ocr: true
195
+ });
196
+ } catch (err) {
197
+ if (err.name === 'AbortError') {
198
+ console.log('Parsing was cancelled.');
199
+ } else {
200
+ console.error('Parsing failed:', err);
201
+ }
202
+ }
203
+ ```
204
+
205
+ > [!IMPORTANT]
206
+ > **AbortError Propagation**
207
+ > When parsing is cancelled via `AbortSignal`, the parser rejects with a standard `AbortError` (a `DOMException` or an Error with `name: 'AbortError'`).
208
+ > This error is *not* wrapped in standard OfficeParser error types so that you can reliably detect cancellation using `error.name === 'AbortError'`.
209
+
210
+ > [!NOTE]
211
+ > **Worker Cleanup on Abort**
212
+ > If an OCR job is actively running in the background when the signal is aborted, `officeParser` automatically terminates the Tesseract worker process immediately and removes it from the pool to prevent thread/memory leaks.
213
+
214
+ ### Custom OCR Timeouts
215
+
216
+ To prevent the parser from hanging indefinitely due to slow network connections (when downloading Tesseract language datasets) or complex image processing, you can configure granular timeouts under `ocrConfig.timeout`.
217
+
218
+ ```js
219
+ const ast = await officeParser.parseOffice('scanned_document.pdf', {
220
+ ocr: true,
221
+ ocrConfig: {
222
+ timeout: {
223
+ workerLoad: 30000, // 30s max to load worker & download language training files
224
+ recognition: 15000, // 15s max per image text recognition
225
+ autoTerminate: 10000 // 10s of inactivity before terminating idle workers
226
+ }
227
+ }
228
+ });
229
+ ```
230
+
231
+ > [!TIP]
232
+ > **Non-Fatal Timeout Recovery**
233
+ > If `workerLoad` or `recognition` timeouts are exceeded, the parser will log a warning in `ast.warnings` and **continue parsing the rest of the document**. The overall promise resolves successfully with the text extracted from the document layers (rather than failing the entire parse).
234
+
181
235
  ### `ast.to()` — Generate from AST
182
236
 
183
237
  The preferred way to convert a parsed AST to another format. Returns a `ConversionResult`.
@@ -366,13 +420,19 @@ interface OfficeChunk {
366
420
  ```text
367
421
  OfficeParserAST
368
422
  ├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (11 formats)
369
- ├── metadata: { author, title, created, modified, customProperties, styleMap, ... }
423
+ ├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
370
424
  ├── content: [ OfficeContentNode ]
371
- │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | ...
425
+ │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
372
426
  │ ├── text: string (concatenated text of node + all descendants)
373
- │ ├── children: [ OfficeContentNode ] (recursive)
427
+ │ ├── children: [ OfficeContentNode ] (recursive structural children)
428
+ │ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
429
+ │ ├── comments: [ OfficeContentNode ] (inline comments attached to this node)
374
430
  │ ├── formatting: { bold, italic, underline, color, size, font, alignment, ... }
375
- │ └── metadata: { level, listId, row, col, rowSpan, colSpan, style, ... }
431
+ │ └── metadata: { level, listId, row, col, rowSpan, colSpan, backgroundColor, style, ... }
432
+ ├── auxiliary?: OfficeAuxiliaryContent (out-of-band layout elements)
433
+ │ ├── headers?: OfficeContentNode[] (DOCX headers)
434
+ │ ├── footers?: OfficeContentNode[] (DOCX footers)
435
+ │ └── slideMasters?: OfficeContentNode[] (PPTX slide masters)
376
436
  ├── attachments: [ OfficeAttachment ] (populated when extractAttachments: true)
377
437
  │ ├── type: 'image' | 'chart'
378
438
  │ ├── name: string
@@ -382,7 +442,7 @@ OfficeParserAST
382
442
  │ └── chartData?: { title, dataSets, labels }
383
443
  ├── warnings: OfficeIssue[] (non-fatal issues from the parsing phase)
384
444
  ├── to(format, config?) (format: 'html'|'md'|'text'|'csv'|'rtf'|'pdf'|'chunks', returns { value, messages })
385
- └── toText() (Deprecated: use .to('text') instead)
445
+ └── ~~toText()~~ (Deprecated: use .to('text') instead)
386
446
  ```
387
447
 
388
448
  ### `OfficeIssue` — Warning / Error Object
@@ -498,17 +558,21 @@ ast.metadata = {
498
558
  created?: Date
499
559
  modified?: Date
500
560
  description?: string
501
- customProperties?: Record<string, any> // user-defined metadata from the document
502
- styleMap?: Record<string, TextFormatting> // named styles → formatting definitions
503
- formatting?: TextFormatting // document-wide defaults
561
+ keywords?: string // NEW: Keywords from document properties
562
+ customProperties?: Record<string, any> // User-defined metadata from the document
563
+ nativeProperties?: Record<string, any> // NEW: All format-specific raw metadata
564
+ styleMap?: Record<string, TextFormatting> // Named styles → formatting definitions
565
+ formatting?: TextFormatting // Document-wide defaults
504
566
  }
505
567
  ```
506
568
 
507
- **Accessing custom properties:**
569
+ **Accessing native properties (format-specific metadata):**
508
570
  ```js
509
571
  const ast = await officeParser.parseOffice('contract.docx');
510
- console.log(ast.metadata.customProperties);
511
- // { "ProjectID": "ABC-123", "InternalReview": true }
572
+ console.log(ast.metadata.nativeProperties);
573
+ // DOCX: { Pages: 5, Application: 'Microsoft Word' }
574
+ // HTML: { description: 'My page', 'og:title': 'Title' }
575
+ // PDF: { Title: 'Report', XMP: { ... } }
512
576
  ```
513
577
 
514
578
  ---
@@ -532,6 +596,61 @@ const headings = ast.content.filter(n => n.type === 'heading' && n.metadata?.lev
532
596
  console.log(headings.map(h => h.text));
533
597
  ```
534
598
 
599
+ ### Extract comments
600
+ ```ts
601
+ // Comments can be attached to any nested node, so we must traverse recursively
602
+ const printComments = (nodes: OfficeContentNode[]) => {
603
+ nodes.forEach(node => {
604
+ if (node.comments) {
605
+ node.comments.forEach(c => {
606
+ console.log(`Comment by ${c.metadata?.author}: ${c.text}`);
607
+ });
608
+ }
609
+ if (node.children) {
610
+ printComments(node.children);
611
+ }
612
+ });
613
+ };
614
+
615
+ printComments(ast.content);
616
+ ```
617
+
618
+ Set `ignoreComments: true` to skip extraction.
619
+
620
+ ### Extract footnotes, endnotes & slide notes
621
+ ```ts
622
+ // Slide speaker notes (PPTX) live on the slide node itself
623
+ const slide = ast.content.find(n => n.type === 'slide');
624
+ console.log(slide?.notes?.map(n => n.text));
625
+
626
+ // Footnotes and endnotes (DOCX/RTF) can be deeply nested, so we traverse recursively:
627
+ const printNotes = (nodes: OfficeContentNode[]) => {
628
+ nodes.forEach(node => {
629
+ if (node.notes) {
630
+ node.notes.forEach(note => console.log(note.text));
631
+ }
632
+ if (node.children) {
633
+ printNotes(node.children);
634
+ }
635
+ });
636
+ };
637
+
638
+ printNotes(ast.content);
639
+ ```
640
+
641
+ > [!IMPORTANT]
642
+ > `putNotesAtLast` is **deprecated**. Notes are always attached via `node.notes` — this flag has no effect and will be removed in a future major version.
643
+
644
+ ### Access headers, footers & slide masters
645
+ ```ts
646
+ // These are NOT in ast.content — use ast.auxiliary
647
+ console.log(ast.auxiliary?.headers?.map(h => h.text)); // DOCX headers
648
+ console.log(ast.auxiliary?.footers?.map(f => f.text)); // DOCX footers
649
+ console.log(ast.auxiliary?.slideMasters?.length); // PPTX slide masters
650
+ ```
651
+
652
+ Set `ignoreHeadersAndFooters: true` or `ignoreSlideMasters: true` to skip extraction.
653
+
535
654
  ### Extract images with OCR text
536
655
  ```js
537
656
  const ast = await officeParser.parseOffice('report.docx', { extractAttachments: true, ocr: true });
@@ -596,8 +715,11 @@ Pass as the second argument to `parseOffice(file, config)`.
596
715
  | Option | Type | Default | Description |
597
716
  |--------|------|---------|-------------|
598
717
  | `newlineDelimiter` | `string` | `'\n'` | Delimiter inserted between lines in text output |
599
- | `ignoreNotes` | `boolean` | `false` | Ignore speaker notes (PPTX/ODP) |
600
- | `putNotesAtLast` | `boolean` | `false` | Collect all notes at the end instead of inline |
718
+ | `ignoreNotes` | `boolean` | `false` | Ignore footnotes/endnotes (DOCX, RTF) and speaker notes (PPTX/ODP) |
719
+ | `ignoreComments` | `boolean` | `false` | **New**: Ignore inline comments/annotations (DOCX, XLSX, PPTX) — by default attached via `node.comments[]` |
720
+ | `ignoreHeadersAndFooters` | `boolean` | `false` | **New**: Skip DOCX headers & footers (populated in `ast.auxiliary.headers/footers` by default) |
721
+ | `ignoreSlideMasters` | `boolean` | `false` | **New**: Skip PPTX slide masters (populated in `ast.auxiliary.slideMasters` by default) |
722
+ | ~~`putNotesAtLast`~~ | `boolean` | `false` | **Deprecated**: Notes are now attached via `node.notes[]`. This flag has no effect |
601
723
  | `extractAttachments` | `boolean` | `false` | Populate `ast.attachments` with Base64 images/charts |
602
724
  | `ocr` | `boolean` | `false` | Run Tesseract OCR on images (requires `extractAttachments: true`) |
603
725
  | `ocrConfig` | `OcrConfig` | `{}` | OCR worker pool settings — see [OCR section](#ocr-scheduler--resource-management) |
@@ -610,7 +732,8 @@ Pass as the second argument to `parseOffice(file, config)`.
610
732
  | `csvDelimiter` | `string` | `','` | Input delimiter when parsing CSV files |
611
733
  | `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
612
734
  | `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
613
- | `outputErrorToConsole` | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
735
+ | `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
736
+ | ~~`outputErrorToConsole`~~ | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
614
737
 
615
738
  ---
616
739
 
@@ -630,6 +753,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
630
753
  | `styleMap` | `string[] \| StructuredStyleMapping[]` | `[]` | Custom semantic style mappings |
631
754
  | `onNode` | `(node) => string \| false \| void` | — | Per-node callback for filtering, overriding, or mutating |
632
755
  | `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal generation issues |
756
+ | `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel the generation operation (rejects with AbortError) |
633
757
 
634
758
  ---
635
759
 
@@ -708,6 +832,12 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
708
832
  |--------|------|---------|-------------|
709
833
  | `standalone` | `boolean` | `true` | Wrap output in a full `<html>` document with CSS |
710
834
  | `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
835
+ | `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
836
+ | `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block — use to override built-in styles |
837
+ | `injections.headStart` | `string` | `''` | Raw HTML injected after `<head>` |
838
+ | `injections.headEnd` | `string` | `''` | Raw HTML injected before `</head>` |
839
+ | `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
840
+ | `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
711
841
 
712
842
  ### MdGeneratorConfig
713
843
 
@@ -724,6 +854,8 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
724
854
  | Option | Type | Default | Description |
725
855
  |--------|------|---------|-------------|
726
856
  | `format` | `string` | `'A4'` | Paper format (`'A4'`, `'Letter'`, `'Legal'`, etc.) |
857
+ | `width` | `string \| number` | `''` | Paper width (e.g., `'5in'`, `'3cm'`) or pixels |
858
+ | `height` | `string \| number` | `''` | Paper height (e.g., `'5in'`, `'3cm'`) or pixels |
727
859
  | `landscape` | `boolean` | `false` | Landscape page orientation |
728
860
  | `printBackground` | `boolean` | `true` | Print background graphics |
729
861
  | `margin` | `object` | `{0,0,0,0}` | Page margins (`top`, `right`, `bottom`, `left`) |
@@ -732,6 +864,7 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
732
864
  | `footerTemplate` | `string` | `''` | HTML template for the print footer |
733
865
  | `scale` | `number` | `1` | Rendering scale factor |
734
866
  | `launchOptions` | `object` | headless defaults | Puppeteer launch options (e.g., `executablePath`) |
867
+ | `timeout` | `number` | `30000` | PDF rendering timeout in milliseconds. Set to `0` to disable. |
735
868
 
736
869
  ### CsvGeneratorConfig
737
870
 
@@ -807,6 +940,7 @@ Configuration for `OfficeConverter.convert(file, format, config)`.
807
940
  | `maxChunkSize` | `number` | `2000` | Max characters even if similarity stays high |
808
941
  | `bufferSize` | `number` | `1` | Surrounding sentences used when computing similarity |
809
942
  | `embeddingBatchSize` | `number` | `50` | Sentences per embedding API batch |
943
+ | `timeout` | `number` | `10000` | Timeout in milliseconds for individual embedding API calls. Set to `0` to disable. |
810
944
 
811
945
  ---
812
946
 
@@ -826,7 +960,8 @@ When `ocr: true` is set, `officeParser` maintains an intelligent **Smart Worker
826
960
  | `workerPath` | `string` | `''` | Custom path to Tesseract worker script |
827
961
  | `corePath` | `string` | `''` | Custom path to Tesseract core script |
828
962
  | `langPath` | `string` | `''` | Custom path for language data files |
829
- | `autoTerminateTimeout` | `number` | `10000` | Inactivity timeout in ms before auto-teardown (0 = disabled) |
963
+ | `timeout` | `OcrTimeoutConfig` | `{}` | Consolidated timeouts: `autoTerminate`, `workerLoad`, `recognition` |
964
+ | ~~`autoTerminateTimeout`~~ | `number` | `10000` | **Deprecated.** Use `timeout.autoTerminate` instead |
830
965
 
831
966
  See all language codes at [tesseract-ocr.github.io](https://tesseract-ocr.github.io/tessdoc/Data-Files).
832
967
 
@@ -927,7 +1062,6 @@ For a full debugging guide, visit the [Live Documentation](https://harshankur.gi
927
1062
 
928
1063
  1. **ODT/ODS Charts**: May show inaccurate data when the chart references external cell ranges or uses complex layout-based data.
929
1064
  2. **PDF Images (Browser)**: Extracted as BMP files for cross-platform compatibility. Conversion is automatic.
930
- 3. **RTF Notes**: `putNotesAtLast` has no effect for RTF files; footnotes and endnotes are always appended at the end.
931
1065
 
932
1066
  ---
933
1067
 
@@ -15,5 +15,5 @@ export declare class OfficeGenerator {
15
15
  */
16
16
  static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
17
17
  type: T;
18
- }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
18
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
19
19
  }
@@ -25,24 +25,33 @@ class OfficeGenerator {
25
25
  * @throws {Error} If the destination format is unsupported
26
26
  */
27
27
  static async generate(ast, destination, config) {
28
+ let generator;
28
29
  switch (destination.toLowerCase()) {
29
30
  case 'text':
30
- return new TextGenerator_js_1.TextGenerator(ast, config).generate();
31
+ generator = new TextGenerator_js_1.TextGenerator(ast, config);
32
+ break;
31
33
  case 'md':
32
- return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
34
+ generator = new MarkdownGenerator_js_1.MarkdownGenerator(ast, config);
35
+ break;
33
36
  case 'html':
34
- return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
37
+ generator = new HtmlGenerator_js_1.HtmlGenerator(ast, config);
38
+ break;
35
39
  case 'pdf':
36
- return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
40
+ generator = new PdfGenerator_js_1.PdfGenerator(ast, config);
41
+ break;
37
42
  case 'csv':
38
- return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
43
+ generator = new CsvGenerator_js_1.CsvGenerator(ast, config);
44
+ break;
39
45
  case 'rtf':
40
- return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
46
+ generator = new RtfGenerator_js_1.RtfGenerator(ast, config);
47
+ break;
41
48
  case 'chunks':
42
- return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
49
+ generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
50
+ break;
43
51
  default:
44
52
  throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
45
53
  }
54
+ return generator.generate();
46
55
  }
47
56
  }
48
57
  exports.OfficeGenerator = OfficeGenerator;
@@ -241,6 +241,12 @@ class OfficeParser {
241
241
  return result;
242
242
  }
243
243
  catch (error) {
244
+ // AbortError must pass through untouched so callers can distinguish a
245
+ // deliberate cancellation (err.name === 'AbortError') from a real parse failure.
246
+ // getWrappedError always creates a plain new Error(), which would strip the
247
+ // AbortError identity and break any instanceof / name checks on the caller side.
248
+ if (error?.name === 'AbortError')
249
+ throw error;
244
250
  const wrappedError = (0, errorUtils_js_1.getWrappedError)(error, internalConfig, filePath);
245
251
  if (callback)
246
252
  callback(undefined, wrappedError);
package/dist/cli.d.ts CHANGED
@@ -15,6 +15,10 @@
15
15
  * --ocrLanguage=eng OCR language (default: eng)
16
16
  * --extractAttachments=true Extract embedded attachments
17
17
  * --ignoreNotes=true Ignore footnotes/endnotes
18
+ * --ignoreComments=true Ignore inline comments
19
+ * --ignoreHeadersAndFooters=true Ignore headers and footers
20
+ * --ignoreSlideMasters=true Ignore slide masters
21
+ * --ignoreInternalLinks=true Ignore internal links
18
22
  * --putNotesAtLast=true Move notes to end of document
19
23
  * --includeRawContent=true Include raw content in AST
20
24
  * --outputErrorToConsole=true Log errors to console
package/dist/cli.js CHANGED
@@ -16,6 +16,10 @@
16
16
  * --ocrLanguage=eng OCR language (default: eng)
17
17
  * --extractAttachments=true Extract embedded attachments
18
18
  * --ignoreNotes=true Ignore footnotes/endnotes
19
+ * --ignoreComments=true Ignore inline comments
20
+ * --ignoreHeadersAndFooters=true Ignore headers and footers
21
+ * --ignoreSlideMasters=true Ignore slide masters
22
+ * --ignoreInternalLinks=true Ignore internal links
19
23
  * --putNotesAtLast=true Move notes to end of document
20
24
  * --includeRawContent=true Include raw content in AST
21
25
  * --outputErrorToConsole=true Log errors to console
@@ -83,9 +87,10 @@ if (fileArg) {
83
87
  const lowerValue = value.toLowerCase();
84
88
  const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
85
89
  const knownBooleans = new Set([
86
- 'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'putNotesAtLast',
87
- 'includeRawContent', 'outputErrorToConsole', 'serializeRawContent',
88
- 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
90
+ 'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'ignoreComments',
91
+ 'ignoreHeadersAndFooters', 'ignoreSlideMasters', 'ignoreInternalLinks',
92
+ 'putNotesAtLast', 'includeRawContent', 'outputErrorToConsole',
93
+ 'serializeRawContent', 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
89
94
  ]);
90
95
  if (cleanKey === 'format') {
91
96
  outputFormat = value;
@@ -184,6 +189,10 @@ else {
184
189
  console.log(' --ocrLanguage=eng OCR language (default: eng)');
185
190
  console.log(' --extractAttachments=true Extract embedded attachments');
186
191
  console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
192
+ console.log(' --ignoreComments=true Ignore inline comments');
193
+ console.log(' --ignoreHeadersAndFooters=true Ignore headers and footers');
194
+ console.log(' --ignoreSlideMasters=true Ignore slide masters');
195
+ console.log(' --ignoreInternalLinks=true Ignore internal links');
187
196
  console.log(' --putNotesAtLast=true Move notes to end of document');
188
197
  console.log(' --includeRawContent=true Include raw content in AST');
189
198
  console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
package/dist/defaults.js CHANGED
@@ -13,6 +13,12 @@ exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = /[.!?。!?]/;
13
13
  * Common abbreviations that should not trigger a sentence split when followed by a period.
14
14
  */
15
15
  exports.DEFAULT_ABBREVIATIONS = ['Mr', 'Dr', 'Ms', 'Inc', 'Ltd', 'Prof', 'Sr', 'Jr', 'vs', 'etc'];
16
+ /** Default timeout values for OCR */
17
+ const DEFAULT_OCR_TIMEOUT = {
18
+ autoTerminate: 10000,
19
+ workerLoad: 60000,
20
+ recognition: 30000,
21
+ };
16
22
  /**
17
23
  * Default configuration for OCR.
18
24
  */
@@ -21,7 +27,12 @@ const DEFAULT_OCR_CONFIG = {
21
27
  workerPath: '',
22
28
  corePath: '',
23
29
  langPath: '',
24
- autoTerminateTimeout: 10000,
30
+ // Preferred: consolidated timeout object. New code should always read from here.
31
+ timeout: DEFAULT_OCR_TIMEOUT,
32
+ // Kept for backward compatibility. When timeout.autoTerminate is set (as above),
33
+ // the ocrUtils resolution logic will prefer timeout.autoTerminate over this flat field.
34
+ autoTerminateTimeout: DEFAULT_OCR_TIMEOUT.autoTerminate,
35
+ abortSignal: null,
25
36
  };
26
37
  /**
27
38
  * Default configuration for the OfficeParser.
@@ -31,12 +42,16 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
31
42
  onWarning: () => { },
32
43
  newlineDelimiter: '\n',
33
44
  ignoreNotes: false,
45
+ ignoreComments: false,
46
+ ignoreHeadersAndFooters: false,
47
+ ignoreSlideMasters: false,
34
48
  putNotesAtLast: false,
35
49
  extractAttachments: false,
36
50
  includeRawContent: false,
37
51
  ocr: false,
38
52
  ocrLanguage: 'eng',
39
53
  ocrConfig: DEFAULT_OCR_CONFIG,
54
+ abortSignal: null,
40
55
  serializeRawContent: true,
41
56
  preserveXmlWhitespace: false,
42
57
  pdfWorkerSrc: DEFAULT_PDF_WORKER_SRC,
@@ -51,6 +66,14 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
51
66
  const DEFAULT_HTML_GENERATOR_CONFIG = {
52
67
  standalone: true,
53
68
  chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
69
+ containerWidth: 'auto',
70
+ customCss: '',
71
+ injections: {
72
+ headStart: '',
73
+ headEnd: '',
74
+ bodyStart: '',
75
+ bodyEnd: '',
76
+ }
54
77
  };
55
78
  /**
56
79
  * Default configuration for PDF generation.
@@ -75,6 +98,7 @@ const DEFAULT_PDF_GENERATOR_CONFIG = {
75
98
  headless: true,
76
99
  args: ['--no-sandbox', '--disable-setuid-sandbox']
77
100
  },
101
+ timeout: 30000,
78
102
  };
79
103
  /**
80
104
  * Default configuration for CSV generation.
@@ -143,6 +167,7 @@ exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = {
143
167
  lengthFunction: (text) => text.length,
144
168
  sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
145
169
  abbreviations: exports.DEFAULT_ABBREVIATIONS,
170
+ timeout: 10000,
146
171
  };
147
172
  /**
148
173
  * The resolved default chunking config (uses document-structure as default strategy).
@@ -162,6 +187,7 @@ exports.DEFAULT_GENERATOR_CONFIG = {
162
187
  includeImages: true,
163
188
  includeCharts: true,
164
189
  ignoreInternalLinks: false,
190
+ abortSignal: null,
165
191
  htmlConfig: DEFAULT_HTML_GENERATOR_CONFIG,
166
192
  mdConfig: DEFAULT_MD_GENERATOR_CONFIG,
167
193
  pdfConfig: DEFAULT_PDF_GENERATOR_CONFIG,
@@ -1,10 +1,10 @@
1
- import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType } from '../types.js';
1
+ import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType, UniversalGeneratorFormat } from '../types.js';
2
2
  import { StyleMapper } from '../utils/styleMapper.js';
3
3
  /**
4
4
  * Base class for all document generators.
5
5
  * Provides common traversal logic and configuration handling.
6
6
  */
7
- export declare abstract class BaseGenerator<D extends string = string> {
7
+ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat = UniversalGeneratorFormat> {
8
8
  protected destination: D;
9
9
  protected config: FullGeneratorConfig;
10
10
  protected ast: OfficeParserAST;
@@ -24,7 +24,7 @@ export declare abstract class BaseGenerator<D extends string = string> {
24
24
  /**
25
25
  * Entry point for generation.
26
26
  */
27
- abstract generate(): Promise<ConversionResult>;
27
+ abstract generate(): Promise<ConversionResult<D>>;
28
28
  /**
29
29
  * Centralized logic for handling the onNode callback.
30
30
  * Evaluates the callback and returns a result that tells the generator how to proceed.
@@ -39,6 +39,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
39
39
  * Note: ConversionResult.value is a JSON string of OfficeChunk[] for the 'chunks' destination.
40
40
  */
41
41
  async generate() {
42
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
42
43
  let chunks;
43
44
  switch (this.chunkConfig.strategy) {
44
45
  case 'fixed-size':
@@ -186,6 +187,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
186
187
  return this.finalizeChunks(chunks, config);
187
188
  }
188
189
  async processNodeForStructure(node, config, splitBy, maxChunkSize, measure, chunks, contextStack) {
190
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
189
191
  // Check for node override or skip
190
192
  const override = await this.handleOnNode(node);
191
193
  if (override === false)
@@ -391,7 +393,14 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
391
393
  renderedRows.push(row.text ?? '');
392
394
  continue;
393
395
  }
394
- const cells = row.children.map(cell => (cell.text ?? '').replace(/\n/g, ' ').trim());
396
+ const getCellText = (cell) => {
397
+ if (cell.text)
398
+ return cell.text;
399
+ if (!cell.children || cell.children.length === 0)
400
+ return '';
401
+ return cell.children.map(c => getCellText(c)).join(' ');
402
+ };
403
+ const cells = row.children.map(cell => getCellText(cell).replace(/\n/g, ' ').trim());
395
404
  renderedRows.push(`| ${cells.join(' | ')} |`);
396
405
  }
397
406
  return renderedRows.join('\n');
@@ -412,7 +421,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
412
421
  if (sentences.length === 0)
413
422
  return [];
414
423
  // Embed all sentences in batches to avoid rate limiting
415
- const embeddings = await this.batchEmbeddings(sentences, config.embeddingFunction, batchSize);
424
+ const embeddings = await this.batchEmbeddings(sentences, config.embeddingFunction, batchSize, config.timeout);
416
425
  // Calculate cosine similarity between adjacent sentence windows
417
426
  const chunks = [];
418
427
  let currentSentences = [];
@@ -479,6 +488,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
479
488
  let currentPage;
480
489
  let currentSheet;
481
490
  const walk = async (node) => {
491
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
482
492
  const override = await this.handleOnNode(node);
483
493
  if (override === false)
484
494
  return;
@@ -533,6 +543,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
533
543
  let currentPage;
534
544
  let currentSheet;
535
545
  const walk = async (node) => {
546
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
536
547
  const override = await this.handleOnNode(node);
537
548
  if (override === false)
538
549
  return;
@@ -602,11 +613,27 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
602
613
  /**
603
614
  * Helper to process embeddings in sequential batches to avoid API rate limits and memory issues.
604
615
  */
605
- async batchEmbeddings(sentences, embedFn, batchSize = 50) {
616
+ async batchEmbeddings(sentences, embedFn, batchSize = 50, timeoutMs) {
606
617
  const results = [];
607
618
  for (let i = 0; i < sentences.length; i += batchSize) {
619
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
608
620
  const batch = sentences.slice(i, i + batchSize);
609
- const batchResults = await Promise.all(batch.map(s => embedFn(s.text)));
621
+ const batchPromises = batch.map(s => {
622
+ const call = embedFn(s.text);
623
+ if (timeoutMs !== undefined && timeoutMs > 0) {
624
+ let timerId;
625
+ const timeoutPromise = new Promise((_, reject) => {
626
+ timerId = setTimeout(() => {
627
+ reject(new Error(`Embedding call timed out after ${timeoutMs}ms`));
628
+ }, timeoutMs);
629
+ });
630
+ return Promise.race([call, timeoutPromise]).finally(() => {
631
+ clearTimeout(timerId);
632
+ });
633
+ }
634
+ return call;
635
+ });
636
+ const batchResults = await Promise.all(batchPromises);
610
637
  results.push(...batchResults);
611
638
  }
612
639
  return results;
@@ -10,7 +10,7 @@ export declare class CsvGenerator extends BaseGenerator<'csv'> {
10
10
  *
11
11
  * @returns A CSV string or a ZIP archive containing multiple CSVs
12
12
  */
13
- generate(): Promise<ConversionResult>;
13
+ generate(): Promise<ConversionResult<'csv'>>;
14
14
  /**
15
15
  * Recursively finds all nodes that can be treated as sheets (sheet or table).
16
16
  */
@@ -12,7 +12,7 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
12
12
  *
13
13
  * @returns An HTML string
14
14
  */
15
- generate(): Promise<ConversionResult>;
15
+ generate(): Promise<ConversionResult<'html'>>;
16
16
  private renderMetaTags;
17
17
  private renderMetadataSummary;
18
18
  /**
@@ -33,5 +33,6 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
33
33
  private getInlineStyles;
34
34
  private getPremiumStyles;
35
35
  protected slugify(text: string): string;
36
+ private getColumnLetter;
36
37
  private escape;
37
38
  }