officeparser 6.1.1 → 7.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +301 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +74 -31
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +828 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +783 -5
  54. package/dist/types.js +73 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.d.ts +8 -3
  60. package/dist/utils/envUtils.js +117 -34
  61. package/dist/utils/errorUtils.d.ts +17 -29
  62. package/dist/utils/errorUtils.js +110 -52
  63. package/dist/utils/moduleLoader.js +19 -11
  64. package/dist/utils/ocrUtils.js +2 -1
  65. package/dist/utils/sheetUtils.d.ts +7 -0
  66. package/dist/utils/sheetUtils.js +35 -0
  67. package/dist/utils/styleMapper.d.ts +36 -0
  68. package/dist/utils/styleMapper.js +224 -0
  69. package/dist/utils/xmlUtils.d.ts +0 -8
  70. package/dist/utils/xmlUtils.js +2 -1
  71. package/package.json +26 -7
@@ -12,6 +12,9 @@
12
12
  * - ODT, ODP, ODS (OpenDocument formats)
13
13
  * - PDF (Portable Document Format)
14
14
  * - RTF (Rich Text Format)
15
+ * - CSV (Comma-Separated Values)
16
+ * - MD (Markdown)
17
+ * - HTML (HyperText Markup Language)
15
18
  *
16
19
  * **Usage:**
17
20
  * ```typescript
@@ -35,13 +38,18 @@
35
38
  */
36
39
  Object.defineProperty(exports, "__esModule", { value: true });
37
40
  exports.OfficeParser = void 0;
38
- const envUtils_js_1 = require("./utils/envUtils.js");
41
+ const CsvParser_js_1 = require("./parsers/CsvParser.js");
39
42
  const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
43
+ const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
44
+ const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
40
45
  const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
41
46
  const PdfParser_js_1 = require("./parsers/PdfParser.js");
42
47
  const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
43
48
  const RtfParser_js_1 = require("./parsers/RtfParser.js");
44
49
  const WordParser_js_1 = require("./parsers/WordParser.js");
50
+ const types_js_1 = require("./types.js");
51
+ const configUtils_js_1 = require("./utils/configUtils.js");
52
+ const envUtils_js_1 = require("./utils/envUtils.js");
45
53
  const errorUtils_js_1 = require("./utils/errorUtils.js");
46
54
  const moduleLoader_js_1 = require("./utils/moduleLoader.js");
47
55
  const ocrUtils_js_1 = require("./utils/ocrUtils.js");
@@ -72,6 +80,9 @@ class OfficeParser {
72
80
  * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
73
81
  * - `.pdf` → PdfParser (PDF.js)
74
82
  * - `.rtf` → RtfParser (custom RTF parser)
83
+ * - `.csv` → CsvParser
84
+ * - `.md` → MarkdownParser
85
+ * - `.html` → HtmlParser
75
86
  *
76
87
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
77
88
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -107,28 +118,20 @@ class OfficeParser {
107
118
  else {
108
119
  actualConfig = configOrCallback || {};
109
120
  }
110
- const internalConfig = {
111
- ignoreNotes: false,
112
- newlineDelimiter: '\n',
113
- putNotesAtLast: false,
114
- outputErrorToConsole: false,
115
- extractAttachments: false,
116
- ocr: false,
117
- ocrLanguage: 'eng',
118
- includeRawContent: false,
119
- serializeRawContent: true,
120
- preserveXmlWhitespace: false,
121
- pdfWorkerSrc: '',
122
- ocrConfig: {},
123
- includeBreakNodes: false,
124
- ...actualConfig
121
+ const internalConfig = (0, configUtils_js_1.resolveParserConfig)(actualConfig);
122
+ const parsingWarnings = [];
123
+ const originalOnWarning = internalConfig.onWarning;
124
+ internalConfig.onWarning = (issue) => {
125
+ parsingWarnings.push(issue);
126
+ if (originalOnWarning)
127
+ originalOnWarning(issue);
125
128
  };
126
129
  let buffer = Buffer.alloc(0);
127
- let ext = '';
130
+ let ext = internalConfig.fileType ?? '';
128
131
  let filePath;
129
132
  try {
130
133
  if (!file) {
131
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
134
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
132
135
  }
133
136
  if (file instanceof ArrayBuffer) {
134
137
  buffer = Buffer.from(file);
@@ -144,29 +147,59 @@ class OfficeParser {
144
147
  // shim 'fs' so it won't crash at build time.
145
148
  const fs = await import('fs');
146
149
  if (!fs.existsSync(file)) {
147
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
150
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
148
151
  }
149
152
  if (fs.lstatSync(file).isDirectory()) {
150
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
153
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
151
154
  }
152
155
  buffer = fs.readFileSync(file);
153
- ext = file.split('.').pop()?.toLowerCase() || '';
156
+ ext = ext || file.split('.').pop() || '';
154
157
  }
155
158
  else {
156
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
159
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
157
160
  }
158
- if (!ext) {
159
- const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
160
- const type = await fileTypeFromBuffer(buffer);
161
- if (type) {
162
- ext = type.ext.toLowerCase();
161
+ // Attempt to detect file type from buffer only if extension is unknown.
162
+ // This matches v6 behavior and prevents crashes in older Node environments
163
+ // where file-type 22.x might be incompatible.
164
+ if (buffer.length > 0 && !ext) {
165
+ try {
166
+ const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
167
+ const type = await fileTypeFromBuffer(buffer);
168
+ if (type) {
169
+ ext = type.ext;
170
+ }
171
+ else {
172
+ // If no extension could be detected and none was provided,
173
+ // it might be a text-based format (csv, md, html) which
174
+ // lack magic bytes. We'll let the switch default handle it.
175
+ }
176
+ }
177
+ catch (error) {
178
+ // Log warning but don't crash; the switch below will handle unsupported/missing ext
179
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
180
+ }
181
+ }
182
+ else if (buffer.length > 0 && ext) {
183
+ // If extension is known, we can optionally verify it, but we wrap it
184
+ // in a try-catch to avoid breaking Node 18 if file-type fails to load.
185
+ try {
186
+ const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
187
+ const type = await fileTypeFromBuffer(buffer);
188
+ if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
189
+ // Mismatch found between authoritative extension and detected content
190
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
191
+ }
163
192
  }
164
- else {
165
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
193
+ catch (error) {
194
+ // Log warning so user knows verification could not be performed
195
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
166
196
  }
167
197
  }
198
+ if (!ext) {
199
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
200
+ }
168
201
  let result;
169
- switch (ext) {
202
+ switch (ext.toLowerCase()) {
170
203
  case 'docx':
171
204
  result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
172
205
  break;
@@ -187,9 +220,19 @@ class OfficeParser {
187
220
  case 'rtf':
188
221
  result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
189
222
  break;
223
+ case 'csv':
224
+ result = await (0, CsvParser_js_1.parseCsv)(buffer, internalConfig);
225
+ break;
226
+ case 'html':
227
+ result = await (0, HtmlParser_js_1.parseHtml)(buffer, internalConfig);
228
+ break;
229
+ case 'md':
230
+ result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
231
+ break;
190
232
  default:
191
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
233
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
192
234
  }
235
+ result.warnings = parsingWarnings;
193
236
  if (callback)
194
237
  callback(result);
195
238
  return result;
package/dist/cli.d.ts CHANGED
@@ -8,7 +8,9 @@
8
8
  * officeparser file.docx --ocr=true --extractAttachments=true
9
9
  *
10
10
  * Options (--key=value):
11
- * --toText=true Output plain text instead of JSON AST
11
+ * --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
12
+ * --output=path Save result to a file
13
+ * --toText=true Legacy flag for plain text output
12
14
  * --ocr=true Enable OCR for images
13
15
  * --ocrLanguage=eng OCR language (default: eng)
14
16
  * --extractAttachments=true Extract embedded attachments
package/dist/cli.js CHANGED
@@ -9,7 +9,9 @@
9
9
  * officeparser file.docx --ocr=true --extractAttachments=true
10
10
  *
11
11
  * Options (--key=value):
12
- * --toText=true Output plain text instead of JSON AST
12
+ * --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
13
+ * --output=path Save result to a file
14
+ * --toText=true Legacy flag for plain text output
13
15
  * --ocr=true Enable OCR for images
14
16
  * --ocrLanguage=eng OCR language (default: eng)
15
17
  * --extractAttachments=true Extract embedded attachments
@@ -18,12 +20,49 @@
18
20
  * --includeRawContent=true Include raw content in AST
19
21
  * --outputErrorToConsole=true Log errors to console
20
22
  */
23
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
24
+ if (k2 === undefined) k2 = k;
25
+ var desc = Object.getOwnPropertyDescriptor(m, k);
26
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
27
+ desc = { enumerable: true, get: function() { return m[k]; } };
28
+ }
29
+ Object.defineProperty(o, k2, desc);
30
+ }) : (function(o, m, k, k2) {
31
+ if (k2 === undefined) k2 = k;
32
+ o[k2] = m[k];
33
+ }));
34
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
35
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
36
+ }) : function(o, v) {
37
+ o["default"] = v;
38
+ });
39
+ var __importStar = (this && this.__importStar) || (function () {
40
+ var ownKeys = function(o) {
41
+ ownKeys = Object.getOwnPropertyNames || function (o) {
42
+ var ar = [];
43
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
44
+ return ar;
45
+ };
46
+ return ownKeys(o);
47
+ };
48
+ return function (mod) {
49
+ if (mod && mod.__esModule) return mod;
50
+ var result = {};
51
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
52
+ __setModuleDefault(result, mod);
53
+ return result;
54
+ };
55
+ })();
21
56
  Object.defineProperty(exports, "__esModule", { value: true });
22
57
  const OfficeParser_js_1 = require("./OfficeParser.js");
58
+ const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
59
+ const fs = __importStar(require("fs"));
23
60
  const args = process.argv.slice(2);
24
61
  let fileArg;
25
62
  let toText = false;
26
63
  let verbose = false;
64
+ let outputFormat;
65
+ let outputFile;
27
66
  const configArgs = [];
28
67
  function isConfigOption(arg) {
29
68
  return arg.startsWith('--') && arg.includes('=');
@@ -43,34 +82,75 @@ if (fileArg) {
43
82
  const cleanKey = key.replace('--', '');
44
83
  const lowerValue = value.toLowerCase();
45
84
  const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
46
- if (cleanKey === 'toText') {
47
- if (boolValue !== undefined)
48
- toText = boolValue;
49
- else
50
- console.warn(`Invalid value for toText: ${value}`);
85
+ const knownBooleans = new Set([
86
+ 'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'putNotesAtLast',
87
+ 'includeRawContent', 'outputErrorToConsole', 'serializeRawContent',
88
+ 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
89
+ ]);
90
+ if (cleanKey === 'format') {
91
+ outputFormat = value;
51
92
  }
52
- else if (cleanKey === 'verbose') {
53
- if (boolValue !== undefined)
54
- verbose = boolValue;
55
- else
56
- console.warn(`Invalid value for verbose: ${value}`);
93
+ else if (cleanKey === 'output') {
94
+ outputFile = value;
57
95
  }
58
96
  else {
59
- // @ts-ignore
60
- if (boolValue !== undefined)
61
- config[cleanKey] = boolValue;
62
- // @ts-ignore
63
- else
97
+ if (boolValue !== undefined) {
98
+ if (cleanKey === 'toText')
99
+ toText = boolValue;
100
+ else if (cleanKey === 'verbose') {
101
+ verbose = boolValue;
102
+ if (verbose)
103
+ config.outputErrorToConsole = true;
104
+ }
105
+ else {
106
+ // @ts-ignore
107
+ config[cleanKey] = boolValue;
108
+ }
109
+ }
110
+ else if (knownBooleans.has(cleanKey)) {
111
+ console.warn(`Invalid boolean value for --${cleanKey}: ${value}. Using default.`);
112
+ }
113
+ else {
114
+ // @ts-ignore
64
115
  config[cleanKey] = value;
116
+ }
65
117
  }
66
118
  });
67
119
  OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
68
120
  .then(async (ast) => {
69
- if (toText) {
70
- process.stdout.write(ast.toText() + '\n');
121
+ let output;
122
+ if (outputFormat) {
123
+ const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat);
124
+ if (Array.isArray(result.value)) {
125
+ output = JSON.stringify(result.value, null, 2);
126
+ }
127
+ else {
128
+ output = result.value;
129
+ }
130
+ }
131
+ else if (toText) {
132
+ output = ast.toText();
133
+ }
134
+ else {
135
+ output = JSON.stringify(ast, null, 2);
136
+ }
137
+ if (outputFile) {
138
+ if (output instanceof Uint8Array) {
139
+ fs.writeFileSync(outputFile, output);
140
+ }
141
+ else {
142
+ fs.writeFileSync(outputFile, output, 'utf8');
143
+ }
144
+ if (verbose)
145
+ console.log(`Output written to ${outputFile}`);
71
146
  }
72
147
  else {
73
- process.stdout.write(JSON.stringify(ast, null, 2) + '\n');
148
+ if (output instanceof Uint8Array) {
149
+ process.stdout.write(output);
150
+ }
151
+ else {
152
+ process.stdout.write(output + '\n');
153
+ }
74
154
  }
75
155
  // Ensure OCR workers are terminated for clean CLI exit
76
156
  if (config.ocr) {
@@ -97,7 +177,9 @@ else {
97
177
  console.log('Usage: officeparser <file> [--option=value]');
98
178
  console.log('');
99
179
  console.log('Options:');
100
- console.log(' --toText=true Output plain text instead of JSON AST');
180
+ console.log(' --format=json|md|html|rtf|csv|text|pdf|chunks Convert to specified format');
181
+ console.log(' --output=file.ext Save output to file instead of stdout');
182
+ console.log(' --toText=true Output plain text instead of JSON AST (legacy)');
101
183
  console.log(' --ocr=true Enable OCR for images');
102
184
  console.log(' --ocrLanguage=eng OCR language (default: eng)');
103
185
  console.log(' --extractAttachments=true Extract embedded attachments');
@@ -111,7 +193,9 @@ else {
111
193
  console.log('');
112
194
  console.log('Examples:');
113
195
  console.log(' officeparser document.docx');
114
- console.log(' officeparser document.docx --toText=true');
115
- console.log(' officeparser report.pdf --ocr=true --extractAttachments=true');
196
+ console.log(' officeparser document.docx --format=html --output=doc.html');
197
+ console.log(' officeparser document.docx --format=md');
198
+ console.log(' officeparser report.pdf --ocr=true --format=text');
199
+ console.log(' officeparser data.xlsx --format=csv --output=data.csv');
116
200
  console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
117
201
  }
@@ -0,0 +1,41 @@
1
+ import { DeepRequired, DocumentStructureChunkingConfig, FixedSizeChunkingConfig, FullGeneratorConfig, OfficeParserConfig, SemanticChunkingConfig } from './types.js';
2
+ /**
3
+ * The default regex used for identifying sentence boundaries.
4
+ * When this default is used, the generator employs a high-fidelity "robust"
5
+ * segmenter that accounts for common abbreviations (Mr., Dr., etc.).
6
+ */
7
+ export declare const DEFAULT_SENTENCE_BOUNDARY_REGEX: RegExp;
8
+ /**
9
+ * Common abbreviations that should not trigger a sentence split when followed by a period.
10
+ */
11
+ export declare const DEFAULT_ABBREVIATIONS: string[];
12
+ /**
13
+ * Default configuration for the OfficeParser.
14
+ */
15
+ export declare const DEFAULT_OFFICE_PARSER_CONFIG: DeepRequired<OfficeParserConfig>;
16
+ /**
17
+ * Default configuration for Fixed-Size chunking.
18
+ */
19
+ export declare const DEFAULT_FIXED_SIZE_CHUNKING_CONFIG: Required<Omit<FixedSizeChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
20
+ sentenceBoundaryRegex: string | RegExp;
21
+ abbreviations: string[];
22
+ };
23
+ /**
24
+ * Default configuration for Document-Structure chunking.
25
+ */
26
+ export declare const DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG: Required<Omit<DocumentStructureChunkingConfig, 'sentenceBoundaryRegex' | 'abbreviations'>> & {
27
+ sentenceBoundaryRegex: string | RegExp;
28
+ abbreviations: string[];
29
+ };
30
+ /**
31
+ * Default configuration for Semantic chunking.
32
+ * Note: `embeddingFunction` has no meaningful default and must be provided by the user.
33
+ */
34
+ export declare const DEFAULT_SEMANTIC_CHUNKING_CONFIG: Required<Omit<SemanticChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
35
+ sentenceBoundaryRegex: string | RegExp;
36
+ abbreviations: string[];
37
+ };
38
+ /**
39
+ * Default configuration for the OfficeGenerator.
40
+ */
41
+ export declare const DEFAULT_GENERATOR_CONFIG: FullGeneratorConfig;
@@ -0,0 +1,172 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.DEFAULT_GENERATOR_CONFIG = exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = exports.DEFAULT_OFFICE_PARSER_CONFIG = exports.DEFAULT_ABBREVIATIONS = exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = void 0;
4
+ const PDFJS_VERSION = '5.6.205';
5
+ const DEFAULT_PDF_WORKER_SRC = `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
6
+ /**
7
+ * The default regex used for identifying sentence boundaries.
8
+ * When this default is used, the generator employs a high-fidelity "robust"
9
+ * segmenter that accounts for common abbreviations (Mr., Dr., etc.).
10
+ */
11
+ exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = /[.!?。!?]/;
12
+ /**
13
+ * Common abbreviations that should not trigger a sentence split when followed by a period.
14
+ */
15
+ exports.DEFAULT_ABBREVIATIONS = ['Mr', 'Dr', 'Ms', 'Inc', 'Ltd', 'Prof', 'Sr', 'Jr', 'vs', 'etc'];
16
+ /**
17
+ * Default configuration for OCR.
18
+ */
19
+ const DEFAULT_OCR_CONFIG = {
20
+ language: 'eng',
21
+ workerPath: '',
22
+ corePath: '',
23
+ langPath: '',
24
+ autoTerminateTimeout: 10000,
25
+ };
26
+ /**
27
+ * Default configuration for the OfficeParser.
28
+ */
29
+ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
30
+ outputErrorToConsole: false,
31
+ onWarning: () => { },
32
+ newlineDelimiter: '\n',
33
+ ignoreNotes: false,
34
+ putNotesAtLast: false,
35
+ extractAttachments: false,
36
+ includeRawContent: false,
37
+ ocr: false,
38
+ ocrLanguage: 'eng',
39
+ ocrConfig: DEFAULT_OCR_CONFIG,
40
+ serializeRawContent: true,
41
+ preserveXmlWhitespace: false,
42
+ pdfWorkerSrc: DEFAULT_PDF_WORKER_SRC,
43
+ includeBreakNodes: false,
44
+ ignoreInternalLinks: false,
45
+ fileType: null,
46
+ csvDelimiter: ',',
47
+ };
48
+ /**
49
+ * Default configuration for HTML generation.
50
+ */
51
+ const DEFAULT_HTML_GENERATOR_CONFIG = {
52
+ standalone: true,
53
+ chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
54
+ };
55
+ /**
56
+ * Default configuration for PDF generation.
57
+ */
58
+ const DEFAULT_PDF_GENERATOR_CONFIG = {
59
+ format: 'A4',
60
+ width: '',
61
+ height: '',
62
+ landscape: false,
63
+ printBackground: true,
64
+ scale: 1,
65
+ margin: {
66
+ top: 0,
67
+ right: 0,
68
+ bottom: 0,
69
+ left: 0
70
+ },
71
+ displayHeaderFooter: false,
72
+ headerTemplate: '',
73
+ footerTemplate: '',
74
+ launchOptions: {
75
+ headless: true,
76
+ args: ['--no-sandbox', '--disable-setuid-sandbox']
77
+ },
78
+ };
79
+ /**
80
+ * Default configuration for CSV generation.
81
+ */
82
+ const DEFAULT_CSV_GENERATOR_CONFIG = {
83
+ sheets: '',
84
+ mergeSheets: true,
85
+ columnDelimiter: ',',
86
+ };
87
+ /**
88
+ * Default configuration for Markdown generation.
89
+ */
90
+ const DEFAULT_MD_GENERATOR_CONFIG = {
91
+ fallbackToHtml: true,
92
+ };
93
+ /**
94
+ * Default configuration for plain text generation.
95
+ */
96
+ const DEFAULT_TEXT_GENERATOR_CONFIG = {
97
+ newlineDelimiter: '\n',
98
+ preserveLayout: false,
99
+ };
100
+ /**
101
+ * Default configuration for Fixed-Size chunking.
102
+ */
103
+ exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = {
104
+ strategy: 'fixed-size',
105
+ chunkSize: 1000,
106
+ chunkOverlap: 200,
107
+ separators: ['\n\n', '\n', ' ', ''],
108
+ stripWhitespace: true,
109
+ includeMetadata: true,
110
+ addStartIndex: false,
111
+ lengthFunction: (text) => text.length,
112
+ sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
113
+ abbreviations: exports.DEFAULT_ABBREVIATIONS,
114
+ };
115
+ /**
116
+ * Default configuration for Document-Structure chunking.
117
+ */
118
+ exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = {
119
+ strategy: 'document-structure',
120
+ splitBy: 'paragraph',
121
+ maxChunkSize: 1000,
122
+ tableSplitStrategy: 'row',
123
+ stripWhitespace: true,
124
+ includeMetadata: true,
125
+ addStartIndex: false,
126
+ lengthFunction: (text) => text.length,
127
+ sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
128
+ abbreviations: exports.DEFAULT_ABBREVIATIONS,
129
+ };
130
+ /**
131
+ * Default configuration for Semantic chunking.
132
+ * Note: `embeddingFunction` has no meaningful default and must be provided by the user.
133
+ */
134
+ exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = {
135
+ strategy: 'semantic',
136
+ similarityThreshold: 0.8,
137
+ maxChunkSize: 2000,
138
+ bufferSize: 1,
139
+ embeddingBatchSize: 50,
140
+ stripWhitespace: true,
141
+ includeMetadata: true,
142
+ addStartIndex: false,
143
+ lengthFunction: (text) => text.length,
144
+ sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
145
+ abbreviations: exports.DEFAULT_ABBREVIATIONS,
146
+ };
147
+ /**
148
+ * The resolved default chunking config (uses document-structure as default strategy).
149
+ */
150
+ const DEFAULT_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG;
151
+ /**
152
+ * Default configuration for the OfficeGenerator.
153
+ */
154
+ exports.DEFAULT_GENERATOR_CONFIG = {
155
+ onNode: () => { },
156
+ onWarning: () => { },
157
+ styleMap: [],
158
+ includeFormatting: true,
159
+ generateIds: true,
160
+ renderMetadata: false,
161
+ ignoreDefaultStyleMap: false,
162
+ includeImages: true,
163
+ includeCharts: true,
164
+ ignoreInternalLinks: false,
165
+ htmlConfig: DEFAULT_HTML_GENERATOR_CONFIG,
166
+ mdConfig: DEFAULT_MD_GENERATOR_CONFIG,
167
+ pdfConfig: DEFAULT_PDF_GENERATOR_CONFIG,
168
+ csvConfig: DEFAULT_CSV_GENERATOR_CONFIG,
169
+ textConfig: DEFAULT_TEXT_GENERATOR_CONFIG,
170
+ rtfConfig: {},
171
+ chunksConfig: DEFAULT_CHUNKING_CONFIG,
172
+ };
@@ -0,0 +1,58 @@
1
+ import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType } from '../types.js';
2
+ import { StyleMapper } from '../utils/styleMapper.js';
3
+ /**
4
+ * Base class for all document generators.
5
+ * Provides common traversal logic and configuration handling.
6
+ */
7
+ export declare abstract class BaseGenerator<D extends string = string> {
8
+ protected destination: D;
9
+ protected config: FullGeneratorConfig;
10
+ protected ast: OfficeParserAST;
11
+ protected messages: OfficeIssue[];
12
+ protected styleMapper: StyleMapper;
13
+ constructor(destination: D, ast: OfficeParserAST, config?: GeneratorConfig<D> | FullGeneratorConfig);
14
+ /**
15
+ * Retrieves the semantic mapping for a node, respecting the includeFormatting flag.
16
+ * Per design requirements: Style mapping is bypassed if formatting is disabled.
17
+ */
18
+ protected getSemanticMapping(node: OfficeContentNode): {
19
+ tag: string;
20
+ classes: string[];
21
+ attributes: Record<string, string>;
22
+ fresh: boolean;
23
+ } | undefined;
24
+ /**
25
+ * Entry point for generation.
26
+ */
27
+ abstract generate(): Promise<ConversionResult>;
28
+ /**
29
+ * Centralized logic for handling the onNode callback.
30
+ * Evaluates the callback and returns a result that tells the generator how to proceed.
31
+ *
32
+ * @returns
33
+ * - `string`: Use this as the node's output, skip default processing.
34
+ * - `false`: Skip this node and its subtree.
35
+ * - `void`: Proceed with default processing.
36
+ */
37
+ protected handleOnNode(node: OfficeContentNode): Promise<string | false | void>;
38
+ /**
39
+ * Recursively processes nodes and builds output.
40
+ *
41
+ * @param node - The current node being processed
42
+ * @param processor - A function that takes a node and its children's output and returns the node's output string.
43
+ * @returns The generated string for this node and its subtree.
44
+ */
45
+ protected processNodeRecursive(node: OfficeContentNode, processor: (node: OfficeContentNode, childrenOutput: string) => string | Promise<string>): Promise<string>;
46
+ /**
47
+ * Helper to generate a unique ID (slug) from text.
48
+ */
49
+ protected slugify(text: string): string;
50
+ /**
51
+ * Recursively extracts plain text from a node and its children.
52
+ */
53
+ protected getNodeText(node: OfficeContentNode): string;
54
+ /**
55
+ * Reports a warning to the user and collects it for the final result.
56
+ */
57
+ protected warn(type: OfficeWarningType, info?: any, node?: OfficeContentNode): void;
58
+ }