officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -0,0 +1,19 @@
1
+ import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
2
+ /**
3
+ * Main generator class providing document conversion functionality.
4
+ */
5
+ export declare class OfficeGenerator {
6
+ /**
7
+ * Generates a file of the specified type from an AST.
8
+ * This is the single source of truth for generation logic.
9
+ *
10
+ * @param ast - The OfficeParserAST to generate from
11
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
12
+ * @param config - Optional configuration for the generator
13
+ * @returns A promise resolving to the ConversionResult containing the value and messages
14
+ * @throws {Error} If the destination format is unsupported
15
+ */
16
+ static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
17
+ type: T;
18
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
19
+ }
@@ -0,0 +1,48 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.OfficeGenerator = void 0;
4
+ const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
5
+ const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
6
+ const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
7
+ const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
8
+ const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
9
+ const RtfGenerator_js_1 = require("./generators/RtfGenerator.js");
10
+ const TextGenerator_js_1 = require("./generators/TextGenerator.js");
11
+ const types_js_1 = require("./types.js");
12
+ const errorUtils_js_1 = require("./utils/errorUtils.js");
13
+ /**
14
+ * Main generator class providing document conversion functionality.
15
+ */
16
+ class OfficeGenerator {
17
+ /**
18
+ * Generates a file of the specified type from an AST.
19
+ * This is the single source of truth for generation logic.
20
+ *
21
+ * @param ast - The OfficeParserAST to generate from
22
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
23
+ * @param config - Optional configuration for the generator
24
+ * @returns A promise resolving to the ConversionResult containing the value and messages
25
+ * @throws {Error} If the destination format is unsupported
26
+ */
27
+ static async generate(ast, destination, config) {
28
+ switch (destination.toLowerCase()) {
29
+ case 'text':
30
+ return new TextGenerator_js_1.TextGenerator(ast, config).generate();
31
+ case 'md':
32
+ return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
33
+ case 'html':
34
+ return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
35
+ case 'pdf':
36
+ return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
37
+ case 'csv':
38
+ return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
39
+ case 'rtf':
40
+ return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
41
+ case 'chunks':
42
+ return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
43
+ default:
44
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
45
+ }
46
+ }
47
+ }
48
+ exports.OfficeGenerator = OfficeGenerator;
@@ -11,6 +11,9 @@
11
11
  * - ODT, ODP, ODS (OpenDocument formats)
12
12
  * - PDF (Portable Document Format)
13
13
  * - RTF (Rich Text Format)
14
+ * - CSV (Comma-Separated Values)
15
+ * - MD (Markdown)
16
+ * - HTML (HyperText Markup Language)
14
17
  *
15
18
  * **Usage:**
16
19
  * ```typescript
@@ -60,6 +63,9 @@ export declare class OfficeParser {
60
63
  * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
61
64
  * - `.pdf` → PdfParser (PDF.js)
62
65
  * - `.rtf` → RtfParser (custom RTF parser)
66
+ * - `.csv` → CsvParser
67
+ * - `.md` → MarkdownParser
68
+ * - `.html` → HtmlParser
63
69
  *
64
70
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
65
71
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -12,6 +12,9 @@
12
12
  * - ODT, ODP, ODS (OpenDocument formats)
13
13
  * - PDF (Portable Document Format)
14
14
  * - RTF (Rich Text Format)
15
+ * - CSV (Comma-Separated Values)
16
+ * - MD (Markdown)
17
+ * - HTML (HyperText Markup Language)
15
18
  *
16
19
  * **Usage:**
17
20
  * ```typescript
@@ -35,13 +38,18 @@
35
38
  */
36
39
  Object.defineProperty(exports, "__esModule", { value: true });
37
40
  exports.OfficeParser = void 0;
38
- const envUtils_js_1 = require("./utils/envUtils.js");
41
+ const CsvParser_js_1 = require("./parsers/CsvParser.js");
39
42
  const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
43
+ const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
44
+ const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
40
45
  const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
41
46
  const PdfParser_js_1 = require("./parsers/PdfParser.js");
42
47
  const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
43
48
  const RtfParser_js_1 = require("./parsers/RtfParser.js");
44
49
  const WordParser_js_1 = require("./parsers/WordParser.js");
50
+ const types_js_1 = require("./types.js");
51
+ const configUtils_js_1 = require("./utils/configUtils.js");
52
+ const envUtils_js_1 = require("./utils/envUtils.js");
45
53
  const errorUtils_js_1 = require("./utils/errorUtils.js");
46
54
  const moduleLoader_js_1 = require("./utils/moduleLoader.js");
47
55
  const ocrUtils_js_1 = require("./utils/ocrUtils.js");
@@ -72,6 +80,9 @@ class OfficeParser {
72
80
  * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
73
81
  * - `.pdf` → PdfParser (PDF.js)
74
82
  * - `.rtf` → RtfParser (custom RTF parser)
83
+ * - `.csv` → CsvParser
84
+ * - `.md` → MarkdownParser
85
+ * - `.html` → HtmlParser
75
86
  *
76
87
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
77
88
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -107,27 +118,20 @@ class OfficeParser {
107
118
  else {
108
119
  actualConfig = configOrCallback || {};
109
120
  }
110
- const internalConfig = {
111
- ignoreNotes: false,
112
- newlineDelimiter: '\n',
113
- putNotesAtLast: false,
114
- outputErrorToConsole: false,
115
- extractAttachments: false,
116
- ocr: false,
117
- ocrLanguage: 'eng',
118
- includeRawContent: false,
119
- serializeRawContent: true,
120
- preserveXmlWhitespace: false,
121
- pdfWorkerSrc: '',
122
- ocrConfig: {},
123
- ...actualConfig
121
+ const internalConfig = (0, configUtils_js_1.resolveParserConfig)(actualConfig);
122
+ const parsingWarnings = [];
123
+ const originalOnWarning = internalConfig.onWarning;
124
+ internalConfig.onWarning = (issue) => {
125
+ parsingWarnings.push(issue);
126
+ if (originalOnWarning)
127
+ originalOnWarning(issue);
124
128
  };
125
129
  let buffer = Buffer.alloc(0);
126
- let ext = '';
130
+ let ext = internalConfig.fileType ?? '';
127
131
  let filePath;
128
132
  try {
129
133
  if (!file) {
130
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
134
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
131
135
  }
132
136
  if (file instanceof ArrayBuffer) {
133
137
  buffer = Buffer.from(file);
@@ -143,29 +147,42 @@ class OfficeParser {
143
147
  // shim 'fs' so it won't crash at build time.
144
148
  const fs = await import('fs');
145
149
  if (!fs.existsSync(file)) {
146
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
150
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
147
151
  }
148
152
  if (fs.lstatSync(file).isDirectory()) {
149
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
153
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
150
154
  }
151
155
  buffer = fs.readFileSync(file);
152
- ext = file.split('.').pop()?.toLowerCase() || '';
156
+ ext = ext || file.split('.').pop() || '';
153
157
  }
154
158
  else {
155
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
159
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
156
160
  }
157
- if (!ext) {
161
+ // Always attempt to detect file type from buffer if it exists,
162
+ // but respect the authoritative 'ext' if it was already set.
163
+ if (buffer.length > 0) {
158
164
  const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
159
165
  const type = await fileTypeFromBuffer(buffer);
160
- if (type) {
161
- ext = type.ext.toLowerCase();
166
+ if (!ext) {
167
+ if (type) {
168
+ ext = type.ext;
169
+ }
170
+ else {
171
+ // If no extension could be detected and none was provided,
172
+ // it might be a text-based format (csv, md, html) which
173
+ // lack magic bytes. We'll let the switch default handle it.
174
+ }
162
175
  }
163
- else {
164
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
176
+ else if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
177
+ // Mismatch found between authoritative extension and detected content
178
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
165
179
  }
166
180
  }
181
+ if (!ext) {
182
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
183
+ }
167
184
  let result;
168
- switch (ext) {
185
+ switch (ext.toLowerCase()) {
169
186
  case 'docx':
170
187
  result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
171
188
  break;
@@ -186,9 +203,19 @@ class OfficeParser {
186
203
  case 'rtf':
187
204
  result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
188
205
  break;
206
+ case 'csv':
207
+ result = await (0, CsvParser_js_1.parseCsv)(buffer, internalConfig);
208
+ break;
209
+ case 'html':
210
+ result = await (0, HtmlParser_js_1.parseHtml)(buffer, internalConfig);
211
+ break;
212
+ case 'md':
213
+ result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
214
+ break;
189
215
  default:
190
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
216
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
191
217
  }
218
+ result.warnings = parsingWarnings;
192
219
  if (callback)
193
220
  callback(result);
194
221
  return result;
package/dist/cli.d.ts CHANGED
@@ -8,7 +8,9 @@
8
8
  * officeparser file.docx --ocr=true --extractAttachments=true
9
9
  *
10
10
  * Options (--key=value):
11
- * --toText=true Output plain text instead of JSON AST
11
+ * --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
12
+ * --output=path Save result to a file
13
+ * --toText=true Legacy flag for plain text output
12
14
  * --ocr=true Enable OCR for images
13
15
  * --ocrLanguage=eng OCR language (default: eng)
14
16
  * --extractAttachments=true Extract embedded attachments
package/dist/cli.js CHANGED
@@ -9,7 +9,9 @@
9
9
  * officeparser file.docx --ocr=true --extractAttachments=true
10
10
  *
11
11
  * Options (--key=value):
12
- * --toText=true Output plain text instead of JSON AST
12
+ * --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
13
+ * --output=path Save result to a file
14
+ * --toText=true Legacy flag for plain text output
13
15
  * --ocr=true Enable OCR for images
14
16
  * --ocrLanguage=eng OCR language (default: eng)
15
17
  * --extractAttachments=true Extract embedded attachments
@@ -18,12 +20,49 @@
18
20
  * --includeRawContent=true Include raw content in AST
19
21
  * --outputErrorToConsole=true Log errors to console
20
22
  */
23
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
24
+ if (k2 === undefined) k2 = k;
25
+ var desc = Object.getOwnPropertyDescriptor(m, k);
26
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
27
+ desc = { enumerable: true, get: function() { return m[k]; } };
28
+ }
29
+ Object.defineProperty(o, k2, desc);
30
+ }) : (function(o, m, k, k2) {
31
+ if (k2 === undefined) k2 = k;
32
+ o[k2] = m[k];
33
+ }));
34
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
35
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
36
+ }) : function(o, v) {
37
+ o["default"] = v;
38
+ });
39
+ var __importStar = (this && this.__importStar) || (function () {
40
+ var ownKeys = function(o) {
41
+ ownKeys = Object.getOwnPropertyNames || function (o) {
42
+ var ar = [];
43
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
44
+ return ar;
45
+ };
46
+ return ownKeys(o);
47
+ };
48
+ return function (mod) {
49
+ if (mod && mod.__esModule) return mod;
50
+ var result = {};
51
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
52
+ __setModuleDefault(result, mod);
53
+ return result;
54
+ };
55
+ })();
21
56
  Object.defineProperty(exports, "__esModule", { value: true });
22
57
  const OfficeParser_js_1 = require("./OfficeParser.js");
58
+ const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
59
+ const fs = __importStar(require("fs"));
23
60
  const args = process.argv.slice(2);
24
61
  let fileArg;
25
62
  let toText = false;
26
63
  let verbose = false;
64
+ let outputFormat;
65
+ let outputFile;
27
66
  const configArgs = [];
28
67
  function isConfigOption(arg) {
29
68
  return arg.startsWith('--') && arg.includes('=');
@@ -43,34 +82,75 @@ if (fileArg) {
43
82
  const cleanKey = key.replace('--', '');
44
83
  const lowerValue = value.toLowerCase();
45
84
  const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
46
- if (cleanKey === 'toText') {
47
- if (boolValue !== undefined)
48
- toText = boolValue;
49
- else
50
- console.warn(`Invalid value for toText: ${value}`);
85
+ const knownBooleans = new Set([
86
+ 'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'putNotesAtLast',
87
+ 'includeRawContent', 'outputErrorToConsole', 'serializeRawContent',
88
+ 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
89
+ ]);
90
+ if (cleanKey === 'format') {
91
+ outputFormat = value;
51
92
  }
52
- else if (cleanKey === 'verbose') {
53
- if (boolValue !== undefined)
54
- verbose = boolValue;
55
- else
56
- console.warn(`Invalid value for verbose: ${value}`);
93
+ else if (cleanKey === 'output') {
94
+ outputFile = value;
57
95
  }
58
96
  else {
59
- // @ts-ignore
60
- if (boolValue !== undefined)
61
- config[cleanKey] = boolValue;
62
- // @ts-ignore
63
- else
97
+ if (boolValue !== undefined) {
98
+ if (cleanKey === 'toText')
99
+ toText = boolValue;
100
+ else if (cleanKey === 'verbose') {
101
+ verbose = boolValue;
102
+ if (verbose)
103
+ config.outputErrorToConsole = true;
104
+ }
105
+ else {
106
+ // @ts-ignore
107
+ config[cleanKey] = boolValue;
108
+ }
109
+ }
110
+ else if (knownBooleans.has(cleanKey)) {
111
+ console.warn(`Invalid boolean value for --${cleanKey}: ${value}. Using default.`);
112
+ }
113
+ else {
114
+ // @ts-ignore
64
115
  config[cleanKey] = value;
116
+ }
65
117
  }
66
118
  });
67
119
  OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
68
120
  .then(async (ast) => {
69
- if (toText) {
70
- process.stdout.write(ast.toText() + '\n');
121
+ let output;
122
+ if (outputFormat) {
123
+ const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat);
124
+ if (Array.isArray(result.value)) {
125
+ output = JSON.stringify(result.value, null, 2);
126
+ }
127
+ else {
128
+ output = result.value;
129
+ }
130
+ }
131
+ else if (toText) {
132
+ output = ast.toText();
133
+ }
134
+ else {
135
+ output = JSON.stringify(ast, null, 2);
136
+ }
137
+ if (outputFile) {
138
+ if (output instanceof Uint8Array) {
139
+ fs.writeFileSync(outputFile, output);
140
+ }
141
+ else {
142
+ fs.writeFileSync(outputFile, output, 'utf8');
143
+ }
144
+ if (verbose)
145
+ console.log(`Output written to ${outputFile}`);
71
146
  }
72
147
  else {
73
- process.stdout.write(JSON.stringify(ast, null, 2) + '\n');
148
+ if (output instanceof Uint8Array) {
149
+ process.stdout.write(output);
150
+ }
151
+ else {
152
+ process.stdout.write(output + '\n');
153
+ }
74
154
  }
75
155
  // Ensure OCR workers are terminated for clean CLI exit
76
156
  if (config.ocr) {
@@ -97,7 +177,9 @@ else {
97
177
  console.log('Usage: officeparser <file> [--option=value]');
98
178
  console.log('');
99
179
  console.log('Options:');
100
- console.log(' --toText=true Output plain text instead of JSON AST');
180
+ console.log(' --format=json|md|html|rtf|csv|text|pdf|chunks Convert to specified format');
181
+ console.log(' --output=file.ext Save output to file instead of stdout');
182
+ console.log(' --toText=true Output plain text instead of JSON AST (legacy)');
101
183
  console.log(' --ocr=true Enable OCR for images');
102
184
  console.log(' --ocrLanguage=eng OCR language (default: eng)');
103
185
  console.log(' --extractAttachments=true Extract embedded attachments');
@@ -106,11 +188,14 @@ else {
106
188
  console.log(' --includeRawContent=true Include raw content in AST');
107
189
  console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
108
190
  console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
191
+ console.log(' --includeBreakNodes=false Include break nodes (DOCX only, default: false)');
109
192
  console.log(' --verbose=true Show full error stack traces');
110
193
  console.log('');
111
194
  console.log('Examples:');
112
195
  console.log(' officeparser document.docx');
113
- console.log(' officeparser document.docx --toText=true');
114
- console.log(' officeparser report.pdf --ocr=true --extractAttachments=true');
196
+ console.log(' officeparser document.docx --format=html --output=doc.html');
197
+ console.log(' officeparser document.docx --format=md');
198
+ console.log(' officeparser report.pdf --ocr=true --format=text');
199
+ console.log(' officeparser data.xlsx --format=csv --output=data.csv');
115
200
  console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
116
201
  }
@@ -0,0 +1,41 @@
1
+ import { DeepRequired, DocumentStructureChunkingConfig, FixedSizeChunkingConfig, FullGeneratorConfig, OfficeParserConfig, SemanticChunkingConfig } from './types.js';
2
+ /**
3
+ * The default regex used for identifying sentence boundaries.
4
+ * When this default is used, the generator employs a high-fidelity "robust"
5
+ * segmenter that accounts for common abbreviations (Mr., Dr., etc.).
6
+ */
7
+ export declare const DEFAULT_SENTENCE_BOUNDARY_REGEX: RegExp;
8
+ /**
9
+ * Common abbreviations that should not trigger a sentence split when followed by a period.
10
+ */
11
+ export declare const DEFAULT_ABBREVIATIONS: string[];
12
+ /**
13
+ * Default configuration for the OfficeParser.
14
+ */
15
+ export declare const DEFAULT_OFFICE_PARSER_CONFIG: DeepRequired<OfficeParserConfig>;
16
+ /**
17
+ * Default configuration for Fixed-Size chunking.
18
+ */
19
+ export declare const DEFAULT_FIXED_SIZE_CHUNKING_CONFIG: Required<Omit<FixedSizeChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
20
+ sentenceBoundaryRegex: string | RegExp;
21
+ abbreviations: string[];
22
+ };
23
+ /**
24
+ * Default configuration for Document-Structure chunking.
25
+ */
26
+ export declare const DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG: Required<Omit<DocumentStructureChunkingConfig, 'sentenceBoundaryRegex' | 'abbreviations'>> & {
27
+ sentenceBoundaryRegex: string | RegExp;
28
+ abbreviations: string[];
29
+ };
30
+ /**
31
+ * Default configuration for Semantic chunking.
32
+ * Note: `embeddingFunction` has no meaningful default and must be provided by the user.
33
+ */
34
+ export declare const DEFAULT_SEMANTIC_CHUNKING_CONFIG: Required<Omit<SemanticChunkingConfig, 'embeddingFunction' | 'sentenceBoundaryRegex' | 'abbreviations'>> & {
35
+ sentenceBoundaryRegex: string | RegExp;
36
+ abbreviations: string[];
37
+ };
38
+ /**
39
+ * Default configuration for the OfficeGenerator.
40
+ */
41
+ export declare const DEFAULT_GENERATOR_CONFIG: FullGeneratorConfig;
@@ -0,0 +1,172 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.DEFAULT_GENERATOR_CONFIG = exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = exports.DEFAULT_OFFICE_PARSER_CONFIG = exports.DEFAULT_ABBREVIATIONS = exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = void 0;
4
+ const PDFJS_VERSION = '5.6.205';
5
+ const DEFAULT_PDF_WORKER_SRC = `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
6
+ /**
7
+ * The default regex used for identifying sentence boundaries.
8
+ * When this default is used, the generator employs a high-fidelity "robust"
9
+ * segmenter that accounts for common abbreviations (Mr., Dr., etc.).
10
+ */
11
+ exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = /[.!?。!?]/;
12
+ /**
13
+ * Common abbreviations that should not trigger a sentence split when followed by a period.
14
+ */
15
+ exports.DEFAULT_ABBREVIATIONS = ['Mr', 'Dr', 'Ms', 'Inc', 'Ltd', 'Prof', 'Sr', 'Jr', 'vs', 'etc'];
16
+ /**
17
+ * Default configuration for OCR.
18
+ */
19
+ const DEFAULT_OCR_CONFIG = {
20
+ language: 'eng',
21
+ workerPath: '',
22
+ corePath: '',
23
+ langPath: '',
24
+ autoTerminateTimeout: 10000,
25
+ };
26
+ /**
27
+ * Default configuration for the OfficeParser.
28
+ */
29
+ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
30
+ outputErrorToConsole: false,
31
+ onWarning: () => { },
32
+ newlineDelimiter: '\n',
33
+ ignoreNotes: false,
34
+ putNotesAtLast: false,
35
+ extractAttachments: false,
36
+ includeRawContent: false,
37
+ ocr: false,
38
+ ocrLanguage: 'eng',
39
+ ocrConfig: DEFAULT_OCR_CONFIG,
40
+ serializeRawContent: true,
41
+ preserveXmlWhitespace: false,
42
+ pdfWorkerSrc: DEFAULT_PDF_WORKER_SRC,
43
+ includeBreakNodes: false,
44
+ ignoreInternalLinks: false,
45
+ fileType: null,
46
+ csvDelimiter: ',',
47
+ };
48
+ /**
49
+ * Default configuration for HTML generation.
50
+ */
51
+ const DEFAULT_HTML_GENERATOR_CONFIG = {
52
+ standalone: true,
53
+ chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
54
+ };
55
+ /**
56
+ * Default configuration for PDF generation.
57
+ */
58
+ const DEFAULT_PDF_GENERATOR_CONFIG = {
59
+ format: 'A4',
60
+ width: '',
61
+ height: '',
62
+ landscape: false,
63
+ printBackground: true,
64
+ scale: 1,
65
+ margin: {
66
+ top: 0,
67
+ right: 0,
68
+ bottom: 0,
69
+ left: 0
70
+ },
71
+ displayHeaderFooter: false,
72
+ headerTemplate: '',
73
+ footerTemplate: '',
74
+ launchOptions: {
75
+ headless: true,
76
+ args: ['--no-sandbox', '--disable-setuid-sandbox']
77
+ },
78
+ };
79
+ /**
80
+ * Default configuration for CSV generation.
81
+ */
82
+ const DEFAULT_CSV_GENERATOR_CONFIG = {
83
+ sheets: '',
84
+ mergeSheets: true,
85
+ columnDelimiter: ',',
86
+ };
87
+ /**
88
+ * Default configuration for Markdown generation.
89
+ */
90
+ const DEFAULT_MD_GENERATOR_CONFIG = {
91
+ fallbackToHtml: true,
92
+ };
93
+ /**
94
+ * Default configuration for plain text generation.
95
+ */
96
+ const DEFAULT_TEXT_GENERATOR_CONFIG = {
97
+ newlineDelimiter: '\n',
98
+ preserveLayout: false,
99
+ };
100
+ /**
101
+ * Default configuration for Fixed-Size chunking.
102
+ */
103
+ exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = {
104
+ strategy: 'fixed-size',
105
+ chunkSize: 1000,
106
+ chunkOverlap: 200,
107
+ separators: ['\n\n', '\n', ' ', ''],
108
+ stripWhitespace: true,
109
+ includeMetadata: true,
110
+ addStartIndex: false,
111
+ lengthFunction: (text) => text.length,
112
+ sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
113
+ abbreviations: exports.DEFAULT_ABBREVIATIONS,
114
+ };
115
+ /**
116
+ * Default configuration for Document-Structure chunking.
117
+ */
118
+ exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = {
119
+ strategy: 'document-structure',
120
+ splitBy: 'paragraph',
121
+ maxChunkSize: 1000,
122
+ tableSplitStrategy: 'row',
123
+ stripWhitespace: true,
124
+ includeMetadata: true,
125
+ addStartIndex: false,
126
+ lengthFunction: (text) => text.length,
127
+ sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
128
+ abbreviations: exports.DEFAULT_ABBREVIATIONS,
129
+ };
130
+ /**
131
+ * Default configuration for Semantic chunking.
132
+ * Note: `embeddingFunction` has no meaningful default and must be provided by the user.
133
+ */
134
+ exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = {
135
+ strategy: 'semantic',
136
+ similarityThreshold: 0.8,
137
+ maxChunkSize: 2000,
138
+ bufferSize: 1,
139
+ embeddingBatchSize: 50,
140
+ stripWhitespace: true,
141
+ includeMetadata: true,
142
+ addStartIndex: false,
143
+ lengthFunction: (text) => text.length,
144
+ sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
145
+ abbreviations: exports.DEFAULT_ABBREVIATIONS,
146
+ };
147
+ /**
148
+ * The resolved default chunking config (uses document-structure as default strategy).
149
+ */
150
+ const DEFAULT_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG;
151
+ /**
152
+ * Default configuration for the OfficeGenerator.
153
+ */
154
+ exports.DEFAULT_GENERATOR_CONFIG = {
155
+ onNode: () => { },
156
+ onWarning: () => { },
157
+ styleMap: [],
158
+ includeFormatting: true,
159
+ generateIds: true,
160
+ renderMetadata: false,
161
+ ignoreDefaultStyleMap: false,
162
+ includeImages: true,
163
+ includeCharts: true,
164
+ ignoreInternalLinks: false,
165
+ htmlConfig: DEFAULT_HTML_GENERATOR_CONFIG,
166
+ mdConfig: DEFAULT_MD_GENERATOR_CONFIG,
167
+ pdfConfig: DEFAULT_PDF_GENERATOR_CONFIG,
168
+ csvConfig: DEFAULT_CSV_GENERATOR_CONFIG,
169
+ textConfig: DEFAULT_TEXT_GENERATOR_CONFIG,
170
+ rtfConfig: {},
171
+ chunksConfig: DEFAULT_CHUNKING_CONFIG,
172
+ };