officeparser 7.1.0 → 7.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +152 -56
  2. package/dist/OfficeGenerator.d.ts +6 -2
  3. package/dist/OfficeGenerator.js +30 -9
  4. package/dist/OfficeParser.d.ts +1 -1
  5. package/dist/OfficeParser.js +1 -1
  6. package/dist/cli.d.ts +18 -12
  7. package/dist/cli.js +255 -81
  8. package/dist/defaults.js +12 -1
  9. package/dist/generators/BaseGenerator.d.ts +4 -3
  10. package/dist/generators/BaseGenerator.js +13 -1
  11. package/dist/generators/ChunkingGenerator.js +32 -5
  12. package/dist/generators/CsvGenerator.d.ts +1 -1
  13. package/dist/generators/HtmlGenerator.d.ts +2 -1
  14. package/dist/generators/HtmlGenerator.js +481 -42
  15. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  16. package/dist/generators/MarkdownGenerator.js +35 -2
  17. package/dist/generators/PdfGenerator.d.ts +1 -1
  18. package/dist/generators/PdfGenerator.js +0 -6
  19. package/dist/generators/RtfGenerator.d.ts +2 -1
  20. package/dist/generators/RtfGenerator.js +49 -6
  21. package/dist/generators/TextGenerator.d.ts +1 -1
  22. package/dist/generators/TextGenerator.js +6 -0
  23. package/dist/officeparser.browser.d.ts +267 -54
  24. package/dist/officeparser.browser.iife.js +599 -187
  25. package/dist/officeparser.browser.mjs +599 -187
  26. package/dist/parsers/CsvParser.js +1 -1
  27. package/dist/parsers/ExcelParser.js +63 -19
  28. package/dist/parsers/HtmlParser.js +10 -1
  29. package/dist/parsers/MarkdownParser.js +13 -10
  30. package/dist/parsers/OpenOfficeParser.js +57 -34
  31. package/dist/parsers/PdfParser.js +28 -3
  32. package/dist/parsers/PowerPointParser.js +164 -40
  33. package/dist/parsers/RtfParser.js +28 -24
  34. package/dist/parsers/WordParser.js +154 -11
  35. package/dist/sbom.cdx.json +100 -100
  36. package/dist/types.d.ts +268 -53
  37. package/dist/types.js +4 -0
  38. package/dist/utils/astUtils.d.ts +2 -2
  39. package/dist/utils/astUtils.js +2 -1
  40. package/dist/utils/configUtils.d.ts +5 -0
  41. package/dist/utils/configUtils.js +55 -1
  42. package/dist/utils/errorUtils.js +3 -1
  43. package/dist/utils/moduleLoader.js +55 -11
  44. package/dist/utils/xmlUtils.d.ts +9 -0
  45. package/dist/utils/xmlUtils.js +53 -1
  46. package/package.json +6 -3
package/dist/cli.js CHANGED
@@ -5,20 +5,26 @@
5
5
  *
6
6
  * Allows running officeparser from the command line:
7
7
  * npx officeparser file.docx
8
- * officeparser file.docx --toText=true
9
- * officeparser file.docx --ocr=true --extractAttachments=true
8
+ * officeparser file.docx --to=text
9
+ * officeparser file.docx --ocr --extractAttachments
10
10
  *
11
- * Options (--key=value):
12
- * --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
11
+ * Options (--key=value, --key value, or bare flags):
12
+ * --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
13
13
  * --output=path Save result to a file
14
- * --toText=true Legacy flag for plain text output
15
- * --ocr=true Enable OCR for images
16
- * --ocrLanguage=eng OCR language (default: eng)
17
- * --extractAttachments=true Extract embedded attachments
18
- * --ignoreNotes=true Ignore footnotes/endnotes
19
- * --putNotesAtLast=true Move notes to end of document
20
- * --includeRawContent=true Include raw content in AST
21
- * --outputErrorToConsole=true Log errors to console
14
+ * --fileType=docx|xlsx|... Override file type detection
15
+ * --ocr Enable OCR for images (default: false)
16
+ * --ocrConfig.language=eng OCR language (default: eng)
17
+ * --extractAttachments Extract embedded attachments (default: false)
18
+ * --ignoreNotes Ignore footnotes/endnotes/speaker notes (default: false)
19
+ * --ignoreComments Ignore inline comments (default: false)
20
+ * --ignoreHeadersAndFooters Ignore headers and footers (default: false)
21
+ * --ignoreSlideMasters Ignore slide masters (default: false)
22
+ * --ignoreInternalLinks Ignore internal links (default: false)
23
+ * --includeRawContent Include raw content in AST (default: false)
24
+ * --serializeRawContent Include stringified XML in metadata (default: true)
25
+ * --preserveXmlWhitespace Keep raw formatting space (default: false)
26
+ * --includeBreakNodes Include break nodes (DOCX only, default: false)
27
+ * --verbose Show full error stack traces and warning logs
22
28
  */
23
29
  var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
24
30
  if (k2 === undefined) k2 = k;
@@ -59,68 +65,219 @@ const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
59
65
  const fs = __importStar(require("fs"));
60
66
  const args = process.argv.slice(2);
61
67
  let fileArg;
62
- let toText = false;
68
+ let showHelp = false;
69
+ let toFlagOption;
70
+ let formatFlagOption;
71
+ let toTextOption;
63
72
  let verbose = false;
64
- let outputFormat;
65
73
  let outputFile;
66
- const configArgs = [];
67
- function isConfigOption(arg) {
68
- return arg.startsWith('--') && arg.includes('=');
69
- }
70
- args.forEach(arg => {
71
- if (isConfigOption(arg)) {
72
- configArgs.push(arg);
73
- }
74
- else if (!fileArg) {
75
- fileArg = arg;
74
+ // Parser and Generator configuration objects that will be populated by command line options.
75
+ const config = {};
76
+ const generatorConfig = {};
77
+ // Known boolean configurations to validate user input against.
78
+ const knownParserBooleans = new Set([
79
+ 'ocr', 'extractAttachments', 'ignoreNotes', 'ignoreComments',
80
+ 'ignoreHeadersAndFooters', 'ignoreSlideMasters', 'ignoreInternalLinks',
81
+ 'includeRawContent', 'serializeRawContent', 'preserveXmlWhitespace', 'includeBreakNodes'
82
+ ]);
83
+ const knownGeneratorBooleans = new Set([
84
+ 'includeFormatting', 'generateIds', 'renderMetadata', 'includeImages', 'includeCharts', 'ignoreInternalLinks'
85
+ ]);
86
+ // Prefixes used to identify configurations targeted for the generator instead of the parser.
87
+ const generatorPrefixes = [
88
+ 'generatorConfig.', 'htmlConfig.', 'csvConfig.', 'textConfig.', 'mdConfig.', 'pdfConfig.', 'rtfConfig.', 'chunksConfig.'
89
+ ];
90
+ // Trackers to detect if deprecated/legacy options were used to log helpful warnings.
91
+ let usedFormat = false;
92
+ let usedToText = false;
93
+ let usedOcrLanguage = false;
94
+ let usedPutNotesAtLast = false;
95
+ let usedOutputErrorToConsole = false;
96
+ // Parse the arguments list
97
+ for (let i = 0; i < args.length; i++) {
98
+ const arg = args[i];
99
+ // Help flags trigger immediate termination of parsing and print help
100
+ if (arg === '-h' || arg === '--help') {
101
+ showHelp = true;
102
+ break;
76
103
  }
77
- });
78
- if (fileArg) {
79
- const config = {};
80
- configArgs.forEach(arg => {
81
- const [key, value] = arg.split('=');
82
- const cleanKey = key.replace('--', '');
83
- const lowerValue = value.toLowerCase();
104
+ if (arg.startsWith('--')) {
105
+ let cleanKey;
106
+ let val;
107
+ // Support --key=value syntax
108
+ if (arg.includes('=')) {
109
+ const idx = arg.indexOf('=');
110
+ cleanKey = arg.slice(2, idx);
111
+ val = arg.slice(idx + 1);
112
+ }
113
+ // Support negation shorthand: --no-ocr sets ocr to false
114
+ else if (arg.startsWith('--no-')) {
115
+ cleanKey = arg.slice(5);
116
+ val = 'false';
117
+ }
118
+ // Support space-separated options or bare flags
119
+ else {
120
+ cleanKey = arg.slice(2);
121
+ const isKnownBoolean = knownParserBooleans.has(cleanKey) ||
122
+ knownGeneratorBooleans.has(cleanKey) ||
123
+ cleanKey === 'verbose' ||
124
+ cleanKey === 'toText' ||
125
+ cleanKey === 'outputErrorToConsole';
126
+ const isNextBool = i + 1 < args.length &&
127
+ (args[i + 1].toLowerCase() === 'true' || args[i + 1].toLowerCase() === 'false');
128
+ // If the next arg is not another option flag, treat it as the value (e.g., --to html)
129
+ // But if this is a known boolean, only consume the next argument if it is a valid boolean string.
130
+ if (i + 1 < args.length && !args[i + 1].startsWith('-') && (!isKnownBoolean || isNextBool)) {
131
+ val = args[i + 1];
132
+ i++;
133
+ }
134
+ // Bare presence implies true (e.g., --ocr is true)
135
+ else {
136
+ val = 'true';
137
+ }
138
+ }
139
+ // Parse boolean strings to raw boolean types
140
+ const lowerValue = val.toLowerCase();
84
141
  const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
85
- const knownBooleans = new Set([
86
- 'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'putNotesAtLast',
87
- 'includeRawContent', 'outputErrorToConsole', 'serializeRawContent',
88
- 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
89
- ]);
90
- if (cleanKey === 'format') {
91
- outputFormat = value;
142
+ // Map core CLI options to variables
143
+ if (cleanKey === 'to') {
144
+ toFlagOption = val;
145
+ }
146
+ else if (cleanKey === 'format') {
147
+ formatFlagOption = val;
148
+ usedFormat = true;
92
149
  }
93
150
  else if (cleanKey === 'output') {
94
- outputFile = value;
151
+ outputFile = val;
152
+ }
153
+ else if (cleanKey === 'toText') {
154
+ toTextOption = boolValue !== undefined ? boolValue : true;
155
+ usedToText = true;
156
+ }
157
+ else if (cleanKey === 'verbose') {
158
+ verbose = boolValue !== undefined ? boolValue : true;
159
+ }
160
+ else if (cleanKey === 'ocrLanguage') {
161
+ config.ocrLanguage = val;
162
+ usedOcrLanguage = true;
163
+ }
164
+ else if (cleanKey === 'putNotesAtLast') {
165
+ config.putNotesAtLast = boolValue !== undefined ? boolValue : true;
166
+ usedPutNotesAtLast = true;
167
+ }
168
+ else if (cleanKey === 'outputErrorToConsole') {
169
+ verbose = boolValue !== undefined ? boolValue : true;
170
+ usedOutputErrorToConsole = true;
95
171
  }
96
172
  else {
97
- if (boolValue !== undefined) {
98
- if (cleanKey === 'toText')
99
- toText = boolValue;
100
- else if (cleanKey === 'verbose') {
101
- verbose = boolValue;
102
- if (verbose)
103
- config.outputErrorToConsole = true;
173
+ // Check if the flag belongs to generatorConfig or a specific sub-generator (e.g., htmlConfig)
174
+ const isGeneratorOption = knownGeneratorBooleans.has(cleanKey) || generatorPrefixes.some(pref => cleanKey.startsWith(pref));
175
+ const target = isGeneratorOption ? generatorConfig : config;
176
+ let path = cleanKey;
177
+ // Strip generatorConfig prefix to flatten it onto the generatorConfig object
178
+ if (isGeneratorOption && cleanKey.startsWith('generatorConfig.')) {
179
+ path = cleanKey.slice('generatorConfig.'.length);
180
+ }
181
+ // Support nested dot-notation parsing (e.g., --ocrConfig.language=fra)
182
+ if (path.includes('.')) {
183
+ const parts = path.split('.');
184
+ let current = target;
185
+ for (let j = 0; j < parts.length - 1; j++) {
186
+ const part = parts[j];
187
+ if (!current[part])
188
+ current[part] = {};
189
+ current = current[part];
190
+ }
191
+ const lastPart = parts[parts.length - 1];
192
+ current[lastPart] = boolValue !== undefined ? boolValue : val;
193
+ }
194
+ // Flat key assignment
195
+ else {
196
+ if (boolValue !== undefined) {
197
+ target[path] = boolValue;
198
+ }
199
+ else if (knownParserBooleans.has(path) || (isGeneratorOption && knownGeneratorBooleans.has(path))) {
200
+ console.warn(`Invalid boolean value for --${cleanKey}: ${val}. Using default.`);
104
201
  }
105
202
  else {
106
- // @ts-ignore
107
- config[cleanKey] = boolValue;
203
+ target[path] = val;
108
204
  }
109
205
  }
110
- else if (knownBooleans.has(cleanKey)) {
111
- console.warn(`Invalid boolean value for --${cleanKey}: ${value}. Using default.`);
112
- }
113
- else {
114
- // @ts-ignore
115
- config[cleanKey] = value;
206
+ }
207
+ }
208
+ else {
209
+ // First positional argument that is not a flag is treated as the input file path
210
+ if (!fileArg) {
211
+ fileArg = arg;
212
+ }
213
+ }
214
+ }
215
+ if (fileArg && !showHelp) {
216
+ // Resolve output format prioritizing: --to > --format > --toText
217
+ let outputFormat;
218
+ if (toFlagOption) {
219
+ outputFormat = toFlagOption;
220
+ }
221
+ else if (formatFlagOption) {
222
+ outputFormat = formatFlagOption;
223
+ }
224
+ else if (toTextOption === true) {
225
+ outputFormat = 'text';
226
+ }
227
+ // Display warning messages for any deprecated CLI options used
228
+ if (usedFormat) {
229
+ console.warn('Warning: --format is deprecated. Use --to instead.');
230
+ }
231
+ if (usedToText) {
232
+ console.warn('Warning: --toText is deprecated. Use --to=text instead.');
233
+ }
234
+ if (usedOcrLanguage) {
235
+ console.warn('Warning: --ocrLanguage is deprecated. Use --ocrConfig.language instead.');
236
+ }
237
+ if (usedPutNotesAtLast) {
238
+ console.warn('Warning: --putNotesAtLast is deprecated and will be ignored by all parsers.');
239
+ }
240
+ if (usedOutputErrorToConsole) {
241
+ console.warn('Warning: --outputErrorToConsole is deprecated. Use --verbose instead.');
242
+ }
243
+ // Intercept parser warning callbacks to format and print issues when verbose is enabled
244
+ const originalOnWarning = config.onWarning;
245
+ config.onWarning = (issue) => {
246
+ if (verbose) {
247
+ const severity = issue.type === 'error' ? 'Error' : 'Warning';
248
+ console.error(`[OfficeParser ${severity}] [${issue.code}]: ${issue.message}`);
249
+ if (issue.details) {
250
+ console.error(issue.details);
116
251
  }
117
252
  }
118
- });
253
+ if (originalOnWarning)
254
+ originalOnWarning(issue);
255
+ };
256
+ // Propagate newlineDelimiter and csvDelimiter if configured flatly but not in generator sub-configs
257
+ if (config.newlineDelimiter !== undefined) {
258
+ if (!generatorConfig.textConfig)
259
+ generatorConfig.textConfig = {};
260
+ if (generatorConfig.textConfig.newlineDelimiter === undefined) {
261
+ generatorConfig.textConfig.newlineDelimiter = config.newlineDelimiter;
262
+ }
263
+ }
264
+ if (config.csvDelimiter !== undefined) {
265
+ if (!generatorConfig.csvConfig)
266
+ generatorConfig.csvConfig = {};
267
+ if (generatorConfig.csvConfig.columnDelimiter === undefined) {
268
+ generatorConfig.csvConfig.columnDelimiter = config.csvDelimiter;
269
+ }
270
+ }
271
+ // Run the main parser
119
272
  OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
120
273
  .then(async (ast) => {
121
274
  let output;
122
- if (outputFormat) {
123
- const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat);
275
+ // Generate JSON output or convert AST using OfficeGenerator
276
+ if (outputFormat === 'json') {
277
+ output = JSON.stringify(ast, null, 2);
278
+ }
279
+ else if (outputFormat) {
280
+ const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat, generatorConfig);
124
281
  if (Array.isArray(result.value)) {
125
282
  output = JSON.stringify(result.value, null, 2);
126
283
  }
@@ -128,12 +285,10 @@ if (fileArg) {
128
285
  output = result.value;
129
286
  }
130
287
  }
131
- else if (toText) {
132
- output = ast.toText();
133
- }
134
288
  else {
135
289
  output = JSON.stringify(ast, null, 2);
136
290
  }
291
+ // Write generated output to output file or print to standard output
137
292
  if (outputFile) {
138
293
  if (output instanceof Uint8Array) {
139
294
  fs.writeFileSync(outputFile, output);
@@ -158,15 +313,16 @@ if (fileArg) {
158
313
  }
159
314
  })
160
315
  .catch(async (err) => {
316
+ // Handle and display parsing error messages
161
317
  console.error(`Error parsing file "${fileArg}":`);
162
318
  if (verbose) {
163
319
  console.error(err);
164
320
  }
165
321
  else {
166
322
  console.error(err.message || err);
167
- console.error('Use --verbose=true for full stack trace.');
323
+ console.error('Use --verbose for full stack trace.');
168
324
  }
169
- // Ensure OCR workers are terminated even on error
325
+ // Ensure OCR workers are terminated even on error to prevent process hang
170
326
  if (config.ocr) {
171
327
  await OfficeParser_js_1.OfficeParser.terminateOcr();
172
328
  }
@@ -174,28 +330,46 @@ if (fileArg) {
174
330
  });
175
331
  }
176
332
  else {
177
- console.log('Usage: officeparser <file> [--option=value]');
333
+ console.log('Usage: officeparser <file> [options]');
178
334
  console.log('');
179
335
  console.log('Options:');
180
- console.log(' --format=json|md|html|rtf|csv|text|pdf|chunks Convert to specified format');
181
- console.log(' --output=file.ext Save output to file instead of stdout');
182
- console.log(' --toText=true Output plain text instead of JSON AST (legacy)');
183
- console.log(' --ocr=true Enable OCR for images');
184
- console.log(' --ocrLanguage=eng OCR language (default: eng)');
185
- console.log(' --extractAttachments=true Extract embedded attachments');
186
- console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
187
- console.log(' --putNotesAtLast=true Move notes to end of document');
188
- console.log(' --includeRawContent=true Include raw content in AST');
189
- console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
190
- console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
191
- console.log(' --includeBreakNodes=false Include break nodes (DOCX only, default: false)');
192
- console.log(' --verbose=true Show full error stack traces');
336
+ console.log(' --to=json|text|md|html|pdf|csv|rtf|chunks Target conversion format (default: json)');
337
+ console.log(' --output=file.ext Save output to file instead of stdout');
338
+ console.log(' --fileType=docx|xlsx|pptx|odt|... Explicitly override input file type detection');
339
+ console.log(' --ocr Enable OCR for images (default: false)');
340
+ console.log(' --ocrConfig.language=eng OCR language (default: eng)');
341
+ console.log(' --extractAttachments Extract embedded attachments (default: false)');
342
+ console.log(' --ignoreNotes Ignore footnotes/endnotes/speaker notes (default: false)');
343
+ console.log(' --ignoreComments Ignore inline comments (default: false)');
344
+ console.log(' --ignoreHeadersAndFooters Ignore headers and footers (default: false)');
345
+ console.log(' --ignoreSlideMasters Ignore slide masters (default: false)');
346
+ console.log(' --ignoreInternalLinks Ignore internal links (default: false)');
347
+ console.log(' --includeRawContent Include raw content in AST (default: false)');
348
+ console.log(' --serializeRawContent Serialize raw XML content (default: true)');
349
+ console.log(' --preserveXmlWhitespace Keep raw formatting space (default: false)');
350
+ console.log(' --includeBreakNodes Include break nodes (DOCX only, default: false)');
351
+ console.log(' --verbose Show full error stack traces and warning logs');
352
+ console.log(' --newlineDelimiter=string Delimiter string between blocks/lines (default: \\n)');
353
+ console.log(' --csvDelimiter=char Custom CSV delimiter (default: ,)');
354
+ console.log('');
355
+ console.log('High-Value Generator Options:');
356
+ console.log(' --includeFormatting Include font formatting like bold/italic (default: true)');
357
+ console.log(' --renderMetadata Render metadata in output content (default: false)');
358
+ console.log(' --htmlConfig.containerWidth=value HTML container width (auto | px | % | vw etc., default: auto)');
359
+ console.log('');
360
+ console.log('Advanced Nested Config Examples:');
361
+ console.log(' --pdfConfig.format=Letter Configure Puppeteer PDF format (A4 | Letter | Legal etc.)');
362
+ console.log(' --chunksConfig.strategy=fixed-size Chunking strategy (fixed-size | document-structure | semantic)');
363
+ console.log('');
364
+ console.log('Format Syntax:');
365
+ console.log(' Flags can be written as --flag (presence implies true), --no-flag (negation),');
366
+ console.log(' --flag=value, or --flag value.');
193
367
  console.log('');
194
368
  console.log('Examples:');
195
369
  console.log(' officeparser document.docx');
196
- console.log(' officeparser document.docx --format=html --output=doc.html');
197
- console.log(' officeparser document.docx --format=md');
198
- console.log(' officeparser report.pdf --ocr=true --format=text');
199
- console.log(' officeparser data.xlsx --format=csv --output=data.csv');
200
- console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
370
+ console.log(' officeparser document.docx --to html --output doc.html');
371
+ console.log(' officeparser document.docx --to md');
372
+ console.log(' officeparser report.pdf --ocr --ocrConfig.language eng --to text');
373
+ console.log(' officeparser data.xlsx --to csv --output data.csv --csvDelimiter ";"');
374
+ console.log(' officeparser image_doc --fileType docx --to json');
201
375
  }
package/dist/defaults.js CHANGED
@@ -42,6 +42,9 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
42
42
  onWarning: () => { },
43
43
  newlineDelimiter: '\n',
44
44
  ignoreNotes: false,
45
+ ignoreComments: false,
46
+ ignoreHeadersAndFooters: false,
47
+ ignoreSlideMasters: false,
45
48
  putNotesAtLast: false,
46
49
  extractAttachments: false,
47
50
  includeRawContent: false,
@@ -63,6 +66,14 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
63
66
  const DEFAULT_HTML_GENERATOR_CONFIG = {
64
67
  standalone: true,
65
68
  chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
69
+ containerWidth: 'auto',
70
+ customCss: '',
71
+ injections: {
72
+ headStart: '',
73
+ headEnd: '',
74
+ bodyStart: '',
75
+ bodyEnd: '',
76
+ }
66
77
  };
67
78
  /**
68
79
  * Default configuration for PDF generation.
@@ -108,7 +119,7 @@ const DEFAULT_MD_GENERATOR_CONFIG = {
108
119
  */
109
120
  const DEFAULT_TEXT_GENERATOR_CONFIG = {
110
121
  newlineDelimiter: '\n',
111
- preserveLayout: false,
122
+ preserveLayout: true,
112
123
  };
113
124
  /**
114
125
  * Default configuration for Fixed-Size chunking.
@@ -1,15 +1,16 @@
1
- import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType } from '../types.js';
1
+ import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType, UniversalGeneratorFormat } from '../types.js';
2
2
  import { StyleMapper } from '../utils/styleMapper.js';
3
3
  /**
4
4
  * Base class for all document generators.
5
5
  * Provides common traversal logic and configuration handling.
6
6
  */
7
- export declare abstract class BaseGenerator<D extends string = string> {
7
+ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat = UniversalGeneratorFormat> {
8
8
  protected destination: D;
9
9
  protected config: FullGeneratorConfig;
10
10
  protected ast: OfficeParserAST;
11
11
  protected messages: OfficeIssue[];
12
12
  protected styleMapper: StyleMapper;
13
+ protected collectedNotes: OfficeContentNode[];
13
14
  constructor(destination: D, ast: OfficeParserAST, config?: GeneratorConfig<D> | FullGeneratorConfig);
14
15
  /**
15
16
  * Retrieves the semantic mapping for a node, respecting the includeFormatting flag.
@@ -24,7 +25,7 @@ export declare abstract class BaseGenerator<D extends string = string> {
24
25
  /**
25
26
  * Entry point for generation.
26
27
  */
27
- abstract generate(): Promise<ConversionResult>;
28
+ abstract generate(): Promise<ConversionResult<D>>;
28
29
  /**
29
30
  * Centralized logic for handling the onNode callback.
30
31
  * Evaluates the callback and returns a result that tells the generator how to proceed.
@@ -14,6 +14,7 @@ class BaseGenerator {
14
14
  ast;
15
15
  messages = [];
16
16
  styleMapper;
17
+ collectedNotes = [];
17
18
  constructor(destination, ast, config) {
18
19
  this.destination = destination;
19
20
  this.config = (0, configUtils_js_1.resolveGeneratorConfig)(destination, ast.config, config);
@@ -65,7 +66,18 @@ class BaseGenerator {
65
66
  childrenOutput += await this.processNodeRecursive(child, processor);
66
67
  }
67
68
  }
68
- return await processor(node, childrenOutput);
69
+ if (node.notes && node.notes.length > 0) {
70
+ if (node.type !== 'slide') {
71
+ this.collectedNotes.push(...node.notes);
72
+ }
73
+ }
74
+ let result = await processor(node, childrenOutput);
75
+ if (node.type === 'slide' && node.notes && node.notes.length > 0) {
76
+ for (const note of node.notes) {
77
+ result += await this.processNodeRecursive(note, processor);
78
+ }
79
+ }
80
+ return result;
69
81
  }
70
82
  /**
71
83
  * Helper to generate a unique ID (slug) from text.
@@ -214,10 +214,17 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
214
214
  || splitBy === 'sheet' && node.type === 'sheet';
215
215
  if (isForcedSplit) {
216
216
  // Process children of the container as individual chunks within the boundary
217
- if (node.children) {
217
+ if (node.children || node.notes) {
218
218
  const innerChunks = [];
219
- for (const child of node.children) {
220
- await this.processNodeForStructure(child, config, 'paragraph', maxChunkSize, measure, innerChunks, contextStack);
219
+ if (node.children) {
220
+ for (const child of node.children) {
221
+ await this.processNodeForStructure(child, config, 'paragraph', maxChunkSize, measure, innerChunks, contextStack);
222
+ }
223
+ }
224
+ if (node.notes) {
225
+ for (const note of node.notes) {
226
+ await this.processNodeForStructure(note, config, 'paragraph', maxChunkSize, measure, innerChunks, contextStack);
227
+ }
221
228
  }
222
229
  for (const ic of innerChunks) {
223
230
  ic.metadata.slideNumber = contextStack.slideNumber;
@@ -268,12 +275,17 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
268
275
  }
269
276
  return;
270
277
  }
271
- // Recurse into children for container nodes
278
+ // Recurse into children and notes for container nodes
272
279
  if (node.children) {
273
280
  for (const child of node.children) {
274
281
  await this.processNodeForStructure(child, config, splitBy, maxChunkSize, measure, chunks, contextStack);
275
282
  }
276
283
  }
284
+ if (node.notes) {
285
+ for (const note of node.notes) {
286
+ await this.processNodeForStructure(note, config, splitBy, maxChunkSize, measure, chunks, contextStack);
287
+ }
288
+ }
277
289
  }
278
290
  isStructuralBoundary(node, splitBy) {
279
291
  if (splitBy === 'heading')
@@ -393,7 +405,14 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
393
405
  renderedRows.push(row.text ?? '');
394
406
  continue;
395
407
  }
396
- const cells = row.children.map(cell => (cell.text ?? '').replace(/\n/g, ' ').trim());
408
+ const getCellText = (cell) => {
409
+ if (cell.text)
410
+ return cell.text;
411
+ if (!cell.children || cell.children.length === 0)
412
+ return '';
413
+ return cell.children.map(c => getCellText(c)).join(' ');
414
+ };
415
+ const cells = row.children.map(cell => getCellText(cell).replace(/\n/g, ' ').trim());
397
416
  renderedRows.push(`| ${cells.join(' | ')} |`);
398
417
  }
399
418
  return renderedRows.join('\n');
@@ -517,6 +536,10 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
517
536
  for (const child of node.children)
518
537
  await walk(child);
519
538
  }
539
+ if (node.notes) {
540
+ for (const note of node.notes)
541
+ await walk(note);
542
+ }
520
543
  };
521
544
  for (const node of this.ast.content)
522
545
  await walk(node);
@@ -561,6 +584,10 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
561
584
  for (const child of node.children)
562
585
  await walk(child);
563
586
  }
587
+ if (node.notes) {
588
+ for (const note of node.notes)
589
+ await walk(note);
590
+ }
564
591
  };
565
592
  for (const node of this.ast.content)
566
593
  await walk(node);
@@ -10,7 +10,7 @@ export declare class CsvGenerator extends BaseGenerator<'csv'> {
10
10
  *
11
11
  * @returns A CSV string or a ZIP archive containing multiple CSVs
12
12
  */
13
- generate(): Promise<ConversionResult>;
13
+ generate(): Promise<ConversionResult<'csv'>>;
14
14
  /**
15
15
  * Recursively finds all nodes that can be treated as sheets (sheet or table).
16
16
  */
@@ -12,7 +12,7 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
12
12
  *
13
13
  * @returns An HTML string
14
14
  */
15
- generate(): Promise<ConversionResult>;
15
+ generate(): Promise<ConversionResult<'html'>>;
16
16
  private renderMetaTags;
17
17
  private renderMetadataSummary;
18
18
  /**
@@ -33,5 +33,6 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
33
33
  private getInlineStyles;
34
34
  private getPremiumStyles;
35
35
  protected slugify(text: string): string;
36
+ private getColumnLetter;
36
37
  private escape;
37
38
  }