officeparser 7.2.0 → 7.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -5,24 +5,26 @@
5
5
  *
6
6
  * Allows running officeparser from the command line:
7
7
  * npx officeparser file.docx
8
- * officeparser file.docx --toText=true
9
- * officeparser file.docx --ocr=true --extractAttachments=true
8
+ * officeparser file.docx --to=text
9
+ * officeparser file.docx --ocr --extractAttachments
10
10
  *
11
- * Options (--key=value):
12
- * --format=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format
11
+ * Options (--key=value, --key value, or bare flags):
12
+ * --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
13
13
  * --output=path Save result to a file
14
- * --toText=true Legacy flag for plain text output
15
- * --ocr=true Enable OCR for images
16
- * --ocrLanguage=eng OCR language (default: eng)
17
- * --extractAttachments=true Extract embedded attachments
18
- * --ignoreNotes=true Ignore footnotes/endnotes
19
- * --ignoreComments=true Ignore inline comments
20
- * --ignoreHeadersAndFooters=true Ignore headers and footers
21
- * --ignoreSlideMasters=true Ignore slide masters
22
- * --ignoreInternalLinks=true Ignore internal links
23
- * --putNotesAtLast=true Move notes to end of document
24
- * --includeRawContent=true Include raw content in AST
25
- * --outputErrorToConsole=true Log errors to console
14
+ * --fileType=docx|xlsx|... Override file type detection
15
+ * --ocr Enable OCR for images (default: false)
16
+ * --ocrConfig.language=eng OCR language (default: eng)
17
+ * --extractAttachments Extract embedded attachments (default: false)
18
+ * --ignoreNotes Ignore footnotes/endnotes/speaker notes (default: false)
19
+ * --ignoreComments Ignore inline comments (default: false)
20
+ * --ignoreHeadersAndFooters Ignore headers and footers (default: false)
21
+ * --ignoreSlideMasters Ignore slide masters (default: false)
22
+ * --ignoreInternalLinks Ignore internal links (default: false)
23
+ * --includeRawContent Include raw content in AST (default: false)
24
+ * --serializeRawContent Include stringified XML in metadata (default: true)
25
+ * --preserveXmlWhitespace Keep raw formatting space (default: false)
26
+ * --includeBreakNodes Include break nodes (DOCX only, default: false)
27
+ * --verbose Show full error stack traces and warning logs
26
28
  */
27
29
  var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
28
30
  if (k2 === undefined) k2 = k;
@@ -63,69 +65,219 @@ const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
63
65
  const fs = __importStar(require("fs"));
64
66
  const args = process.argv.slice(2);
65
67
  let fileArg;
66
- let toText = false;
68
+ let showHelp = false;
69
+ let toFlagOption;
70
+ let formatFlagOption;
71
+ let toTextOption;
67
72
  let verbose = false;
68
- let outputFormat;
69
73
  let outputFile;
70
- const configArgs = [];
71
- function isConfigOption(arg) {
72
- return arg.startsWith('--') && arg.includes('=');
73
- }
74
- args.forEach(arg => {
75
- if (isConfigOption(arg)) {
76
- configArgs.push(arg);
77
- }
78
- else if (!fileArg) {
79
- fileArg = arg;
74
+ // Parser and Generator configuration objects that will be populated by command line options.
75
+ const config = {};
76
+ const generatorConfig = {};
77
+ // Known boolean configurations to validate user input against.
78
+ const knownParserBooleans = new Set([
79
+ 'ocr', 'extractAttachments', 'ignoreNotes', 'ignoreComments',
80
+ 'ignoreHeadersAndFooters', 'ignoreSlideMasters', 'ignoreInternalLinks',
81
+ 'includeRawContent', 'serializeRawContent', 'preserveXmlWhitespace', 'includeBreakNodes'
82
+ ]);
83
+ const knownGeneratorBooleans = new Set([
84
+ 'includeFormatting', 'generateIds', 'renderMetadata', 'includeImages', 'includeCharts', 'ignoreInternalLinks'
85
+ ]);
86
+ // Prefixes used to identify configurations targeted for the generator instead of the parser.
87
+ const generatorPrefixes = [
88
+ 'generatorConfig.', 'htmlConfig.', 'csvConfig.', 'textConfig.', 'mdConfig.', 'pdfConfig.', 'rtfConfig.', 'chunksConfig.'
89
+ ];
90
+ // Trackers to detect if deprecated/legacy options were used to log helpful warnings.
91
+ let usedFormat = false;
92
+ let usedToText = false;
93
+ let usedOcrLanguage = false;
94
+ let usedPutNotesAtLast = false;
95
+ let usedOutputErrorToConsole = false;
96
+ // Parse the arguments list
97
+ for (let i = 0; i < args.length; i++) {
98
+ const arg = args[i];
99
+ // Help flags trigger immediate termination of parsing and print help
100
+ if (arg === '-h' || arg === '--help') {
101
+ showHelp = true;
102
+ break;
80
103
  }
81
- });
82
- if (fileArg) {
83
- const config = {};
84
- configArgs.forEach(arg => {
85
- const [key, value] = arg.split('=');
86
- const cleanKey = key.replace('--', '');
87
- const lowerValue = value.toLowerCase();
104
+ if (arg.startsWith('--')) {
105
+ let cleanKey;
106
+ let val;
107
+ // Support --key=value syntax
108
+ if (arg.includes('=')) {
109
+ const idx = arg.indexOf('=');
110
+ cleanKey = arg.slice(2, idx);
111
+ val = arg.slice(idx + 1);
112
+ }
113
+ // Support negation shorthand: --no-ocr sets ocr to false
114
+ else if (arg.startsWith('--no-')) {
115
+ cleanKey = arg.slice(5);
116
+ val = 'false';
117
+ }
118
+ // Support space-separated options or bare flags
119
+ else {
120
+ cleanKey = arg.slice(2);
121
+ const isKnownBoolean = knownParserBooleans.has(cleanKey) ||
122
+ knownGeneratorBooleans.has(cleanKey) ||
123
+ cleanKey === 'verbose' ||
124
+ cleanKey === 'toText' ||
125
+ cleanKey === 'outputErrorToConsole';
126
+ const isNextBool = i + 1 < args.length &&
127
+ (args[i + 1].toLowerCase() === 'true' || args[i + 1].toLowerCase() === 'false');
128
+ // If the next arg is not another option flag, treat it as the value (e.g., --to html)
129
+ // But if this is a known boolean, only consume the next argument if it is a valid boolean string.
130
+ if (i + 1 < args.length && !args[i + 1].startsWith('-') && (!isKnownBoolean || isNextBool)) {
131
+ val = args[i + 1];
132
+ i++;
133
+ }
134
+ // Bare presence implies true (e.g., --ocr is true)
135
+ else {
136
+ val = 'true';
137
+ }
138
+ }
139
+ // Parse boolean strings to raw boolean types
140
+ const lowerValue = val.toLowerCase();
88
141
  const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
89
- const knownBooleans = new Set([
90
- 'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'ignoreComments',
91
- 'ignoreHeadersAndFooters', 'ignoreSlideMasters', 'ignoreInternalLinks',
92
- 'putNotesAtLast', 'includeRawContent', 'outputErrorToConsole',
93
- 'serializeRawContent', 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
94
- ]);
95
- if (cleanKey === 'format') {
96
- outputFormat = value;
142
+ // Map core CLI options to variables
143
+ if (cleanKey === 'to') {
144
+ toFlagOption = val;
145
+ }
146
+ else if (cleanKey === 'format') {
147
+ formatFlagOption = val;
148
+ usedFormat = true;
97
149
  }
98
150
  else if (cleanKey === 'output') {
99
- outputFile = value;
151
+ outputFile = val;
152
+ }
153
+ else if (cleanKey === 'toText') {
154
+ toTextOption = boolValue !== undefined ? boolValue : true;
155
+ usedToText = true;
156
+ }
157
+ else if (cleanKey === 'verbose') {
158
+ verbose = boolValue !== undefined ? boolValue : true;
159
+ }
160
+ else if (cleanKey === 'ocrLanguage') {
161
+ config.ocrLanguage = val;
162
+ usedOcrLanguage = true;
163
+ }
164
+ else if (cleanKey === 'putNotesAtLast') {
165
+ config.putNotesAtLast = boolValue !== undefined ? boolValue : true;
166
+ usedPutNotesAtLast = true;
167
+ }
168
+ else if (cleanKey === 'outputErrorToConsole') {
169
+ verbose = boolValue !== undefined ? boolValue : true;
170
+ usedOutputErrorToConsole = true;
100
171
  }
101
172
  else {
102
- if (boolValue !== undefined) {
103
- if (cleanKey === 'toText')
104
- toText = boolValue;
105
- else if (cleanKey === 'verbose') {
106
- verbose = boolValue;
107
- if (verbose)
108
- config.outputErrorToConsole = true;
173
+ // Check if the flag belongs to generatorConfig or a specific sub-generator (e.g., htmlConfig)
174
+ const isGeneratorOption = knownGeneratorBooleans.has(cleanKey) || generatorPrefixes.some(pref => cleanKey.startsWith(pref));
175
+ const target = isGeneratorOption ? generatorConfig : config;
176
+ let path = cleanKey;
177
+ // Strip generatorConfig prefix to flatten it onto the generatorConfig object
178
+ if (isGeneratorOption && cleanKey.startsWith('generatorConfig.')) {
179
+ path = cleanKey.slice('generatorConfig.'.length);
180
+ }
181
+ // Support nested dot-notation parsing (e.g., --ocrConfig.language=fra)
182
+ if (path.includes('.')) {
183
+ const parts = path.split('.');
184
+ let current = target;
185
+ for (let j = 0; j < parts.length - 1; j++) {
186
+ const part = parts[j];
187
+ if (!current[part])
188
+ current[part] = {};
189
+ current = current[part];
190
+ }
191
+ const lastPart = parts[parts.length - 1];
192
+ current[lastPart] = boolValue !== undefined ? boolValue : val;
193
+ }
194
+ // Flat key assignment
195
+ else {
196
+ if (boolValue !== undefined) {
197
+ target[path] = boolValue;
198
+ }
199
+ else if (knownParserBooleans.has(path) || (isGeneratorOption && knownGeneratorBooleans.has(path))) {
200
+ console.warn(`Invalid boolean value for --${cleanKey}: ${val}. Using default.`);
109
201
  }
110
202
  else {
111
- // @ts-ignore
112
- config[cleanKey] = boolValue;
203
+ target[path] = val;
113
204
  }
114
205
  }
115
- else if (knownBooleans.has(cleanKey)) {
116
- console.warn(`Invalid boolean value for --${cleanKey}: ${value}. Using default.`);
117
- }
118
- else {
119
- // @ts-ignore
120
- config[cleanKey] = value;
206
+ }
207
+ }
208
+ else {
209
+ // First positional argument that is not a flag is treated as the input file path
210
+ if (!fileArg) {
211
+ fileArg = arg;
212
+ }
213
+ }
214
+ }
215
+ if (fileArg && !showHelp) {
216
+ // Resolve output format prioritizing: --to > --format > --toText
217
+ let outputFormat;
218
+ if (toFlagOption) {
219
+ outputFormat = toFlagOption;
220
+ }
221
+ else if (formatFlagOption) {
222
+ outputFormat = formatFlagOption;
223
+ }
224
+ else if (toTextOption === true) {
225
+ outputFormat = 'text';
226
+ }
227
+ // Display warning messages for any deprecated CLI options used
228
+ if (usedFormat) {
229
+ console.warn('Warning: --format is deprecated. Use --to instead.');
230
+ }
231
+ if (usedToText) {
232
+ console.warn('Warning: --toText is deprecated. Use --to=text instead.');
233
+ }
234
+ if (usedOcrLanguage) {
235
+ console.warn('Warning: --ocrLanguage is deprecated. Use --ocrConfig.language instead.');
236
+ }
237
+ if (usedPutNotesAtLast) {
238
+ console.warn('Warning: --putNotesAtLast is deprecated and will be ignored by all parsers.');
239
+ }
240
+ if (usedOutputErrorToConsole) {
241
+ console.warn('Warning: --outputErrorToConsole is deprecated. Use --verbose instead.');
242
+ }
243
+ // Intercept parser warning callbacks to format and print issues when verbose is enabled
244
+ const originalOnWarning = config.onWarning;
245
+ config.onWarning = (issue) => {
246
+ if (verbose) {
247
+ const severity = issue.type === 'error' ? 'Error' : 'Warning';
248
+ console.error(`[OfficeParser ${severity}] [${issue.code}]: ${issue.message}`);
249
+ if (issue.details) {
250
+ console.error(issue.details);
121
251
  }
122
252
  }
123
- });
253
+ if (originalOnWarning)
254
+ originalOnWarning(issue);
255
+ };
256
+ // Propagate newlineDelimiter and csvDelimiter if configured flatly but not in generator sub-configs
257
+ if (config.newlineDelimiter !== undefined) {
258
+ if (!generatorConfig.textConfig)
259
+ generatorConfig.textConfig = {};
260
+ if (generatorConfig.textConfig.newlineDelimiter === undefined) {
261
+ generatorConfig.textConfig.newlineDelimiter = config.newlineDelimiter;
262
+ }
263
+ }
264
+ if (config.csvDelimiter !== undefined) {
265
+ if (!generatorConfig.csvConfig)
266
+ generatorConfig.csvConfig = {};
267
+ if (generatorConfig.csvConfig.columnDelimiter === undefined) {
268
+ generatorConfig.csvConfig.columnDelimiter = config.csvDelimiter;
269
+ }
270
+ }
271
+ // Run the main parser
124
272
  OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
125
273
  .then(async (ast) => {
126
274
  let output;
127
- if (outputFormat) {
128
- const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat);
275
+ // Generate JSON output or convert AST using OfficeGenerator
276
+ if (outputFormat === 'json') {
277
+ output = JSON.stringify(ast, null, 2);
278
+ }
279
+ else if (outputFormat) {
280
+ const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, outputFormat, generatorConfig);
129
281
  if (Array.isArray(result.value)) {
130
282
  output = JSON.stringify(result.value, null, 2);
131
283
  }
@@ -133,12 +285,10 @@ if (fileArg) {
133
285
  output = result.value;
134
286
  }
135
287
  }
136
- else if (toText) {
137
- output = ast.toText();
138
- }
139
288
  else {
140
289
  output = JSON.stringify(ast, null, 2);
141
290
  }
291
+ // Write generated output to output file or print to standard output
142
292
  if (outputFile) {
143
293
  if (output instanceof Uint8Array) {
144
294
  fs.writeFileSync(outputFile, output);
@@ -163,15 +313,16 @@ if (fileArg) {
163
313
  }
164
314
  })
165
315
  .catch(async (err) => {
316
+ // Handle and display parsing error messages
166
317
  console.error(`Error parsing file "${fileArg}":`);
167
318
  if (verbose) {
168
319
  console.error(err);
169
320
  }
170
321
  else {
171
322
  console.error(err.message || err);
172
- console.error('Use --verbose=true for full stack trace.');
323
+ console.error('Use --verbose for full stack trace.');
173
324
  }
174
- // Ensure OCR workers are terminated even on error
325
+ // Ensure OCR workers are terminated even on error to prevent process hang
175
326
  if (config.ocr) {
176
327
  await OfficeParser_js_1.OfficeParser.terminateOcr();
177
328
  }
@@ -179,32 +330,46 @@ if (fileArg) {
179
330
  });
180
331
  }
181
332
  else {
182
- console.log('Usage: officeparser <file> [--option=value]');
333
+ console.log('Usage: officeparser <file> [options]');
183
334
  console.log('');
184
335
  console.log('Options:');
185
- console.log(' --format=json|md|html|rtf|csv|text|pdf|chunks Convert to specified format');
186
- console.log(' --output=file.ext Save output to file instead of stdout');
187
- console.log(' --toText=true Output plain text instead of JSON AST (legacy)');
188
- console.log(' --ocr=true Enable OCR for images');
189
- console.log(' --ocrLanguage=eng OCR language (default: eng)');
190
- console.log(' --extractAttachments=true Extract embedded attachments');
191
- console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
192
- console.log(' --ignoreComments=true Ignore inline comments');
193
- console.log(' --ignoreHeadersAndFooters=true Ignore headers and footers');
194
- console.log(' --ignoreSlideMasters=true Ignore slide masters');
195
- console.log(' --ignoreInternalLinks=true Ignore internal links');
196
- console.log(' --putNotesAtLast=true Move notes to end of document');
197
- console.log(' --includeRawContent=true Include raw content in AST');
198
- console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
199
- console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
200
- console.log(' --includeBreakNodes=false Include break nodes (DOCX only, default: false)');
201
- console.log(' --verbose=true Show full error stack traces');
336
+ console.log(' --to=json|text|md|html|pdf|csv|rtf|chunks Target conversion format (default: json)');
337
+ console.log(' --output=file.ext Save output to file instead of stdout');
338
+ console.log(' --fileType=docx|xlsx|pptx|odt|... Explicitly override input file type detection');
339
+ console.log(' --ocr Enable OCR for images (default: false)');
340
+ console.log(' --ocrConfig.language=eng OCR language (default: eng)');
341
+ console.log(' --extractAttachments Extract embedded attachments (default: false)');
342
+ console.log(' --ignoreNotes Ignore footnotes/endnotes/speaker notes (default: false)');
343
+ console.log(' --ignoreComments Ignore inline comments (default: false)');
344
+ console.log(' --ignoreHeadersAndFooters Ignore headers and footers (default: false)');
345
+ console.log(' --ignoreSlideMasters Ignore slide masters (default: false)');
346
+ console.log(' --ignoreInternalLinks Ignore internal links (default: false)');
347
+ console.log(' --includeRawContent Include raw content in AST (default: false)');
348
+ console.log(' --serializeRawContent Serialize raw XML content (default: true)');
349
+ console.log(' --preserveXmlWhitespace Keep raw formatting space (default: false)');
350
+ console.log(' --includeBreakNodes Include break nodes (DOCX only, default: false)');
351
+ console.log(' --verbose Show full error stack traces and warning logs');
352
+ console.log(' --newlineDelimiter=string Delimiter string between blocks/lines (default: \\n)');
353
+ console.log(' --csvDelimiter=char Custom CSV delimiter (default: ,)');
354
+ console.log('');
355
+ console.log('High-Value Generator Options:');
356
+ console.log(' --includeFormatting Include font formatting like bold/italic (default: true)');
357
+ console.log(' --renderMetadata Render metadata in output content (default: false)');
358
+ console.log(' --htmlConfig.containerWidth=value HTML container width (auto | px | % | vw etc., default: auto)');
359
+ console.log('');
360
+ console.log('Advanced Nested Config Examples:');
361
+ console.log(' --pdfConfig.format=Letter Configure Puppeteer PDF format (A4 | Letter | Legal etc.)');
362
+ console.log(' --chunksConfig.strategy=fixed-size Chunking strategy (fixed-size | document-structure | semantic)');
363
+ console.log('');
364
+ console.log('Format Syntax:');
365
+ console.log(' Flags can be written as --flag (presence implies true), --no-flag (negation),');
366
+ console.log(' --flag=value, or --flag value.');
202
367
  console.log('');
203
368
  console.log('Examples:');
204
369
  console.log(' officeparser document.docx');
205
- console.log(' officeparser document.docx --format=html --output=doc.html');
206
- console.log(' officeparser document.docx --format=md');
207
- console.log(' officeparser report.pdf --ocr=true --format=text');
208
- console.log(' officeparser data.xlsx --format=csv --output=data.csv');
209
- console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
370
+ console.log(' officeparser document.docx --to html --output doc.html');
371
+ console.log(' officeparser document.docx --to md');
372
+ console.log(' officeparser report.pdf --ocr --ocrConfig.language eng --to text');
373
+ console.log(' officeparser data.xlsx --to csv --output data.csv --csvDelimiter ";"');
374
+ console.log(' officeparser image_doc --fileType docx --to json');
210
375
  }
package/dist/defaults.js CHANGED
@@ -59,6 +59,10 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
59
59
  ignoreInternalLinks: false,
60
60
  fileType: null,
61
61
  csvDelimiter: ',',
62
+ decompressionLimits: {
63
+ maxUncompressedBytes: 512 * 1024 * 1024,
64
+ maxZipEntries: 10000,
65
+ },
62
66
  };
63
67
  /**
64
68
  * Default configuration for HTML generation.
@@ -119,7 +123,7 @@ const DEFAULT_MD_GENERATOR_CONFIG = {
119
123
  */
120
124
  const DEFAULT_TEXT_GENERATOR_CONFIG = {
121
125
  newlineDelimiter: '\n',
122
- preserveLayout: false,
126
+ preserveLayout: true,
123
127
  };
124
128
  /**
125
129
  * Default configuration for Fixed-Size chunking.
@@ -10,6 +10,7 @@ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat =
10
10
  protected ast: OfficeParserAST;
11
11
  protected messages: OfficeIssue[];
12
12
  protected styleMapper: StyleMapper;
13
+ protected collectedNotes: OfficeContentNode[];
13
14
  constructor(destination: D, ast: OfficeParserAST, config?: GeneratorConfig<D> | FullGeneratorConfig);
14
15
  /**
15
16
  * Retrieves the semantic mapping for a node, respecting the includeFormatting flag.
@@ -14,6 +14,7 @@ class BaseGenerator {
14
14
  ast;
15
15
  messages = [];
16
16
  styleMapper;
17
+ collectedNotes = [];
17
18
  constructor(destination, ast, config) {
18
19
  this.destination = destination;
19
20
  this.config = (0, configUtils_js_1.resolveGeneratorConfig)(destination, ast.config, config);
@@ -65,7 +66,18 @@ class BaseGenerator {
65
66
  childrenOutput += await this.processNodeRecursive(child, processor);
66
67
  }
67
68
  }
68
- return await processor(node, childrenOutput);
69
+ if (node.notes && node.notes.length > 0) {
70
+ if (node.type !== 'slide') {
71
+ this.collectedNotes.push(...node.notes);
72
+ }
73
+ }
74
+ let result = await processor(node, childrenOutput);
75
+ if (node.type === 'slide' && node.notes && node.notes.length > 0) {
76
+ for (const note of node.notes) {
77
+ result += await this.processNodeRecursive(note, processor);
78
+ }
79
+ }
80
+ return result;
69
81
  }
70
82
  /**
71
83
  * Helper to generate a unique ID (slug) from text.
@@ -214,10 +214,17 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
214
214
  || splitBy === 'sheet' && node.type === 'sheet';
215
215
  if (isForcedSplit) {
216
216
  // Process children of the container as individual chunks within the boundary
217
- if (node.children) {
217
+ if (node.children || node.notes) {
218
218
  const innerChunks = [];
219
- for (const child of node.children) {
220
- await this.processNodeForStructure(child, config, 'paragraph', maxChunkSize, measure, innerChunks, contextStack);
219
+ if (node.children) {
220
+ for (const child of node.children) {
221
+ await this.processNodeForStructure(child, config, 'paragraph', maxChunkSize, measure, innerChunks, contextStack);
222
+ }
223
+ }
224
+ if (node.notes) {
225
+ for (const note of node.notes) {
226
+ await this.processNodeForStructure(note, config, 'paragraph', maxChunkSize, measure, innerChunks, contextStack);
227
+ }
221
228
  }
222
229
  for (const ic of innerChunks) {
223
230
  ic.metadata.slideNumber = contextStack.slideNumber;
@@ -268,12 +275,17 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
268
275
  }
269
276
  return;
270
277
  }
271
- // Recurse into children for container nodes
278
+ // Recurse into children and notes for container nodes
272
279
  if (node.children) {
273
280
  for (const child of node.children) {
274
281
  await this.processNodeForStructure(child, config, splitBy, maxChunkSize, measure, chunks, contextStack);
275
282
  }
276
283
  }
284
+ if (node.notes) {
285
+ for (const note of node.notes) {
286
+ await this.processNodeForStructure(note, config, splitBy, maxChunkSize, measure, chunks, contextStack);
287
+ }
288
+ }
277
289
  }
278
290
  isStructuralBoundary(node, splitBy) {
279
291
  if (splitBy === 'heading')
@@ -524,6 +536,10 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
524
536
  for (const child of node.children)
525
537
  await walk(child);
526
538
  }
539
+ if (node.notes) {
540
+ for (const note of node.notes)
541
+ await walk(note);
542
+ }
527
543
  };
528
544
  for (const node of this.ast.content)
529
545
  await walk(node);
@@ -568,6 +584,10 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
568
584
  for (const child of node.children)
569
585
  await walk(child);
570
586
  }
587
+ if (node.notes) {
588
+ for (const note of node.notes)
589
+ await walk(note);
590
+ }
571
591
  };
572
592
  for (const node of this.ast.content)
573
593
  await walk(node);
@@ -624,7 +644,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
624
644
  let timerId;
625
645
  const timeoutPromise = new Promise((_, reject) => {
626
646
  timerId = setTimeout(() => {
627
- reject(new Error(`Embedding call timed out after ${timeoutMs}ms`));
647
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT, this.config, timeoutMs));
628
648
  }, timeoutMs);
629
649
  });
630
650
  return Promise.race([call, timeoutPromise]).finally(() => {
@@ -27,7 +27,14 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
27
27
  containerClass = 'presentation-container';
28
28
  else if (isPdf)
29
29
  containerClass = 'pdf-container';
30
- const bodyContent = await this.processNodeArray(this.ast.content);
30
+ let bodyContent = await this.processNodeArray(this.ast.content);
31
+ if (this.collectedNotes.length > 0) {
32
+ let notesHtml = '';
33
+ for (const note of this.collectedNotes) {
34
+ notesHtml += await this.processNodeRecursive(note, this.nodeProcessor.bind(this));
35
+ }
36
+ bodyContent += `\n<div class="document-notes-section">\n<hr class="page-break">\n${notesHtml}\n</div>\n`;
37
+ }
31
38
  const metadataBlock = this.config.renderMetadata ? this.renderMetadataSummary() : '';
32
39
  let title = 'Document';
33
40
  let metaTags = '';
@@ -385,9 +392,19 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
385
392
  // Fallback for nodes that have text property but no children (e.g. simple paragraphs)
386
393
  childrenOutput = this.escape(node.text);
387
394
  }
388
- const result = await processor(node, childrenOutput);
395
+ if (node.notes && node.notes.length > 0) {
396
+ if (node.type !== 'slide') {
397
+ this.collectedNotes.push(...node.notes);
398
+ }
399
+ }
400
+ let result = await processor(node, childrenOutput);
389
401
  if (isTable)
390
402
  this.tableNestingLevel--;
403
+ if (node.type === 'slide' && node.notes && node.notes.length > 0) {
404
+ for (const note of node.notes) {
405
+ result += await this.processNodeRecursive(note, processor);
406
+ }
407
+ }
391
408
  return result;
392
409
  }
393
410
  /**
@@ -452,7 +469,7 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
452
469
  src = `data:${attachment.mimeType || 'image/png'};base64,${attachment.data}`;
453
470
  }
454
471
  }
455
- const img = `<img src="${src}" alt="${this.escape(node.text || meta?.altText || '')}"${className}${mappedAttrs}${styleAttr}>`;
472
+ const img = `<img src="${this.escape(src)}" alt="${this.escape(node.text || meta?.altText || '')}"${className}${mappedAttrs}${styleAttr}>`;
456
473
  const content = this.config.includeFormatting ? `<div class="image-container">${img}<div class="caption">${this.escape(attachmentName || '')}</div></div>` : img;
457
474
  return `${extraAnchors}<div${idAttr}>${content}</div>`;
458
475
  }
@@ -218,6 +218,11 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
218
218
  const anchors = this.renderAnchors(node.metadata);
219
219
  return `\n---\n\n${anchors}${anchors ? '\n' : ''}${childrenOutput}\n\n`;
220
220
  }
221
+ case 'note': {
222
+ const meta = node.metadata;
223
+ const typeLabel = meta?.noteType === 'footnote' ? 'Footnote' : (meta?.noteType === 'endnote' ? 'Endnote' : 'Note');
224
+ return `> **${typeLabel}:** ${childrenOutput.trim()}\n\n`;
225
+ }
221
226
  default:
222
227
  return childrenOutput;
223
228
  }
@@ -241,6 +246,13 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
241
246
  }
242
247
  output += result;
243
248
  }
249
+ if (this.collectedNotes.length > 0) {
250
+ let notesMd = '\n\n---\n\n### Notes\n\n';
251
+ for (const note of this.collectedNotes) {
252
+ notesMd += await this.processNodeRecursive(note, processor);
253
+ }
254
+ output += notesMd;
255
+ }
244
256
  return {
245
257
  value: (output + '\n\n' + this.hoistedContent.join('\n\n')).trim(),
246
258
  messages: this.messages
@@ -267,7 +279,18 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
267
279
  childrenOutput += await this.processNodeRecursive(child, processor);
268
280
  }
269
281
  }
270
- return await processor(node, childrenOutput);
282
+ if (node.notes && node.notes.length > 0) {
283
+ if (node.type !== 'slide') {
284
+ this.collectedNotes.push(...node.notes);
285
+ }
286
+ }
287
+ let result = await processor(node, childrenOutput);
288
+ if (node.type === 'slide' && node.notes && node.notes.length > 0) {
289
+ for (const note of node.notes) {
290
+ result += await this.processNodeRecursive(note, processor);
291
+ }
292
+ }
293
+ return result;
271
294
  }
272
295
  /**
273
296
  * Merges adjacent text nodes with identical formatting and metadata.
@@ -284,9 +307,17 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
284
307
  current.text = (current.text || '') + (node.text || '');
285
308
  if (current.rawContent && node.rawContent)
286
309
  current.rawContent += node.rawContent;
310
+ if (node.notes && node.notes.length > 0) {
311
+ if (!current.notes)
312
+ current.notes = [];
313
+ current.notes.push(...node.notes);
314
+ }
287
315
  }
288
316
  else {
289
317
  current = { ...node }; // Clone
318
+ if (node.notes) {
319
+ current.notes = [...node.notes];
320
+ }
290
321
  result.push(current);
291
322
  }
292
323
  }