officeparser 6.1.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +219 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -29
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +826 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +781 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +27 -8
@@ -0,0 +1,683 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.ChunkingGenerator = void 0;
4
+ const defaults_js_1 = require("../defaults.js");
5
+ const types_js_1 = require("../types.js");
6
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
7
+ const BaseGenerator_js_1 = require("./BaseGenerator.js");
8
+ /**
9
+ * Generates a list of OfficeChunk objects from an AST for use in RAG pipelines.
10
+ * Supports three strategies: 'fixed-size', 'document-structure', and 'semantic'.
11
+ */
12
+ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
13
+ /** The resolved chunking config (with defaults applied). */
14
+ chunkConfig;
15
+ /** Whether the user provided an explicit sentence boundary regex. */
16
+ isCustomRegex;
17
+ constructor(ast, config) {
18
+ super('chunks', ast, config);
19
+ this.chunkConfig = this.resolveChunkingConfig(this.config.chunksConfig);
20
+ // Track if the user explicitly provided a regex (vs using the library default)
21
+ this.isCustomRegex = !!config?.chunksConfig?.sentenceBoundaryRegex;
22
+ }
23
+ /**
24
+ * Merges the user's chunking config with the appropriate defaults for the chosen strategy.
25
+ */
26
+ resolveChunkingConfig(userChunkConfig) {
27
+ const strategy = userChunkConfig?.strategy ?? 'document-structure';
28
+ switch (strategy) {
29
+ case 'fixed-size':
30
+ return { ...defaults_js_1.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG, ...userChunkConfig };
31
+ case 'semantic':
32
+ return { ...defaults_js_1.DEFAULT_SEMANTIC_CHUNKING_CONFIG, ...userChunkConfig };
33
+ case 'document-structure':
34
+ return { ...defaults_js_1.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG, ...userChunkConfig };
35
+ }
36
+ }
37
+ /**
38
+ * Main entry point. Routes to the correct strategy implementation.
39
+ * Note: ConversionResult.value is a JSON string of OfficeChunk[] for the 'chunks' destination.
40
+ */
41
+ async generate() {
42
+ let chunks;
43
+ switch (this.chunkConfig.strategy) {
44
+ case 'fixed-size':
45
+ chunks = await this.generateFixedSize(this.chunkConfig);
46
+ break;
47
+ case 'semantic':
48
+ chunks = await this.generateSemantic(this.chunkConfig);
49
+ break;
50
+ case 'document-structure':
51
+ chunks = await this.generateDocumentStructure(this.chunkConfig);
52
+ break;
53
+ }
54
+ return {
55
+ value: chunks,
56
+ messages: this.messages,
57
+ };
58
+ }
59
+ // ─── Strategy 1: Fixed-Size ────────────────────────────────────────────────
60
+ /**
61
+ * Splits the full document text into fixed-size chunks with optional overlap.
62
+ * Attempts to split on natural separators before hard-cutting.
63
+ */
64
+ async generateFixedSize(config) {
65
+ const { chunkSize, chunkOverlap, separators, lengthFunction: measure } = config;
66
+ // Build a flat text with positional map from top-level nodes
67
+ const { text: fullText, nodeMap } = await this.buildFlatTextWithPositions();
68
+ const rawChunks = this.splitTextRecursively(fullText, chunkSize, chunkOverlap, separators, measure);
69
+ return rawChunks.map(({ text, start, end }) => {
70
+ const chunk = { text, metadata: { sourceType: this.ast.type } };
71
+ if (config.includeMetadata ?? true) {
72
+ this.enrichMetadataFromPosition(chunk, nodeMap, start);
73
+ }
74
+ if (config.addStartIndex) {
75
+ chunk.startIndex = start;
76
+ chunk.endIndex = end;
77
+ }
78
+ return chunk;
79
+ });
80
+ }
81
+ /**
82
+ * Recursively tries separators to split text into chunks of at most `chunkSize`,
83
+ * with `chunkOverlap` characters of overlap between consecutive chunks.
84
+ */
85
+ splitTextRecursively(text, chunkSize, chunkOverlap, separators, measure) {
86
+ if (measure(text) <= chunkSize) {
87
+ return text.trim() ? [{ text, start: 0, end: text.length }] : [];
88
+ }
89
+ let chosenSep;
90
+ let nextSeparators = [];
91
+ let parts = [];
92
+ let actualSep = '';
93
+ // 1. Try splitting into sentences first
94
+ const sentences = this.splitIntoSentences(text);
95
+ if (sentences.length > 1) {
96
+ parts = sentences;
97
+ actualSep = ' ';
98
+ chosenSep = 'SENTENCE_SPLIT'; // Internal marker to avoid hard-cut
99
+ nextSeparators = separators;
100
+ }
101
+ else {
102
+ // 2. Fallback to the character separators provided in config
103
+ for (let i = 0; i < separators.length; i++) {
104
+ const sep = separators[i];
105
+ parts = sep ? text.split(sep) : [...text];
106
+ if (parts.length > 1) {
107
+ chosenSep = sep;
108
+ actualSep = sep;
109
+ nextSeparators = separators.slice(i + 1);
110
+ break;
111
+ }
112
+ }
113
+ }
114
+ if (chosenSep === undefined) {
115
+ // Last resort: hard cut
116
+ const results = [];
117
+ let offset = 0;
118
+ while (offset < text.length) {
119
+ const slice = text.slice(offset, offset + chunkSize);
120
+ results.push({ text: slice, start: offset, end: offset + slice.length });
121
+ offset += Math.max(1, chunkSize - chunkOverlap);
122
+ }
123
+ return results;
124
+ }
125
+ const results = [];
126
+ let currentChunk = '';
127
+ let currentStart = 0;
128
+ let absoluteOffset = 0;
129
+ for (let i = 0; i < parts.length; i++) {
130
+ const part = parts[i];
131
+ const candidate = currentChunk ? currentChunk + actualSep + part : part;
132
+ if (measure(candidate) <= chunkSize) {
133
+ currentChunk = candidate;
134
+ }
135
+ else {
136
+ if (currentChunk.trim()) {
137
+ // If currentChunk is still too big (should only happen if it's a single part), recurse!
138
+ if (measure(currentChunk) > chunkSize) {
139
+ const subResults = this.splitTextRecursively(currentChunk, chunkSize, chunkOverlap, nextSeparators, measure);
140
+ for (const r of subResults) {
141
+ results.push({ text: r.text, start: currentStart + r.start, end: currentStart + r.end });
142
+ }
143
+ }
144
+ else {
145
+ results.push({ text: currentChunk, start: currentStart, end: currentStart + currentChunk.length });
146
+ }
147
+ }
148
+ // Start next chunk
149
+ if (chunkOverlap > 0 && currentChunk.length > chunkOverlap && measure(currentChunk) <= chunkSize) {
150
+ const overlapText = currentChunk.slice(-chunkOverlap);
151
+ currentStart = absoluteOffset - chunkOverlap;
152
+ currentChunk = overlapText + actualSep + part;
153
+ }
154
+ else {
155
+ currentStart = absoluteOffset;
156
+ currentChunk = part;
157
+ }
158
+ }
159
+ absoluteOffset += part.length + actualSep.length;
160
+ }
161
+ if (currentChunk.trim()) {
162
+ if (measure(currentChunk) > chunkSize) {
163
+ const subResults = this.splitTextRecursively(currentChunk, chunkSize, chunkOverlap, nextSeparators, measure);
164
+ for (const r of subResults) {
165
+ results.push({ text: r.text, start: currentStart + r.start, end: currentStart + r.end });
166
+ }
167
+ }
168
+ else {
169
+ results.push({ text: currentChunk, start: currentStart, end: currentStart + currentChunk.length });
170
+ }
171
+ }
172
+ return results;
173
+ }
174
+ // ─── Strategy 2: Document Structure ───────────────────────────────────────
175
+ /**
176
+ * Walks the AST and splits at the designated structural boundaries (slide, page, heading, paragraph).
177
+ */
178
+ async generateDocumentStructure(config) {
179
+ const { splitBy, maxChunkSize, lengthFunction: measure } = config;
180
+ const chunks = [];
181
+ const contextStack = {};
182
+ // Walk top-level nodes and decide where to cut
183
+ for (const node of this.ast.content) {
184
+ await this.processNodeForStructure(node, config, splitBy, maxChunkSize, measure, chunks, contextStack);
185
+ }
186
+ return this.finalizeChunks(chunks, config);
187
+ }
188
+ async processNodeForStructure(node, config, splitBy, maxChunkSize, measure, chunks, contextStack) {
189
+ // Check for node override or skip
190
+ const override = await this.handleOnNode(node);
191
+ if (override === false)
192
+ return;
193
+ // Update context from structural container nodes
194
+ if (node.type === 'slide') {
195
+ const meta = node.metadata;
196
+ contextStack.slideNumber = meta?.slideNumber;
197
+ }
198
+ else if (node.type === 'page') {
199
+ const meta = node.metadata;
200
+ contextStack.pageNumber = meta?.pageNumber;
201
+ }
202
+ else if (node.type === 'sheet') {
203
+ const meta = node.metadata;
204
+ contextStack.sheetName = meta?.sheetName;
205
+ }
206
+ else if (node.type === 'heading') {
207
+ contextStack.heading = node.text;
208
+ }
209
+ const isStructuralBoundary = this.isStructuralBoundary(node, splitBy);
210
+ const isForcedSplit = splitBy === 'slide' && node.type === 'slide'
211
+ || splitBy === 'page' && node.type === 'page'
212
+ || splitBy === 'sheet' && node.type === 'sheet';
213
+ if (isForcedSplit) {
214
+ // Process children of the container as individual chunks within the boundary
215
+ if (node.children) {
216
+ const innerChunks = [];
217
+ for (const child of node.children) {
218
+ await this.processNodeForStructure(child, config, 'paragraph', maxChunkSize, measure, innerChunks, contextStack);
219
+ }
220
+ for (const ic of innerChunks) {
221
+ ic.metadata.slideNumber = contextStack.slideNumber;
222
+ ic.metadata.pageNumber = contextStack.pageNumber;
223
+ ic.metadata.sheetName = contextStack.sheetName;
224
+ chunks.push(ic);
225
+ }
226
+ }
227
+ return;
228
+ }
229
+ if (node.type === 'table') {
230
+ await this.processTableNode(node, config, maxChunkSize, measure, chunks, contextStack);
231
+ return;
232
+ }
233
+ const isContentNode = node.type === 'paragraph' || node.type === 'heading' || node.type === 'list' || node.type === 'code' || node.type === 'cell' || (node.text && (!node.children || node.children.length === 0));
234
+ if (isStructuralBoundary || isContentNode) {
235
+ const text = typeof override === 'string' ? override : (node.text ?? '');
236
+ const isWhitespaceOnly = !text.trim() && !text.includes('\u00A0');
237
+ if (isWhitespaceOnly && text.length > 0) {
238
+ // Log skipped empty nodes if debugging
239
+ // Only warn for non-cell nodes to reduce spreadsheet noise
240
+ if (node.type !== 'cell') {
241
+ this.warn(types_js_1.OfficeWarningType.WHITESPACE_NODE_SKIPPED, node.type, node);
242
+ }
243
+ return;
244
+ }
245
+ if (node.type === 'heading')
246
+ contextStack.heading = text;
247
+ const chunk = {
248
+ text,
249
+ metadata: {
250
+ sourceType: this.ast.type,
251
+ closestHeading: contextStack.heading,
252
+ slideNumber: contextStack.slideNumber,
253
+ pageNumber: contextStack.pageNumber,
254
+ sheetName: contextStack.sheetName,
255
+ },
256
+ };
257
+ // If this chunk is too big, further split it
258
+ if (measure(text) > maxChunkSize) {
259
+ const subChunks = this.splitTextRecursively(text, maxChunkSize, 0, ['\n\n', '\n', ' ', ''], measure);
260
+ for (const sub of subChunks) {
261
+ chunks.push({ ...chunk, text: sub.text });
262
+ }
263
+ }
264
+ else {
265
+ chunks.push(chunk);
266
+ }
267
+ return;
268
+ }
269
+ // Recurse into children for container nodes
270
+ if (node.children) {
271
+ for (const child of node.children) {
272
+ await this.processNodeForStructure(child, config, splitBy, maxChunkSize, measure, chunks, contextStack);
273
+ }
274
+ }
275
+ }
276
+ isStructuralBoundary(node, splitBy) {
277
+ if (splitBy === 'heading')
278
+ return node.type === 'heading';
279
+ if (splitBy === 'paragraph')
280
+ return node.type === 'paragraph' || node.type === 'heading' || node.type === 'cell';
281
+ return false;
282
+ }
283
+ /**
284
+ * Handles table chunking with the configured tableSplitStrategy.
285
+ * 'row': keeps header row attached to every chunk.
286
+ * 'flatten': converts table to text and splits normally.
287
+ */
288
+ async processTableNode(node, config, maxChunkSize, measure, chunks, contextStack) {
289
+ const strategy = config.tableSplitStrategy;
290
+ if (strategy === 'flatten' || !node.children || node.children.length === 0) {
291
+ // Flatten: treat as plain text
292
+ const text = node.text ?? '';
293
+ if (!text.trim())
294
+ return;
295
+ chunks.push({
296
+ text,
297
+ metadata: {
298
+ sourceType: this.ast.type,
299
+ closestHeading: contextStack.heading,
300
+ slideNumber: contextStack.slideNumber,
301
+ pageNumber: contextStack.pageNumber,
302
+ sheetName: contextStack.sheetName,
303
+ },
304
+ });
305
+ return;
306
+ }
307
+ // 'row' strategy: extract header row(s) and chunk remaining rows
308
+ const rows = node.children; // Each child is a 'row' node
309
+ const headerRows = [];
310
+ const dataRows = [];
311
+ // Heuristic: first row is the header
312
+ if (rows.length > 0) {
313
+ const firstRow = rows[0];
314
+ const override = await this.handleOnNode(firstRow);
315
+ if (override !== false) {
316
+ // If overridden, we use the string as header, but we still treat it as a header
317
+ headerRows.push(firstRow); // We still add the node to get metadata later if needed, but renderRowsAsText will handle it?
318
+ // Actually renderRowsAsText needs to be updated too.
319
+ }
320
+ for (let i = 1; i < rows.length; i++) {
321
+ dataRows.push(rows[i]);
322
+ }
323
+ }
324
+ const headerText = await this.renderRowsAsText(headerRows);
325
+ const baseMetadata = {
326
+ sourceType: this.ast.type,
327
+ closestHeading: contextStack.heading,
328
+ slideNumber: contextStack.slideNumber,
329
+ pageNumber: contextStack.pageNumber,
330
+ sheetName: contextStack.sheetName,
331
+ isTableChunk: true,
332
+ };
333
+ // Group data rows into chunks
334
+ let currentRows = [];
335
+ let currentSize = headerText ? measure(headerText) : 0;
336
+ const flushCurrentRows = async () => {
337
+ if (currentRows.length === 0)
338
+ return;
339
+ const rowText = await this.renderRowsAsText(currentRows);
340
+ const chunkText = headerText ? `${headerText}\n${rowText}` : rowText;
341
+ if (chunkText.trim()) {
342
+ if (measure(chunkText) > maxChunkSize) {
343
+ // Fallback to recursive splitting for oversized table chunks
344
+ const subChunks = this.splitTextRecursively(chunkText, maxChunkSize, 0, ['\n', ' ', ''], measure);
345
+ for (const sub of subChunks) {
346
+ chunks.push({ text: sub.text, metadata: { ...baseMetadata } });
347
+ }
348
+ }
349
+ else {
350
+ chunks.push({ text: chunkText, metadata: { ...baseMetadata } });
351
+ }
352
+ }
353
+ currentRows = [];
354
+ currentSize = headerText ? measure(headerText) : 0;
355
+ };
356
+ for (const row of dataRows) {
357
+ const override = await this.handleOnNode(row);
358
+ if (override === false)
359
+ continue;
360
+ const rowText = typeof override === 'string' ? override : await this.renderRowsAsText([row]);
361
+ const rowSize = measure(rowText);
362
+ if (currentSize + rowSize > maxChunkSize && currentRows.length > 0) {
363
+ await flushCurrentRows();
364
+ }
365
+ if (typeof override === 'string') {
366
+ // If row was overridden, we flush current, then add the override as a chunk.
367
+ await flushCurrentRows();
368
+ const chunkText = headerText ? `${headerText}\n${override}` : override;
369
+ chunks.push({ text: chunkText, metadata: { ...baseMetadata } });
370
+ continue;
371
+ }
372
+ currentRows.push(row);
373
+ currentSize += rowSize;
374
+ }
375
+ await flushCurrentRows();
376
+ }
377
+ /**
378
+ * Renders a list of row nodes as a pipe-separated text string.
379
+ */
380
+ async renderRowsAsText(rows) {
381
+ const renderedRows = [];
382
+ for (const row of rows) {
383
+ const override = await this.handleOnNode(row);
384
+ if (override === false)
385
+ continue;
386
+ if (typeof override === 'string') {
387
+ renderedRows.push(override);
388
+ continue;
389
+ }
390
+ if (!row.children) {
391
+ renderedRows.push(row.text ?? '');
392
+ continue;
393
+ }
394
+ const cells = row.children.map(cell => (cell.text ?? '').replace(/\n/g, ' ').trim());
395
+ renderedRows.push(`| ${cells.join(' | ')} |`);
396
+ }
397
+ return renderedRows.join('\n');
398
+ }
399
+ // ─── Strategy 3: Semantic ─────────────────────────────────────────────────
400
+ /**
401
+ * Splits document into semantically coherent chunks using cosine similarity
402
+ * between sentence embeddings. A new chunk begins when similarity drops
403
+ * below `similarityThreshold`.
404
+ */
405
+ async generateSemantic(config) {
406
+ const { similarityThreshold: threshold, maxChunkSize, bufferSize, lengthFunction: measure, embeddingBatchSize: batchSize } = config;
407
+ if (typeof config.embeddingFunction !== 'function') {
408
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION, this.config);
409
+ }
410
+ // Extract all leaf-level text sentences from the AST
411
+ const sentences = await this.extractSentences();
412
+ if (sentences.length === 0)
413
+ return [];
414
+ // Embed all sentences in batches to avoid rate limiting
415
+ const embeddings = await this.batchEmbeddings(sentences, config.embeddingFunction, batchSize);
416
+ // Calculate cosine similarity between adjacent sentence windows
417
+ const chunks = [];
418
+ let currentSentences = [];
419
+ let currentSize = 0;
420
+ for (let i = 0; i < sentences.length; i++) {
421
+ const sentenceText = sentences[i].text;
422
+ currentSentences.push(sentences[i]);
423
+ currentSize += measure(sentenceText);
424
+ // Check if we should split here
425
+ const isLast = i === sentences.length - 1;
426
+ const exceedsMax = currentSize > maxChunkSize;
427
+ let shouldSplit = isLast || exceedsMax;
428
+ if (!shouldSplit && i < sentences.length - 1) {
429
+ // Compare the current window with the next window
430
+ const currentWindowEnd = Math.min(i + bufferSize, sentences.length - 1);
431
+ const nextWindowStart = i + 1;
432
+ const nextWindowEnd = Math.min(i + 1 + bufferSize, sentences.length - 1);
433
+ const currentEmbedding = this.averageEmbeddings(embeddings.slice(Math.max(0, i - bufferSize + 1), currentWindowEnd + 1));
434
+ const nextEmbedding = this.averageEmbeddings(embeddings.slice(nextWindowStart, nextWindowEnd + 1));
435
+ const similarity = this.cosineSimilarity(currentEmbedding, nextEmbedding);
436
+ if (similarity < threshold) {
437
+ shouldSplit = true;
438
+ }
439
+ }
440
+ if (shouldSplit && currentSentences.length > 0) {
441
+ const chunkText = currentSentences.map(s => s.text).join(' ');
442
+ const firstSentence = currentSentences[0];
443
+ const baseMetadata = {
444
+ sourceType: this.ast.type,
445
+ closestHeading: firstSentence.closestHeading,
446
+ slideNumber: firstSentence.slideNumber,
447
+ pageNumber: firstSentence.pageNumber,
448
+ sheetName: firstSentence.sheetName,
449
+ };
450
+ if (measure(chunkText) > maxChunkSize) {
451
+ // Fallback to recursive splitting for oversized semantic chunks
452
+ const subChunks = this.splitTextRecursively(chunkText, maxChunkSize, 0, ['\n', ' ', ''], measure);
453
+ for (const sub of subChunks) {
454
+ chunks.push({
455
+ text: sub.text,
456
+ metadata: { ...baseMetadata }
457
+ });
458
+ }
459
+ }
460
+ else {
461
+ chunks.push({
462
+ text: chunkText,
463
+ metadata: baseMetadata
464
+ });
465
+ }
466
+ currentSentences = [];
467
+ currentSize = 0;
468
+ }
469
+ }
470
+ return this.finalizeChunks(chunks, config);
471
+ }
472
+ /**
473
+ * Extracts all text sentences from the AST with their contextual metadata.
474
+ */
475
+ async extractSentences() {
476
+ const results = [];
477
+ let currentHeading;
478
+ let currentSlide;
479
+ let currentPage;
480
+ let currentSheet;
481
+ const walk = async (node) => {
482
+ const override = await this.handleOnNode(node);
483
+ if (override === false)
484
+ return;
485
+ if (node.type === 'heading')
486
+ currentHeading = node.text;
487
+ if (node.type === 'slide')
488
+ currentSlide = node.metadata?.slideNumber;
489
+ if (node.type === 'page')
490
+ currentPage = node.metadata?.pageNumber;
491
+ if (node.type === 'sheet')
492
+ currentSheet = node.metadata?.sheetName;
493
+ const isContentNode = node.type === 'paragraph' || node.type === 'heading' || node.type === 'list' || node.type === 'cell' || (node.text && (!node.children || node.children.length === 0));
494
+ if (isContentNode) {
495
+ const text = (typeof override === 'string' ? override : (node.text ?? '')).trim();
496
+ if (!text)
497
+ return;
498
+ // Split paragraph text into individual sentences for finer-grained similarity
499
+ const sentences = this.splitIntoSentences(text);
500
+ for (const sentence of sentences) {
501
+ if (sentence.trim()) {
502
+ results.push({
503
+ text: sentence.trim(),
504
+ closestHeading: currentHeading,
505
+ slideNumber: currentSlide,
506
+ pageNumber: currentPage,
507
+ sheetName: currentSheet,
508
+ });
509
+ }
510
+ }
511
+ return; // don't recurse into children; we already have the text
512
+ }
513
+ if (node.children) {
514
+ for (const child of node.children)
515
+ await walk(child);
516
+ }
517
+ };
518
+ for (const node of this.ast.content)
519
+ await walk(node);
520
+ return results;
521
+ }
522
+ // ─── Shared Utilities ─────────────────────────────────────────────────────
523
+ /**
524
+ * Builds a flat text string from the entire document and a map of
525
+ * character offsets to AST node metadata for position-based metadata lookups.
526
+ */
527
+ async buildFlatTextWithPositions() {
528
+ const parts = [];
529
+ const nodeMap = [];
530
+ let offset = 0;
531
+ let currentHeading;
532
+ let currentSlide;
533
+ let currentPage;
534
+ let currentSheet;
535
+ const walk = async (node) => {
536
+ const override = await this.handleOnNode(node);
537
+ if (override === false)
538
+ return;
539
+ if (node.type === 'heading')
540
+ currentHeading = node.text;
541
+ if (node.type === 'slide')
542
+ currentSlide = node.metadata?.slideNumber;
543
+ if (node.type === 'page')
544
+ currentPage = node.metadata?.pageNumber;
545
+ if (node.type === 'sheet')
546
+ currentSheet = node.metadata?.sheetName;
547
+ const isContentNode = node.type === 'paragraph' || node.type === 'heading' || node.type === 'list' || node.type === 'code' || node.type === 'cell' || (node.text && (!node.children || node.children.length === 0));
548
+ if (isContentNode) {
549
+ const nodeText = typeof override === 'string' ? override : (node.text || '');
550
+ const txt = nodeText + '\n';
551
+ nodeMap.push({ offset, heading: currentHeading, slideNumber: currentSlide, pageNumber: currentPage, sheetName: currentSheet });
552
+ parts.push(txt);
553
+ offset += txt.length;
554
+ return;
555
+ }
556
+ if (node.children) {
557
+ for (const child of node.children)
558
+ await walk(child);
559
+ }
560
+ };
561
+ for (const node of this.ast.content)
562
+ await walk(node);
563
+ return { text: parts.join(''), nodeMap };
564
+ }
565
+ /**
566
+ * Finds the closest AST metadata for a given character position.
567
+ */
568
+ enrichMetadataFromPosition(chunk, nodeMap, charOffset) {
569
+ let best = nodeMap[0];
570
+ for (const entry of nodeMap) {
571
+ if (entry.offset <= charOffset)
572
+ best = entry;
573
+ else
574
+ break;
575
+ }
576
+ if (best) {
577
+ chunk.metadata.closestHeading = best.heading;
578
+ chunk.metadata.slideNumber = best.slideNumber;
579
+ chunk.metadata.pageNumber = best.pageNumber;
580
+ chunk.metadata.sheetName = best.sheetName;
581
+ }
582
+ }
583
+ /**
584
+ * Applies final post-processing: strips whitespace, sets sourceType.
585
+ */
586
+ finalizeChunks(chunks, config) {
587
+ if (chunks.length === 0) {
588
+ this.warn(types_js_1.OfficeWarningType.EMPTY_CHUNK_GENERATED, this.chunkConfig.strategy);
589
+ }
590
+ return chunks
591
+ .map(chunk => {
592
+ const text = config.stripWhitespace !== false ? chunk.text.trim() : chunk.text;
593
+ const result = { text, metadata: { sourceType: this.ast.type } };
594
+ if (config.includeMetadata !== false) {
595
+ result.metadata = chunk.metadata;
596
+ }
597
+ return result;
598
+ })
599
+ .filter(chunk => chunk.text.length > 0);
600
+ }
601
+ // ─── Embedding Math Utilities ──────────────────────────────────────────────
602
+ /**
603
+ * Helper to process embeddings in sequential batches to avoid API rate limits and memory issues.
604
+ */
605
+ async batchEmbeddings(sentences, embedFn, batchSize = 50) {
606
+ const results = [];
607
+ for (let i = 0; i < sentences.length; i += batchSize) {
608
+ const batch = sentences.slice(i, i + batchSize);
609
+ const batchResults = await Promise.all(batch.map(s => embedFn(s.text)));
610
+ results.push(...batchResults);
611
+ }
612
+ return results;
613
+ }
614
+ cosineSimilarity(a, b) {
615
+ if (a.length !== b.length || a.length === 0)
616
+ return 0;
617
+ let dot = 0, normA = 0, normB = 0;
618
+ for (let i = 0; i < a.length; i++) {
619
+ dot += a[i] * b[i];
620
+ normA += a[i] * a[i];
621
+ normB += b[i] * b[i];
622
+ }
623
+ const denom = Math.sqrt(normA) * Math.sqrt(normB);
624
+ return denom === 0 ? 0 : dot / denom;
625
+ }
626
+ averageEmbeddings(embeddings) {
627
+ if (embeddings.length === 0)
628
+ return [];
629
+ const len = embeddings[0].length;
630
+ const avg = new Array(len).fill(0);
631
+ for (const emb of embeddings) {
632
+ for (let i = 0; i < len; i++)
633
+ avg[i] += emb[i];
634
+ }
635
+ return avg.map(v => v / embeddings.length);
636
+ }
637
+ /**
638
+ * Robustly splits text into sentences, respecting abbreviations and non-Western punctuation.
639
+ */
640
+ splitIntoSentences(text) {
641
+ if (this.isCustomRegex) {
642
+ const userRegex = this.chunkConfig.sentenceBoundaryRegex;
643
+ const regex = typeof userRegex === 'string' ? new RegExp(userRegex, 'g') : userRegex;
644
+ // Split while keeping the separator if possible, or just split
645
+ return text.split(regex).map(s => s.trim()).filter(Boolean);
646
+ }
647
+ const abbreviations = this.chunkConfig.abbreviations;
648
+ const sentences = [];
649
+ let start = 0;
650
+ // Japanese full stop: 。 Exclamation: ! Question: ?
651
+ // Western: . ! ?
652
+ const markRegex = /[.!?。!?]/g;
653
+ let match;
654
+ while ((match = markRegex.exec(text)) !== null) {
655
+ const mark = match[0];
656
+ const pos = match.index;
657
+ const nextChar = text[pos + 1];
658
+ const isAtEnd = pos === text.length - 1;
659
+ const isFollowedByWhitespace = !nextChar || /\s/.test(nextChar);
660
+ const isJapaneseMark = /[。!?]/.test(mark);
661
+ if (isFollowedByWhitespace || isJapaneseMark) {
662
+ // Check for abbreviations (only for period)
663
+ if (mark === '.') {
664
+ const prevSpace = text.lastIndexOf(' ', pos - 1);
665
+ let lastWord = text.substring(prevSpace + 1, pos);
666
+ // Strip punctuation like quotes, parentheses, brackets
667
+ lastWord = lastWord.replace(/^[^\w]+|[^\w]+$/g, '');
668
+ if (abbreviations.includes(lastWord))
669
+ continue;
670
+ }
671
+ sentences.push(text.substring(start, pos + 1).trim());
672
+ start = pos + 1;
673
+ }
674
+ }
675
+ if (start < text.length) {
676
+ const remaining = text.substring(start).trim();
677
+ if (remaining)
678
+ sentences.push(remaining);
679
+ }
680
+ return sentences.length > 0 ? sentences : [text];
681
+ }
682
+ }
683
+ exports.ChunkingGenerator = ChunkingGenerator;