@retrivora-ai/rag-engine 0.4.4 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +32 -57
  2. package/dist/{ChromaDBProvider-GI7TB7GJ.mjs → ChromaDBProvider-APQVJ5F7.mjs} +2 -2
  3. package/dist/{DocumentChunker-3yElxTO3.d.mts → DocumentChunker-C-sCZPhi.d.mts} +6 -6
  4. package/dist/{DocumentChunker-3yElxTO3.d.ts → DocumentChunker-C-sCZPhi.d.ts} +6 -6
  5. package/dist/{MilvusProvider-WDVTFB7D.mjs → MilvusProvider-35US67MS.mjs} +2 -2
  6. package/dist/{MongoDBProvider-ZKW34AEL.mjs → MongoDBProvider-COVYZDP6.mjs} +2 -2
  7. package/dist/{PineconeProvider-BE2JWSPD.mjs → PineconeProvider-AWFJQDZL.mjs} +2 -2
  8. package/dist/{PostgreSQLProvider-5HHTK4SU.mjs → PostgreSQLProvider-IEYRJ7XJ.mjs} +2 -2
  9. package/dist/{QdrantProvider-XVDVBNIG.mjs → QdrantProvider-M6TQYZRO.mjs} +2 -2
  10. package/dist/{RagConfig-BgRDL9Vy.d.mts → RagConfig-DRJO4hGU.d.mts} +12 -1
  11. package/dist/{RagConfig-BgRDL9Vy.d.ts → RagConfig-DRJO4hGU.d.ts} +12 -1
  12. package/dist/{RedisProvider-EK2R2PQH.mjs → RedisProvider-3G5PBLZ4.mjs} +2 -2
  13. package/dist/{SimpleGraphProvider-M6T7SE7D.mjs → SimpleGraphProvider-UK7DJW37.mjs} +1 -1
  14. package/dist/{UniversalVectorProvider-YIDRX6VT.mjs → UniversalVectorProvider-FYQ3B2PW.mjs} +3 -3
  15. package/dist/{WeaviateProvider-4CAPQ7UY.mjs → WeaviateProvider-ITHO36IL.mjs} +2 -2
  16. package/dist/{chunk-5KNBWQM6.mjs → chunk-4A47RCG2.mjs} +5 -1
  17. package/dist/{chunk-EDLTMSNY.mjs → chunk-67AJ6SMD.mjs} +1 -1
  18. package/dist/{chunk-IWHCAQEA.mjs → chunk-7SOSCZGS.mjs} +68 -7
  19. package/dist/{chunk-LJWWPTWE.mjs → chunk-FLOSGE6A.mjs} +76 -14
  20. package/dist/{chunk-H6RKMU7W.mjs → chunk-NXUCKY5L.mjs} +1 -1
  21. package/dist/{chunk-KTS3LLHY.mjs → chunk-OOQXNLXD.mjs} +5 -5
  22. package/dist/{chunk-OKY5P6RA.mjs → chunk-P4HAQ7KB.mjs} +1186 -1345
  23. package/dist/chunk-QMIKLALV.mjs +57 -0
  24. package/dist/{chunk-3QWAK3RZ.mjs → chunk-TYHTZIDP.mjs} +6 -2
  25. package/dist/{chunk-GQT5LF4G.mjs → chunk-U6KHVZLF.mjs} +2 -2
  26. package/dist/{chunk-RK2UDJA2.mjs → chunk-WGSZNY3X.mjs} +1 -1
  27. package/dist/{chunk-XCNXPECE.mjs → chunk-ZNBKHNJ4.mjs} +55 -1
  28. package/dist/handlers/index.d.mts +2 -2
  29. package/dist/handlers/index.d.ts +2 -2
  30. package/dist/handlers/index.js +1417 -1377
  31. package/dist/handlers/index.mjs +3 -3
  32. package/dist/index-CrGMwXfO.d.ts +112 -0
  33. package/dist/index-v669iV-k.d.mts +112 -0
  34. package/dist/index.d.mts +5 -5
  35. package/dist/index.d.ts +5 -5
  36. package/dist/index.mjs +2 -2
  37. package/dist/server.d.mts +104 -158
  38. package/dist/server.d.ts +104 -158
  39. package/dist/server.js +1419 -1379
  40. package/dist/server.mjs +12 -12
  41. package/package.json +5 -1
  42. package/src/components/DocumentUpload.tsx +1 -1
  43. package/src/config/RagConfig.ts +7 -0
  44. package/src/core/ConfigValidator.ts +66 -492
  45. package/src/core/LangChainAgent.ts +78 -0
  46. package/src/core/Pipeline.ts +210 -222
  47. package/src/core/ProviderHealthCheck.ts +35 -406
  48. package/src/core/ProviderInterfaces.ts +37 -0
  49. package/src/core/ProviderRegistry.ts +70 -55
  50. package/src/core/QueryProcessor.ts +173 -0
  51. package/src/llm/ILLMProvider.ts +10 -0
  52. package/src/llm/LLMFactory.ts +33 -13
  53. package/src/llm/providers/AnthropicProvider.ts +55 -15
  54. package/src/llm/providers/GeminiProvider.ts +51 -0
  55. package/src/llm/providers/OllamaProvider.ts +100 -15
  56. package/src/llm/providers/OpenAIProvider.ts +60 -11
  57. package/src/providers/vectordb/BaseVectorProvider.ts +11 -0
  58. package/src/providers/vectordb/MilvusProvider.ts +4 -0
  59. package/src/providers/vectordb/MongoDBProvider.ts +75 -11
  60. package/src/providers/vectordb/PineconeProvider.ts +60 -5
  61. package/src/providers/vectordb/PostgreSQLProvider.ts +84 -14
  62. package/src/providers/vectordb/QdrantProvider.ts +4 -0
  63. package/src/providers/vectordb/WeaviateProvider.ts +8 -4
  64. package/src/rag/DocumentChunker.ts +15 -19
  65. package/src/rag/EntityExtractor.ts +1 -1
  66. package/src/rag/LlamaIndexIngestor.ts +61 -0
  67. package/src/rag/Reranker.ts +20 -0
  68. package/src/server.ts +1 -1
  69. package/src/types/index.ts +9 -0
  70. package/src/utils/DocumentParser.ts +1 -1
  71. package/dist/chunk-FWCSY2DS.mjs +0 -37
  72. package/dist/index-7qeLTPBL.d.mts +0 -114
  73. package/dist/index-DowY4_K0.d.ts +0 -114
@@ -0,0 +1,78 @@
1
+ import { Pipeline } from "./Pipeline";
2
+ import { RagConfig } from "../config/RagConfig";
3
+
4
+ /**
5
+ * LangChainAgent — Orchestration layer that turns the RAG pipeline into an Agent.
6
+ *
7
+ * Uses dynamic imports to remain an optional dependency.
8
+ */
9
+ export class LangChainAgent {
10
+ private executor?: unknown;
11
+
12
+ constructor(private pipeline: Pipeline, private config: RagConfig) {}
13
+
14
+ /**
15
+ * Initializes the agent with the RAG pipeline as a primary tool.
16
+ * Dynamically imports LangChain dependencies to avoid build errors if missing.
17
+ */
18
+ async initialize(chatModel: unknown) {
19
+ try {
20
+ // Dynamic imports for optional dependency support using template strings to hide from TS static analysis
21
+ const { DynamicTool } = await import(`${"@langchain/core/tools"}`);
22
+ const { ChatPromptTemplate, MessagesPlaceholder } = await import(`${"@langchain/core/prompts"}`);
23
+ const { AgentExecutor, createOpenAIFunctionsAgent } = await import(`${"langchain/agents"}`);
24
+
25
+ // 1. Define the RAG Search Tool
26
+ const searchTool = new DynamicTool({
27
+ name: "document_search",
28
+ description: "Use this tool to search through the knowledge base and document repository. Input should be a specific search query.",
29
+ func: async (query: string) => {
30
+ const response = await this.pipeline.ask(query);
31
+ return `Search Results:\n${response.reply}\n\nSources Used: ${JSON.stringify(response.sources.map(s => s.id))}`;
32
+ },
33
+ });
34
+
35
+ const tools = [searchTool];
36
+
37
+ // 2. Define the Agent Prompt
38
+ const prompt = ChatPromptTemplate.fromMessages([
39
+ ["system", this.config.llm.systemPrompt || "You are a helpful AI assistant with access to a document search tool."],
40
+ new MessagesPlaceholder("chat_history"),
41
+ ["human", "{input}"],
42
+ new MessagesPlaceholder("agent_scratchpad"),
43
+ ]);
44
+
45
+ // 3. Create the Agent
46
+ const agent = await createOpenAIFunctionsAgent({
47
+ llm: chatModel,
48
+ tools,
49
+ prompt,
50
+ });
51
+
52
+ this.executor = new AgentExecutor({
53
+ agent,
54
+ tools,
55
+ });
56
+ } catch (error) {
57
+ console.error("[LangChainAgent] Failed to initialize. Ensure 'langchain' and '@langchain/core' are installed.");
58
+ throw error;
59
+ }
60
+ }
61
+
62
+ /**
63
+ * Run the agentic flow.
64
+ */
65
+ async run(input: string, chatHistory: unknown[] = []) {
66
+ if (!this.executor) {
67
+ throw new Error("[LangChainAgent] Agent not initialized. Call initialize() first.");
68
+ }
69
+
70
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
71
+ const response = await (this.executor as any).invoke({
72
+ input,
73
+ chat_history: chatHistory,
74
+ });
75
+
76
+ return response.output;
77
+ }
78
+ }
@@ -1,165 +1,17 @@
1
1
  import { BaseVectorProvider } from '../providers/vectordb/BaseVectorProvider';
2
2
  import { BaseGraphProvider } from '../providers/graphdb/BaseGraphProvider';
3
3
  import { ILLMProvider, ChatMessage } from '../llm/ILLMProvider';
4
- import { DocumentChunker } from '../rag/DocumentChunker';
4
+ import { DocumentChunker, Chunk } from '../rag/DocumentChunker';
5
5
  import { EntityExtractor } from '../rag/EntityExtractor';
6
+ import { Reranker } from '../rag/Reranker';
7
+ import { LlamaIndexIngestor } from '../rag/LlamaIndexIngestor';
8
+ import { LangChainAgent } from './LangChainAgent';
6
9
  import { RagConfig } from '../config/RagConfig';
7
10
  import { ProviderRegistry } from './ProviderRegistry';
8
11
  import { BatchProcessor, BatchOptions } from './BatchProcessor';
9
12
  import { EmbeddingStrategyResolver } from '../config/EmbeddingStrategy';
10
- import { IngestDocument, ChatResponse, GraphSearchResult } from '../types';
11
-
12
- interface QueryFieldHint {
13
- field?: string;
14
- value: string;
15
- }
16
-
17
- interface QueryFilter {
18
- metadata?: Record<string, string>;
19
- keywords?: string[];
20
- queryText?: string;
21
- }
22
-
23
- function normalizeHintValue(value: string): string {
24
- return value.replace(/\s+/g, ' ').trim();
25
- }
26
-
27
- function isLikelyPromptPhrase(value: string): boolean {
28
- return /^(what|which|who|where|when|why|how)\b/i.test(value.trim());
29
- }
30
-
31
- function extractQueryFieldHints(question: string): QueryFieldHint[] {
32
- if (!question.trim()) return [];
33
-
34
- const hints = new Map<string, QueryFieldHint>();
35
-
36
- const addHint = (value: string, field?: string) => {
37
- const normalizedValue = normalizeHintValue(value);
38
- if (!normalizedValue) return;
39
-
40
- const normalizedField = field
41
- ? field
42
- .toLowerCase()
43
- .replace(/[^a-z0-9]+/g, ' ')
44
- .trim()
45
- : undefined;
46
-
47
- const key = `${normalizedField ?? '*'}::${normalizedValue.toLowerCase()}`;
48
- if (!hints.has(key)) {
49
- hints.set(key, {
50
- value: normalizedValue,
51
- ...(normalizedField ? { field: normalizedField } : {}),
52
- });
53
- }
54
- };
55
-
56
- for (const match of question.matchAll(/["']([^"']{2,100})["']/g)) {
57
- addHint(match[1]);
58
- }
59
-
60
- const naturalQuestionPatterns = [
61
- /\b(?:what|which)\s+(?:is|are|was|were)\s+(?:the\s+)?([^?.!,]{1,60}?)\s+of\s+["']?([^"'\n?.!,]{2,120})["']?(?=[?.!,]|$)/gi,
62
- /\b(?:who|what)\s+(?:is|are|was|were)\s+["']?([^"'\n?.!,]{2,120})["']?(?=[?.!,]|$)/gi,
63
- /\b(?:about|for|regarding)\s+["']?([^"'\n?.!,]{2,120})["']?(?=[?.!,]|$)/gi,
64
- ];
65
-
66
- // Targeted patterns: map common person/company question forms to a `name` metadata hint
67
- const personCompanyPatterns = [
68
- /\bcompany(?:\s+name)?\s+(?:of|for)\s+["']?([^"'\n?.!,]{2,120})["']?(?=[?.!,]|$)/gi,
69
- /\b(?:which|what)\s+company\s+does\s+["']?([^"'\n?.!,]{2,120})["']?\s+work(?:\s+for|\s+at)?(?=[?.!,]|$)/gi,
70
- /\bwhere\s+does\s+["']?([^"'\n?.!,]{2,120})["']?\s+work(?:\s+for|\s+at)?(?=[?.!,]|$)/gi,
71
- ];
72
-
73
- for (const pattern of personCompanyPatterns) {
74
- for (const match of question.matchAll(pattern)) {
75
- const name = match[1];
76
- if (name) addHint(name, 'name');
77
- }
78
- }
79
-
80
- // Universal patterns: capture common data types across domains so the system
81
- // can construct metadata filters for arbitrary datasets. These are intentionally
82
- // generic and provider-agnostic (e.g., `email`, `phone`, `date`, `amount`, `id`).
83
- const universalPatterns: Array<{ regex: RegExp; field?: string; group?: number }> = [
84
- { regex: /([a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,})/gi, field: 'email', group: 1 },
85
- { regex: /(\+?\d[\d\-\.\s\(\)]{6,}\d)/g, field: 'phone', group: 1 },
86
- { regex: /(\b\d{4}-\d{2}-\d{2}\b|\b\d{1,2}\/\d{1,2}\/\d{2,4}\b)/g, field: 'date', group: 1 },
87
- { regex: /(\$\s?\d{1,3}(?:,\d{3})*(?:\.\d+)?)/g, field: 'amount', group: 1 },
88
- { regex: /\b(ID|id|identifier)[: ]\s*([A-Za-z0-9\-]{3,})\b/gi, field: 'id', group: 2 },
89
- // Generic quoted phrase / proper-noun sequences as keywords (already partially handled above)
90
- { regex: /"([^"]{2,120})"/g, group: 1 },
91
- { regex: /'([^']{2,120})'/g, group: 1 },
92
- ];
93
-
94
- for (const p of universalPatterns) {
95
- for (const match of question.matchAll(p.regex)) {
96
- const val = p.group ? match[p.group] ?? match[0] : match[0];
97
- if (!val) continue;
98
- if (p.field) addHint(val, p.field);
99
- else addHint(val);
100
- }
101
- }
102
-
103
- for (const pattern of naturalQuestionPatterns) {
104
- for (const match of question.matchAll(pattern)) {
105
- const value = match[2] ?? match[1];
106
- if (value) addHint(value);
107
- }
108
- }
109
-
110
- const fieldPattern = `([^\\n:=?.!,]{1,60}?)`;
111
- const valuePattern = `([^\\n?.!,]{1,120}?)`;
112
- const fieldValuePatterns = [
113
- new RegExp(`\\b${fieldPattern}\\s*(?:=|:)\\s*["']?${valuePattern}["']?(?=[?.!,]|$)`, 'gi'),
114
- new RegExp(`\\b${fieldPattern}\\s+(?:is|are|was|were|equals?|equal to|named|called)\\s+["']?${valuePattern}["']?(?=[?.!,]|$)`, 'gi'),
115
- new RegExp(`\\bwith\\s+${fieldPattern}\\s+["']?${valuePattern}["']?(?=[?.!,]|$)`, 'gi'),
116
- ];
117
-
118
- for (const pattern of fieldValuePatterns) {
119
- for (const match of question.matchAll(pattern)) {
120
- const field = normalizeHintValue(match[1] ?? '');
121
- const value = match[2] ?? '';
122
-
123
- if (field && !isLikelyPromptPhrase(field)) {
124
- addHint(value, field);
125
- } else {
126
- addHint(value);
127
- }
128
- }
129
- }
130
-
131
- for (const match of question.matchAll(/\b[A-Z][a-z]+(?:\s+[A-Z][a-z]+){0,3}\b/g)) {
132
- addHint(match[0]);
133
- }
134
-
135
- return [...hints.values()];
136
- }
137
-
138
- function buildQueryFilter(question: string, hints: QueryFieldHint[]): QueryFilter {
139
- const filter: QueryFilter = { metadata: {}, keywords: [], queryText: question };
140
-
141
- for (const hint of hints) {
142
- if (hint.field) {
143
- // prefer last-seen value for a field; providers may interpret filters differently
144
- filter.metadata![hint.field] = hint.value;
145
- } else {
146
- // treat as keyword / named entity
147
- filter.keywords!.push(hint.value);
148
- }
149
- }
150
-
151
- // Also extract quoted phrases (if not already captured) as keywords
152
- for (const match of question.matchAll(/\b[A-Z][a-z]+(?:\s+[A-Z][a-z]+){0,3}\b/g)) {
153
- const term = normalizeHintValue(match[0]);
154
- if (term && !filter.keywords!.includes(term)) filter.keywords!.push(term);
155
- }
156
-
157
- // Remove empty metadata object when no fields present to keep filter minimal
158
- if (Object.keys(filter.metadata || {}).length === 0) delete filter.metadata;
159
- if (filter.keywords && filter.keywords.length === 0) delete filter.keywords;
160
-
161
- return filter;
162
- }
13
+ import { QueryProcessor } from './QueryProcessor';
14
+ import { IngestDocument, ChatResponse, VectorMatch, GraphSearchResult, RetrievalResult, UpsertDocument } from '../types';
163
15
 
164
16
  /**
165
17
  * Pipeline — orchestrates the RAG flow: Embed → Search → Augment → Generate.
@@ -177,16 +29,22 @@ export class Pipeline {
177
29
  private llmProvider!: ILLMProvider;
178
30
  private embeddingProvider!: ILLMProvider;
179
31
  private chunker: DocumentChunker;
32
+ private llamaIngestor?: LlamaIndexIngestor;
180
33
  private entityExtractor?: EntityExtractor;
181
- private config: RagConfig;
34
+ private reranker: Reranker;
35
+ private agent?: LangChainAgent;
36
+ private embeddingCache = new Map<string, number[]>();
182
37
  private initialised = false;
183
38
 
184
- constructor(config: RagConfig) {
185
- this.config = config;
39
+ constructor(private config: RagConfig) {
186
40
  this.chunker = new DocumentChunker(
187
41
  config.rag?.chunkSize ?? 1000,
188
42
  config.rag?.chunkOverlap ?? 200
189
43
  );
44
+ if (config.rag?.chunkingStrategy === 'llamaindex') {
45
+ this.llamaIngestor = new LlamaIndexIngestor();
46
+ }
47
+ this.reranker = new Reranker();
190
48
  }
191
49
 
192
50
  async initialize(): Promise<void> {
@@ -195,9 +53,7 @@ export class Pipeline {
195
53
  // Resolve vector DB provider
196
54
  this.vectorDB = await ProviderRegistry.createVectorProvider(this.config.vectorDb);
197
55
 
198
- // Resolve LLM + embedding providers using EmbeddingStrategyResolver.
199
- // This handles all cases: integrated (same provider for chat+embed),
200
- // separate (e.g. Anthropic + OpenAI embedding), and external.
56
+ // Resolve LLM + embedding providers
201
57
  const { llmProvider, embeddingProvider } = await EmbeddingStrategyResolver.resolve(
202
58
  this.config.llm,
203
59
  this.config.embedding
@@ -213,6 +69,13 @@ export class Pipeline {
213
69
  }
214
70
 
215
71
  await this.vectorDB.initialize();
72
+
73
+ // Initialize Agentic Layer if configured
74
+ if (this.config.rag?.architecture === 'agentic') {
75
+ this.agent = new LangChainAgent(this, this.config);
76
+ await this.agent.initialize(this.llmProvider as unknown);
77
+ }
78
+
216
79
  this.initialised = true;
217
80
  }
218
81
 
@@ -230,28 +93,9 @@ export class Pipeline {
230
93
 
231
94
  for (const doc of documents) {
232
95
  try {
233
- const chunks = this.chunker.chunk(doc.content, {
234
- docId: doc.docId,
235
- metadata: doc.metadata,
236
- });
237
-
238
- // Batch embed with retry
239
- const embedBatchOptions: BatchOptions = {
240
- batchSize: 50,
241
- maxRetries: 3,
242
- initialDelayMs: 100,
243
- };
244
-
245
- const vectors = await BatchProcessor.mapWithRetry(
246
- chunks.map(c => c.content),
247
- (text) => this.embeddingProvider.embed(text, { taskType: 'document' }),
248
- embedBatchOptions
249
- );
250
-
251
- if (vectors.length !== chunks.length) {
252
- throw new Error(`Embedding mismatch: got ${vectors.length} vectors for ${chunks.length} chunks`);
253
- }
254
-
96
+ const chunks = await this.prepareChunks(doc);
97
+ const vectors = await this.processEmbeddings(chunks);
98
+
255
99
  const upsertDocs = chunks.map((chunk, i) => ({
256
100
  id: chunk.id,
257
101
  vector: vectors[i],
@@ -259,36 +103,15 @@ export class Pipeline {
259
103
  metadata: chunk.metadata,
260
104
  }));
261
105
 
262
- // Batch upsert with retry
263
- const upsertBatchOptions: BatchOptions = {
264
- batchSize: 100,
265
- maxRetries: 3,
266
- initialDelayMs: 100,
267
- };
268
-
269
- const upsertResult = await BatchProcessor.processBatch(
270
- upsertDocs,
271
- (batch) => this.vectorDB.batchUpsert(batch, ns),
272
- upsertBatchOptions
273
- );
274
-
275
- if (upsertResult.errors.length > 0) {
276
- console.warn(`[Pipeline] Ingestion: ${upsertResult.errors.length} batches failed for doc ${doc.docId}`);
277
- }
106
+ const totalProcessed = await this.processUpserts(upsertDocs, ns);
278
107
 
279
108
  results.push({
280
109
  docId: doc.docId,
281
- chunksIngested: upsertResult.totalProcessed,
110
+ chunksIngested: totalProcessed,
282
111
  });
283
112
 
284
- // Graph ingestion
285
113
  if (this.graphDB && this.entityExtractor) {
286
- console.log(`[Pipeline] Extracting entities for doc ${doc.docId}...`);
287
- for (const chunk of chunks) {
288
- const { nodes, edges } = await this.entityExtractor.extract(chunk.content);
289
- if (nodes.length > 0) await this.graphDB.addNodes(nodes);
290
- if (edges.length > 0) await this.graphDB.addEdges(edges);
291
- }
114
+ await this.processGraphIngestion(doc.docId, chunks);
292
115
  }
293
116
  } catch (error) {
294
117
  console.error(`[Pipeline] Failed to ingest document ${doc.docId}:`, error);
@@ -299,8 +122,132 @@ export class Pipeline {
299
122
  return results;
300
123
  }
301
124
 
125
+ /**
126
+ * Step 1: Chunk the document content.
127
+ */
128
+ private async prepareChunks(doc: IngestDocument): Promise<Chunk[]> {
129
+ if (this.llamaIngestor) {
130
+ return await this.llamaIngestor.chunk(doc.content, {
131
+ docId: doc.docId,
132
+ metadata: doc.metadata,
133
+ chunkSize: this.config.rag?.chunkSize,
134
+ chunkOverlap: this.config.rag?.chunkOverlap,
135
+ });
136
+ }
137
+
138
+ return this.chunker.chunk(doc.content, {
139
+ docId: doc.docId,
140
+ metadata: doc.metadata,
141
+ });
142
+ }
143
+
144
+ /**
145
+ * Step 2: Generate embeddings for chunks with retry logic.
146
+ */
147
+ private async processEmbeddings(chunks: Chunk[]): Promise<number[][]> {
148
+ const embedBatchOptions: BatchOptions = {
149
+ batchSize: 50,
150
+ maxRetries: 3,
151
+ initialDelayMs: 100,
152
+ };
153
+
154
+ const vectors = await BatchProcessor.mapWithRetry(
155
+ chunks.map(c => c.content),
156
+ (text) => this.embeddingProvider.embed(text, { taskType: 'document' }),
157
+ embedBatchOptions
158
+ );
159
+
160
+ if (vectors.length !== chunks.length) {
161
+ throw new Error(`Embedding mismatch: got ${vectors.length} vectors for ${chunks.length} chunks`);
162
+ }
163
+
164
+ return vectors;
165
+ }
166
+
167
+ /**
168
+ * Step 3: Upsert chunks to vector database with retry logic.
169
+ */
170
+ private async processUpserts(upsertDocs: UpsertDocument[], namespace: string): Promise<number> {
171
+ const upsertBatchOptions: BatchOptions = {
172
+ batchSize: 100,
173
+ maxRetries: 3,
174
+ initialDelayMs: 100,
175
+ };
176
+
177
+ const upsertResult = await BatchProcessor.processBatch(
178
+ upsertDocs,
179
+ (batch) => this.vectorDB.batchUpsert(batch, namespace),
180
+ upsertBatchOptions
181
+ );
182
+
183
+ if (upsertResult.errors.length > 0) {
184
+ console.warn(`[Pipeline] Ingestion: ${upsertResult.errors.length} batches failed`);
185
+ }
186
+
187
+ return upsertResult.totalProcessed;
188
+ }
189
+
190
+ /**
191
+ * Step 4: Optional graph-based entity extraction and ingestion.
192
+ */
193
+ private async processGraphIngestion(docId: string | number, chunks: Chunk[]): Promise<void> {
194
+ console.log(`[Pipeline] Extracting entities for doc ${docId} (${chunks.length} chunks)...`);
195
+
196
+ const extractionOptions: BatchOptions = {
197
+ batchSize: 2, // Low concurrency for LLM extraction
198
+ maxRetries: 1,
199
+ initialDelayMs: 500,
200
+ };
201
+
202
+ await BatchProcessor.processBatch(
203
+ chunks,
204
+ async (batch) => {
205
+ for (const chunk of batch) {
206
+ try {
207
+ const { nodes, edges } = await this.entityExtractor!.extract(chunk.content);
208
+ if (nodes.length > 0) await this.graphDB!.addNodes(nodes);
209
+ if (edges.length > 0) await this.graphDB!.addEdges(edges);
210
+ } catch (err) {
211
+ console.warn(`[Pipeline] Entity extraction failed for chunk:`, err);
212
+ }
213
+ }
214
+ },
215
+ extractionOptions
216
+ );
217
+ }
218
+
302
219
  async ask(question: string, history: ChatMessage[] = [], namespace?: string): Promise<ChatResponse> {
303
220
  await this.initialize();
221
+
222
+ if (this.config.rag?.architecture === 'agentic' && this.agent) {
223
+ console.log('[Pipeline] 🤖 Executing in Agentic Mode...');
224
+ const agentReply = await this.agent.run(question, history);
225
+ return { reply: agentReply, sources: [] };
226
+ }
227
+
228
+ const stream = this.askStream(question, history, namespace);
229
+ let reply = '';
230
+ let sources: VectorMatch[] = [];
231
+ let graphData: GraphSearchResult | undefined;
232
+
233
+ for await (const chunk of stream) {
234
+ if (typeof chunk === 'string') {
235
+ reply += chunk;
236
+ } else if ('sources' in chunk) {
237
+ sources = chunk.sources;
238
+ graphData = chunk.graphData;
239
+ }
240
+ }
241
+
242
+ return { reply, sources, graphData };
243
+ }
244
+
245
+ /**
246
+ * High-performance streaming RAG flow.
247
+ * Yields text chunks first, then the retrieval metadata at the end.
248
+ */
249
+ async *askStream(question: string, history: ChatMessage[] = [], namespace?: string): AsyncIterable<string | ChatResponse> {
250
+ await this.initialize();
304
251
  const ns = namespace ?? this.config.projectId;
305
252
  const topK = this.config.rag?.topK ?? 5;
306
253
  const scoreThreshold = this.config.rag?.scoreThreshold ?? 0.0;
@@ -308,24 +255,27 @@ export class Pipeline {
308
255
  try {
309
256
  let searchQuery = question;
310
257
 
311
- // 1. Query Transformation
258
+ // 1. Query Pre-processing
312
259
  if (this.config.rag?.useQueryTransformation) {
313
260
  searchQuery = await this.rewriteQuery(question, history);
314
261
  }
315
262
 
316
- // 2. Vector Retrieval
317
- const queryVector = await this.embeddingProvider.embed(searchQuery, { taskType: 'query' });
318
- const fieldHints = extractQueryFieldHints(question);
319
- const filter = buildQueryFilter(question, fieldHints) as Record<string, unknown>;
320
- filter.__entityHints = fieldHints;
263
+ // 2. Parallel Retrieval
264
+ const hints = QueryProcessor.extractQueryFieldHints(question);
265
+ const filter = QueryProcessor.buildQueryFilter(question, hints);
321
266
 
322
- const rawMatches = await this.vectorDB.query(queryVector, topK, ns, filter);
323
- const sources = rawMatches.filter((m) => m.score >= scoreThreshold);
267
+ const { sources: rawSources, graphData } = await this.retrieve(searchQuery, {
268
+ namespace: ns,
269
+ topK: topK * 2,
270
+ filter
271
+ });
324
272
 
325
- // 3. Graph Retrieval
326
- let graphData: GraphSearchResult | undefined;
327
- if (this.graphDB && this.config.rag?.useGraphRetrieval) {
328
- graphData = await this.graphDB.query(searchQuery);
273
+ // 3. Reranking
274
+ let sources = rawSources.filter((m) => m.score >= scoreThreshold);
275
+ if (this.config.rag?.useReranking) {
276
+ sources = await this.reranker.rerank(sources, question, topK);
277
+ } else {
278
+ sources = sources.slice(0, topK);
329
279
  }
330
280
 
331
281
  // 4. Context Augmentation
@@ -340,16 +290,54 @@ export class Pipeline {
340
290
  context = `GRAPH KNOWLEDGE:\n${graphContext}\n\nVECTOR CONTEXT:\n${context}`;
341
291
  }
342
292
 
343
- // 5. Generation
293
+ // 5. Generation (Streaming)
344
294
  const messages: ChatMessage[] = [...history, { role: 'user', content: question }];
345
- const reply = await this.llmProvider.chat(messages, context);
295
+
296
+ if (this.llmProvider.chatStream) {
297
+ for await (const chunk of this.llmProvider.chatStream(messages, context)) {
298
+ yield chunk;
299
+ }
300
+ } else {
301
+ const reply = await this.llmProvider.chat(messages, context);
302
+ yield reply;
303
+ }
304
+
305
+ // Yield retrieval metadata at the end so UI can show sources
306
+ yield { reply: '', sources, graphData };
346
307
 
347
- return { reply, sources, graphData };
348
308
  } catch (error) {
349
- throw new Error(`[Pipeline] Chat failed: ${error instanceof Error ? error.message : String(error)}`);
309
+ throw new Error(`[Pipeline] Stream failed: ${error instanceof Error ? error.message : String(error)}`);
350
310
  }
351
311
  }
352
312
 
313
+ /**
314
+ * Universal retrieval method combining all enabled providers.
315
+ */
316
+ async retrieve(query: string, options: { namespace?: string; topK?: number; filter?: Record<string, unknown> }): Promise<RetrievalResult> {
317
+ const ns = options.namespace ?? this.config.projectId;
318
+ const topK = options.topK ?? 5;
319
+
320
+ // Performance Optimization: Cache query embeddings to avoid redundant LLM calls
321
+ const cacheKey = `${ns}::${query}`;
322
+ let queryVector = this.embeddingCache.get(cacheKey);
323
+
324
+ const [retrievedVector, graphData] = await Promise.all([
325
+ queryVector ? Promise.resolve(queryVector) : this.embeddingProvider.embed(query, { taskType: 'query' }),
326
+ this.graphDB && this.config.rag?.useGraphRetrieval
327
+ ? this.graphDB.query(query)
328
+ : Promise.resolve(undefined)
329
+ ]);
330
+
331
+ if (!queryVector) {
332
+ this.embeddingCache.set(cacheKey, retrievedVector);
333
+ queryVector = retrievedVector;
334
+ }
335
+
336
+ const sources = await this.vectorDB.query(queryVector, topK, ns, options.filter);
337
+
338
+ return { sources, graphData };
339
+ }
340
+
353
341
  /**
354
342
  * Rewrite the user query for better retrieval performance.
355
343
  */