@airoom/nextmin-node 2.0.1 → 2.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +151 -0
- package/dist/api/apiRouter.d.ts +28 -1
- package/dist/api/apiRouter.js +138 -8
- package/dist/api/router/mountBatchRoutes.d.ts +2 -0
- package/dist/api/router/mountBatchRoutes.js +90 -0
- package/dist/api/router/mountCrudRoutes.js +79 -20
- package/dist/api/router/setupAuthRoutes.js +475 -38
- package/dist/api/router/setupChatWidgetRoutes.d.ts +3 -0
- package/dist/api/router/setupChatWidgetRoutes.js +207 -0
- package/dist/api/router/setupFileRoutes.js +258 -12
- package/dist/api/router/utils.d.ts +4 -1
- package/dist/api/router/utils.js +25 -7
- package/dist/cli.d.ts +1 -0
- package/dist/cli.js +145 -64
- package/dist/database/DatabaseAdapter.d.ts +2 -0
- package/dist/database/NMAdapter.d.ts +11 -0
- package/dist/database/NMAdapter.js +688 -73
- package/dist/database/QueryEngine.js +7 -4
- package/dist/files/FileStorageAdapter.d.ts +6 -0
- package/dist/files/LocalFileStorageAdapter.d.ts +6 -0
- package/dist/files/LocalFileStorageAdapter.js +39 -3
- package/dist/files/S3FileStorageAdapter.d.ts +8 -1
- package/dist/files/S3FileStorageAdapter.js +86 -11
- package/dist/files/filename.js +6 -4
- package/dist/index.d.ts +1 -0
- package/dist/index.js +3 -1
- package/dist/models/BaseModel.d.ts +19 -2
- package/dist/models/BaseModel.js +4 -4
- package/dist/policy/authorize.d.ts +1 -1
- package/dist/policy/authorize.js +64 -13
- package/dist/schemas/Users.json +20 -10
- package/dist/services/AggregateService.d.ts +8 -0
- package/dist/services/AggregateService.js +87 -0
- package/dist/services/IndexingService.d.ts +24 -0
- package/dist/services/IndexingService.js +555 -0
- package/dist/services/LocalAIService.d.ts +34 -0
- package/dist/services/LocalAIService.js +471 -0
- package/dist/services/OpenAIService.d.ts +16 -0
- package/dist/services/OpenAIService.js +586 -0
- package/dist/utils/Events.js +1 -1
- package/dist/utils/Logger.d.ts +2 -1
- package/dist/utils/Logger.js +15 -6
- package/dist/utils/SchemaLoader.d.ts +2 -2
- package/dist/utils/SchemaLoader.js +44 -6
- package/package.json +12 -4
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
export declare class LocalAIService {
|
|
2
|
+
private static llamaInstance;
|
|
3
|
+
private static modelInstance;
|
|
4
|
+
private static contextInstance;
|
|
5
|
+
private static activeSessions;
|
|
6
|
+
private static cleanupInterval;
|
|
7
|
+
constructor(localModelPath?: string);
|
|
8
|
+
private static initCleanupTimer;
|
|
9
|
+
/**
|
|
10
|
+
* Lazily loads the Llama model and context into memory to ensure fast response times
|
|
11
|
+
* and avoid reading the multi-gigabyte GGUF model file from disk repeatedly.
|
|
12
|
+
*/
|
|
13
|
+
private getLlamaModelAndContext;
|
|
14
|
+
/**
|
|
15
|
+
* Simple TF-IDF local search on scraped pages stored in .nextmin/scraped-pages
|
|
16
|
+
*/
|
|
17
|
+
/**
|
|
18
|
+
* Simple TF-IDF local search on scraped pages stored in .nextmin/scraped-pages
|
|
19
|
+
* with smart paragraph-based chunking and path boosting.
|
|
20
|
+
*/
|
|
21
|
+
private localSearch;
|
|
22
|
+
/**
|
|
23
|
+
* Generates chat answer using local node-llama-cpp execution.
|
|
24
|
+
*/
|
|
25
|
+
generateAnswer(history: Array<{
|
|
26
|
+
role: string;
|
|
27
|
+
content: string;
|
|
28
|
+
}>, currentMessage: string, systemPromptText?: string, localModelPath?: string, threads?: number, threadId?: string, onTextChunk?: (chunk: string) => void): Promise<string>;
|
|
29
|
+
/**
|
|
30
|
+
* Prewarm helper to load the model and context in the background
|
|
31
|
+
*/
|
|
32
|
+
prewarm(localModelPath?: string, threads?: number): Promise<void>;
|
|
33
|
+
private importLlamaCpp;
|
|
34
|
+
}
|
|
@@ -0,0 +1,471 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
+
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
+
};
|
|
5
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.LocalAIService = void 0;
|
|
7
|
+
const fs_1 = __importDefault(require("fs"));
|
|
8
|
+
const path_1 = __importDefault(require("path"));
|
|
9
|
+
const Logger_1 = __importDefault(require("../utils/Logger"));
|
|
10
|
+
class LocalAIService {
|
|
11
|
+
constructor(localModelPath) {
|
|
12
|
+
// Constructor kept for signature compatibility
|
|
13
|
+
}
|
|
14
|
+
static initCleanupTimer() {
|
|
15
|
+
if (LocalAIService.cleanupInterval)
|
|
16
|
+
return;
|
|
17
|
+
LocalAIService.cleanupInterval = setInterval(() => {
|
|
18
|
+
const now = Date.now();
|
|
19
|
+
const TTL = 10 * 60 * 1000; // 10 minutes
|
|
20
|
+
for (const [threadId, entry] of LocalAIService.activeSessions.entries()) {
|
|
21
|
+
if (now - entry.lastAccessed > TTL) {
|
|
22
|
+
try {
|
|
23
|
+
const seq = entry.sequence || entry.session?.contextSequence || entry.session;
|
|
24
|
+
if (seq && typeof seq.dispose === 'function') {
|
|
25
|
+
seq.dispose();
|
|
26
|
+
}
|
|
27
|
+
Logger_1.default.info('LocalAIService', `Disposed inactive chat sequence for thread: ${threadId}`);
|
|
28
|
+
}
|
|
29
|
+
catch (err) {
|
|
30
|
+
Logger_1.default.warn('LocalAIService', `Failed to dispose inactive sequence for thread: ${threadId}`, err);
|
|
31
|
+
}
|
|
32
|
+
LocalAIService.activeSessions.delete(threadId);
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
}, 60000);
|
|
36
|
+
if (LocalAIService.cleanupInterval?.unref) {
|
|
37
|
+
LocalAIService.cleanupInterval.unref();
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Lazily loads the Llama model and context into memory to ensure fast response times
|
|
42
|
+
* and avoid reading the multi-gigabyte GGUF model file from disk repeatedly.
|
|
43
|
+
*/
|
|
44
|
+
async getLlamaModelAndContext(localModelPath, threads) {
|
|
45
|
+
if (LocalAIService.modelInstance && LocalAIService.contextInstance) {
|
|
46
|
+
return { model: LocalAIService.modelInstance, context: LocalAIService.contextInstance };
|
|
47
|
+
}
|
|
48
|
+
Logger_1.default.info('LocalAIService', 'Initializing node-llama-cpp and loading model/context...');
|
|
49
|
+
const modelPath = localModelPath || process.env.LOCAL_MODEL_PATH || path_1.default.join(process.cwd(), 'model', 'gemma-4-E2B-it-Q3_K_M.gguf');
|
|
50
|
+
if (!fs_1.default.existsSync(modelPath)) {
|
|
51
|
+
throw new Error(`Local model not found at: ${modelPath}. Please place the Gemma GGUF model file at this location.`);
|
|
52
|
+
}
|
|
53
|
+
try {
|
|
54
|
+
const { getLlama } = await this.importLlamaCpp();
|
|
55
|
+
const llama = await getLlama({
|
|
56
|
+
gpu: false
|
|
57
|
+
});
|
|
58
|
+
LocalAIService.llamaInstance = llama;
|
|
59
|
+
const model = await llama.loadModel({
|
|
60
|
+
modelPath: modelPath
|
|
61
|
+
});
|
|
62
|
+
LocalAIService.modelInstance = model;
|
|
63
|
+
Logger_1.default.info('LocalAIService', `Model loaded successfully from ${modelPath}`);
|
|
64
|
+
const threadCount = threads || parseInt(process.env.LLAMA_THREADS || '3', 10);
|
|
65
|
+
const context = await model.createContext({
|
|
66
|
+
threads: threadCount,
|
|
67
|
+
contextSize: 4096,
|
|
68
|
+
sequences: 2
|
|
69
|
+
});
|
|
70
|
+
LocalAIService.contextInstance = context;
|
|
71
|
+
Logger_1.default.info('LocalAIService', `Context created successfully with ${threadCount} threads and 2 sequence slots.`);
|
|
72
|
+
return { model, context };
|
|
73
|
+
}
|
|
74
|
+
catch (err) {
|
|
75
|
+
Logger_1.default.error('LocalAIService', 'Failed to initialize node-llama-cpp or load the model/context', err);
|
|
76
|
+
throw err;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* Simple TF-IDF local search on scraped pages stored in .nextmin/scraped-pages
|
|
81
|
+
*/
|
|
82
|
+
/**
|
|
83
|
+
* Simple TF-IDF local search on scraped pages stored in .nextmin/scraped-pages
|
|
84
|
+
* with smart paragraph-based chunking and path boosting.
|
|
85
|
+
*/
|
|
86
|
+
async localSearch(query, limit = 6) {
|
|
87
|
+
const dir = path_1.default.join(process.cwd(), '.nextmin', 'scraped-pages');
|
|
88
|
+
if (!fs_1.default.existsSync(dir)) {
|
|
89
|
+
Logger_1.default.warn('LocalAIService', `Scraped pages directory does not exist at ${dir}`);
|
|
90
|
+
return '';
|
|
91
|
+
}
|
|
92
|
+
const files = fs_1.default.readdirSync(dir).filter(f => f.endsWith('.txt'));
|
|
93
|
+
if (files.length === 0) {
|
|
94
|
+
Logger_1.default.warn('LocalAIService', 'No scraped pages found for local search.');
|
|
95
|
+
return '';
|
|
96
|
+
}
|
|
97
|
+
// Helper to tokenize and normalize words (supports Unicode/Bengali characters)
|
|
98
|
+
const tokenize = (text) => {
|
|
99
|
+
return text
|
|
100
|
+
.toLowerCase()
|
|
101
|
+
.replace(/[^\p{L}\p{M}\p{N}\s-]/gu, ' ')
|
|
102
|
+
.split(/[\s_]+/)
|
|
103
|
+
.filter(w => w.length > 2);
|
|
104
|
+
};
|
|
105
|
+
// Fuzzy-normalize the query: fix common typos and aliases for known product terms.
|
|
106
|
+
// Simple edit-distance check: if a word is within 1-2 chars of a known term, correct it.
|
|
107
|
+
const knownTerms = ['nextmin', 'nextmin-node', 'nextmin-react', 'nmadapter'];
|
|
108
|
+
const normalizeQuery = (q) => {
|
|
109
|
+
return q.replace(/\b\w+\b/g, (word) => {
|
|
110
|
+
const lower = word.toLowerCase();
|
|
111
|
+
for (const term of knownTerms) {
|
|
112
|
+
if (lower === term)
|
|
113
|
+
return word; // exact match, keep
|
|
114
|
+
// Simple typo check: same length ±1, and share most chars
|
|
115
|
+
if (Math.abs(lower.length - term.length) <= 1) {
|
|
116
|
+
let mismatches = 0;
|
|
117
|
+
const shorter = lower.length < term.length ? lower : term;
|
|
118
|
+
const longer = lower.length < term.length ? term : lower;
|
|
119
|
+
let si = 0;
|
|
120
|
+
for (let li = 0; li < longer.length && si < shorter.length; li++) {
|
|
121
|
+
if (longer[li] !== shorter[si]) {
|
|
122
|
+
mismatches++;
|
|
123
|
+
}
|
|
124
|
+
else {
|
|
125
|
+
si++;
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
mismatches += (longer.length - si); // remaining unmatched chars
|
|
129
|
+
if (mismatches <= 2 && shorter.length >= 5)
|
|
130
|
+
return term; // correct to known term
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
return word;
|
|
134
|
+
});
|
|
135
|
+
};
|
|
136
|
+
const normalizedQuery = normalizeQuery(query);
|
|
137
|
+
if (normalizedQuery !== query) {
|
|
138
|
+
Logger_1.default.info('LocalAIService', `Query normalized: "${query}" → "${normalizedQuery}"`);
|
|
139
|
+
}
|
|
140
|
+
const queryTokens = tokenize(normalizedQuery);
|
|
141
|
+
if (queryTokens.length === 0)
|
|
142
|
+
return '';
|
|
143
|
+
const chunks = [];
|
|
144
|
+
const termDocCount = new Map();
|
|
145
|
+
// 1. Process files into semantic chunks
|
|
146
|
+
for (const file of files) {
|
|
147
|
+
try {
|
|
148
|
+
const rawContent = fs_1.default.readFileSync(path_1.default.join(dir, file), 'utf-8');
|
|
149
|
+
// Decode HTML entities so scraped text is clean for TF-IDF and LLM context
|
|
150
|
+
let content = rawContent
|
|
151
|
+
.replace(/&/g, '&')
|
|
152
|
+
.replace(/</g, '<')
|
|
153
|
+
.replace(/>/g, '>')
|
|
154
|
+
.replace(/"/g, '"')
|
|
155
|
+
.replace(/'/g, "'")
|
|
156
|
+
.replace(/ /g, ' ');
|
|
157
|
+
// Clean repeating header menu and footer disclaimer to keep context 100% focused on page content
|
|
158
|
+
content = content.replace(/\[Home\]\(\/\)[\s\S]*?open navigation menu/gi, '');
|
|
159
|
+
content = content.replace(/Disclaimer: Information on Doctors24[\s\S]*/gi, '');
|
|
160
|
+
let name = file.replace(/^page-/, '').replace(/\.txt$/, '');
|
|
161
|
+
if (name.startsWith('_'))
|
|
162
|
+
name = name.slice(1);
|
|
163
|
+
const urlPath = '/' + name.replace(/_/g, '/');
|
|
164
|
+
// Tokenize path name and boost it (3x weight multiplier)
|
|
165
|
+
const pathTokens = tokenize(name.replace(/_/g, ' '));
|
|
166
|
+
const boostedPathTokens = [];
|
|
167
|
+
for (const token of pathTokens) {
|
|
168
|
+
boostedPathTokens.push(token, token, token);
|
|
169
|
+
}
|
|
170
|
+
// Split by paragraphs
|
|
171
|
+
const paragraphs = content.split(/\n\s*\n+/);
|
|
172
|
+
let currentChunkText = '';
|
|
173
|
+
const saveChunk = (text) => {
|
|
174
|
+
const cleanText = text.trim();
|
|
175
|
+
if (cleanText.length < 20)
|
|
176
|
+
return;
|
|
177
|
+
const bodyTokens = tokenize(cleanText);
|
|
178
|
+
const chunkTokens = [...bodyTokens, ...boostedPathTokens];
|
|
179
|
+
chunks.push({
|
|
180
|
+
file,
|
|
181
|
+
urlPath,
|
|
182
|
+
content: cleanText,
|
|
183
|
+
tokens: chunkTokens,
|
|
184
|
+
length: chunkTokens.length
|
|
185
|
+
});
|
|
186
|
+
};
|
|
187
|
+
for (const para of paragraphs) {
|
|
188
|
+
const cleanPara = para.trim();
|
|
189
|
+
if (!cleanPara)
|
|
190
|
+
continue;
|
|
191
|
+
// Group smaller paragraphs to make substantial chunks of 800-1200 chars
|
|
192
|
+
if (currentChunkText.length + cleanPara.length < 1000) {
|
|
193
|
+
currentChunkText += (currentChunkText ? '\n\n' : '') + cleanPara;
|
|
194
|
+
}
|
|
195
|
+
else {
|
|
196
|
+
if (currentChunkText) {
|
|
197
|
+
saveChunk(currentChunkText);
|
|
198
|
+
}
|
|
199
|
+
currentChunkText = cleanPara;
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
if (currentChunkText) {
|
|
203
|
+
saveChunk(currentChunkText);
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
catch (err) {
|
|
207
|
+
// Skip unreadable files
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
const totalChunks = chunks.length;
|
|
211
|
+
if (totalChunks === 0)
|
|
212
|
+
return '';
|
|
213
|
+
// 2. Count document frequency of each token in the chunk corpus
|
|
214
|
+
for (const chunk of chunks) {
|
|
215
|
+
const uniqueTokens = new Set(chunk.tokens);
|
|
216
|
+
for (const token of uniqueTokens) {
|
|
217
|
+
termDocCount.set(token, (termDocCount.get(token) || 0) + 1);
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
// 3. Compute TF-IDF score for each chunk
|
|
221
|
+
const scores = [];
|
|
222
|
+
for (const chunk of chunks) {
|
|
223
|
+
let score = 0;
|
|
224
|
+
for (const token of queryTokens) {
|
|
225
|
+
const tf = chunk.tokens.filter(t => t === token).length;
|
|
226
|
+
if (tf > 0) {
|
|
227
|
+
const docFreq = termDocCount.get(token) || 1;
|
|
228
|
+
const idf = Math.log(1 + totalChunks / docFreq);
|
|
229
|
+
score += tf * idf;
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
if (score > 0) {
|
|
233
|
+
// Normalize score by square root of length to prevent long chunk bias
|
|
234
|
+
const normScore = score / Math.sqrt(chunk.length + 1);
|
|
235
|
+
// Match ratio check using only INFORMATIVE query tokens (high IDF).
|
|
236
|
+
// Tokens appearing in >50% of all chunks (like "nextmin" in every nav bar)
|
|
237
|
+
// are corpus stop-words and should not count toward relevance matching.
|
|
238
|
+
const bodyTokenSet = new Set(tokenize(chunk.content));
|
|
239
|
+
const informativeQueryTokens = [...new Set(queryTokens)].filter(t => {
|
|
240
|
+
const docFreq = termDocCount.get(t) || 0;
|
|
241
|
+
return docFreq > 0 && docFreq / totalChunks < 0.5; // appears in <50% of chunks
|
|
242
|
+
});
|
|
243
|
+
// If there are no informative tokens in the query (e.g. pure greeting),
|
|
244
|
+
// let it through — the greeting check handles that case later.
|
|
245
|
+
if (informativeQueryTokens.length === 0) {
|
|
246
|
+
scores.push({ chunk, score: normScore });
|
|
247
|
+
continue;
|
|
248
|
+
}
|
|
249
|
+
const matchedCount = informativeQueryTokens.filter(t => bodyTokenSet.has(t)).length;
|
|
250
|
+
const matchRatio = matchedCount / informativeQueryTokens.length;
|
|
251
|
+
// Require either 2+ informative token matches, OR ≥50% match ratio (for short queries).
|
|
252
|
+
// Single coincidental matches (e.g. "price" in an API code example) don't count.
|
|
253
|
+
const meetsThreshold = matchedCount >= 2 || (informativeQueryTokens.length <= 2 && matchRatio >= 0.5);
|
|
254
|
+
if (meetsThreshold) {
|
|
255
|
+
scores.push({ chunk, score: normScore });
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
// 4. Sort, apply minimum score threshold, and return top chunks
|
|
260
|
+
scores.sort((a, b) => b.score - a.score);
|
|
261
|
+
// Minimum relevance threshold: if the best chunk score is too low,
|
|
262
|
+
// none of the content is actually relevant to this query — return empty
|
|
263
|
+
// so the caller can show the friendly fallback message instead.
|
|
264
|
+
const MIN_SCORE_THRESHOLD = 0.15;
|
|
265
|
+
const relevantScores = scores.filter(s => s.score >= MIN_SCORE_THRESHOLD);
|
|
266
|
+
if (relevantScores.length === 0) {
|
|
267
|
+
Logger_1.default.info('LocalAIService', `RAG: no chunks above threshold (${MIN_SCORE_THRESHOLD}) for query: "${query}". Best score: ${scores[0]?.score.toFixed(4) ?? 'none'}`);
|
|
268
|
+
return '';
|
|
269
|
+
}
|
|
270
|
+
const topMatches = relevantScores.slice(0, limit);
|
|
271
|
+
Logger_1.default.info('LocalAIService', `RAG retrieved ${topMatches.length} relevant chunks for query: "${query}"`);
|
|
272
|
+
topMatches.forEach((m, i) => {
|
|
273
|
+
Logger_1.default.info('LocalAIService', ` Chunk ${i + 1}: score=${m.score.toFixed(4)} page=${m.chunk.urlPath} snippet="${m.chunk.content.slice(0, 80).replace(/\n/g, ' ')}..."`);
|
|
274
|
+
});
|
|
275
|
+
const contextParts = [];
|
|
276
|
+
// Limit to top 3 chunks and truncate each to 600 chars to stay within Gemma's context window
|
|
277
|
+
for (const match of topMatches.slice(0, 3)) {
|
|
278
|
+
const c = match.chunk;
|
|
279
|
+
const truncated = c.content.length > 600 ? c.content.slice(0, 600) + '...' : c.content;
|
|
280
|
+
contextParts.push(`SOURCE PAGE: ${c.urlPath}\n${truncated}`);
|
|
281
|
+
}
|
|
282
|
+
return contextParts.join('\n\n');
|
|
283
|
+
}
|
|
284
|
+
/**
|
|
285
|
+
* Generates chat answer using local node-llama-cpp execution.
|
|
286
|
+
*/
|
|
287
|
+
async generateAnswer(history, currentMessage, systemPromptText, localModelPath, threads, threadId, onTextChunk) {
|
|
288
|
+
try {
|
|
289
|
+
// Get all previous user messages from history
|
|
290
|
+
const userHistory = history.filter(h => h.role === 'user').map(h => h.content);
|
|
291
|
+
const searchQuery = userHistory.length > 0
|
|
292
|
+
? `${userHistory[userHistory.length - 1]} ${currentMessage}`
|
|
293
|
+
: currentMessage;
|
|
294
|
+
// 1. Perform local TF-IDF matching on the scraped pages for context using the context-aware query
|
|
295
|
+
const context = await this.localSearch(searchQuery);
|
|
296
|
+
// If no context found, check if it's a greeting/chit-chat first.
|
|
297
|
+
// Greetings still get a warm LLM reply; substantive questions get a friendly fallback.
|
|
298
|
+
if (!context) {
|
|
299
|
+
const lowerMsg = currentMessage.toLowerCase().trim();
|
|
300
|
+
const isGreeting = /^(hi|hello|hey|howdy|hiya|yo|sup|good\s+(morning|afternoon|evening|day)|how are you|what can you do|who are you|what('s| is) up)/.test(lowerMsg);
|
|
301
|
+
if (!isGreeting) {
|
|
302
|
+
Logger_1.default.warn('LocalAIService', `No context found for query: "${currentMessage}". Returning friendly fallback.`);
|
|
303
|
+
return "Thanks for asking! 😊 I don't have enough information on that just yet, but we're working on it. Please check back soon or visit our website for the latest details!";
|
|
304
|
+
}
|
|
305
|
+
Logger_1.default.info('LocalAIService', `No context for greeting query: "${currentMessage}". Routing to LLM.`);
|
|
306
|
+
}
|
|
307
|
+
// Build system prompt that grounds the model in the retrieved context
|
|
308
|
+
const contextBlock = context
|
|
309
|
+
? `\n\n===== RELEVANT WEBSITE CONTENT =====\n${context}\n===== END OF WEBSITE CONTENT =====`
|
|
310
|
+
: '';
|
|
311
|
+
// Build standard RAG guidelines to prevent link hallucinations and force descriptions
|
|
312
|
+
const ragInstructions = `
|
|
313
|
+
CRITICAL RAG INSTRUCTIONS:
|
|
314
|
+
- Base your answers on the website content provided and the system instructions.
|
|
315
|
+
- STRICT URL FORMATTING RULES:
|
|
316
|
+
* For any link, you MUST use the exact path from the SOURCE PAGE line (which will always start with "/"). Example: If the source page is "SOURCE PAGE: /some-path", the link MUST be exactly "[base-url]/some-path". Do NOT change, nest, or modify the prefix or structure.
|
|
317
|
+
* DO NOT make up or guess any URL paths or slugs. Only use the exact path provided in the "SOURCE PAGE: [path]" line.
|
|
318
|
+
- Do not guess, make up, or hallucinate any URL paths, details, or names that are not explicitly shown in the website content or defined in the system instructions.
|
|
319
|
+
- When suggesting a page, section, or link, always write a short description first to explain it before sharing the link. Do not output raw links.
|
|
320
|
+
- Only output up to 3-5 highly relevant links.
|
|
321
|
+
- If the user asks for a specific URL, check if it's in the website content or system instructions. If it is, output it directly.`;
|
|
322
|
+
const baseSystemPrompt = systemPromptText || `You are a helpful website assistant. Answer questions using the website content provided.`;
|
|
323
|
+
const systemWithContext = `${baseSystemPrompt}\n\n${ragInstructions}${contextBlock}`;
|
|
324
|
+
// 2. Build the user prompt — inject context at message level too so Gemma/Llama can't miss it
|
|
325
|
+
const userPrompt = context
|
|
326
|
+
? `Using the website content provided in the system instructions, answer this question:\n\n${currentMessage}`
|
|
327
|
+
: currentMessage;
|
|
328
|
+
// 3. Initialize/fetch Llama model and context
|
|
329
|
+
const { context: contextInstance } = await this.getLlamaModelAndContext(localModelPath, threads);
|
|
330
|
+
const useCache = !!threadId;
|
|
331
|
+
let sequence;
|
|
332
|
+
let tempSequence = null;
|
|
333
|
+
if (useCache && LocalAIService.activeSessions.has(threadId)) {
|
|
334
|
+
const entry = LocalAIService.activeSessions.get(threadId);
|
|
335
|
+
sequence = entry.sequence;
|
|
336
|
+
if (entry.session && !sequence) {
|
|
337
|
+
sequence = entry.session.contextSequence;
|
|
338
|
+
}
|
|
339
|
+
entry.sequence = sequence;
|
|
340
|
+
delete entry.session;
|
|
341
|
+
entry.lastAccessed = Date.now();
|
|
342
|
+
Logger_1.default.info('LocalAIService', `Reusing existing chat sequence for thread: ${threadId}`);
|
|
343
|
+
}
|
|
344
|
+
else {
|
|
345
|
+
const MAX_SEQUENCES = 2;
|
|
346
|
+
if (useCache && LocalAIService.activeSessions.size >= MAX_SEQUENCES) {
|
|
347
|
+
let oldestThreadId = null;
|
|
348
|
+
let oldestTime = Infinity;
|
|
349
|
+
for (const [tid, entry] of LocalAIService.activeSessions.entries()) {
|
|
350
|
+
if (entry.lastAccessed < oldestTime) {
|
|
351
|
+
oldestTime = entry.lastAccessed;
|
|
352
|
+
oldestThreadId = tid;
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
if (oldestThreadId) {
|
|
356
|
+
try {
|
|
357
|
+
const entry = LocalAIService.activeSessions.get(oldestThreadId);
|
|
358
|
+
const seqToDispose = entry.sequence || entry.session?.contextSequence || entry.session;
|
|
359
|
+
if (seqToDispose && typeof seqToDispose.dispose === 'function') {
|
|
360
|
+
seqToDispose.dispose();
|
|
361
|
+
}
|
|
362
|
+
Logger_1.default.info('LocalAIService', `Evicted oldest active sequence ${oldestThreadId} to free a sequence slot.`);
|
|
363
|
+
}
|
|
364
|
+
catch (err) {
|
|
365
|
+
Logger_1.default.warn('LocalAIService', `Failed to dispose evicted sequence ${oldestThreadId}`, err);
|
|
366
|
+
}
|
|
367
|
+
LocalAIService.activeSessions.delete(oldestThreadId);
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
sequence = contextInstance.getSequence();
|
|
371
|
+
if (!useCache) {
|
|
372
|
+
tempSequence = sequence;
|
|
373
|
+
}
|
|
374
|
+
if (useCache) {
|
|
375
|
+
LocalAIService.activeSessions.set(threadId, { sequence, lastAccessed: Date.now() });
|
|
376
|
+
Logger_1.default.info('LocalAIService', `Created new chat sequence for thread: ${threadId}`);
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
const { LlamaChatSession } = await this.importLlamaCpp();
|
|
380
|
+
const session = new LlamaChatSession({
|
|
381
|
+
contextSequence: sequence,
|
|
382
|
+
systemPrompt: systemWithContext
|
|
383
|
+
});
|
|
384
|
+
// 4. Format history for node-llama-cpp v3
|
|
385
|
+
const llamaHistory = [];
|
|
386
|
+
let tempUserText = '';
|
|
387
|
+
for (const msg of history) {
|
|
388
|
+
if (msg.role === 'user') {
|
|
389
|
+
tempUserText = msg.content;
|
|
390
|
+
}
|
|
391
|
+
else if (msg.role === 'model' || msg.role === 'assistant') {
|
|
392
|
+
if (tempUserText) {
|
|
393
|
+
llamaHistory.push({ role: 'user', text: tempUserText });
|
|
394
|
+
tempUserText = '';
|
|
395
|
+
}
|
|
396
|
+
llamaHistory.push({ role: 'model', response: [msg.content] });
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
if (llamaHistory.length > 0) {
|
|
400
|
+
session.setChatHistory(llamaHistory);
|
|
401
|
+
}
|
|
402
|
+
// Extract frontend/base URL to guide link formatting
|
|
403
|
+
const baseUrl = process.env.FRONTEND_URL ? process.env.FRONTEND_URL.replace(/\/$/, '') : 'https://doctors24.bd';
|
|
404
|
+
// Build the final prompt — context goes DIRECTLY in the message for maximum reliability.
|
|
405
|
+
const finalPrompt = context
|
|
406
|
+
? `Here is the relevant website content:\n\n${context}\n\n---\n\nAnswer the following question naturally, conversationally, and informatively as our support agent. Provide a helpful description or summary first to explain the content before sharing any links. Do not just output the URL on its own.
|
|
407
|
+
|
|
408
|
+
IMPORTANT ROUTING & LINK RULES:
|
|
409
|
+
1. When mentioning any page, section, or link from the website content, you MUST format it as a full clickable URL using the base URL "${baseUrl}".
|
|
410
|
+
2. Use the exact path provided in the SOURCE PAGE line. For example, if it says "SOURCE PAGE: /some-path", the URL MUST be exactly "${baseUrl}/some-path". Do NOT modify the path prefixes, names, or slugs.
|
|
411
|
+
|
|
412
|
+
Question: ${currentMessage}`
|
|
413
|
+
: currentMessage;
|
|
414
|
+
Logger_1.default.info('LocalAIService', `In-process prompt to model: "${currentMessage}"`);
|
|
415
|
+
let response;
|
|
416
|
+
try {
|
|
417
|
+
response = await session.prompt(finalPrompt, {
|
|
418
|
+
onTextChunk: onTextChunk || undefined,
|
|
419
|
+
temperature: 0.7,
|
|
420
|
+
repeatPenalty: {
|
|
421
|
+
penalty: 1.15
|
|
422
|
+
}
|
|
423
|
+
});
|
|
424
|
+
}
|
|
425
|
+
catch (promptErr) {
|
|
426
|
+
if (tempSequence) {
|
|
427
|
+
tempSequence.dispose();
|
|
428
|
+
}
|
|
429
|
+
Logger_1.default.warn('LocalAIService', `Model returned empty/error output: ${promptErr.message}. Using friendly fallback.`);
|
|
430
|
+
return "Thanks for asking! 😊 I don't have enough information on that just yet, but we're working on it. Please check back soon or visit our website for the latest details!";
|
|
431
|
+
}
|
|
432
|
+
// Guard: if response is empty string, also return fallback
|
|
433
|
+
if (!response || !response.trim()) {
|
|
434
|
+
if (tempSequence) {
|
|
435
|
+
tempSequence.dispose();
|
|
436
|
+
}
|
|
437
|
+
Logger_1.default.warn('LocalAIService', 'Model returned empty response. Using friendly fallback.');
|
|
438
|
+
return "Thanks for asking! 😊 I don't have enough information on that just yet, but we're working on it. Please check back soon or visit our website for the latest details!";
|
|
439
|
+
}
|
|
440
|
+
// 5. Dispose temporary sequence to prevent RAM memory leaks if not caching
|
|
441
|
+
if (tempSequence) {
|
|
442
|
+
tempSequence.dispose();
|
|
443
|
+
}
|
|
444
|
+
return response;
|
|
445
|
+
}
|
|
446
|
+
catch (err) {
|
|
447
|
+
Logger_1.default.error('LocalAIService', 'Failed to generate answer from local model', err);
|
|
448
|
+
throw err;
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
/**
|
|
452
|
+
* Prewarm helper to load the model and context in the background
|
|
453
|
+
*/
|
|
454
|
+
async prewarm(localModelPath, threads) {
|
|
455
|
+
try {
|
|
456
|
+
await this.getLlamaModelAndContext(localModelPath, threads);
|
|
457
|
+
}
|
|
458
|
+
catch (err) {
|
|
459
|
+
Logger_1.default.warn('LocalAIService', 'Prewarm failed to load model/context', err);
|
|
460
|
+
}
|
|
461
|
+
}
|
|
462
|
+
async importLlamaCpp() {
|
|
463
|
+
return eval('import("node-llama-cpp")');
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
exports.LocalAIService = LocalAIService;
|
|
467
|
+
LocalAIService.llamaInstance = null;
|
|
468
|
+
LocalAIService.modelInstance = null;
|
|
469
|
+
LocalAIService.contextInstance = null;
|
|
470
|
+
LocalAIService.activeSessions = new Map();
|
|
471
|
+
LocalAIService.cleanupInterval = null;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export declare class OpenAIService {
|
|
2
|
+
private static getConfigPath;
|
|
3
|
+
private static readOpenAIMetadata;
|
|
4
|
+
private static writeOpenAIMetadata;
|
|
5
|
+
/**
|
|
6
|
+
* Syncs local scraped text files to the OpenAI Vector Store.
|
|
7
|
+
* Uses parallel file uploads and batch attachments to optimize speed.
|
|
8
|
+
*/
|
|
9
|
+
syncFilesToVectorStore(force?: boolean): Promise<void>;
|
|
10
|
+
/**
|
|
11
|
+
* Generates a streaming answer from OpenAI Responses API.
|
|
12
|
+
* Note: For user chat queries, we DO NOT retry on 429 because we want to fail-fast
|
|
13
|
+
* and fall back to the local Gemma offline model immediately without blocking.
|
|
14
|
+
*/
|
|
15
|
+
generateAnswerStream(localThreadId: string, message: string, systemPrompt: string, onTextChunk: (chunk: string) => void, disableFileSearch?: boolean): Promise<string>;
|
|
16
|
+
}
|