@airoom/nextmin-node 2.0.2 → 2.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +151 -0
- package/dist/api/apiRouter.d.ts +20 -0
- package/dist/api/apiRouter.js +122 -7
- package/dist/api/router/mountCrudRoutes.js +69 -18
- package/dist/api/router/setupAuthRoutes.js +475 -38
- package/dist/api/router/setupChatWidgetRoutes.d.ts +3 -0
- package/dist/api/router/setupChatWidgetRoutes.js +207 -0
- package/dist/api/router/setupFileRoutes.js +221 -16
- package/dist/api/router/utils.d.ts +2 -1
- package/dist/api/router/utils.js +8 -6
- package/dist/cli.d.ts +1 -0
- package/dist/cli.js +145 -64
- package/dist/database/DatabaseAdapter.d.ts +1 -0
- package/dist/database/NMAdapter.d.ts +6 -0
- package/dist/database/NMAdapter.js +437 -86
- package/dist/database/QueryEngine.js +7 -4
- package/dist/files/FileStorageAdapter.d.ts +1 -0
- package/dist/files/LocalFileStorageAdapter.d.ts +1 -0
- package/dist/files/LocalFileStorageAdapter.js +25 -3
- package/dist/files/S3FileStorageAdapter.d.ts +3 -1
- package/dist/files/S3FileStorageAdapter.js +72 -11
- package/dist/files/filename.js +6 -4
- package/dist/index.d.ts +1 -0
- package/dist/index.js +3 -1
- package/dist/models/BaseModel.d.ts +4 -0
- package/dist/policy/authorize.d.ts +1 -1
- package/dist/policy/authorize.js +64 -13
- package/dist/schemas/Users.json +20 -10
- package/dist/services/IndexingService.d.ts +24 -0
- package/dist/services/IndexingService.js +555 -0
- package/dist/services/LocalAIService.d.ts +34 -0
- package/dist/services/LocalAIService.js +471 -0
- package/dist/services/OpenAIService.d.ts +16 -0
- package/dist/services/OpenAIService.js +586 -0
- package/dist/utils/SchemaLoader.d.ts +1 -1
- package/dist/utils/SchemaLoader.js +40 -4
- package/package.json +12 -4
|
@@ -0,0 +1,555 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
+
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
+
};
|
|
5
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.IndexingService = void 0;
|
|
7
|
+
exports.reindex = reindex;
|
|
8
|
+
const fs_1 = __importDefault(require("fs"));
|
|
9
|
+
const path_1 = __importDefault(require("path"));
|
|
10
|
+
const crypto_1 = __importDefault(require("crypto"));
|
|
11
|
+
const https_1 = __importDefault(require("https"));
|
|
12
|
+
const http_1 = __importDefault(require("http"));
|
|
13
|
+
const Logger_1 = __importDefault(require("../utils/Logger"));
|
|
14
|
+
const OpenAIService_1 = require("./OpenAIService");
|
|
15
|
+
// Custom HTTPS agent that skips TLS verification — needed for sites with
|
|
16
|
+
// expired or self-signed certificates (e.g. Cloudflare Pages custom domains).
|
|
17
|
+
const insecureHttpsAgent = new https_1.default.Agent({ rejectUnauthorized: false });
|
|
18
|
+
/**
|
|
19
|
+
* Fetch a URL using Node's built-in http/https modules, bypassing TLS errors.
|
|
20
|
+
* Returns a Response-compatible object with ok, status, headers, text() and json().
|
|
21
|
+
*/
|
|
22
|
+
function fetchInsecure(url, options = {}) {
|
|
23
|
+
return new Promise((resolve, reject) => {
|
|
24
|
+
const parsedUrl = new URL(url);
|
|
25
|
+
const isHttps = parsedUrl.protocol === 'https:';
|
|
26
|
+
const lib = isHttps ? https_1.default : http_1.default;
|
|
27
|
+
const reqOptions = {
|
|
28
|
+
hostname: parsedUrl.hostname,
|
|
29
|
+
port: parsedUrl.port || (isHttps ? 443 : 80),
|
|
30
|
+
path: parsedUrl.pathname + parsedUrl.search,
|
|
31
|
+
method: 'GET',
|
|
32
|
+
headers: options.headers || {},
|
|
33
|
+
agent: isHttps ? insecureHttpsAgent : undefined,
|
|
34
|
+
timeout: options.timeout || 30000,
|
|
35
|
+
};
|
|
36
|
+
const req = lib.request(reqOptions, (res) => {
|
|
37
|
+
const chunks = [];
|
|
38
|
+
res.on('data', (chunk) => chunks.push(chunk));
|
|
39
|
+
res.on('end', () => {
|
|
40
|
+
const body = Buffer.concat(chunks).toString('utf-8');
|
|
41
|
+
const statusCode = res.statusCode || 0;
|
|
42
|
+
const headers = res.headers;
|
|
43
|
+
resolve({
|
|
44
|
+
ok: statusCode >= 200 && statusCode < 300,
|
|
45
|
+
status: statusCode,
|
|
46
|
+
headers: {
|
|
47
|
+
get: (k) => {
|
|
48
|
+
const v = headers[k.toLowerCase()];
|
|
49
|
+
return Array.isArray(v) ? v[0] : (v ?? null);
|
|
50
|
+
}
|
|
51
|
+
},
|
|
52
|
+
text: async () => body,
|
|
53
|
+
});
|
|
54
|
+
});
|
|
55
|
+
res.on('error', reject);
|
|
56
|
+
});
|
|
57
|
+
req.on('timeout', () => {
|
|
58
|
+
req.destroy();
|
|
59
|
+
reject(new Error(`Request to ${url} timed out after ${options.timeout || 30000}ms`));
|
|
60
|
+
});
|
|
61
|
+
req.on('error', reject);
|
|
62
|
+
req.end();
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
// Helper for fetch with timeout — wraps fetchInsecure with a consistent interface.
|
|
66
|
+
async function fetchWithTimeout(url, options = {}) {
|
|
67
|
+
// Use global fetch if it has been stubbed (like in unit tests)
|
|
68
|
+
if (typeof global !== 'undefined' && typeof global.fetch === 'function') {
|
|
69
|
+
try {
|
|
70
|
+
return await global.fetch(url, options);
|
|
71
|
+
}
|
|
72
|
+
catch (err) {
|
|
73
|
+
// Fallback to fetchInsecure if global fetch fails
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
return fetchInsecure(url, options);
|
|
77
|
+
}
|
|
78
|
+
// Simple HTML text stripper
|
|
79
|
+
function stripHtml(html) {
|
|
80
|
+
// Preprocess links to preserve their text and destination in Markdown format
|
|
81
|
+
const processed = html.replace(/<a[^>]+href=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi, (match, href, text) => {
|
|
82
|
+
const cleanText = text.replace(/<[^>]*>/g, ' ').replace(/\s+/g, ' ').trim();
|
|
83
|
+
if (!cleanText)
|
|
84
|
+
return '';
|
|
85
|
+
return ` [${cleanText}](${href}) `;
|
|
86
|
+
});
|
|
87
|
+
return processed
|
|
88
|
+
.replace(/<script[^>]*>([\s\S]*?)<\/script>/gi, ' ')
|
|
89
|
+
.replace(/<style[^>]*>([\s\S]*?)<\/style>/gi, ' ')
|
|
90
|
+
.replace(/<[^>]*>/g, ' ')
|
|
91
|
+
.replace(/\s+/g, ' ')
|
|
92
|
+
.trim();
|
|
93
|
+
}
|
|
94
|
+
class IndexingService {
|
|
95
|
+
constructor(chatWidget, force = false, dbAdapter) {
|
|
96
|
+
this.itemKeyToId = new Map();
|
|
97
|
+
this.chatWidget = chatWidget;
|
|
98
|
+
this.force = force;
|
|
99
|
+
this.dbAdapter = dbAdapter;
|
|
100
|
+
}
|
|
101
|
+
getManifestPath() {
|
|
102
|
+
const dir = path_1.default.join(process.cwd(), '.nextmin');
|
|
103
|
+
if (!fs_1.default.existsSync(dir))
|
|
104
|
+
fs_1.default.mkdirSync(dir, { recursive: true });
|
|
105
|
+
return path_1.default.join(dir, 'indexing-manifest.json');
|
|
106
|
+
}
|
|
107
|
+
readManifest() {
|
|
108
|
+
const filePath = this.getManifestPath();
|
|
109
|
+
if (!fs_1.default.existsSync(filePath))
|
|
110
|
+
return {};
|
|
111
|
+
try {
|
|
112
|
+
const content = fs_1.default.readFileSync(filePath, 'utf-8');
|
|
113
|
+
const parsed = JSON.parse(content);
|
|
114
|
+
return parsed.hashes || {};
|
|
115
|
+
}
|
|
116
|
+
catch {
|
|
117
|
+
return {};
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
writeManifest(hashes) {
|
|
121
|
+
const filePath = this.getManifestPath();
|
|
122
|
+
try {
|
|
123
|
+
fs_1.default.writeFileSync(filePath, JSON.stringify({ lastIndexed: new Date().toISOString(), hashes }, null, 2), 'utf-8');
|
|
124
|
+
}
|
|
125
|
+
catch (err) {
|
|
126
|
+
Logger_1.default.error('IndexingService', 'Failed to write indexing-manifest.json', err);
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
getScrapedPagesDir() {
|
|
130
|
+
const dir = path_1.default.join(process.cwd(), '.nextmin', 'scraped-pages');
|
|
131
|
+
if (!fs_1.default.existsSync(dir))
|
|
132
|
+
fs_1.default.mkdirSync(dir, { recursive: true });
|
|
133
|
+
return dir;
|
|
134
|
+
}
|
|
135
|
+
writeScrapedPageLocal(key, content) {
|
|
136
|
+
const dir = this.getScrapedPagesDir();
|
|
137
|
+
const safeName = key.replace(/[^a-zA-Z0-9_\-.]/g, '_');
|
|
138
|
+
const filePath = path_1.default.join(dir, `${safeName}.txt`);
|
|
139
|
+
fs_1.default.writeFileSync(filePath, content, 'utf-8');
|
|
140
|
+
return filePath;
|
|
141
|
+
}
|
|
142
|
+
deleteScrapedPageLocal(key) {
|
|
143
|
+
const dir = this.getScrapedPagesDir();
|
|
144
|
+
const safeName = key.replace(/[^a-zA-Z0-9_\-.]/g, '_');
|
|
145
|
+
const filePath = path_1.default.join(dir, `${safeName}.txt`);
|
|
146
|
+
if (fs_1.default.existsSync(filePath)) {
|
|
147
|
+
try {
|
|
148
|
+
fs_1.default.unlinkSync(filePath);
|
|
149
|
+
}
|
|
150
|
+
catch (e) {
|
|
151
|
+
Logger_1.default.warn('IndexingService', `Failed to delete local scraped file ${filePath}`, e);
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
// ---------- Web Page Scraping / Crawler API ----------
|
|
156
|
+
async fetchSitemapUrls(sitemapUrl, baseUrl, visited = new Set()) {
|
|
157
|
+
if (visited.has(sitemapUrl))
|
|
158
|
+
return [];
|
|
159
|
+
visited.add(sitemapUrl);
|
|
160
|
+
Logger_1.default.info('IndexingService', `Fetching sitemap: ${sitemapUrl}`);
|
|
161
|
+
try {
|
|
162
|
+
const res = await fetchWithTimeout(sitemapUrl, {
|
|
163
|
+
headers: { 'User-Agent': 'NextMinCrawler/1.0' },
|
|
164
|
+
timeout: 30000
|
|
165
|
+
});
|
|
166
|
+
if (!res.ok) {
|
|
167
|
+
Logger_1.default.warn('IndexingService', `Sitemap fetch failed (${res.status}): ${sitemapUrl}`);
|
|
168
|
+
return [];
|
|
169
|
+
}
|
|
170
|
+
const xml = await res.text();
|
|
171
|
+
const locRegex = /<loc>([^<]+)<\/loc>/gi;
|
|
172
|
+
let match;
|
|
173
|
+
const subSitemaps = [];
|
|
174
|
+
const paths = [];
|
|
175
|
+
const siteOrigin = new URL(baseUrl).origin;
|
|
176
|
+
const isSitemapIndex = xml.includes('<sitemapindex') || xml.includes('<sitemap>');
|
|
177
|
+
while ((match = locRegex.exec(xml)) !== null) {
|
|
178
|
+
const loc = match[1].trim().replace(/&/g, '&');
|
|
179
|
+
try {
|
|
180
|
+
const urlObj = new URL(loc);
|
|
181
|
+
if (urlObj.origin === siteOrigin) {
|
|
182
|
+
let path = urlObj.pathname;
|
|
183
|
+
if (urlObj.search)
|
|
184
|
+
path += urlObj.search;
|
|
185
|
+
if (path !== '/' && path.endsWith('/')) {
|
|
186
|
+
path = path.slice(0, -1);
|
|
187
|
+
}
|
|
188
|
+
if (isSitemapIndex) {
|
|
189
|
+
subSitemaps.push(loc);
|
|
190
|
+
}
|
|
191
|
+
else {
|
|
192
|
+
paths.push(path);
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
catch {
|
|
197
|
+
// Ignore invalid URLs
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
if (subSitemaps.length > 0) {
|
|
201
|
+
const results = await Promise.all(subSitemaps.map(subUrl => this.fetchSitemapUrls(subUrl, baseUrl, visited)));
|
|
202
|
+
return [...paths, ...results.flat()];
|
|
203
|
+
}
|
|
204
|
+
return paths;
|
|
205
|
+
}
|
|
206
|
+
catch (err) {
|
|
207
|
+
Logger_1.default.error('IndexingService', `Error fetching sitemap ${sitemapUrl}`, err);
|
|
208
|
+
return [];
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
async crawlSite(siteUrl, startPaths = ['/'], concurrency = 5) {
|
|
212
|
+
const visited = new Set();
|
|
213
|
+
const results = new Map();
|
|
214
|
+
const base = siteUrl.endsWith('/') ? siteUrl.slice(0, -1) : siteUrl;
|
|
215
|
+
// 1. Try to discover URLs via Sitemap first
|
|
216
|
+
let sitemapPaths = null;
|
|
217
|
+
const sitemapUrl = this.chatWidget.sitemapUrl || `${base}/sitemap.xml`;
|
|
218
|
+
Logger_1.default.info('IndexingService', `Checking for sitemap at ${sitemapUrl}...`);
|
|
219
|
+
try {
|
|
220
|
+
const discovered = await this.fetchSitemapUrls(sitemapUrl, base);
|
|
221
|
+
if (discovered && discovered.length > 0) {
|
|
222
|
+
sitemapPaths = [...new Set(discovered)];
|
|
223
|
+
Logger_1.default.info('IndexingService', `Discovered ${sitemapPaths.length} URLs from sitemap.`);
|
|
224
|
+
}
|
|
225
|
+
else {
|
|
226
|
+
Logger_1.default.info('IndexingService', `No URLs found in sitemap or sitemap is unreachable.`);
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
catch (err) {
|
|
230
|
+
Logger_1.default.warn('IndexingService', `Failed to discover sitemap URLs. Falling back to recursive crawler.`, err);
|
|
231
|
+
}
|
|
232
|
+
const useSitemap = sitemapPaths !== null && sitemapPaths.length > 0;
|
|
233
|
+
const queue = useSitemap ? sitemapPaths : [...startPaths];
|
|
234
|
+
Logger_1.default.info('IndexingService', `Starting concurrent web scrape of ${base} with ${concurrency} threads (${useSitemap ? 'Sitemap Mode' : 'Link-following Mode'})...`);
|
|
235
|
+
const activeFetches = new Set();
|
|
236
|
+
const runWorker = async () => {
|
|
237
|
+
while (true) {
|
|
238
|
+
let currentPath = null;
|
|
239
|
+
for (let i = 0; i < queue.length; i++) {
|
|
240
|
+
const p = queue[i];
|
|
241
|
+
if (!visited.has(p) && !activeFetches.has(p)) {
|
|
242
|
+
currentPath = p;
|
|
243
|
+
queue.splice(i, 1);
|
|
244
|
+
break;
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
if (currentPath === null) {
|
|
248
|
+
if (activeFetches.size === 0 && queue.length === 0) {
|
|
249
|
+
break;
|
|
250
|
+
}
|
|
251
|
+
await new Promise(resolve => setTimeout(resolve, 50));
|
|
252
|
+
continue;
|
|
253
|
+
}
|
|
254
|
+
activeFetches.add(currentPath);
|
|
255
|
+
const fullUrl = `${base}${currentPath}`;
|
|
256
|
+
try {
|
|
257
|
+
Logger_1.default.info('IndexingService', `Fetching page for scraping: ${fullUrl}`);
|
|
258
|
+
const res = await fetchWithTimeout(fullUrl, {
|
|
259
|
+
headers: {
|
|
260
|
+
'User-Agent': 'NextMinCrawler/1.0'
|
|
261
|
+
},
|
|
262
|
+
timeout: 30000
|
|
263
|
+
});
|
|
264
|
+
visited.add(currentPath);
|
|
265
|
+
activeFetches.delete(currentPath);
|
|
266
|
+
if (!res.ok) {
|
|
267
|
+
Logger_1.default.warn('IndexingService', `Failed to fetch page ${fullUrl} (${res.status})`);
|
|
268
|
+
continue;
|
|
269
|
+
}
|
|
270
|
+
const contentType = res.headers.get('content-type') || '';
|
|
271
|
+
if (!contentType.includes('text/html')) {
|
|
272
|
+
Logger_1.default.debug('IndexingService', `Skipping non-HTML URL: ${fullUrl} (${contentType})`);
|
|
273
|
+
continue;
|
|
274
|
+
}
|
|
275
|
+
const html = await res.text();
|
|
276
|
+
const text = stripHtml(html);
|
|
277
|
+
if (text && text.length >= 20) {
|
|
278
|
+
results.set(currentPath, text);
|
|
279
|
+
}
|
|
280
|
+
// ONLY extract and follow links if we are NOT using the sitemap list
|
|
281
|
+
if (!useSitemap) {
|
|
282
|
+
const links = this.extractInternalLinks(html, currentPath, base);
|
|
283
|
+
for (const link of links) {
|
|
284
|
+
if (!visited.has(link) && !activeFetches.has(link) && !queue.includes(link)) {
|
|
285
|
+
queue.push(link);
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
catch (err) {
|
|
291
|
+
visited.add(currentPath);
|
|
292
|
+
activeFetches.delete(currentPath);
|
|
293
|
+
Logger_1.default.error('IndexingService', `Error crawling page ${fullUrl}`, err);
|
|
294
|
+
}
|
|
295
|
+
await new Promise(resolve => setTimeout(resolve, 100));
|
|
296
|
+
}
|
|
297
|
+
};
|
|
298
|
+
const workers = Array.from({ length: concurrency }, () => runWorker());
|
|
299
|
+
await Promise.all(workers);
|
|
300
|
+
Logger_1.default.info('IndexingService', `Scrape complete. Scraped ${results.size} pages.`);
|
|
301
|
+
return results;
|
|
302
|
+
}
|
|
303
|
+
extractInternalLinks(html, currentPath, baseUrl) {
|
|
304
|
+
const links = [];
|
|
305
|
+
const hrefRegex = /<a[^>]+href=["']([^"']+)["']/gi;
|
|
306
|
+
let match;
|
|
307
|
+
while ((match = hrefRegex.exec(html)) !== null) {
|
|
308
|
+
let href = match[1].trim();
|
|
309
|
+
if (!href || href.startsWith('#') || href.startsWith('mailto:') || href.startsWith('tel:') || href.toLowerCase().startsWith('javascript:')) {
|
|
310
|
+
continue;
|
|
311
|
+
}
|
|
312
|
+
try {
|
|
313
|
+
const resolved = new URL(href, `${baseUrl}${currentPath}`);
|
|
314
|
+
if (resolved.origin === new URL(baseUrl).origin) {
|
|
315
|
+
let path = resolved.pathname;
|
|
316
|
+
if (resolved.search)
|
|
317
|
+
path += resolved.search;
|
|
318
|
+
if (path !== '/' && path.endsWith('/')) {
|
|
319
|
+
path = path.slice(0, -1);
|
|
320
|
+
}
|
|
321
|
+
const isAsset = /\.(png|jpe?g|gif|svg|css|js|ico|pdf|zip|mp4|json|xml)$/i.test(resolved.pathname);
|
|
322
|
+
if (!isAsset) {
|
|
323
|
+
links.push(path);
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
catch {
|
|
328
|
+
// Ignore invalid URL
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
return [...new Set(links)];
|
|
332
|
+
}
|
|
333
|
+
// ---------- Database Schema Indexing ----------
|
|
334
|
+
async indexDatabaseSchemas() {
|
|
335
|
+
const results = new Map();
|
|
336
|
+
if (!this.dbAdapter)
|
|
337
|
+
return results;
|
|
338
|
+
const schemas = this.chatWidget.whitelabelSchemas || [];
|
|
339
|
+
if (schemas.length === 0)
|
|
340
|
+
return results;
|
|
341
|
+
Logger_1.default.info('IndexingService', `Indexing ${schemas.length} database schemas: ${schemas.join(', ')}...`);
|
|
342
|
+
const siteUrl = this.chatWidget.siteUrl || 'https://doctors24.bd';
|
|
343
|
+
for (const schemaName of schemas) {
|
|
344
|
+
try {
|
|
345
|
+
const collectionName = schemaName.toLowerCase();
|
|
346
|
+
let docs = [];
|
|
347
|
+
if (typeof this.dbAdapter.read === 'function') {
|
|
348
|
+
docs = await this.dbAdapter.read(collectionName, {}, 50000, 0);
|
|
349
|
+
}
|
|
350
|
+
else if (this.dbAdapter.db && typeof this.dbAdapter.db.collection === 'function') {
|
|
351
|
+
docs = await this.dbAdapter.db.collection(collectionName).find({}).toArray();
|
|
352
|
+
}
|
|
353
|
+
if (!Array.isArray(docs))
|
|
354
|
+
continue;
|
|
355
|
+
Logger_1.default.info('IndexingService', `Schema '${schemaName}': found ${docs.length} database records.`);
|
|
356
|
+
for (const doc of docs) {
|
|
357
|
+
if (!doc)
|
|
358
|
+
continue;
|
|
359
|
+
const slug = doc.slug || doc._id || doc.id;
|
|
360
|
+
if (!slug)
|
|
361
|
+
continue;
|
|
362
|
+
let entityType = schemaName.replace(/s$/i, '').toLowerCase();
|
|
363
|
+
if (entityType === 'specialitie' || entityType === 'specialty')
|
|
364
|
+
entityType = 'speciality';
|
|
365
|
+
const canonicalPath = `/${entityType}/${slug}`;
|
|
366
|
+
const fullUrl = `${siteUrl}${canonicalPath}`;
|
|
367
|
+
const title = doc.fullName || doc.name || doc.title || doc.headline || `${schemaName} Record`;
|
|
368
|
+
let apptText = '';
|
|
369
|
+
if (doc.appointments) {
|
|
370
|
+
try {
|
|
371
|
+
const apps = typeof doc.appointments === 'string' ? JSON.parse(doc.appointments) : doc.appointments;
|
|
372
|
+
if (Array.isArray(apps)) {
|
|
373
|
+
apptText = apps
|
|
374
|
+
.map((a) => `- Hospital/Chamber: ${a.hospital?.name || a.name || ''}\n Address: ${a.hospital?.address || a.hospital?.physicalLocation || a.address || ''}\n Visiting Hours: ${a.visitingHours || ''}\n Phone / Appointment: ${a.phone || a.hospital?.telephone || doc.phone || ''}`)
|
|
375
|
+
.join('\n\n');
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
catch { }
|
|
379
|
+
}
|
|
380
|
+
let formattedDetails = '';
|
|
381
|
+
if (doc.degree || doc.fellowshipsOrTrainings)
|
|
382
|
+
formattedDetails += `- Qualifications: ${doc.degree || ''} ${doc.fellowshipsOrTrainings || ''}\n`;
|
|
383
|
+
if (doc.speciality || doc.specialty)
|
|
384
|
+
formattedDetails += `- Specialty: ${doc.speciality || doc.specialty || ''}\n`;
|
|
385
|
+
if (doc.designationAndDepartment || doc.designation)
|
|
386
|
+
formattedDetails += `- Designation: ${doc.designationAndDepartment || doc.designation || ''}\n`;
|
|
387
|
+
if (doc.workPlace || doc.workplace)
|
|
388
|
+
formattedDetails += `- Workplace: ${doc.workPlace || doc.workplace || ''}\n`;
|
|
389
|
+
if (doc.phone || doc.telephone)
|
|
390
|
+
formattedDetails += `- Primary Phone: ${doc.phone || doc.telephone || ''}\n`;
|
|
391
|
+
if (doc.address || doc.physicalLocation)
|
|
392
|
+
formattedDetails += `- Address: ${doc.address || doc.physicalLocation || ''}\n`;
|
|
393
|
+
const aboutText = stripHtml(doc.aboutYou || doc.aboutHospital || doc.description || doc.body || doc.content || '');
|
|
394
|
+
const content = `DIRECTORY DOCUMENT
|
|
395
|
+
PAGE_TITLE: ${title} | ${schemaName}
|
|
396
|
+
CANONICAL_URL: ${fullUrl}
|
|
397
|
+
|
|
398
|
+
INFORMATION:
|
|
399
|
+
- Name/Title: ${title}
|
|
400
|
+
${formattedDetails}
|
|
401
|
+
${apptText ? `\nCHAMBER & APPOINTMENT DETAILS:\n${apptText}\n` : ''}
|
|
402
|
+
${aboutText ? `\nABOUT:\n${aboutText}\n` : ''}
|
|
403
|
+
EXPLICIT ROUTING INSTRUCTION FOR ASSISTANT:
|
|
404
|
+
When users ask about this item or entity, always direct them to canonical URL: ${fullUrl}
|
|
405
|
+
`;
|
|
406
|
+
results.set(canonicalPath, content);
|
|
407
|
+
}
|
|
408
|
+
}
|
|
409
|
+
catch (err) {
|
|
410
|
+
Logger_1.default.error('IndexingService', `Failed to index database schema '${schemaName}'`, err);
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
return results;
|
|
414
|
+
}
|
|
415
|
+
// ---------- Entrypoint ----------
|
|
416
|
+
async runIndex() {
|
|
417
|
+
if (!this.chatWidget.enabled) {
|
|
418
|
+
Logger_1.default.info('IndexingService', 'Chat widget is disabled. Skipping index.');
|
|
419
|
+
return;
|
|
420
|
+
}
|
|
421
|
+
Logger_1.default.info('IndexingService', 'Starting LOCAL-ONLY indexing pipeline...');
|
|
422
|
+
const activeKeys = new Set();
|
|
423
|
+
const manifestHashes = this.force ? {} : this.readManifest();
|
|
424
|
+
const currentManifest = { ...manifestHashes };
|
|
425
|
+
let hasErrors = false;
|
|
426
|
+
const allPageEntries = new Map();
|
|
427
|
+
// 1. Scrape Site Pages (Crawler)
|
|
428
|
+
if (this.chatWidget.siteUrl && this.chatWidget.sourceType !== 'db') {
|
|
429
|
+
const paths = this.chatWidget.scrapePaths || ['/'];
|
|
430
|
+
try {
|
|
431
|
+
const concurrency = this.chatWidget.concurrency || 5;
|
|
432
|
+
const scrapedPages = await this.crawlSite(this.chatWidget.siteUrl, paths, concurrency);
|
|
433
|
+
for (const [k, v] of scrapedPages.entries())
|
|
434
|
+
allPageEntries.set(k, v);
|
|
435
|
+
}
|
|
436
|
+
catch (err) {
|
|
437
|
+
Logger_1.default.error('IndexingService', 'Site crawl step failed', err);
|
|
438
|
+
hasErrors = true;
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
// 2. Index Database Schemas (DB Exporter)
|
|
442
|
+
if (this.dbAdapter && (this.chatWidget.sourceType === 'db' || this.chatWidget.sourceType === 'both' || !this.chatWidget.sourceType)) {
|
|
443
|
+
try {
|
|
444
|
+
const dbPages = await this.indexDatabaseSchemas();
|
|
445
|
+
for (const [k, v] of dbPages.entries())
|
|
446
|
+
allPageEntries.set(k, v);
|
|
447
|
+
}
|
|
448
|
+
catch (err) {
|
|
449
|
+
Logger_1.default.error('IndexingService', 'Database schema indexing failed', err);
|
|
450
|
+
hasErrors = true;
|
|
451
|
+
}
|
|
452
|
+
}
|
|
453
|
+
if (allPageEntries.size > 0) {
|
|
454
|
+
Logger_1.default.info('IndexingService', `Processing ${allPageEntries.size} total page entries...`);
|
|
455
|
+
const concurrency = this.chatWidget.concurrency || 5;
|
|
456
|
+
try {
|
|
457
|
+
let processedCount = 0;
|
|
458
|
+
let skippedCount = 0;
|
|
459
|
+
const entries = Array.from(allPageEntries.entries());
|
|
460
|
+
let index = 0;
|
|
461
|
+
// Concurrent Indexing Workers
|
|
462
|
+
const indexWorker = async () => {
|
|
463
|
+
while (true) {
|
|
464
|
+
let entry = null;
|
|
465
|
+
if (index < entries.length) {
|
|
466
|
+
entry = entries[index++];
|
|
467
|
+
}
|
|
468
|
+
if (!entry)
|
|
469
|
+
break;
|
|
470
|
+
const [pathname, text] = entry;
|
|
471
|
+
try {
|
|
472
|
+
const key = `page-${pathname.replace(/[^a-zA-Z0-9_\-]/g, '_') || 'index'}`;
|
|
473
|
+
activeKeys.add(key);
|
|
474
|
+
const hash = crypto_1.default.createHash('md5').update(text).digest('hex');
|
|
475
|
+
if (manifestHashes[key] === hash) {
|
|
476
|
+
skippedCount++;
|
|
477
|
+
currentManifest[key] = hash;
|
|
478
|
+
continue;
|
|
479
|
+
}
|
|
480
|
+
// Save locally
|
|
481
|
+
this.writeScrapedPageLocal(key, text);
|
|
482
|
+
processedCount++;
|
|
483
|
+
// Incrementally write manifest to disk to save progress immediately
|
|
484
|
+
currentManifest[key] = hash;
|
|
485
|
+
this.writeManifest(currentManifest);
|
|
486
|
+
await new Promise(resolve => setTimeout(resolve, 50)); // Gentle throttle
|
|
487
|
+
}
|
|
488
|
+
catch (err) {
|
|
489
|
+
Logger_1.default.error('IndexingService', `Failed to index scraped page ${pathname}`, err);
|
|
490
|
+
hasErrors = true;
|
|
491
|
+
}
|
|
492
|
+
}
|
|
493
|
+
};
|
|
494
|
+
Logger_1.default.info('IndexingService', `Processing in parallel using ${concurrency} concurrent workers...`);
|
|
495
|
+
const indexWorkers = Array.from({ length: concurrency }, () => indexWorker());
|
|
496
|
+
await Promise.all(indexWorkers);
|
|
497
|
+
Logger_1.default.info('IndexingService', `Completed indexing scraped pages: ${processedCount} updated, ${skippedCount} unchanged (loaded from cache).`);
|
|
498
|
+
}
|
|
499
|
+
catch (err) {
|
|
500
|
+
Logger_1.default.error('IndexingService', 'Site crawl step failed', err);
|
|
501
|
+
hasErrors = true;
|
|
502
|
+
}
|
|
503
|
+
}
|
|
504
|
+
// 2. Prune deleted files
|
|
505
|
+
if (!hasErrors) {
|
|
506
|
+
for (const key of Object.keys(manifestHashes)) {
|
|
507
|
+
if (!activeKeys.has(key)) {
|
|
508
|
+
Logger_1.default.info('IndexingService', `Pruning deleted document: ${key}`);
|
|
509
|
+
// Remove local file
|
|
510
|
+
this.deleteScrapedPageLocal(key);
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
else {
|
|
515
|
+
Logger_1.default.info('IndexingService', 'Skipping pruning stage due to errors encountered during crawling/indexing.');
|
|
516
|
+
}
|
|
517
|
+
// 3. Update final local manifest
|
|
518
|
+
const finalManifest = {};
|
|
519
|
+
for (const key of activeKeys) {
|
|
520
|
+
if (currentManifest[key]) {
|
|
521
|
+
finalManifest[key] = currentManifest[key];
|
|
522
|
+
}
|
|
523
|
+
}
|
|
524
|
+
// If errors occurred, preserve unvisited entries in the manifest
|
|
525
|
+
if (hasErrors) {
|
|
526
|
+
for (const key of Object.keys(manifestHashes)) {
|
|
527
|
+
if (!activeKeys.has(key)) {
|
|
528
|
+
finalManifest[key] = manifestHashes[key];
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
}
|
|
532
|
+
this.writeManifest(finalManifest);
|
|
533
|
+
// Sync newly indexed files to OpenAI Vector Store if enabled and API Key is set
|
|
534
|
+
if (this.chatWidget.enableOpenai !== false && process.env.OPENAI_API_KEY) {
|
|
535
|
+
try {
|
|
536
|
+
Logger_1.default.info('IndexingService', 'Triggering OpenAI Vector Store synchronization...');
|
|
537
|
+
const openAI = new OpenAIService_1.OpenAIService();
|
|
538
|
+
await openAI.syncFilesToVectorStore(this.force);
|
|
539
|
+
}
|
|
540
|
+
catch (err) {
|
|
541
|
+
Logger_1.default.error('IndexingService', `OpenAI file sync failed: ${err.message || err}`);
|
|
542
|
+
hasErrors = true;
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
if (hasErrors) {
|
|
546
|
+
throw new Error('Indexing completed with errors. Please check the logs.');
|
|
547
|
+
}
|
|
548
|
+
Logger_1.default.info('IndexingService', `Incremental indexing complete.`);
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
exports.IndexingService = IndexingService;
|
|
552
|
+
async function reindex(config) {
|
|
553
|
+
const indexer = new IndexingService(config.chatWidget, config.force, config.dbAdapter);
|
|
554
|
+
await indexer.runIndex();
|
|
555
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
export declare class LocalAIService {
|
|
2
|
+
private static llamaInstance;
|
|
3
|
+
private static modelInstance;
|
|
4
|
+
private static contextInstance;
|
|
5
|
+
private static activeSessions;
|
|
6
|
+
private static cleanupInterval;
|
|
7
|
+
constructor(localModelPath?: string);
|
|
8
|
+
private static initCleanupTimer;
|
|
9
|
+
/**
|
|
10
|
+
* Lazily loads the Llama model and context into memory to ensure fast response times
|
|
11
|
+
* and avoid reading the multi-gigabyte GGUF model file from disk repeatedly.
|
|
12
|
+
*/
|
|
13
|
+
private getLlamaModelAndContext;
|
|
14
|
+
/**
|
|
15
|
+
* Simple TF-IDF local search on scraped pages stored in .nextmin/scraped-pages
|
|
16
|
+
*/
|
|
17
|
+
/**
|
|
18
|
+
* Simple TF-IDF local search on scraped pages stored in .nextmin/scraped-pages
|
|
19
|
+
* with smart paragraph-based chunking and path boosting.
|
|
20
|
+
*/
|
|
21
|
+
private localSearch;
|
|
22
|
+
/**
|
|
23
|
+
* Generates chat answer using local node-llama-cpp execution.
|
|
24
|
+
*/
|
|
25
|
+
generateAnswer(history: Array<{
|
|
26
|
+
role: string;
|
|
27
|
+
content: string;
|
|
28
|
+
}>, currentMessage: string, systemPromptText?: string, localModelPath?: string, threads?: number, threadId?: string, onTextChunk?: (chunk: string) => void): Promise<string>;
|
|
29
|
+
/**
|
|
30
|
+
* Prewarm helper to load the model and context in the background
|
|
31
|
+
*/
|
|
32
|
+
prewarm(localModelPath?: string, threads?: number): Promise<void>;
|
|
33
|
+
private importLlamaCpp;
|
|
34
|
+
}
|