@airoom/nextmin-node 2.0.1 → 2.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +151 -0
  2. package/dist/api/apiRouter.d.ts +28 -1
  3. package/dist/api/apiRouter.js +138 -8
  4. package/dist/api/router/mountBatchRoutes.d.ts +2 -0
  5. package/dist/api/router/mountBatchRoutes.js +90 -0
  6. package/dist/api/router/mountCrudRoutes.js +79 -20
  7. package/dist/api/router/setupAuthRoutes.js +475 -38
  8. package/dist/api/router/setupChatWidgetRoutes.d.ts +3 -0
  9. package/dist/api/router/setupChatWidgetRoutes.js +207 -0
  10. package/dist/api/router/setupFileRoutes.js +258 -12
  11. package/dist/api/router/utils.d.ts +4 -1
  12. package/dist/api/router/utils.js +25 -7
  13. package/dist/cli.d.ts +1 -0
  14. package/dist/cli.js +145 -64
  15. package/dist/database/DatabaseAdapter.d.ts +2 -0
  16. package/dist/database/NMAdapter.d.ts +11 -0
  17. package/dist/database/NMAdapter.js +688 -73
  18. package/dist/database/QueryEngine.js +7 -4
  19. package/dist/files/FileStorageAdapter.d.ts +6 -0
  20. package/dist/files/LocalFileStorageAdapter.d.ts +6 -0
  21. package/dist/files/LocalFileStorageAdapter.js +39 -3
  22. package/dist/files/S3FileStorageAdapter.d.ts +8 -1
  23. package/dist/files/S3FileStorageAdapter.js +86 -11
  24. package/dist/files/filename.js +6 -4
  25. package/dist/index.d.ts +1 -0
  26. package/dist/index.js +3 -1
  27. package/dist/models/BaseModel.d.ts +19 -2
  28. package/dist/models/BaseModel.js +4 -4
  29. package/dist/policy/authorize.d.ts +1 -1
  30. package/dist/policy/authorize.js +64 -13
  31. package/dist/schemas/Users.json +20 -10
  32. package/dist/services/AggregateService.d.ts +8 -0
  33. package/dist/services/AggregateService.js +87 -0
  34. package/dist/services/IndexingService.d.ts +24 -0
  35. package/dist/services/IndexingService.js +555 -0
  36. package/dist/services/LocalAIService.d.ts +34 -0
  37. package/dist/services/LocalAIService.js +471 -0
  38. package/dist/services/OpenAIService.d.ts +16 -0
  39. package/dist/services/OpenAIService.js +586 -0
  40. package/dist/utils/Events.js +1 -1
  41. package/dist/utils/Logger.d.ts +2 -1
  42. package/dist/utils/Logger.js +15 -6
  43. package/dist/utils/SchemaLoader.d.ts +2 -2
  44. package/dist/utils/SchemaLoader.js +44 -6
  45. package/package.json +12 -4
@@ -0,0 +1,87 @@
1
+ "use strict";
2
+ var __importDefault = (this && this.__importDefault) || function (mod) {
3
+ return (mod && mod.__esModule) ? mod : { "default": mod };
4
+ };
5
+ Object.defineProperty(exports, "__esModule", { value: true });
6
+ exports.AggregateService = void 0;
7
+ const Events_1 = require("../utils/Events");
8
+ const Logger_1 = __importDefault(require("../utils/Logger"));
9
+ class AggregateService {
10
+ constructor(ctx) {
11
+ this.ctx = ctx;
12
+ this.init();
13
+ }
14
+ init() {
15
+ Events_1.events.on(Events_1.Events.AFTER_CREATE, (payload) => this.handleUpdate(payload.modelName, payload.result, 'create'));
16
+ Events_1.events.on(Events_1.Events.AFTER_UPDATE, (payload) => this.handleUpdate(payload.modelName, payload.result, 'update'));
17
+ Events_1.events.on(Events_1.Events.AFTER_DELETE, (payload) => this.handleUpdate(payload.modelName, payload.result, 'delete'));
18
+ Logger_1.default.info('AggregateService', 'Initialized and listening for CRUD events');
19
+ }
20
+ async handleUpdate(sourceModelName, doc, action) {
21
+ if (!doc)
22
+ return;
23
+ // Find SCHEMAS that have aggregates dependent on sourceModelName
24
+ const allSchemas = Object.values(this.ctx.liveSchemas);
25
+ for (const targetSchema of allSchemas) {
26
+ if (!targetSchema.aggregates)
27
+ continue;
28
+ const ags = targetSchema.aggregates.filter((a) => a.sourceModel.toLowerCase() === sourceModelName.toLowerCase());
29
+ if (ags.length === 0)
30
+ continue;
31
+ for (const ag of ags) {
32
+ const parentId = doc[ag.foreignField];
33
+ if (!parentId)
34
+ continue;
35
+ // Recalculate and update target document (Doctor)
36
+ await this.recalculate(targetSchema.modelName, parentId, ag);
37
+ }
38
+ }
39
+ }
40
+ async recalculate(targetModelName, targetId, ag) {
41
+ try {
42
+ const sourceModel = this.ctx.getModel(ag.sourceModel.toLowerCase());
43
+ const targetModel = this.ctx.getModel(targetModelName.toLowerCase());
44
+ const filter = { [ag.foreignField]: targetId };
45
+ let result = 0;
46
+ if (ag.op === 'count') {
47
+ result = await sourceModel.count(filter, true);
48
+ }
49
+ else if (ag.op === 'avg' || ag.op === 'sum') {
50
+ const adapter = this.ctx.dbAdapter;
51
+ if (adapter.dataSource.options.type === 'mongodb') {
52
+ const repo = adapter.dataSource.getMongoRepository(ag.sourceModel);
53
+ const aggregation = [
54
+ { $match: adapter.transformQuery(filter, sourceModel.schema) },
55
+ { $group: { _id: null, val: { [`$${ag.op}`]: `$${ag.field}` } } },
56
+ ];
57
+ const res = await repo.aggregate(aggregation).toArray();
58
+ result = res[0]?.val || 0;
59
+ }
60
+ else {
61
+ // SQL
62
+ const repo = adapter.repositories.get(ag.sourceModel.toLowerCase());
63
+ const qb = repo.createQueryBuilder('entity');
64
+ qb.select(`${ag.op === 'avg' ? 'AVG' : 'SUM'}(entity.${ag.field})`, 'val');
65
+ qb.where(adapter.transformQuery(filter, sourceModel.schema));
66
+ const res = await qb.getRawOne();
67
+ result = parseFloat(res?.val || res?.VAL || 0);
68
+ }
69
+ }
70
+ // Update target document (e.g. Doctor)
71
+ const updateData = {};
72
+ const parts = ag.targetField.split('.');
73
+ let curr = updateData;
74
+ for (let i = 0; i < parts.length - 1; i++) {
75
+ curr[parts[i]] = {};
76
+ curr = curr[parts[i]];
77
+ }
78
+ curr[parts[parts.length - 1]] = result;
79
+ await targetModel.update(targetId, updateData, true);
80
+ Logger_1.default.debug('AggregateService', `Updated ${targetModelName}:${targetId} aggregate ${ag.targetField} to ${result}`);
81
+ }
82
+ catch (err) {
83
+ Logger_1.default.error('AggregateService', `Failed to recalculate aggregate for ${targetModelName}:${targetId}`, err);
84
+ }
85
+ }
86
+ }
87
+ exports.AggregateService = AggregateService;
@@ -0,0 +1,24 @@
1
+ import { ChatWidgetConfig } from '../api/apiRouter';
2
+ export declare class IndexingService {
3
+ private chatWidget;
4
+ private force;
5
+ private dbAdapter?;
6
+ private itemKeyToId;
7
+ constructor(chatWidget: ChatWidgetConfig, force?: boolean, dbAdapter?: any);
8
+ private getManifestPath;
9
+ private readManifest;
10
+ private writeManifest;
11
+ private getScrapedPagesDir;
12
+ private writeScrapedPageLocal;
13
+ private deleteScrapedPageLocal;
14
+ private fetchSitemapUrls;
15
+ crawlSite(siteUrl: string, startPaths?: string[], concurrency?: number): Promise<Map<string, string>>;
16
+ private extractInternalLinks;
17
+ private indexDatabaseSchemas;
18
+ runIndex(): Promise<void>;
19
+ }
20
+ export declare function reindex(config: {
21
+ chatWidget: ChatWidgetConfig;
22
+ force?: boolean;
23
+ dbAdapter?: any;
24
+ }): Promise<void>;
@@ -0,0 +1,555 @@
1
+ "use strict";
2
+ var __importDefault = (this && this.__importDefault) || function (mod) {
3
+ return (mod && mod.__esModule) ? mod : { "default": mod };
4
+ };
5
+ Object.defineProperty(exports, "__esModule", { value: true });
6
+ exports.IndexingService = void 0;
7
+ exports.reindex = reindex;
8
+ const fs_1 = __importDefault(require("fs"));
9
+ const path_1 = __importDefault(require("path"));
10
+ const crypto_1 = __importDefault(require("crypto"));
11
+ const https_1 = __importDefault(require("https"));
12
+ const http_1 = __importDefault(require("http"));
13
+ const Logger_1 = __importDefault(require("../utils/Logger"));
14
+ const OpenAIService_1 = require("./OpenAIService");
15
+ // Custom HTTPS agent that skips TLS verification — needed for sites with
16
+ // expired or self-signed certificates (e.g. Cloudflare Pages custom domains).
17
+ const insecureHttpsAgent = new https_1.default.Agent({ rejectUnauthorized: false });
18
+ /**
19
+ * Fetch a URL using Node's built-in http/https modules, bypassing TLS errors.
20
+ * Returns a Response-compatible object with ok, status, headers, text() and json().
21
+ */
22
+ function fetchInsecure(url, options = {}) {
23
+ return new Promise((resolve, reject) => {
24
+ const parsedUrl = new URL(url);
25
+ const isHttps = parsedUrl.protocol === 'https:';
26
+ const lib = isHttps ? https_1.default : http_1.default;
27
+ const reqOptions = {
28
+ hostname: parsedUrl.hostname,
29
+ port: parsedUrl.port || (isHttps ? 443 : 80),
30
+ path: parsedUrl.pathname + parsedUrl.search,
31
+ method: 'GET',
32
+ headers: options.headers || {},
33
+ agent: isHttps ? insecureHttpsAgent : undefined,
34
+ timeout: options.timeout || 30000,
35
+ };
36
+ const req = lib.request(reqOptions, (res) => {
37
+ const chunks = [];
38
+ res.on('data', (chunk) => chunks.push(chunk));
39
+ res.on('end', () => {
40
+ const body = Buffer.concat(chunks).toString('utf-8');
41
+ const statusCode = res.statusCode || 0;
42
+ const headers = res.headers;
43
+ resolve({
44
+ ok: statusCode >= 200 && statusCode < 300,
45
+ status: statusCode,
46
+ headers: {
47
+ get: (k) => {
48
+ const v = headers[k.toLowerCase()];
49
+ return Array.isArray(v) ? v[0] : (v ?? null);
50
+ }
51
+ },
52
+ text: async () => body,
53
+ });
54
+ });
55
+ res.on('error', reject);
56
+ });
57
+ req.on('timeout', () => {
58
+ req.destroy();
59
+ reject(new Error(`Request to ${url} timed out after ${options.timeout || 30000}ms`));
60
+ });
61
+ req.on('error', reject);
62
+ req.end();
63
+ });
64
+ }
65
+ // Helper for fetch with timeout — wraps fetchInsecure with a consistent interface.
66
+ async function fetchWithTimeout(url, options = {}) {
67
+ // Use global fetch if it has been stubbed (like in unit tests)
68
+ if (typeof global !== 'undefined' && typeof global.fetch === 'function') {
69
+ try {
70
+ return await global.fetch(url, options);
71
+ }
72
+ catch (err) {
73
+ // Fallback to fetchInsecure if global fetch fails
74
+ }
75
+ }
76
+ return fetchInsecure(url, options);
77
+ }
78
+ // Simple HTML text stripper
79
+ function stripHtml(html) {
80
+ // Preprocess links to preserve their text and destination in Markdown format
81
+ const processed = html.replace(/<a[^>]+href=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi, (match, href, text) => {
82
+ const cleanText = text.replace(/<[^>]*>/g, ' ').replace(/\s+/g, ' ').trim();
83
+ if (!cleanText)
84
+ return '';
85
+ return ` [${cleanText}](${href}) `;
86
+ });
87
+ return processed
88
+ .replace(/<script[^>]*>([\s\S]*?)<\/script>/gi, ' ')
89
+ .replace(/<style[^>]*>([\s\S]*?)<\/style>/gi, ' ')
90
+ .replace(/<[^>]*>/g, ' ')
91
+ .replace(/\s+/g, ' ')
92
+ .trim();
93
+ }
94
+ class IndexingService {
95
+ constructor(chatWidget, force = false, dbAdapter) {
96
+ this.itemKeyToId = new Map();
97
+ this.chatWidget = chatWidget;
98
+ this.force = force;
99
+ this.dbAdapter = dbAdapter;
100
+ }
101
+ getManifestPath() {
102
+ const dir = path_1.default.join(process.cwd(), '.nextmin');
103
+ if (!fs_1.default.existsSync(dir))
104
+ fs_1.default.mkdirSync(dir, { recursive: true });
105
+ return path_1.default.join(dir, 'indexing-manifest.json');
106
+ }
107
+ readManifest() {
108
+ const filePath = this.getManifestPath();
109
+ if (!fs_1.default.existsSync(filePath))
110
+ return {};
111
+ try {
112
+ const content = fs_1.default.readFileSync(filePath, 'utf-8');
113
+ const parsed = JSON.parse(content);
114
+ return parsed.hashes || {};
115
+ }
116
+ catch {
117
+ return {};
118
+ }
119
+ }
120
+ writeManifest(hashes) {
121
+ const filePath = this.getManifestPath();
122
+ try {
123
+ fs_1.default.writeFileSync(filePath, JSON.stringify({ lastIndexed: new Date().toISOString(), hashes }, null, 2), 'utf-8');
124
+ }
125
+ catch (err) {
126
+ Logger_1.default.error('IndexingService', 'Failed to write indexing-manifest.json', err);
127
+ }
128
+ }
129
+ getScrapedPagesDir() {
130
+ const dir = path_1.default.join(process.cwd(), '.nextmin', 'scraped-pages');
131
+ if (!fs_1.default.existsSync(dir))
132
+ fs_1.default.mkdirSync(dir, { recursive: true });
133
+ return dir;
134
+ }
135
+ writeScrapedPageLocal(key, content) {
136
+ const dir = this.getScrapedPagesDir();
137
+ const safeName = key.replace(/[^a-zA-Z0-9_\-.]/g, '_');
138
+ const filePath = path_1.default.join(dir, `${safeName}.txt`);
139
+ fs_1.default.writeFileSync(filePath, content, 'utf-8');
140
+ return filePath;
141
+ }
142
+ deleteScrapedPageLocal(key) {
143
+ const dir = this.getScrapedPagesDir();
144
+ const safeName = key.replace(/[^a-zA-Z0-9_\-.]/g, '_');
145
+ const filePath = path_1.default.join(dir, `${safeName}.txt`);
146
+ if (fs_1.default.existsSync(filePath)) {
147
+ try {
148
+ fs_1.default.unlinkSync(filePath);
149
+ }
150
+ catch (e) {
151
+ Logger_1.default.warn('IndexingService', `Failed to delete local scraped file ${filePath}`, e);
152
+ }
153
+ }
154
+ }
155
+ // ---------- Web Page Scraping / Crawler API ----------
156
+ async fetchSitemapUrls(sitemapUrl, baseUrl, visited = new Set()) {
157
+ if (visited.has(sitemapUrl))
158
+ return [];
159
+ visited.add(sitemapUrl);
160
+ Logger_1.default.info('IndexingService', `Fetching sitemap: ${sitemapUrl}`);
161
+ try {
162
+ const res = await fetchWithTimeout(sitemapUrl, {
163
+ headers: { 'User-Agent': 'NextMinCrawler/1.0' },
164
+ timeout: 30000
165
+ });
166
+ if (!res.ok) {
167
+ Logger_1.default.warn('IndexingService', `Sitemap fetch failed (${res.status}): ${sitemapUrl}`);
168
+ return [];
169
+ }
170
+ const xml = await res.text();
171
+ const locRegex = /<loc>([^<]+)<\/loc>/gi;
172
+ let match;
173
+ const subSitemaps = [];
174
+ const paths = [];
175
+ const siteOrigin = new URL(baseUrl).origin;
176
+ const isSitemapIndex = xml.includes('<sitemapindex') || xml.includes('<sitemap>');
177
+ while ((match = locRegex.exec(xml)) !== null) {
178
+ const loc = match[1].trim().replace(/&amp;/g, '&');
179
+ try {
180
+ const urlObj = new URL(loc);
181
+ if (urlObj.origin === siteOrigin) {
182
+ let path = urlObj.pathname;
183
+ if (urlObj.search)
184
+ path += urlObj.search;
185
+ if (path !== '/' && path.endsWith('/')) {
186
+ path = path.slice(0, -1);
187
+ }
188
+ if (isSitemapIndex) {
189
+ subSitemaps.push(loc);
190
+ }
191
+ else {
192
+ paths.push(path);
193
+ }
194
+ }
195
+ }
196
+ catch {
197
+ // Ignore invalid URLs
198
+ }
199
+ }
200
+ if (subSitemaps.length > 0) {
201
+ const results = await Promise.all(subSitemaps.map(subUrl => this.fetchSitemapUrls(subUrl, baseUrl, visited)));
202
+ return [...paths, ...results.flat()];
203
+ }
204
+ return paths;
205
+ }
206
+ catch (err) {
207
+ Logger_1.default.error('IndexingService', `Error fetching sitemap ${sitemapUrl}`, err);
208
+ return [];
209
+ }
210
+ }
211
+ async crawlSite(siteUrl, startPaths = ['/'], concurrency = 5) {
212
+ const visited = new Set();
213
+ const results = new Map();
214
+ const base = siteUrl.endsWith('/') ? siteUrl.slice(0, -1) : siteUrl;
215
+ // 1. Try to discover URLs via Sitemap first
216
+ let sitemapPaths = null;
217
+ const sitemapUrl = this.chatWidget.sitemapUrl || `${base}/sitemap.xml`;
218
+ Logger_1.default.info('IndexingService', `Checking for sitemap at ${sitemapUrl}...`);
219
+ try {
220
+ const discovered = await this.fetchSitemapUrls(sitemapUrl, base);
221
+ if (discovered && discovered.length > 0) {
222
+ sitemapPaths = [...new Set(discovered)];
223
+ Logger_1.default.info('IndexingService', `Discovered ${sitemapPaths.length} URLs from sitemap.`);
224
+ }
225
+ else {
226
+ Logger_1.default.info('IndexingService', `No URLs found in sitemap or sitemap is unreachable.`);
227
+ }
228
+ }
229
+ catch (err) {
230
+ Logger_1.default.warn('IndexingService', `Failed to discover sitemap URLs. Falling back to recursive crawler.`, err);
231
+ }
232
+ const useSitemap = sitemapPaths !== null && sitemapPaths.length > 0;
233
+ const queue = useSitemap ? sitemapPaths : [...startPaths];
234
+ Logger_1.default.info('IndexingService', `Starting concurrent web scrape of ${base} with ${concurrency} threads (${useSitemap ? 'Sitemap Mode' : 'Link-following Mode'})...`);
235
+ const activeFetches = new Set();
236
+ const runWorker = async () => {
237
+ while (true) {
238
+ let currentPath = null;
239
+ for (let i = 0; i < queue.length; i++) {
240
+ const p = queue[i];
241
+ if (!visited.has(p) && !activeFetches.has(p)) {
242
+ currentPath = p;
243
+ queue.splice(i, 1);
244
+ break;
245
+ }
246
+ }
247
+ if (currentPath === null) {
248
+ if (activeFetches.size === 0 && queue.length === 0) {
249
+ break;
250
+ }
251
+ await new Promise(resolve => setTimeout(resolve, 50));
252
+ continue;
253
+ }
254
+ activeFetches.add(currentPath);
255
+ const fullUrl = `${base}${currentPath}`;
256
+ try {
257
+ Logger_1.default.info('IndexingService', `Fetching page for scraping: ${fullUrl}`);
258
+ const res = await fetchWithTimeout(fullUrl, {
259
+ headers: {
260
+ 'User-Agent': 'NextMinCrawler/1.0'
261
+ },
262
+ timeout: 30000
263
+ });
264
+ visited.add(currentPath);
265
+ activeFetches.delete(currentPath);
266
+ if (!res.ok) {
267
+ Logger_1.default.warn('IndexingService', `Failed to fetch page ${fullUrl} (${res.status})`);
268
+ continue;
269
+ }
270
+ const contentType = res.headers.get('content-type') || '';
271
+ if (!contentType.includes('text/html')) {
272
+ Logger_1.default.debug('IndexingService', `Skipping non-HTML URL: ${fullUrl} (${contentType})`);
273
+ continue;
274
+ }
275
+ const html = await res.text();
276
+ const text = stripHtml(html);
277
+ if (text && text.length >= 20) {
278
+ results.set(currentPath, text);
279
+ }
280
+ // ONLY extract and follow links if we are NOT using the sitemap list
281
+ if (!useSitemap) {
282
+ const links = this.extractInternalLinks(html, currentPath, base);
283
+ for (const link of links) {
284
+ if (!visited.has(link) && !activeFetches.has(link) && !queue.includes(link)) {
285
+ queue.push(link);
286
+ }
287
+ }
288
+ }
289
+ }
290
+ catch (err) {
291
+ visited.add(currentPath);
292
+ activeFetches.delete(currentPath);
293
+ Logger_1.default.error('IndexingService', `Error crawling page ${fullUrl}`, err);
294
+ }
295
+ await new Promise(resolve => setTimeout(resolve, 100));
296
+ }
297
+ };
298
+ const workers = Array.from({ length: concurrency }, () => runWorker());
299
+ await Promise.all(workers);
300
+ Logger_1.default.info('IndexingService', `Scrape complete. Scraped ${results.size} pages.`);
301
+ return results;
302
+ }
303
+ extractInternalLinks(html, currentPath, baseUrl) {
304
+ const links = [];
305
+ const hrefRegex = /<a[^>]+href=["']([^"']+)["']/gi;
306
+ let match;
307
+ while ((match = hrefRegex.exec(html)) !== null) {
308
+ let href = match[1].trim();
309
+ if (!href || href.startsWith('#') || href.startsWith('mailto:') || href.startsWith('tel:') || href.toLowerCase().startsWith('javascript:')) {
310
+ continue;
311
+ }
312
+ try {
313
+ const resolved = new URL(href, `${baseUrl}${currentPath}`);
314
+ if (resolved.origin === new URL(baseUrl).origin) {
315
+ let path = resolved.pathname;
316
+ if (resolved.search)
317
+ path += resolved.search;
318
+ if (path !== '/' && path.endsWith('/')) {
319
+ path = path.slice(0, -1);
320
+ }
321
+ const isAsset = /\.(png|jpe?g|gif|svg|css|js|ico|pdf|zip|mp4|json|xml)$/i.test(resolved.pathname);
322
+ if (!isAsset) {
323
+ links.push(path);
324
+ }
325
+ }
326
+ }
327
+ catch {
328
+ // Ignore invalid URL
329
+ }
330
+ }
331
+ return [...new Set(links)];
332
+ }
333
+ // ---------- Database Schema Indexing ----------
334
+ async indexDatabaseSchemas() {
335
+ const results = new Map();
336
+ if (!this.dbAdapter)
337
+ return results;
338
+ const schemas = this.chatWidget.whitelabelSchemas || [];
339
+ if (schemas.length === 0)
340
+ return results;
341
+ Logger_1.default.info('IndexingService', `Indexing ${schemas.length} database schemas: ${schemas.join(', ')}...`);
342
+ const siteUrl = this.chatWidget.siteUrl || 'https://doctors24.bd';
343
+ for (const schemaName of schemas) {
344
+ try {
345
+ const collectionName = schemaName.toLowerCase();
346
+ let docs = [];
347
+ if (typeof this.dbAdapter.read === 'function') {
348
+ docs = await this.dbAdapter.read(collectionName, {}, 50000, 0);
349
+ }
350
+ else if (this.dbAdapter.db && typeof this.dbAdapter.db.collection === 'function') {
351
+ docs = await this.dbAdapter.db.collection(collectionName).find({}).toArray();
352
+ }
353
+ if (!Array.isArray(docs))
354
+ continue;
355
+ Logger_1.default.info('IndexingService', `Schema '${schemaName}': found ${docs.length} database records.`);
356
+ for (const doc of docs) {
357
+ if (!doc)
358
+ continue;
359
+ const slug = doc.slug || doc._id || doc.id;
360
+ if (!slug)
361
+ continue;
362
+ let entityType = schemaName.replace(/s$/i, '').toLowerCase();
363
+ if (entityType === 'specialitie' || entityType === 'specialty')
364
+ entityType = 'speciality';
365
+ const canonicalPath = `/${entityType}/${slug}`;
366
+ const fullUrl = `${siteUrl}${canonicalPath}`;
367
+ const title = doc.fullName || doc.name || doc.title || doc.headline || `${schemaName} Record`;
368
+ let apptText = '';
369
+ if (doc.appointments) {
370
+ try {
371
+ const apps = typeof doc.appointments === 'string' ? JSON.parse(doc.appointments) : doc.appointments;
372
+ if (Array.isArray(apps)) {
373
+ apptText = apps
374
+ .map((a) => `- Hospital/Chamber: ${a.hospital?.name || a.name || ''}\n Address: ${a.hospital?.address || a.hospital?.physicalLocation || a.address || ''}\n Visiting Hours: ${a.visitingHours || ''}\n Phone / Appointment: ${a.phone || a.hospital?.telephone || doc.phone || ''}`)
375
+ .join('\n\n');
376
+ }
377
+ }
378
+ catch { }
379
+ }
380
+ let formattedDetails = '';
381
+ if (doc.degree || doc.fellowshipsOrTrainings)
382
+ formattedDetails += `- Qualifications: ${doc.degree || ''} ${doc.fellowshipsOrTrainings || ''}\n`;
383
+ if (doc.speciality || doc.specialty)
384
+ formattedDetails += `- Specialty: ${doc.speciality || doc.specialty || ''}\n`;
385
+ if (doc.designationAndDepartment || doc.designation)
386
+ formattedDetails += `- Designation: ${doc.designationAndDepartment || doc.designation || ''}\n`;
387
+ if (doc.workPlace || doc.workplace)
388
+ formattedDetails += `- Workplace: ${doc.workPlace || doc.workplace || ''}\n`;
389
+ if (doc.phone || doc.telephone)
390
+ formattedDetails += `- Primary Phone: ${doc.phone || doc.telephone || ''}\n`;
391
+ if (doc.address || doc.physicalLocation)
392
+ formattedDetails += `- Address: ${doc.address || doc.physicalLocation || ''}\n`;
393
+ const aboutText = stripHtml(doc.aboutYou || doc.aboutHospital || doc.description || doc.body || doc.content || '');
394
+ const content = `DIRECTORY DOCUMENT
395
+ PAGE_TITLE: ${title} | ${schemaName}
396
+ CANONICAL_URL: ${fullUrl}
397
+
398
+ INFORMATION:
399
+ - Name/Title: ${title}
400
+ ${formattedDetails}
401
+ ${apptText ? `\nCHAMBER & APPOINTMENT DETAILS:\n${apptText}\n` : ''}
402
+ ${aboutText ? `\nABOUT:\n${aboutText}\n` : ''}
403
+ EXPLICIT ROUTING INSTRUCTION FOR ASSISTANT:
404
+ When users ask about this item or entity, always direct them to canonical URL: ${fullUrl}
405
+ `;
406
+ results.set(canonicalPath, content);
407
+ }
408
+ }
409
+ catch (err) {
410
+ Logger_1.default.error('IndexingService', `Failed to index database schema '${schemaName}'`, err);
411
+ }
412
+ }
413
+ return results;
414
+ }
415
+ // ---------- Entrypoint ----------
416
+ async runIndex() {
417
+ if (!this.chatWidget.enabled) {
418
+ Logger_1.default.info('IndexingService', 'Chat widget is disabled. Skipping index.');
419
+ return;
420
+ }
421
+ Logger_1.default.info('IndexingService', 'Starting LOCAL-ONLY indexing pipeline...');
422
+ const activeKeys = new Set();
423
+ const manifestHashes = this.force ? {} : this.readManifest();
424
+ const currentManifest = { ...manifestHashes };
425
+ let hasErrors = false;
426
+ const allPageEntries = new Map();
427
+ // 1. Scrape Site Pages (Crawler)
428
+ if (this.chatWidget.siteUrl && this.chatWidget.sourceType !== 'db') {
429
+ const paths = this.chatWidget.scrapePaths || ['/'];
430
+ try {
431
+ const concurrency = this.chatWidget.concurrency || 5;
432
+ const scrapedPages = await this.crawlSite(this.chatWidget.siteUrl, paths, concurrency);
433
+ for (const [k, v] of scrapedPages.entries())
434
+ allPageEntries.set(k, v);
435
+ }
436
+ catch (err) {
437
+ Logger_1.default.error('IndexingService', 'Site crawl step failed', err);
438
+ hasErrors = true;
439
+ }
440
+ }
441
+ // 2. Index Database Schemas (DB Exporter)
442
+ if (this.dbAdapter && (this.chatWidget.sourceType === 'db' || this.chatWidget.sourceType === 'both' || !this.chatWidget.sourceType)) {
443
+ try {
444
+ const dbPages = await this.indexDatabaseSchemas();
445
+ for (const [k, v] of dbPages.entries())
446
+ allPageEntries.set(k, v);
447
+ }
448
+ catch (err) {
449
+ Logger_1.default.error('IndexingService', 'Database schema indexing failed', err);
450
+ hasErrors = true;
451
+ }
452
+ }
453
+ if (allPageEntries.size > 0) {
454
+ Logger_1.default.info('IndexingService', `Processing ${allPageEntries.size} total page entries...`);
455
+ const concurrency = this.chatWidget.concurrency || 5;
456
+ try {
457
+ let processedCount = 0;
458
+ let skippedCount = 0;
459
+ const entries = Array.from(allPageEntries.entries());
460
+ let index = 0;
461
+ // Concurrent Indexing Workers
462
+ const indexWorker = async () => {
463
+ while (true) {
464
+ let entry = null;
465
+ if (index < entries.length) {
466
+ entry = entries[index++];
467
+ }
468
+ if (!entry)
469
+ break;
470
+ const [pathname, text] = entry;
471
+ try {
472
+ const key = `page-${pathname.replace(/[^a-zA-Z0-9_\-]/g, '_') || 'index'}`;
473
+ activeKeys.add(key);
474
+ const hash = crypto_1.default.createHash('md5').update(text).digest('hex');
475
+ if (manifestHashes[key] === hash) {
476
+ skippedCount++;
477
+ currentManifest[key] = hash;
478
+ continue;
479
+ }
480
+ // Save locally
481
+ this.writeScrapedPageLocal(key, text);
482
+ processedCount++;
483
+ // Incrementally write manifest to disk to save progress immediately
484
+ currentManifest[key] = hash;
485
+ this.writeManifest(currentManifest);
486
+ await new Promise(resolve => setTimeout(resolve, 50)); // Gentle throttle
487
+ }
488
+ catch (err) {
489
+ Logger_1.default.error('IndexingService', `Failed to index scraped page ${pathname}`, err);
490
+ hasErrors = true;
491
+ }
492
+ }
493
+ };
494
+ Logger_1.default.info('IndexingService', `Processing in parallel using ${concurrency} concurrent workers...`);
495
+ const indexWorkers = Array.from({ length: concurrency }, () => indexWorker());
496
+ await Promise.all(indexWorkers);
497
+ Logger_1.default.info('IndexingService', `Completed indexing scraped pages: ${processedCount} updated, ${skippedCount} unchanged (loaded from cache).`);
498
+ }
499
+ catch (err) {
500
+ Logger_1.default.error('IndexingService', 'Site crawl step failed', err);
501
+ hasErrors = true;
502
+ }
503
+ }
504
+ // 2. Prune deleted files
505
+ if (!hasErrors) {
506
+ for (const key of Object.keys(manifestHashes)) {
507
+ if (!activeKeys.has(key)) {
508
+ Logger_1.default.info('IndexingService', `Pruning deleted document: ${key}`);
509
+ // Remove local file
510
+ this.deleteScrapedPageLocal(key);
511
+ }
512
+ }
513
+ }
514
+ else {
515
+ Logger_1.default.info('IndexingService', 'Skipping pruning stage due to errors encountered during crawling/indexing.');
516
+ }
517
+ // 3. Update final local manifest
518
+ const finalManifest = {};
519
+ for (const key of activeKeys) {
520
+ if (currentManifest[key]) {
521
+ finalManifest[key] = currentManifest[key];
522
+ }
523
+ }
524
+ // If errors occurred, preserve unvisited entries in the manifest
525
+ if (hasErrors) {
526
+ for (const key of Object.keys(manifestHashes)) {
527
+ if (!activeKeys.has(key)) {
528
+ finalManifest[key] = manifestHashes[key];
529
+ }
530
+ }
531
+ }
532
+ this.writeManifest(finalManifest);
533
+ // Sync newly indexed files to OpenAI Vector Store if enabled and API Key is set
534
+ if (this.chatWidget.enableOpenai !== false && process.env.OPENAI_API_KEY) {
535
+ try {
536
+ Logger_1.default.info('IndexingService', 'Triggering OpenAI Vector Store synchronization...');
537
+ const openAI = new OpenAIService_1.OpenAIService();
538
+ await openAI.syncFilesToVectorStore(this.force);
539
+ }
540
+ catch (err) {
541
+ Logger_1.default.error('IndexingService', `OpenAI file sync failed: ${err.message || err}`);
542
+ hasErrors = true;
543
+ }
544
+ }
545
+ if (hasErrors) {
546
+ throw new Error('Indexing completed with errors. Please check the logs.');
547
+ }
548
+ Logger_1.default.info('IndexingService', `Incremental indexing complete.`);
549
+ }
550
+ }
551
+ exports.IndexingService = IndexingService;
552
+ async function reindex(config) {
553
+ const indexer = new IndexingService(config.chatWidget, config.force, config.dbAdapter);
554
+ await indexer.runIndex();
555
+ }