pi-duckdb-search 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +173 -0
- package/duckdb-search-build.cjs +360 -0
- package/duckdb-search-test.cjs +576 -0
- package/extensions/duckdb-search.ts +803 -0
- package/package.json +60 -0
|
@@ -0,0 +1,803 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* duckdb-search.ts — DuckDB-powered session transcript search for Pi.
|
|
3
|
+
*
|
|
4
|
+
* v2.4: All SQL paths use prepared statements or escaped literals.
|
|
5
|
+
* Test suite added.
|
|
6
|
+
*
|
|
7
|
+
* Tools:
|
|
8
|
+
* session_search — BM25 keyword search
|
|
9
|
+
* session_semantic — Semantic (vector cosine) search
|
|
10
|
+
* session_hybrid — Combined BM25 + cosine via Reciprocal Rank Fusion
|
|
11
|
+
* session_read — Read a specific entry by file path + line number
|
|
12
|
+
* session_status — Index status and entry count
|
|
13
|
+
*
|
|
14
|
+
* Dependencies:
|
|
15
|
+
* @duckdb/node-api — DuckDB native binary (FTS + FLOAT[] arrays)
|
|
16
|
+
* @huggingface/transformers — ONNX embeddings (all-MiniLM-L6-v2, ~23MB)
|
|
17
|
+
*
|
|
18
|
+
* DuckDB optimizations applied (v2.3):
|
|
19
|
+
* - Read-only mode: access_mode='READ_ONLY' — no WAL, no locks, no MVCC
|
|
20
|
+
* - Prepared statements: connection.run(sql, params) — SQL injection safe
|
|
21
|
+
* - FTS overwrite=true: no need to drop before recreate
|
|
22
|
+
* - ignore_errors=true: skip malformed JSONL lines
|
|
23
|
+
* - Race guard: ensureInProgress flag prevents concurrent builds
|
|
24
|
+
*
|
|
25
|
+
* Verified NOT worth doing:
|
|
26
|
+
* - FTS incremental: not available in DuckDB v1.5.6 (no 'incremental' parameter)
|
|
27
|
+
* - HNSW/vss: brute-force cosine is 13ms at 17K vectors; HNSW persistence is experimental
|
|
28
|
+
* - Appender API: 2x faster inserts but FLOAT[] array binding is broken in node-api
|
|
29
|
+
* - threads=1: actually slower (27ms vs 19ms with default 4 threads)
|
|
30
|
+
* - preserve_insertion_order=false: no size difference at our scale
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
import { existsSync, mkdirSync, readdirSync, statSync } from "node:fs";
|
|
34
|
+
import { join } from "node:path";
|
|
35
|
+
import { Type } from "typebox";
|
|
36
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
37
|
+
|
|
38
|
+
const HOME = process.env.HOME ?? "/home/cpagan";
|
|
39
|
+
const SESSIONS_DIR = join(HOME, ".pi", "agent", "sessions");
|
|
40
|
+
const DB_DIR = join(HOME, ".pi", "agent", "duckdb-search");
|
|
41
|
+
const DB_PATH = join(DB_DIR, "sessions.duckdb");
|
|
42
|
+
const MODEL_ID = "Xenova/all-MiniLM-L6-v2";
|
|
43
|
+
const EMBED_DIM = 384;
|
|
44
|
+
const EXCLUDE_RECENT_MS = 5 * 60 * 1000;
|
|
45
|
+
const MAX_CONTENT_CHARS = 2000;
|
|
46
|
+
const RRF_K = 60;
|
|
47
|
+
|
|
48
|
+
// Lazy module loading
|
|
49
|
+
let duckdbModule: any = null;
|
|
50
|
+
async function getDuckDB() {
|
|
51
|
+
if (duckdbModule) return duckdbModule;
|
|
52
|
+
duckdbModule = await import("@duckdb/node-api");
|
|
53
|
+
return duckdbModule;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
let transformersModule: any = null;
|
|
57
|
+
async function getTransformers() {
|
|
58
|
+
if (transformersModule) return transformersModule;
|
|
59
|
+
transformersModule = await import("@huggingface/transformers");
|
|
60
|
+
return transformersModule;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
interface IndexState {
|
|
64
|
+
ready: boolean;
|
|
65
|
+
indexing: boolean;
|
|
66
|
+
entryCount: number;
|
|
67
|
+
sessionCount: number;
|
|
68
|
+
lastBuilt: string | null;
|
|
69
|
+
error: string | null;
|
|
70
|
+
hasEmbeddings: boolean;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const state: IndexState = {
|
|
74
|
+
ready: false,
|
|
75
|
+
indexing: false,
|
|
76
|
+
entryCount: 0,
|
|
77
|
+
sessionCount: 0,
|
|
78
|
+
lastBuilt: null,
|
|
79
|
+
error: null,
|
|
80
|
+
hasEmbeddings: false,
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
let dbInstance: any = null;
|
|
84
|
+
let dbConn: any = null;
|
|
85
|
+
let embedder: any = null;
|
|
86
|
+
let ensureInProgress = false; // race guard
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Get a read-only DuckDB connection.
|
|
90
|
+
* Read-only mode: no WAL creation, no file locks, no MVCC overhead.
|
|
91
|
+
* The cron build script writes the DB; the extension only reads it.
|
|
92
|
+
*/
|
|
93
|
+
async function getConnection() {
|
|
94
|
+
if (dbConn) return dbConn;
|
|
95
|
+
const { DuckDBInstance } = await getDuckDB();
|
|
96
|
+
mkdirSync(DB_DIR, { recursive: true });
|
|
97
|
+
if (existsSync(DB_PATH)) {
|
|
98
|
+
// Read-only mode when DB exists (normal path)
|
|
99
|
+
dbInstance = await DuckDBInstance.create(DB_PATH, { access_mode: "READ_ONLY" });
|
|
100
|
+
} else {
|
|
101
|
+
// Read-write mode for initial build
|
|
102
|
+
dbInstance = await DuckDBInstance.create(DB_PATH);
|
|
103
|
+
}
|
|
104
|
+
dbConn = await dbInstance.connect();
|
|
105
|
+
return dbConn;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
async function getEmbedder() {
|
|
109
|
+
if (embedder) return embedder;
|
|
110
|
+
const { pipeline } = await getTransformers();
|
|
111
|
+
embedder = await pipeline("feature-extraction", MODEL_ID);
|
|
112
|
+
return embedder;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
async function embed(text: string): Promise<Float32Array> {
|
|
116
|
+
const pipe = await getEmbedder();
|
|
117
|
+
const output = await pipe(text, { pooling: "mean", normalize: true });
|
|
118
|
+
return new Float32Array(output.data);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function findSessionFiles(excludeRecent: boolean): string[] {
|
|
122
|
+
const files: string[] = [];
|
|
123
|
+
const now = Date.now();
|
|
124
|
+
function scan(dir: string) {
|
|
125
|
+
if (!existsSync(dir)) return;
|
|
126
|
+
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
|
127
|
+
const fullPath = join(dir, entry.name);
|
|
128
|
+
if (entry.isDirectory()) {
|
|
129
|
+
scan(fullPath);
|
|
130
|
+
} else if (entry.name.endsWith(".jsonl")) {
|
|
131
|
+
if (excludeRecent) {
|
|
132
|
+
try {
|
|
133
|
+
const st = statSync(fullPath);
|
|
134
|
+
if (now - st.mtimeMs < EXCLUDE_RECENT_MS) continue;
|
|
135
|
+
} catch { /* ignore */ }
|
|
136
|
+
}
|
|
137
|
+
files.push(fullPath);
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
scan(SESSIONS_DIR);
|
|
142
|
+
return files;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Metadata access using prepared statements.
|
|
147
|
+
*/
|
|
148
|
+
async function getMetadata(key: string): Promise<string | null> {
|
|
149
|
+
try {
|
|
150
|
+
const conn = await getConnection();
|
|
151
|
+
const r = await conn.run(
|
|
152
|
+
"SELECT value FROM _metadata WHERE key = ?",
|
|
153
|
+
[key]
|
|
154
|
+
);
|
|
155
|
+
const rows = await r.getRows();
|
|
156
|
+
return rows.length > 0 ? String(rows[0][0]) : null;
|
|
157
|
+
} catch {
|
|
158
|
+
return null;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
async function setMetadata(key: string, value: string): Promise<void> {
|
|
163
|
+
try {
|
|
164
|
+
const conn = await getConnection();
|
|
165
|
+
await conn.run("CREATE TABLE IF NOT EXISTS _metadata (key VARCHAR PRIMARY KEY, value VARCHAR);");
|
|
166
|
+
await conn.run(
|
|
167
|
+
"INSERT OR REPLACE INTO _metadata VALUES (?, ?)",
|
|
168
|
+
[key, value]
|
|
169
|
+
);
|
|
170
|
+
} catch { /* ignore */ }
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
async function checkExistingDatabase(): Promise<boolean> {
|
|
174
|
+
if (!existsSync(DB_PATH)) return false;
|
|
175
|
+
|
|
176
|
+
try {
|
|
177
|
+
const conn = await getConnection();
|
|
178
|
+
|
|
179
|
+
const tableCheck = await conn.run(
|
|
180
|
+
"SELECT COUNT(*) FROM information_schema.tables WHERE table_name = 'session_entries'"
|
|
181
|
+
);
|
|
182
|
+
if (Number((await tableCheck.getRows())[0][0]) === 0) return false;
|
|
183
|
+
|
|
184
|
+
const entryCount = Number(await getMetadata("entry_count") ?? "0");
|
|
185
|
+
const sessionCount = Number(await getMetadata("session_count") ?? "0");
|
|
186
|
+
const hasEmbs = (await getMetadata("has_embeddings")) === "true";
|
|
187
|
+
const lastBuilt = await getMetadata("last_built");
|
|
188
|
+
|
|
189
|
+
if (entryCount === 0) return false;
|
|
190
|
+
|
|
191
|
+
const filesOnDisk = findSessionFiles(true);
|
|
192
|
+
if (sessionCount < filesOnDisk.length * 0.9) return false;
|
|
193
|
+
|
|
194
|
+
state.ready = true;
|
|
195
|
+
state.entryCount = entryCount;
|
|
196
|
+
state.sessionCount = sessionCount;
|
|
197
|
+
state.hasEmbeddings = hasEmbs;
|
|
198
|
+
state.lastBuilt = lastBuilt;
|
|
199
|
+
state.error = null;
|
|
200
|
+
|
|
201
|
+
console.log(`[duckdb-search] Using existing database: ${entryCount} entries, ${sessionCount} sessions, embeddings: ${hasEmbs ? "ready" : "no"}`);
|
|
202
|
+
return true;
|
|
203
|
+
} catch (e) {
|
|
204
|
+
console.error("[duckdb-search] checkExistingDatabase failed:", e);
|
|
205
|
+
return false;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Full content extraction: text + thinking + tool call names.
|
|
211
|
+
* Captures thinking blocks (60% of previously "empty" entries had thinking content).
|
|
212
|
+
*/
|
|
213
|
+
const CONTENT_EXTRACTION_SQL = `
|
|
214
|
+
CASE
|
|
215
|
+
WHEN json_type(json->'message'->'content') = 'ARRAY'
|
|
216
|
+
THEN substring(
|
|
217
|
+
concat(
|
|
218
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].text'), ' '), ''),
|
|
219
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].thinking'), ' '), ''),
|
|
220
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].name'), ' '), '')
|
|
221
|
+
),
|
|
222
|
+
1, ${MAX_CONTENT_CHARS}
|
|
223
|
+
)
|
|
224
|
+
ELSE substring(COALESCE(json->'message'->>'content', json->>'content', ''), 1, ${MAX_CONTENT_CHARS})
|
|
225
|
+
END
|
|
226
|
+
`;
|
|
227
|
+
|
|
228
|
+
async function buildIndex(): Promise<{ entries: number; sessions: number; error?: string }> {
|
|
229
|
+
if (state.indexing) return { entries: state.entryCount, sessions: state.sessionCount, error: "Indexing already in progress" };
|
|
230
|
+
state.indexing = true;
|
|
231
|
+
state.error = null;
|
|
232
|
+
|
|
233
|
+
try {
|
|
234
|
+
const conn = await getConnection();
|
|
235
|
+
await conn.run("INSTALL fts; LOAD fts;");
|
|
236
|
+
|
|
237
|
+
try { await conn.run("PRAGMA drop_fts_index('session_entries');"); } catch { /* no index yet */ }
|
|
238
|
+
await conn.run("DROP TABLE IF EXISTS session_entries;");
|
|
239
|
+
await conn.run("DROP TABLE IF EXISTS session_fts_index;");
|
|
240
|
+
|
|
241
|
+
const files = findSessionFiles(true);
|
|
242
|
+
if (files.length === 0) {
|
|
243
|
+
state.indexing = false;
|
|
244
|
+
state.error = "No JSONL files found in " + SESSIONS_DIR;
|
|
245
|
+
return { entries: 0, sessions: 0, error: state.error };
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
const glob = join(SESSIONS_DIR, "**", "*.jsonl").replace(/\\/g, "/");
|
|
249
|
+
const escapedGlob = esc(glob);
|
|
250
|
+
|
|
251
|
+
await conn.run(`
|
|
252
|
+
CREATE TABLE session_entries AS
|
|
253
|
+
SELECT
|
|
254
|
+
filename,
|
|
255
|
+
line_number,
|
|
256
|
+
json->>'type' AS entry_type,
|
|
257
|
+
json->>'id' AS entry_id,
|
|
258
|
+
json->>'parentId' AS parent_id,
|
|
259
|
+
json->>'timestamp' AS timestamp,
|
|
260
|
+
json->'message'->>'role' AS role,
|
|
261
|
+
json->>'customType' AS custom_type,
|
|
262
|
+
${CONTENT_EXTRACTION_SQL} AS content_text
|
|
263
|
+
FROM (
|
|
264
|
+
SELECT
|
|
265
|
+
filename,
|
|
266
|
+
row_number() OVER (PARTITION BY filename ORDER BY filename) AS line_number,
|
|
267
|
+
json
|
|
268
|
+
FROM read_json_objects('${escapedGlob}', format='newline_delimited', filename=true, ignore_errors=true)
|
|
269
|
+
)
|
|
270
|
+
WHERE json->>'type' IN ('message', 'custom_message')
|
|
271
|
+
`);
|
|
272
|
+
|
|
273
|
+
// FTS with overwrite=true — no need to call drop_fts_index first
|
|
274
|
+
await conn.run("PRAGMA create_fts_index('session_entries', 'entry_id', 'content_text', 'role', stemmer='porter', overwrite=true);");
|
|
275
|
+
|
|
276
|
+
const countResult = await conn.run("SELECT COUNT(*) AS c FROM session_entries");
|
|
277
|
+
const entryCount = Number((await countResult.getRows())[0][0]);
|
|
278
|
+
const sessResult = await conn.run("SELECT COUNT(DISTINCT filename) AS c FROM session_entries");
|
|
279
|
+
const sessionCount = Number((await sessResult.getRows())[0][0]);
|
|
280
|
+
|
|
281
|
+
state.ready = true;
|
|
282
|
+
state.entryCount = entryCount;
|
|
283
|
+
state.sessionCount = sessionCount;
|
|
284
|
+
state.lastBuilt = new Date().toISOString();
|
|
285
|
+
state.hasEmbeddings = false;
|
|
286
|
+
|
|
287
|
+
await setMetadata("entry_count", String(entryCount));
|
|
288
|
+
await setMetadata("session_count", String(sessionCount));
|
|
289
|
+
await setMetadata("last_built", state.lastBuilt);
|
|
290
|
+
await setMetadata("has_embeddings", "false");
|
|
291
|
+
|
|
292
|
+
generateEmbeddings().catch((e) => {
|
|
293
|
+
console.error("[duckdb-search] embedding generation failed:", e);
|
|
294
|
+
});
|
|
295
|
+
|
|
296
|
+
return { entries: entryCount, sessions: sessionCount };
|
|
297
|
+
} catch (e) {
|
|
298
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
299
|
+
state.error = msg;
|
|
300
|
+
state.ready = false;
|
|
301
|
+
return { entries: 0, sessions: 0, error: msg };
|
|
302
|
+
} finally {
|
|
303
|
+
state.indexing = false;
|
|
304
|
+
}
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
async function generateEmbeddings(): Promise<void> {
|
|
308
|
+
const conn = await getConnection();
|
|
309
|
+
await conn.run(`ALTER TABLE session_entries ADD COLUMN IF NOT EXISTS embedding FLOAT[${EMBED_DIM}];`);
|
|
310
|
+
|
|
311
|
+
const result = await conn.run("SELECT entry_id, content_text FROM session_entries WHERE embedding IS NULL AND length(content_text) > 0 ORDER BY rowid;");
|
|
312
|
+
const rows = await result.getRows();
|
|
313
|
+
|
|
314
|
+
if (rows.length === 0) {
|
|
315
|
+
state.hasEmbeddings = true;
|
|
316
|
+
await setMetadata("has_embeddings", "true");
|
|
317
|
+
return;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
console.log(`[duckdb-search] Generating embeddings for ${rows.length} entries...`);
|
|
321
|
+
|
|
322
|
+
await conn.run(`CREATE TABLE IF NOT EXISTS _emb_temp (entry_id VARCHAR, embedding FLOAT[${EMBED_DIM}]);`);
|
|
323
|
+
await conn.run("DELETE FROM _emb_temp;");
|
|
324
|
+
|
|
325
|
+
const batchSize = 64;
|
|
326
|
+
for (let i = 0; i < rows.length; i += batchSize) {
|
|
327
|
+
const batch = rows.slice(i, i + batchSize);
|
|
328
|
+
const placeholders: string[] = [];
|
|
329
|
+
const insertParams: any[] = [];
|
|
330
|
+
|
|
331
|
+
for (const row of batch) {
|
|
332
|
+
const entryId = row[0] as string;
|
|
333
|
+
const text = String(row[1] || "").slice(0, MAX_CONTENT_CHARS);
|
|
334
|
+
if (!text) continue;
|
|
335
|
+
const embedding = await embed(text);
|
|
336
|
+
const arrStr = embeddingToDuckArray(embedding);
|
|
337
|
+
placeholders.push(`(?, ?::FLOAT[${EMBED_DIM}])`);
|
|
338
|
+
insertParams.push(entryId, arrStr);
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
if (placeholders.length > 0) {
|
|
342
|
+
await conn.run(
|
|
343
|
+
`INSERT INTO _emb_temp VALUES ${placeholders.join(",")};`,
|
|
344
|
+
insertParams
|
|
345
|
+
);
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
if ((i + batchSize) % 512 === 0 || i + batchSize >= rows.length) {
|
|
349
|
+
console.log(`[duckdb-search] Embedded ${Math.min(i + batchSize, rows.length)}/${rows.length}`);
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
await conn.run("UPDATE session_entries SET embedding = (SELECT embedding FROM _emb_temp WHERE _emb_temp.entry_id = session_entries.entry_id) WHERE entry_id IN (SELECT entry_id FROM _emb_temp);");
|
|
354
|
+
await conn.run("DROP TABLE _emb_temp;");
|
|
355
|
+
await conn.run("CHECKPOINT;");
|
|
356
|
+
|
|
357
|
+
state.hasEmbeddings = true;
|
|
358
|
+
await setMetadata("has_embeddings", "true");
|
|
359
|
+
console.log(`[duckdb-search] Embeddings complete: ${rows.length} entries`);
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
/**
|
|
363
|
+
* Ensure the index is ready — with race guard.
|
|
364
|
+
* ensureInProgress flag prevents concurrent buildIndex() calls
|
|
365
|
+
* if session_start event and first search fire simultaneously.
|
|
366
|
+
*/
|
|
367
|
+
async function ensureIndex(): Promise<boolean> {
|
|
368
|
+
if (state.ready) return true;
|
|
369
|
+
if (state.indexing || ensureInProgress) return false;
|
|
370
|
+
|
|
371
|
+
ensureInProgress = true;
|
|
372
|
+
try {
|
|
373
|
+
const existing = await checkExistingDatabase();
|
|
374
|
+
if (existing) return true;
|
|
375
|
+
|
|
376
|
+
const result = await buildIndex();
|
|
377
|
+
return !result.error;
|
|
378
|
+
} finally {
|
|
379
|
+
ensureInProgress = false;
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/** Format an embedding as a DuckDB array literal for cosine queries. */
|
|
384
|
+
function embeddingToDuckArray(emb: Float32Array): string {
|
|
385
|
+
return `[${Array.from(emb).map((v) => v.toFixed(6)).join(",")}]`;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** Escape a string for use in a DuckDB single-quoted string literal. */
|
|
389
|
+
function esc(s: string): string {
|
|
390
|
+
return s.replace(/'/g, "''");
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
export default function (pi: ExtensionAPI) {
|
|
394
|
+
// Background build on session_start — loads DB before first search
|
|
395
|
+
pi.on("session_start", async () => {
|
|
396
|
+
ensureIndex().catch((e) => {
|
|
397
|
+
console.error("[duckdb-search] background ensureIndex failed:", e);
|
|
398
|
+
});
|
|
399
|
+
});
|
|
400
|
+
|
|
401
|
+
// session_search — BM25 keyword search (prepared statements)
|
|
402
|
+
pi.registerTool({
|
|
403
|
+
name: "session_search",
|
|
404
|
+
label: "Session Search",
|
|
405
|
+
description:
|
|
406
|
+
"Search past Pi session transcripts using BM25 keyword ranking. " +
|
|
407
|
+
"Returns matching entries with file path, line number, role, timestamp, and snippet. " +
|
|
408
|
+
"Use distinctive keywords for best results. Excludes the current active session.",
|
|
409
|
+
parameters: Type.Object({
|
|
410
|
+
query: Type.String({ description: "Search terms. Multiple words are ANDed for BM25 ranking." }),
|
|
411
|
+
role: Type.Optional(Type.String({ description: "Filter by role: user, assistant, toolResult, or custom" })),
|
|
412
|
+
limit: Type.Optional(Type.Number({ description: "Max results (default 20, max 50)" })),
|
|
413
|
+
offset: Type.Optional(Type.Number({ description: "Pagination offset (default 0)" })),
|
|
414
|
+
}),
|
|
415
|
+
|
|
416
|
+
async execute(_toolCallId, params, _signal, _onUpdate, _ctx) {
|
|
417
|
+
const query = (params.query as string)?.trim();
|
|
418
|
+
if (!query) return { content: [{ type: "text", text: "Error: query is required" }], details: { error: "query required" } };
|
|
419
|
+
|
|
420
|
+
const limit = Math.min(Math.max(1, (params.limit as number) ?? 20), 50);
|
|
421
|
+
const offset = Math.max(0, (params.offset as number) ?? 0);
|
|
422
|
+
const role = (params.role as string) ?? null;
|
|
423
|
+
|
|
424
|
+
const ready = await ensureIndex();
|
|
425
|
+
if (!ready) return { content: [{ type: "text", text: `Index not ready: ${state.error ?? "building..."}` }], details: { error: state.error ?? "indexing" } };
|
|
426
|
+
|
|
427
|
+
try {
|
|
428
|
+
const conn = await getConnection();
|
|
429
|
+
await conn.run("LOAD fts;");
|
|
430
|
+
|
|
431
|
+
// Prepared statement: ? placeholders for SQL injection safety
|
|
432
|
+
let sql: string;
|
|
433
|
+
let sqlParams: any[];
|
|
434
|
+
|
|
435
|
+
if (role) {
|
|
436
|
+
sql = `
|
|
437
|
+
SELECT filename, line_number, role, timestamp, entry_type,
|
|
438
|
+
substring(content_text, 1, 500) AS snippet,
|
|
439
|
+
fts_main_session_entries.match_bm25(entry_id, ?) AS score
|
|
440
|
+
FROM session_entries
|
|
441
|
+
WHERE fts_main_session_entries.match_bm25(entry_id, ?) IS NOT NULL
|
|
442
|
+
AND role = ?
|
|
443
|
+
ORDER BY score DESC
|
|
444
|
+
LIMIT ? OFFSET ?
|
|
445
|
+
`;
|
|
446
|
+
sqlParams = [query, query, role, limit, offset];
|
|
447
|
+
} else {
|
|
448
|
+
sql = `
|
|
449
|
+
SELECT filename, line_number, role, timestamp, entry_type,
|
|
450
|
+
substring(content_text, 1, 500) AS snippet,
|
|
451
|
+
fts_main_session_entries.match_bm25(entry_id, ?) AS score
|
|
452
|
+
FROM session_entries
|
|
453
|
+
WHERE fts_main_session_entries.match_bm25(entry_id, ?) IS NOT NULL
|
|
454
|
+
ORDER BY score DESC
|
|
455
|
+
LIMIT ? OFFSET ?
|
|
456
|
+
`;
|
|
457
|
+
sqlParams = [query, query, limit, offset];
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
const result = await conn.run(sql, sqlParams);
|
|
461
|
+
const rows = await result.getRows();
|
|
462
|
+
|
|
463
|
+
if (rows.length === 0) return { content: [{ type: "text", text: `No results found for: ${query}` }], details: { query, result_count: 0, mode: "bm25" } };
|
|
464
|
+
|
|
465
|
+
const lines: string[] = [`Found ${rows.length} result(s) for: ${query}`, ""];
|
|
466
|
+
for (const row of rows) {
|
|
467
|
+
const [filename, lineNum, entryRole, ts, entryType, snippet, score] = row;
|
|
468
|
+
const shortFile = String(filename).replace(SESSIONS_DIR + "/", "");
|
|
469
|
+
lines.push(`--- ${shortFile}:${lineNum} (score: ${Number(score).toFixed(2)}) ---`);
|
|
470
|
+
lines.push(` role: ${entryRole || "N/A"} | type: ${entryType} | time: ${ts || "N/A"}`);
|
|
471
|
+
lines.push(` text: ${String(snippet || "").slice(0, 300)}`);
|
|
472
|
+
lines.push("");
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
return { content: [{ type: "text", text: lines.join("\n") }], details: { query, result_count: rows.length, limit, offset, mode: "bm25" } };
|
|
476
|
+
} catch (e) {
|
|
477
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
478
|
+
return { content: [{ type: "text", text: `Search error: ${msg}` }], details: { error: msg } };
|
|
479
|
+
}
|
|
480
|
+
},
|
|
481
|
+
});
|
|
482
|
+
|
|
483
|
+
// session_semantic — vector cosine similarity (prepared statements)
|
|
484
|
+
pi.registerTool({
|
|
485
|
+
name: "session_semantic",
|
|
486
|
+
label: "Session Semantic Search",
|
|
487
|
+
description:
|
|
488
|
+
"Search past Pi session transcripts by meaning, not just keywords. " +
|
|
489
|
+
"Uses local ONNX embeddings (all-MiniLM-L6-v2) and cosine similarity. " +
|
|
490
|
+
"Example: searching 'RFC 5549 extended nexthop' can find entries about 'BGP unnumbered'.",
|
|
491
|
+
parameters: Type.Object({
|
|
492
|
+
query: Type.String({ description: "Natural language query — searches by meaning, not exact words." }),
|
|
493
|
+
role: Type.Optional(Type.String({ description: "Filter by role: user, assistant, toolResult, or custom" })),
|
|
494
|
+
limit: Type.Optional(Type.Number({ description: "Max results (default 20, max 50)" })),
|
|
495
|
+
offset: Type.Optional(Type.Number({ description: "Pagination offset (default 0)" })),
|
|
496
|
+
}),
|
|
497
|
+
|
|
498
|
+
async execute(_toolCallId, params, _signal, _onUpdate, _ctx) {
|
|
499
|
+
const query = (params.query as string)?.trim();
|
|
500
|
+
if (!query) return { content: [{ type: "text", text: "Error: query is required" }], details: { error: "query required" } };
|
|
501
|
+
|
|
502
|
+
const limit = Math.min(Math.max(1, (params.limit as number) ?? 20), 50);
|
|
503
|
+
const offset = Math.max(0, (params.offset as number) ?? 0);
|
|
504
|
+
const role = (params.role as string) ?? null;
|
|
505
|
+
|
|
506
|
+
const ready = await ensureIndex();
|
|
507
|
+
if (!ready) return { content: [{ type: "text", text: `Index not ready: ${state.error ?? "building..."}` }], details: { error: state.error ?? "indexing" } };
|
|
508
|
+
|
|
509
|
+
if (!state.hasEmbeddings) return { content: [{ type: "text", text: "Embeddings not yet generated. Run the cron build script or wait for background generation." }], details: { error: "embeddings not ready" } };
|
|
510
|
+
|
|
511
|
+
try {
|
|
512
|
+
const conn = await getConnection();
|
|
513
|
+
const queryEmbedding = await embed(query);
|
|
514
|
+
const arrStr = embeddingToDuckArray(queryEmbedding);
|
|
515
|
+
|
|
516
|
+
// Prepared statement with ? placeholders
|
|
517
|
+
let sql: string;
|
|
518
|
+
let sqlParams: any[];
|
|
519
|
+
|
|
520
|
+
if (role) {
|
|
521
|
+
sql = `
|
|
522
|
+
SELECT filename, line_number, role, timestamp, entry_type,
|
|
523
|
+
substring(content_text, 1, 500) AS snippet,
|
|
524
|
+
array_cosine_similarity(embedding, ?::FLOAT[${EMBED_DIM}]) AS similarity
|
|
525
|
+
FROM session_entries
|
|
526
|
+
WHERE embedding IS NOT NULL AND role = ?
|
|
527
|
+
ORDER BY similarity DESC
|
|
528
|
+
LIMIT ? OFFSET ?
|
|
529
|
+
`;
|
|
530
|
+
sqlParams = [arrStr, role, limit, offset];
|
|
531
|
+
} else {
|
|
532
|
+
sql = `
|
|
533
|
+
SELECT filename, line_number, role, timestamp, entry_type,
|
|
534
|
+
substring(content_text, 1, 500) AS snippet,
|
|
535
|
+
array_cosine_similarity(embedding, ?::FLOAT[${EMBED_DIM}]) AS similarity
|
|
536
|
+
FROM session_entries
|
|
537
|
+
WHERE embedding IS NOT NULL
|
|
538
|
+
ORDER BY similarity DESC
|
|
539
|
+
LIMIT ? OFFSET ?
|
|
540
|
+
`;
|
|
541
|
+
sqlParams = [arrStr, limit, offset];
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
const result = await conn.run(sql, sqlParams);
|
|
545
|
+
const rows = await result.getRows();
|
|
546
|
+
|
|
547
|
+
if (rows.length === 0) return { content: [{ type: "text", text: `No results found for: ${query}` }], details: { query, result_count: 0, mode: "semantic" } };
|
|
548
|
+
|
|
549
|
+
const lines: string[] = [`Found ${rows.length} semantic result(s) for: ${query}`, ""];
|
|
550
|
+
for (const row of rows) {
|
|
551
|
+
const [filename, lineNum, entryRole, ts, entryType, snippet, similarity] = row;
|
|
552
|
+
const shortFile = String(filename).replace(SESSIONS_DIR + "/", "");
|
|
553
|
+
lines.push(`--- ${shortFile}:${lineNum} (similarity: ${Number(similarity).toFixed(3)}) ---`);
|
|
554
|
+
lines.push(` role: ${entryRole || "N/A"} | type: ${entryType} | time: ${ts || "N/A"}`);
|
|
555
|
+
lines.push(` text: ${String(snippet || "").slice(0, 300)}`);
|
|
556
|
+
lines.push("");
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
return { content: [{ type: "text", text: lines.join("\n") }], details: { query, result_count: rows.length, limit, offset, mode: "semantic" } };
|
|
560
|
+
} catch (e) {
|
|
561
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
562
|
+
return { content: [{ type: "text", text: `Semantic search error: ${msg}` }], details: { error: msg } };
|
|
563
|
+
}
|
|
564
|
+
},
|
|
565
|
+
});
|
|
566
|
+
|
|
567
|
+
// session_hybrid — BM25 + cosine via RRF (prepared statements)
|
|
568
|
+
pi.registerTool({
|
|
569
|
+
name: "session_hybrid",
|
|
570
|
+
label: "Session Hybrid Search",
|
|
571
|
+
description:
|
|
572
|
+
"Search session transcripts combining BM25 keyword ranking and semantic cosine similarity " +
|
|
573
|
+
"via Reciprocal Rank Fusion (RRF). Returns the best of both: exact keyword matches AND " +
|
|
574
|
+
"conceptually related entries. Recommended for comprehensive search.",
|
|
575
|
+
parameters: Type.Object({
|
|
576
|
+
query: Type.String({ description: "Search query — keywords for BM25, meaning for semantic, both for hybrid." }),
|
|
577
|
+
role: Type.Optional(Type.String({ description: "Filter by role: user, assistant, toolResult, or custom" })),
|
|
578
|
+
limit: Type.Optional(Type.Number({ description: "Max results (default 20, max 50)" })),
|
|
579
|
+
}),
|
|
580
|
+
|
|
581
|
+
async execute(_toolCallId, params, _signal, _onUpdate, _ctx) {
|
|
582
|
+
const query = (params.query as string)?.trim();
|
|
583
|
+
if (!query) return { content: [{ type: "text", text: "Error: query is required" }], details: { error: "query required" } };
|
|
584
|
+
|
|
585
|
+
const limit = Math.min(Math.max(1, (params.limit as number) ?? 20), 50);
|
|
586
|
+
const role = (params.role as string) ?? null;
|
|
587
|
+
|
|
588
|
+
const ready = await ensureIndex();
|
|
589
|
+
if (!ready) return { content: [{ type: "text", text: `Index not ready: ${state.error ?? "building..."}` }], details: { error: state.error ?? "indexing" } };
|
|
590
|
+
|
|
591
|
+
try {
|
|
592
|
+
const conn = await getConnection();
|
|
593
|
+
await conn.run("LOAD fts;");
|
|
594
|
+
|
|
595
|
+
// BM25 query with prepared statement
|
|
596
|
+
let bm25Sql: string;
|
|
597
|
+
let bm25Params: any[];
|
|
598
|
+
if (role) {
|
|
599
|
+
bm25Sql = `
|
|
600
|
+
SELECT entry_id, filename, line_number, role, timestamp, entry_type,
|
|
601
|
+
substring(content_text, 1, 500) AS snippet
|
|
602
|
+
FROM session_entries
|
|
603
|
+
WHERE fts_main_session_entries.match_bm25(entry_id, ?) IS NOT NULL AND role = ?
|
|
604
|
+
ORDER BY fts_main_session_entries.match_bm25(entry_id, ?) DESC
|
|
605
|
+
LIMIT 50
|
|
606
|
+
`;
|
|
607
|
+
bm25Params = [query, role, query];
|
|
608
|
+
} else {
|
|
609
|
+
bm25Sql = `
|
|
610
|
+
SELECT entry_id, filename, line_number, role, timestamp, entry_type,
|
|
611
|
+
substring(content_text, 1, 500) AS snippet
|
|
612
|
+
FROM session_entries
|
|
613
|
+
WHERE fts_main_session_entries.match_bm25(entry_id, ?) IS NOT NULL
|
|
614
|
+
ORDER BY fts_main_session_entries.match_bm25(entry_id, ?) DESC
|
|
615
|
+
LIMIT 50
|
|
616
|
+
`;
|
|
617
|
+
bm25Params = [query, query];
|
|
618
|
+
}
|
|
619
|
+
const bm25Result = await conn.run(bm25Sql, bm25Params);
|
|
620
|
+
const bm25Rows = await bm25Result.getRows();
|
|
621
|
+
|
|
622
|
+
// Cosine query with prepared statement
|
|
623
|
+
let cosineRows: any[][] = [];
|
|
624
|
+
if (state.hasEmbeddings) {
|
|
625
|
+
const queryEmbedding = await embed(query);
|
|
626
|
+
const arrStr = embeddingToDuckArray(queryEmbedding);
|
|
627
|
+
|
|
628
|
+
let cosineSql: string;
|
|
629
|
+
let cosineParams: any[];
|
|
630
|
+
if (role) {
|
|
631
|
+
cosineSql = `
|
|
632
|
+
SELECT entry_id, filename, line_number, role, timestamp, entry_type,
|
|
633
|
+
substring(content_text, 1, 500) AS snippet
|
|
634
|
+
FROM session_entries
|
|
635
|
+
WHERE embedding IS NOT NULL AND role = ?
|
|
636
|
+
ORDER BY array_cosine_similarity(embedding, ?::FLOAT[${EMBED_DIM}]) DESC
|
|
637
|
+
LIMIT 50
|
|
638
|
+
`;
|
|
639
|
+
cosineParams = [role, arrStr];
|
|
640
|
+
} else {
|
|
641
|
+
cosineSql = `
|
|
642
|
+
SELECT entry_id, filename, line_number, role, timestamp, entry_type,
|
|
643
|
+
substring(content_text, 1, 500) AS snippet
|
|
644
|
+
FROM session_entries
|
|
645
|
+
WHERE embedding IS NOT NULL
|
|
646
|
+
ORDER BY array_cosine_similarity(embedding, ?::FLOAT[${EMBED_DIM}]) DESC
|
|
647
|
+
LIMIT 50
|
|
648
|
+
`;
|
|
649
|
+
cosineParams = [arrStr];
|
|
650
|
+
}
|
|
651
|
+
const cosineResult = await conn.run(cosineSql, cosineParams);
|
|
652
|
+
cosineRows = await cosineResult.getRows();
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
// Reciprocal Rank Fusion
|
|
656
|
+
const rrfScores = new Map<string, { row: any[]; score: number; bm25_rank: number | null; cosine_rank: number | null }>();
|
|
657
|
+
|
|
658
|
+
for (let i = 0; i < bm25Rows.length; i++) {
|
|
659
|
+
const entryId = String(bm25Rows[i][0]);
|
|
660
|
+
const rrf = 1 / (RRF_K + i + 1);
|
|
661
|
+
rrfScores.set(entryId, { row: bm25Rows[i], score: rrf, bm25_rank: i + 1, cosine_rank: null });
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
for (let i = 0; i < cosineRows.length; i++) {
|
|
665
|
+
const entryId = String(cosineRows[i][0]);
|
|
666
|
+
const rrf = 1 / (RRF_K + i + 1);
|
|
667
|
+
const existing = rrfScores.get(entryId);
|
|
668
|
+
if (existing) {
|
|
669
|
+
existing.score += rrf;
|
|
670
|
+
existing.cosine_rank = i + 1;
|
|
671
|
+
} else {
|
|
672
|
+
rrfScores.set(entryId, { row: cosineRows[i], score: rrf, bm25_rank: null, cosine_rank: i + 1 });
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
const fused = Array.from(rrfScores.values()).sort((a, b) => b.score - a.score).slice(0, limit);
|
|
677
|
+
|
|
678
|
+
if (fused.length === 0) return { content: [{ type: "text", text: `No results found for: ${query}` }], details: { query, result_count: 0, mode: "hybrid" } };
|
|
679
|
+
|
|
680
|
+
const lines: string[] = [`Found ${fused.length} hybrid result(s) for: ${query}`, ""];
|
|
681
|
+
for (const item of fused) {
|
|
682
|
+
const [entryId, filename, lineNum, entryRole, ts, entryType, snippet] = item.row;
|
|
683
|
+
const shortFile = String(filename).replace(SESSIONS_DIR + "/", "");
|
|
684
|
+
const rankInfo = [`bm25:${item.bm25_rank ?? "-"}`, `cos:${item.cosine_rank ?? "-"}`].join(" ");
|
|
685
|
+
lines.push(`--- ${shortFile}:${lineNum} (rrf: ${item.score.toFixed(4)} | ${rankInfo}) ---`);
|
|
686
|
+
lines.push(` role: ${entryRole || "N/A"} | type: ${entryType} | time: ${ts || "N/A"}`);
|
|
687
|
+
lines.push(` text: ${String(snippet || "").slice(0, 300)}`);
|
|
688
|
+
lines.push("");
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
return { content: [{ type: "text", text: lines.join("\n") }], details: { query, result_count: fused.length, limit, mode: "hybrid", has_embeddings: state.hasEmbeddings } };
|
|
692
|
+
} catch (e) {
|
|
693
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
694
|
+
return { content: [{ type: "text", text: `Hybrid search error: ${msg}` }], details: { error: msg } };
|
|
695
|
+
}
|
|
696
|
+
},
|
|
697
|
+
});
|
|
698
|
+
|
|
699
|
+
// session_read — read specific entry by path + line (prepared statement)
|
|
700
|
+
pi.registerTool({
|
|
701
|
+
name: "session_read",
|
|
702
|
+
label: "Session Read",
|
|
703
|
+
description:
|
|
704
|
+
"Read a specific entry from a Pi session JSONL file by file path and line number. " +
|
|
705
|
+
"Returns the full decoded entry including role, content, and metadata. " +
|
|
706
|
+
"Use after session_search, session_semantic, or session_hybrid to read full context.",
|
|
707
|
+
parameters: Type.Object({
|
|
708
|
+
path: Type.String({ description: "Absolute path to the .jsonl session file" }),
|
|
709
|
+
line: Type.Number({ description: "Line number (1-indexed) in the JSONL file" }),
|
|
710
|
+
}),
|
|
711
|
+
|
|
712
|
+
async execute(_toolCallId, params, _signal, _onUpdate, _ctx) {
|
|
713
|
+
const filePath = params.path as string;
|
|
714
|
+
const lineNum = params.line as number;
|
|
715
|
+
|
|
716
|
+
if (!filePath || !filePath.endsWith(".jsonl")) return { content: [{ type: "text", text: "Error: path must point to a .jsonl file" }], details: { error: "invalid path" } };
|
|
717
|
+
if (!existsSync(filePath)) return { content: [{ type: "text", text: `Error: file not found: ${filePath}` }], details: { error: "file not found" } };
|
|
718
|
+
|
|
719
|
+
try {
|
|
720
|
+
const conn = await getConnection();
|
|
721
|
+
const result = await conn.run(
|
|
722
|
+
`
|
|
723
|
+
WITH numbered AS (
|
|
724
|
+
SELECT row_number() OVER () AS rn, json
|
|
725
|
+
FROM read_json_objects(?, format='newline_delimited', ignore_errors=true)
|
|
726
|
+
)
|
|
727
|
+
SELECT
|
|
728
|
+
json->>'type' AS entry_type, json->>'id' AS entry_id,
|
|
729
|
+
json->>'parentId' AS parent_id, json->>'timestamp' AS timestamp,
|
|
730
|
+
json->'message'->>'role' AS role,
|
|
731
|
+
json->'message'->>'content' AS content,
|
|
732
|
+
json->>'customType' AS custom_type,
|
|
733
|
+
CAST(json AS VARCHAR) AS raw_json
|
|
734
|
+
FROM numbered WHERE rn = ? LIMIT 1
|
|
735
|
+
`,
|
|
736
|
+
[filePath, Math.max(1, lineNum)]
|
|
737
|
+
);
|
|
738
|
+
|
|
739
|
+
const rows = await result.getRows();
|
|
740
|
+
if (rows.length === 0) return { content: [{ type: "text", text: `No entry at line ${lineNum} in ${filePath}` }], details: { error: "line not found" } };
|
|
741
|
+
|
|
742
|
+
const [entryType, entryId, parentId, ts, role, content, customType, rawJson] = rows[0];
|
|
743
|
+
const lines: string[] = [`Entry at ${filePath}:${lineNum}`, ` type: ${entryType}`];
|
|
744
|
+
if (customType) lines.push(` customType: ${customType}`);
|
|
745
|
+
if (role) lines.push(` role: ${role}`);
|
|
746
|
+
if (ts) lines.push(` timestamp: ${ts}`);
|
|
747
|
+
if (entryId) lines.push(` id: ${entryId}`);
|
|
748
|
+
if (parentId) lines.push(` parentId: ${parentId}`);
|
|
749
|
+
lines.push("");
|
|
750
|
+
lines.push(content ? `content:\n${String(content).slice(0, 2000)}` : `raw_json:\n${String(rawJson).slice(0, 2000)}`);
|
|
751
|
+
|
|
752
|
+
return { content: [{ type: "text", text: lines.join("\n") }], details: { path: filePath, line: lineNum, entry_type: entryType } };
|
|
753
|
+
} catch (e) {
|
|
754
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
755
|
+
return { content: [{ type: "text", text: `Read error: ${msg}` }], details: { error: msg } };
|
|
756
|
+
}
|
|
757
|
+
},
|
|
758
|
+
});
|
|
759
|
+
|
|
760
|
+
// session_status
|
|
761
|
+
pi.registerTool({
|
|
762
|
+
name: "session_status",
|
|
763
|
+
label: "Session Index Status",
|
|
764
|
+
description:
|
|
765
|
+
"Check the DuckDB session search index status: entry count, session count, " +
|
|
766
|
+
"embedding status, last build time, and whether a rebuild is needed.",
|
|
767
|
+
parameters: Type.Object({}),
|
|
768
|
+
|
|
769
|
+
async execute(_toolCallId, _params, _signal, _onUpdate, _ctx) {
|
|
770
|
+
await ensureIndex();
|
|
771
|
+
const files = findSessionFiles(true);
|
|
772
|
+
const fileCount = files.length;
|
|
773
|
+
|
|
774
|
+
const lines: string[] = ["DuckDB Session Search Index Status", ""];
|
|
775
|
+
lines.push(` indexed entries: ${state.entryCount}`);
|
|
776
|
+
lines.push(` indexed sessions: ${state.sessionCount}`);
|
|
777
|
+
lines.push(` JSONL files on disk: ${fileCount} (excluding recent)`);
|
|
778
|
+
lines.push(` index ready: ${state.ready}`);
|
|
779
|
+
lines.push(` indexing: ${state.indexing}`);
|
|
780
|
+
lines.push(` embeddings: ${state.hasEmbeddings ? "ready" : "building or not started"}`);
|
|
781
|
+
lines.push(` last built: ${state.lastBuilt ?? "never"}`);
|
|
782
|
+
lines.push(` database: ${DB_PATH}`);
|
|
783
|
+
lines.push(` model: ${MODEL_ID} (${EMBED_DIM} dims)`);
|
|
784
|
+
lines.push(` tools: session_search (BM25), session_semantic (cosine), session_hybrid (RRF), session_read`);
|
|
785
|
+
if (state.error) lines.push(` error: ${state.error}`);
|
|
786
|
+
|
|
787
|
+
if (state.ready && fileCount > state.sessionCount * 1.1) {
|
|
788
|
+
lines.push("");
|
|
789
|
+
lines.push(` ⚠ Session count mismatch (${state.sessionCount} indexed vs ${fileCount} on disk). Rebuild recommended.`);
|
|
790
|
+
}
|
|
791
|
+
|
|
792
|
+
if (!state.ready && !state.indexing) {
|
|
793
|
+
lines.push("");
|
|
794
|
+
lines.push(" Index not built. Call session_search to trigger build, or run the cron build script.");
|
|
795
|
+
}
|
|
796
|
+
|
|
797
|
+
return {
|
|
798
|
+
content: [{ type: "text", text: lines.join("\n") }],
|
|
799
|
+
details: { entry_count: state.entryCount, session_count: state.sessionCount, file_count: fileCount, ready: state.ready, indexing: state.indexing, has_embeddings: state.hasEmbeddings, last_built: state.lastBuilt, error: state.error },
|
|
800
|
+
};
|
|
801
|
+
},
|
|
802
|
+
});
|
|
803
|
+
}
|