pi-duckdb-search 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +173 -0
- package/duckdb-search-build.cjs +360 -0
- package/duckdb-search-test.cjs +576 -0
- package/extensions/duckdb-search.ts +803 -0
- package/package.json +60 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 cpagan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
# pi-duckdb-search
|
|
2
|
+
|
|
3
|
+
DuckDB-powered session transcript search for the [Pi coding agent](https://pi.dev).
|
|
4
|
+
|
|
5
|
+
Searches your Pi session JSONL files using three modes:
|
|
6
|
+
- **BM25 keyword** — exact term matching with Porter stemming
|
|
7
|
+
- **Semantic** — ONNX embeddings (all-MiniLM-L6-v2) with cosine similarity
|
|
8
|
+
- **Hybrid** — Reciprocal Rank Fusion combining both for the best results
|
|
9
|
+
|
|
10
|
+
Self-contained: no external APIs, no cloud accounts, no network after first model download (~23 MB).
|
|
11
|
+
|
|
12
|
+
## Install
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
pi install npm:pi-duckdb-search
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
## Tools
|
|
19
|
+
|
|
20
|
+
Five tools are registered, callable by the LLM during conversation:
|
|
21
|
+
|
|
22
|
+
| Tool | Mode | Use When |
|
|
23
|
+
|------|------|----------|
|
|
24
|
+
| `session_search` | BM25 keyword | You know exact words |
|
|
25
|
+
| `session_semantic` | Vector cosine | You want meaning-based matches |
|
|
26
|
+
| `session_hybrid` | RRF fusion | **Recommended** — best of both |
|
|
27
|
+
| `session_read` | Direct read | Read a specific entry by path + line |
|
|
28
|
+
| `session_status` | Status | Check index health |
|
|
29
|
+
|
|
30
|
+
## How It Works
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
~/.pi/agent/sessions/**/*.jsonl
|
|
34
|
+
│
|
|
35
|
+
▼ read_json_objects(format='newline_delimited', ignore_errors=true)
|
|
36
|
+
DuckDB database (~47 MB at 137 sessions)
|
|
37
|
+
├── session_entries table (content_text + embedding FLOAT[384])
|
|
38
|
+
├── FTS index (Porter stemmer, BM25 ranking)
|
|
39
|
+
└── _metadata table (entry_count, has_embeddings, last_built)
|
|
40
|
+
│
|
|
41
|
+
▼ connection.run(sql, [params]) — prepared statements
|
|
42
|
+
BM25: fts_main_session_entries.match_bm25(entry_id, ?)
|
|
43
|
+
Cosine: array_cosine_similarity(embedding, ?::FLOAT[384])
|
|
44
|
+
Hybrid: RRF fusion (k=60)
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
The extension opens the database in **read-only mode** — no WAL, no locks. A standalone cron build script creates and maintains the database offline.
|
|
48
|
+
|
|
49
|
+
## Setup
|
|
50
|
+
|
|
51
|
+
### 1. Install dependencies
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
cd ~/.pi/agent && npm install @duckdb/node-api @huggingface/transformers
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
### 2. Set up the cron build (recommended)
|
|
58
|
+
|
|
59
|
+
The cron script builds the database offline (FTS + embeddings) daily at 4am:
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
# Add to crontab:
|
|
63
|
+
0 4 * * * /path/to/node ~/.pi/agent/duckdb-search/duckdb-search-build.cjs >> ~/.pi/agent/duckdb-search/build.log 2>&1
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Manual build (first time or after changes):
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
node ~/.pi/agent/duckdb-search/duckdb-search-build.cjs
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
First build takes ~4 minutes (includes 23 MB model download + ONNX inference on ~17K entries). Subsequent incremental builds take ~2 seconds if no new sessions.
|
|
73
|
+
|
|
74
|
+
### 3. Restart Pi
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
pi # or /reload in an existing session
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
The extension auto-loads the pre-built database on first search. No rebuild on restart.
|
|
81
|
+
|
|
82
|
+
## Performance
|
|
83
|
+
|
|
84
|
+
| Metric | Value |
|
|
85
|
+
|--------|-------|
|
|
86
|
+
| BM25 search latency | ~19 ms |
|
|
87
|
+
| Semantic search latency | ~24 ms |
|
|
88
|
+
| Hybrid search latency | ~35 ms |
|
|
89
|
+
| Database size | ~47 MB (137 sessions, 17K entries) |
|
|
90
|
+
| Full build time | ~4 min (ONNX inference dominates) |
|
|
91
|
+
| Incremental build time | ~2 s (FTS rebuild only, embeddings preserved) |
|
|
92
|
+
| Model download | ~23 MB (one-time, cached in `~/.cache/huggingface/`) |
|
|
93
|
+
|
|
94
|
+
## Content Extraction
|
|
95
|
+
|
|
96
|
+
`content_text` extracts from JSON message content arrays:
|
|
97
|
+
- `text` — assistant/user text
|
|
98
|
+
- `thinking` — reasoning blocks (captured since v2.2; 60% of previously "empty" entries had thinking content)
|
|
99
|
+
- `name` — tool call names
|
|
100
|
+
|
|
101
|
+
Truncated to 2000 characters. Only error responses with `content: []` have no embedding.
|
|
102
|
+
|
|
103
|
+
## Security
|
|
104
|
+
|
|
105
|
+
- All search queries use **prepared statements** (`connection.run(sql, [params])` with `?` placeholders)
|
|
106
|
+
- Database opened in **read-only mode** when pre-built by cron
|
|
107
|
+
- **SQL injection resistant** — tested with `'; DROP TABLE --` payloads
|
|
108
|
+
- 34-assertion test suite covers injection, edge cases, and read-only enforcement
|
|
109
|
+
|
|
110
|
+
## Test Suite
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
NODE_PATH=~/.pi/agent/node_modules node ~/.pi/agent/duckdb-search/duckdb-search-test.cjs
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
34 assertions across 12 test groups:
|
|
117
|
+
- Schema & content extraction
|
|
118
|
+
- BM25 keyword search
|
|
119
|
+
- Semantic cosine search
|
|
120
|
+
- Hybrid RRF search
|
|
121
|
+
- session_read by path + line
|
|
122
|
+
- Metadata persistence
|
|
123
|
+
- SQL injection resistance
|
|
124
|
+
- Edge cases (empty query, missing file, limit clamping, line beyond EOF)
|
|
125
|
+
- FTS overwrite without drop
|
|
126
|
+
- ignore_errors with malformed JSON
|
|
127
|
+
- Content truncation (5000 chars → 2000 chars)
|
|
128
|
+
- Read-only mode enforcement
|
|
129
|
+
|
|
130
|
+
## Architecture Decisions
|
|
131
|
+
|
|
132
|
+
| Decision | Rationale |
|
|
133
|
+
|----------|-----------|
|
|
134
|
+
| **DuckDB over SQLite** | Native `read_json_objects()` reads JSONL directly — no ingestion pipeline |
|
|
135
|
+
| **Cron over lazy build** | Predictable, no race conditions, doesn't block first search |
|
|
136
|
+
| **all-MiniLM-L6-v2** | 23 MB, 384 dims, sufficient at 17K entries |
|
|
137
|
+
| **Brute-force cosine over HNSW** | 13 ms at 17K vectors; HNSW persistence is experimental in DuckDB |
|
|
138
|
+
| **Read-only mode** | No WAL, no locks — cron writes, extension reads |
|
|
139
|
+
| **Prepared statements** | SQL injection safety for all user-supplied query strings |
|
|
140
|
+
|
|
141
|
+
## Configuration
|
|
142
|
+
|
|
143
|
+
| Setting | Default | Description |
|
|
144
|
+
|---------|---------|-------------|
|
|
145
|
+
| `SESSIONS_DIR` | `~/.pi/agent/sessions` | Pi session JSONL directory |
|
|
146
|
+
| `DB_PATH` | `~/.pi/agent/duckdb-search/sessions.duckdb` | DuckDB database file |
|
|
147
|
+
| `MODEL_ID` | `Xenova/all-MiniLM-L6-v2` | ONNX embedding model |
|
|
148
|
+
| `EMBED_DIM` | 384 | Embedding dimensions |
|
|
149
|
+
| `EXCLUDE_RECENT_MS` | 300000 (5 min) | Exclude files modified in last N ms |
|
|
150
|
+
| `MAX_CONTENT_CHARS` | 2000 | Content truncation limit |
|
|
151
|
+
| `RRF_K` | 60 | Reciprocal Rank Fusion constant |
|
|
152
|
+
|
|
153
|
+
## Dependencies
|
|
154
|
+
|
|
155
|
+
| Package | Purpose |
|
|
156
|
+
|---------|---------|
|
|
157
|
+
| `@duckdb/node-api` | DuckDB native binary (FTS + FLOAT[] arrays) |
|
|
158
|
+
| `@huggingface/transformers` | ONNX embeddings (all-MiniLM-L6-v2) |
|
|
159
|
+
|
|
160
|
+
Pi-supplied (declared as peerDependencies, not bundled):
|
|
161
|
+
- `@earendil-works/pi-coding-agent`
|
|
162
|
+
- `typebox`
|
|
163
|
+
|
|
164
|
+
## Limitations
|
|
165
|
+
|
|
166
|
+
1. **FTS full rebuild on incremental builds** — DuckDB v1.5.6 FTS doesn't support incremental index updates. The cron script rebuilds FTS (~1 second) but preserves embeddings.
|
|
167
|
+
2. **No snippet windowing** — Snippets start from the beginning of `content_text`, not the matching term.
|
|
168
|
+
3. **Active session excluded** — Files modified in last 5 minutes are skipped to avoid indexing while JSONL is being written.
|
|
169
|
+
4. **HNSW not used** — Brute-force cosine is 13 ms at 17K vectors. HNSW would matter at 100K+ vectors but DuckDB's vss extension has experimental persistence (data loss risk).
|
|
170
|
+
|
|
171
|
+
## License
|
|
172
|
+
|
|
173
|
+
MIT
|
|
@@ -0,0 +1,360 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* duckdb-search-build.cjs — Standalone build script for DuckDB session search.
|
|
4
|
+
*
|
|
5
|
+
* v2.3: Full content extraction, incremental indexing, ignore_errors, FTS overwrite=true.
|
|
6
|
+
*
|
|
7
|
+
* Stores metadata in _metadata table for the extension to read.
|
|
8
|
+
*
|
|
9
|
+
* Run via cron (daily at 4am):
|
|
10
|
+
* 0 4 * * * /home/cpagan/.local/share/pi-node/node-v22.23.0-linux-x64/bin/node \
|
|
11
|
+
* /home/cpagan/.pi/agent/duckdb-search/duckdb-search-build.cjs >> \
|
|
12
|
+
* /home/cpagan/.pi/agent/duckdb-search/build.log 2>&1
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
const { DuckDBInstance } = require("@duckdb/node-api");
|
|
16
|
+
const path = require("node:path");
|
|
17
|
+
const fs = require("node:fs");
|
|
18
|
+
|
|
19
|
+
const HOME = process.env.HOME || "/home/cpagan";
|
|
20
|
+
const SESSIONS_DIR = path.join(HOME, ".pi", "agent", "sessions");
|
|
21
|
+
const DB_DIR = path.join(HOME, ".pi", "agent", "duckdb-search");
|
|
22
|
+
const DB_PATH = path.join(DB_DIR, "sessions.duckdb");
|
|
23
|
+
const MODEL_ID = "Xenova/all-MiniLM-L6-v2";
|
|
24
|
+
const EMBED_DIM = 384;
|
|
25
|
+
const EXCLUDE_RECENT_MS = 5 * 60 * 1000;
|
|
26
|
+
const MAX_CONTENT_CHARS = 2000;
|
|
27
|
+
|
|
28
|
+
function ts() { return new Date().toISOString(); }
|
|
29
|
+
function log(msg) { console.log(`[${ts()}] ${msg}`); }
|
|
30
|
+
|
|
31
|
+
/** Find all JSONL files, excluding files modified in last 5 minutes. */
|
|
32
|
+
function findSessionFiles() {
|
|
33
|
+
const files = [];
|
|
34
|
+
const now = Date.now();
|
|
35
|
+
function scan(dir) {
|
|
36
|
+
if (!fs.existsSync(dir)) return;
|
|
37
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
38
|
+
const full = path.join(dir, entry.name);
|
|
39
|
+
if (entry.isDirectory()) scan(full);
|
|
40
|
+
else if (entry.name.endsWith(".jsonl")) {
|
|
41
|
+
try {
|
|
42
|
+
const st = fs.statSync(full);
|
|
43
|
+
if (now - st.mtimeMs < EXCLUDE_RECENT_MS) {
|
|
44
|
+
log(` skipping recent file: ${entry.name} (modified ${Math.round((now - st.mtimeMs) / 1000)}s ago)`);
|
|
45
|
+
continue;
|
|
46
|
+
}
|
|
47
|
+
files.push({ path: full, mtime: st.mtimeMs });
|
|
48
|
+
} catch { /* ignore */ }
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
scan(SESSIONS_DIR);
|
|
53
|
+
return files;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
async function getMetadata(conn, key) {
|
|
57
|
+
try {
|
|
58
|
+
const r = await conn.run(`SELECT value FROM _metadata WHERE key = '${key.replace(/'/g, "''")}';`);
|
|
59
|
+
const rows = await r.getRows();
|
|
60
|
+
return rows.length > 0 ? String(rows[0][0]) : null;
|
|
61
|
+
} catch { return null; }
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
async function setMetadata(conn, key, value) {
|
|
65
|
+
await conn.run(`CREATE TABLE IF NOT EXISTS _metadata (key VARCHAR PRIMARY KEY, value VARCHAR);`);
|
|
66
|
+
await conn.run(`INSERT OR REPLACE INTO _metadata VALUES ('${key.replace(/'/g, "''")}', '${value.replace(/'/g, "''")}');`);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
async function loadEmbedder() {
|
|
70
|
+
log("Loading embedding model...");
|
|
71
|
+
const { pipeline } = await import("@huggingface/transformers");
|
|
72
|
+
const embedder = await pipeline("feature-extraction", MODEL_ID);
|
|
73
|
+
log("Model loaded");
|
|
74
|
+
return embedder;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
async function embed(embedder, text) {
|
|
78
|
+
const output = await embedder(text, { pooling: "mean", normalize: true });
|
|
79
|
+
return new Float32Array(output.data);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** Full rebuild from scratch — used when no existing DB or schema changed. */
|
|
83
|
+
async function fullBuild(conn, files) {
|
|
84
|
+
log("Full rebuild (no existing database or schema mismatch)...");
|
|
85
|
+
|
|
86
|
+
try { await conn.run("PRAGMA drop_fts_index('session_entries');"); } catch {}
|
|
87
|
+
await conn.run("DROP TABLE IF EXISTS session_entries;");
|
|
88
|
+
await conn.run("DROP TABLE IF EXISTS _emb_temp;");
|
|
89
|
+
|
|
90
|
+
const glob = path.join(SESSIONS_DIR, "**", "*.jsonl").replace(/\\/g, "/");
|
|
91
|
+
|
|
92
|
+
await conn.run(`
|
|
93
|
+
CREATE TABLE session_entries AS
|
|
94
|
+
SELECT
|
|
95
|
+
filename,
|
|
96
|
+
line_number,
|
|
97
|
+
json->>'type' AS entry_type,
|
|
98
|
+
json->>'id' AS entry_id,
|
|
99
|
+
json->>'parentId' AS parent_id,
|
|
100
|
+
json->>'timestamp' AS timestamp,
|
|
101
|
+
json->'message'->>'role' AS role,
|
|
102
|
+
json->>'customType' AS custom_type,
|
|
103
|
+
CASE
|
|
104
|
+
WHEN json_type(json->'message'->'content') = 'ARRAY'
|
|
105
|
+
THEN substring(
|
|
106
|
+
concat(
|
|
107
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].text'), ' '), ''),
|
|
108
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].thinking'), ' '), ''),
|
|
109
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].name'), ' '), '')
|
|
110
|
+
), 1, ${MAX_CONTENT_CHARS}
|
|
111
|
+
)
|
|
112
|
+
ELSE substring(COALESCE(json->'message'->>'content', json->>'content', ''), 1, ${MAX_CONTENT_CHARS})
|
|
113
|
+
END AS content_text
|
|
114
|
+
FROM (
|
|
115
|
+
SELECT
|
|
116
|
+
filename,
|
|
117
|
+
row_number() OVER (PARTITION BY filename ORDER BY filename) AS line_number,
|
|
118
|
+
json
|
|
119
|
+
FROM read_json_objects('${glob}', format='newline_delimited', filename=true, ignore_errors=true)
|
|
120
|
+
)
|
|
121
|
+
WHERE json->>'type' IN ('message', 'custom_message')
|
|
122
|
+
`);
|
|
123
|
+
|
|
124
|
+
const entryCount = Number((await (await conn.run("SELECT COUNT(*) FROM session_entries")).getRows())[0][0]);
|
|
125
|
+
const sessionCount = Number((await (await conn.run("SELECT COUNT(DISTINCT filename) FROM session_entries")).getRows())[0][0]);
|
|
126
|
+
log(`Indexed ${entryCount} entries from ${sessionCount} sessions`);
|
|
127
|
+
|
|
128
|
+
log("Building FTS index...");
|
|
129
|
+
await conn.run("PRAGMA create_fts_index('session_entries', 'entry_id', 'content_text', 'role', stemmer='porter', overwrite=true);");
|
|
130
|
+
|
|
131
|
+
// Add embedding column
|
|
132
|
+
await conn.run(`ALTER TABLE session_entries ADD COLUMN IF NOT EXISTS embedding FLOAT[${EMBED_DIM}];`);
|
|
133
|
+
|
|
134
|
+
// Generate all embeddings
|
|
135
|
+
const embedder = await loadEmbedder();
|
|
136
|
+
await generateEmbeddingsForNew(conn, embedder);
|
|
137
|
+
|
|
138
|
+
return { entryCount, sessionCount };
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/** Generate embeddings only for entries where embedding IS NULL. */
|
|
142
|
+
async function generateEmbeddingsForNew(conn, embedder) {
|
|
143
|
+
const result = await conn.run("SELECT entry_id, content_text FROM session_entries WHERE embedding IS NULL AND length(content_text) > 0 ORDER BY rowid;");
|
|
144
|
+
const rows = await result.getRows();
|
|
145
|
+
|
|
146
|
+
if (rows.length === 0) {
|
|
147
|
+
log("All entries already have embeddings.");
|
|
148
|
+
return 0;
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
log(`Generating embeddings for ${rows.length} entries...`);
|
|
152
|
+
|
|
153
|
+
await conn.run(`CREATE TABLE IF NOT EXISTS _emb_temp (entry_id VARCHAR, embedding FLOAT[${EMBED_DIM}]);`);
|
|
154
|
+
await conn.run("DELETE FROM _emb_temp;");
|
|
155
|
+
|
|
156
|
+
const batchSize = 64;
|
|
157
|
+
let embedded = 0;
|
|
158
|
+
for (let i = 0; i < rows.length; i += batchSize) {
|
|
159
|
+
const batch = rows.slice(i, i + batchSize);
|
|
160
|
+
const placeholders = [];
|
|
161
|
+
const insertParams = [];
|
|
162
|
+
|
|
163
|
+
for (const row of batch) {
|
|
164
|
+
const entryId = row[0];
|
|
165
|
+
const text = String(row[1] || "").slice(0, MAX_CONTENT_CHARS);
|
|
166
|
+
if (!text) continue;
|
|
167
|
+
const embedding = await embed(embedder, text);
|
|
168
|
+
const arr = Array.from(embedding).map((v) => v.toFixed(6)).join(",");
|
|
169
|
+
placeholders.push(`(?, ?::FLOAT[${EMBED_DIM}])`);
|
|
170
|
+
insertParams.push(entryId, `[${arr}]`);
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
if (placeholders.length > 0) {
|
|
174
|
+
await conn.run(
|
|
175
|
+
`INSERT INTO _emb_temp VALUES ${placeholders.join(",")};`,
|
|
176
|
+
insertParams
|
|
177
|
+
);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
embedded = Math.min(i + batchSize, rows.length);
|
|
181
|
+
if (embedded % 512 < batchSize || embedded >= rows.length) {
|
|
182
|
+
log(` Embedded ${embedded}/${rows.length}`);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
log("Merging embeddings...");
|
|
187
|
+
await conn.run("UPDATE session_entries SET embedding = (SELECT embedding FROM _emb_temp WHERE _emb_temp.entry_id = session_entries.entry_id) WHERE entry_id IN (SELECT entry_id FROM _emb_temp);");
|
|
188
|
+
await conn.run("DROP TABLE _emb_temp;");
|
|
189
|
+
|
|
190
|
+
return rows.length;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/** Incremental build — only process new/modified files. */
|
|
194
|
+
async function incrementalBuild(conn, files) {
|
|
195
|
+
log("Incremental build...");
|
|
196
|
+
|
|
197
|
+
// Get list of already-indexed filenames
|
|
198
|
+
const indexedResult = await conn.run("SELECT DISTINCT filename FROM session_entries;");
|
|
199
|
+
const indexedFiles = new Set((await indexedResult.getRows()).map((r) => String(r[0])));
|
|
200
|
+
log(` Previously indexed: ${indexedFiles.size} files`);
|
|
201
|
+
|
|
202
|
+
// Find new files (on disk but not in DB)
|
|
203
|
+
const newFiles = files.filter((f) => !indexedFiles.has(f.path));
|
|
204
|
+
// Find deleted files (in DB but not on disk)
|
|
205
|
+
const diskPaths = new Set(files.map((f) => f.path));
|
|
206
|
+
const deletedFiles = Array.from(indexedFiles).filter((f) => !diskPaths.has(f));
|
|
207
|
+
|
|
208
|
+
log(` New files: ${newFiles.length}`);
|
|
209
|
+
log(` Deleted files: ${deletedFiles.length}`);
|
|
210
|
+
|
|
211
|
+
// Delete entries from removed files
|
|
212
|
+
for (const filePath of deletedFiles) {
|
|
213
|
+
await conn.run(`DELETE FROM session_entries WHERE filename = '${filePath.replace(/'/g, "''")}';`);
|
|
214
|
+
log(` Deleted entries for: ${path.basename(filePath)}`);
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// Insert entries for new files
|
|
218
|
+
if (newFiles.length > 0) {
|
|
219
|
+
for (const file of newFiles) {
|
|
220
|
+
const escapedPath = file.path.replace(/'/g, "''").replace(/\\/g, "/");
|
|
221
|
+
await conn.run(`
|
|
222
|
+
INSERT INTO session_entries (filename, line_number, entry_type, entry_id, parent_id, timestamp, role, custom_type, content_text)
|
|
223
|
+
SELECT
|
|
224
|
+
filename,
|
|
225
|
+
line_number,
|
|
226
|
+
json->>'type' AS entry_type,
|
|
227
|
+
json->>'id' AS entry_id,
|
|
228
|
+
json->>'parentId' AS parent_id,
|
|
229
|
+
json->>'timestamp' AS timestamp,
|
|
230
|
+
json->'message'->>'role' AS role,
|
|
231
|
+
json->>'customType' AS custom_type,
|
|
232
|
+
CASE
|
|
233
|
+
WHEN json_type(json->'message'->'content') = 'ARRAY'
|
|
234
|
+
THEN substring(
|
|
235
|
+
concat(
|
|
236
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].text'), ' '), ''),
|
|
237
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].thinking'), ' '), ''),
|
|
238
|
+
COALESCE(array_to_string(json_extract_string(json->'message'->'content', '$[*].name'), ' '), '')
|
|
239
|
+
), 1, ${MAX_CONTENT_CHARS}
|
|
240
|
+
)
|
|
241
|
+
ELSE substring(COALESCE(json->'message'->>'content', json->>'content', ''), 1, ${MAX_CONTENT_CHARS})
|
|
242
|
+
END AS content_text
|
|
243
|
+
FROM (
|
|
244
|
+
SELECT
|
|
245
|
+
'${escapedPath}' AS filename,
|
|
246
|
+
row_number() OVER () AS line_number,
|
|
247
|
+
json
|
|
248
|
+
FROM read_json_objects('${escapedPath}', format='newline_delimited', ignore_errors=true)
|
|
249
|
+
)
|
|
250
|
+
WHERE json->>'type' IN ('message', 'custom_message')
|
|
251
|
+
`);
|
|
252
|
+
log(` Inserted: ${path.basename(file.path)}`);
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
// Ensure embedding column exists
|
|
257
|
+
await conn.run(`ALTER TABLE session_entries ADD COLUMN IF NOT EXISTS embedding FLOAT[${EMBED_DIM}];`);
|
|
258
|
+
|
|
259
|
+
// Rebuild FTS index (DuckDB FTS doesn't support incremental)
|
|
260
|
+
log("Rebuilding FTS index...");
|
|
261
|
+
await conn.run("PRAGMA create_fts_index('session_entries', 'entry_id', 'content_text', 'role', stemmer='porter', overwrite=true);");
|
|
262
|
+
|
|
263
|
+
// Generate embeddings for any entries that don't have them (new files or crash recovery)
|
|
264
|
+
let newEmbeddings = 0;
|
|
265
|
+
const needEmbResult = await conn.run("SELECT COUNT(*) FROM session_entries WHERE embedding IS NULL AND length(content_text) > 0;");
|
|
266
|
+
const needEmb = Number((await needEmbResult.getRows())[0][0]);
|
|
267
|
+
if (needEmb > 0) {
|
|
268
|
+
log(`${needEmb} entries need embeddings — generating...`);
|
|
269
|
+
const embedder = await loadEmbedder();
|
|
270
|
+
newEmbeddings = await generateEmbeddingsForNew(conn, embedder);
|
|
271
|
+
} else {
|
|
272
|
+
log("All entries have embeddings.");
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
// Count final state
|
|
276
|
+
const entryCount = Number((await (await conn.run("SELECT COUNT(*) FROM session_entries")).getRows())[0][0]);
|
|
277
|
+
const sessionCount = Number((await (await conn.run("SELECT COUNT(DISTINCT filename) FROM session_entries")).getRows())[0][0]);
|
|
278
|
+
const embCount = Number((await (await conn.run("SELECT COUNT(*) FROM session_entries WHERE embedding IS NOT NULL")).getRows())[0][0]);
|
|
279
|
+
|
|
280
|
+
return { entryCount, sessionCount, newEmbeddings, embCount };
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
async function main() {
|
|
284
|
+
log("=== DuckDB session search build started ===");
|
|
285
|
+
fs.mkdirSync(DB_DIR, { recursive: true });
|
|
286
|
+
|
|
287
|
+
const files = findSessionFiles();
|
|
288
|
+
log(`Found ${files.length} JSONL files (excluding recent)`);
|
|
289
|
+
if (files.length === 0) { log("No files to index. Exiting."); process.exit(0); }
|
|
290
|
+
|
|
291
|
+
const dbExists = fs.existsSync(DB_PATH);
|
|
292
|
+
let conn;
|
|
293
|
+
|
|
294
|
+
if (!dbExists) {
|
|
295
|
+
log("No existing database — creating new...");
|
|
296
|
+
const inst = await DuckDBInstance.create(DB_PATH);
|
|
297
|
+
conn = await inst.connect();
|
|
298
|
+
await conn.run("INSTALL fts; LOAD fts;");
|
|
299
|
+
|
|
300
|
+
const { entryCount, sessionCount } = await fullBuild(conn, files);
|
|
301
|
+
await conn.run("CHECKPOINT;");
|
|
302
|
+
|
|
303
|
+
// Store metadata
|
|
304
|
+
await setMetadata(conn, "entry_count", String(entryCount));
|
|
305
|
+
await setMetadata(conn, "session_count", String(sessionCount));
|
|
306
|
+
await setMetadata(conn, "last_built", ts());
|
|
307
|
+
await setMetadata(conn, "has_embeddings", "true");
|
|
308
|
+
await setMetadata(conn, "build_type", "full");
|
|
309
|
+
|
|
310
|
+
const embCount = Number((await (await conn.run("SELECT COUNT(*) FROM session_entries WHERE embedding IS NOT NULL")).getRows())[0][0]);
|
|
311
|
+
log(`Embeddings: ${embCount}/${entryCount}`);
|
|
312
|
+
|
|
313
|
+
await conn.run("CHECKPOINT;");
|
|
314
|
+
try { await conn.disconnect(); } catch {}
|
|
315
|
+
log(`Database size: ${(fs.statSync(DB_PATH).size / 1024 / 1024).toFixed(1)} MB`);
|
|
316
|
+
log("=== Build complete (full) ===");
|
|
317
|
+
return;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
// Existing DB — try incremental
|
|
321
|
+
log("Opening existing database...");
|
|
322
|
+
const inst = await DuckDBInstance.create(DB_PATH);
|
|
323
|
+
conn = await inst.connect();
|
|
324
|
+
await conn.run("LOAD fts;");
|
|
325
|
+
|
|
326
|
+
// Check if session_entries table exists
|
|
327
|
+
const tableCheck = await conn.run("SELECT COUNT(*) FROM information_schema.tables WHERE table_name = 'session_entries'");
|
|
328
|
+
const hasTable = Number((await tableCheck.getRows())[0][0]) > 0;
|
|
329
|
+
|
|
330
|
+
let result;
|
|
331
|
+
if (!hasTable) {
|
|
332
|
+
result = await fullBuild(conn, files);
|
|
333
|
+
await setMetadata(conn, "build_type", "full");
|
|
334
|
+
} else {
|
|
335
|
+
result = await incrementalBuild(conn, files);
|
|
336
|
+
await setMetadata(conn, "build_type", "incremental");
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
await conn.run("CHECKPOINT;");
|
|
340
|
+
|
|
341
|
+
// Store metadata
|
|
342
|
+
await setMetadata(conn, "entry_count", String(result.entryCount));
|
|
343
|
+
await setMetadata(conn, "session_count", String(result.sessionCount));
|
|
344
|
+
await setMetadata(conn, "last_built", ts());
|
|
345
|
+
await setMetadata(conn, "has_embeddings", "true");
|
|
346
|
+
|
|
347
|
+
// Final stats
|
|
348
|
+
const embCount = Number((await (await conn.run("SELECT COUNT(*) FROM session_entries WHERE embedding IS NOT NULL")).getRows())[0][0]);
|
|
349
|
+
log(`Final: ${result.entryCount} entries, ${result.sessionCount} sessions, ${embCount} embeddings`);
|
|
350
|
+
|
|
351
|
+
try { await conn.disconnect(); } catch {}
|
|
352
|
+
log(`Database size: ${(fs.statSync(DB_PATH).size / 1024 / 1024).toFixed(1)} MB`);
|
|
353
|
+
log("=== Build complete ===");
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
main().catch((e) => {
|
|
357
|
+
log(`FATAL: ${e.message}`);
|
|
358
|
+
log(e.stack || "");
|
|
359
|
+
process.exit(1);
|
|
360
|
+
});
|