@hanhnd/agent-kit 1.0.38 → 1.0.39
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +231 -125
- package/dist/mcp/memory.js +1 -1
- package/dist/server.js +234 -128
- package/dist/services/digest/digest-processor.test.js +1 -3
- package/dist/services/memory/indexer.js +35 -86
- package/dist/services/memory/indexer.test.js +229 -586
- package/dist/services/memory/store.d.ts +4 -11
- package/dist/services/memory/store.js +200 -66
- package/dist/services/memory/store.test.js +359 -108
- package/dist/services/memory/types.d.ts +28 -0
- package/package.json +1 -1
package/dist/server.js
CHANGED
|
@@ -25930,8 +25930,8 @@ function getElementAtPath(obj, path15) {
|
|
|
25930
25930
|
}
|
|
25931
25931
|
function promiseAllObject(promisesObj) {
|
|
25932
25932
|
const keys = Object.keys(promisesObj);
|
|
25933
|
-
const
|
|
25934
|
-
return Promise.all(
|
|
25933
|
+
const promises3 = keys.map((key) => promisesObj[key]);
|
|
25934
|
+
return Promise.all(promises3).then((results) => {
|
|
25935
25935
|
const resolvedObj = {};
|
|
25936
25936
|
for (let i = 0; i < keys.length; i++) {
|
|
25937
25937
|
resolvedObj[keys[i]] = results[i];
|
|
@@ -37016,14 +37016,14 @@ var MemoryIndexer = class {
|
|
|
37016
37016
|
const existingChunkIds = this.store.hashesBySource(source);
|
|
37017
37017
|
let text;
|
|
37018
37018
|
try {
|
|
37019
|
-
text = fs6.
|
|
37019
|
+
text = await fs6.promises.readFile(absolutePath, "utf8");
|
|
37020
37020
|
} catch (err) {
|
|
37021
37021
|
console.warn(`[memory-indexer] Cannot read file ${absolutePath}:`, err);
|
|
37022
37022
|
return { indexed: 0, deleted: 0, skipped: existingChunkIds.size };
|
|
37023
37023
|
}
|
|
37024
37024
|
let fileMtimeAt;
|
|
37025
37025
|
try {
|
|
37026
|
-
fileMtimeAt = fs6.
|
|
37026
|
+
fileMtimeAt = (await fs6.promises.stat(absolutePath)).mtimeMs;
|
|
37027
37027
|
} catch (err) {
|
|
37028
37028
|
console.warn(`[memory-indexer] Cannot stat file ${absolutePath}:`, err);
|
|
37029
37029
|
fileMtimeAt = Date.now();
|
|
@@ -37056,22 +37056,30 @@ var MemoryIndexer = class {
|
|
|
37056
37056
|
}
|
|
37057
37057
|
async indexDirectory(rootDir, opts = {}) {
|
|
37058
37058
|
const totals = { indexed: 0, deleted: 0, skipped: 0 };
|
|
37059
|
-
|
|
37059
|
+
let rootExists;
|
|
37060
|
+
try {
|
|
37061
|
+
await fs6.promises.stat(rootDir);
|
|
37062
|
+
rootExists = true;
|
|
37063
|
+
} catch {
|
|
37064
|
+
rootExists = false;
|
|
37065
|
+
}
|
|
37066
|
+
if (!rootExists) return totals;
|
|
37060
37067
|
const relativeBase = opts.relativeBase ?? this.config.wikiDir;
|
|
37061
37068
|
const excludeFiles = new Set(opts.excludeFiles ?? []);
|
|
37062
37069
|
const files = [];
|
|
37063
|
-
const walk = (dirPath) => {
|
|
37070
|
+
const walk = async (dirPath) => {
|
|
37064
37071
|
let entries;
|
|
37065
37072
|
try {
|
|
37066
|
-
entries = fs6.
|
|
37073
|
+
entries = await fs6.promises.readdir(dirPath, { withFileTypes: true });
|
|
37067
37074
|
} catch (err) {
|
|
37068
37075
|
console.warn(`[memory-indexer] Cannot scan directory ${dirPath}:`, err);
|
|
37069
37076
|
return;
|
|
37070
37077
|
}
|
|
37078
|
+
entries.sort((a, b) => a.name.localeCompare(b.name));
|
|
37071
37079
|
for (const entry of entries) {
|
|
37072
37080
|
const entryPath = path8.join(dirPath, entry.name);
|
|
37073
37081
|
if (entry.isDirectory()) {
|
|
37074
|
-
walk(entryPath);
|
|
37082
|
+
await walk(entryPath);
|
|
37075
37083
|
continue;
|
|
37076
37084
|
}
|
|
37077
37085
|
if (entry.isFile() && /\.md$/i.test(entry.name) && !excludeFiles.has(entry.name)) {
|
|
@@ -37079,7 +37087,7 @@ var MemoryIndexer = class {
|
|
|
37079
37087
|
}
|
|
37080
37088
|
}
|
|
37081
37089
|
};
|
|
37082
|
-
walk(rootDir);
|
|
37090
|
+
await walk(rootDir);
|
|
37083
37091
|
const currentSources = new Set(files.map((file) => path8.relative(relativeBase, file)));
|
|
37084
37092
|
for (const stale of this.store.indexedSources()) {
|
|
37085
37093
|
if (!currentSources.has(stale)) {
|
|
@@ -37097,77 +37105,34 @@ var MemoryIndexer = class {
|
|
|
37097
37105
|
}
|
|
37098
37106
|
async search(query, topK) {
|
|
37099
37107
|
const fetchLimit = topK * FETCH_MULTIPLIER;
|
|
37100
|
-
let
|
|
37108
|
+
let embedding;
|
|
37101
37109
|
if (this.store.vecAvailable) {
|
|
37102
37110
|
try {
|
|
37103
|
-
const
|
|
37104
|
-
|
|
37111
|
+
const embeddings = await this.embedder.embed([query]);
|
|
37112
|
+
embedding = embeddings[0];
|
|
37105
37113
|
} catch (err) {
|
|
37106
37114
|
console.warn("[memory-indexer] Dense search embedding failed:", err);
|
|
37107
37115
|
}
|
|
37108
37116
|
}
|
|
37109
|
-
const
|
|
37110
|
-
|
|
37111
|
-
|
|
37112
|
-
|
|
37113
|
-
|
|
37114
|
-
|
|
37115
|
-
|
|
37116
|
-
|
|
37117
|
-
const unionIndexOf = new Map(unionArray.map((id, idx) => [id, idx]));
|
|
37118
|
-
const recencyOrdered = [...unionArray].sort((a, b) => {
|
|
37119
|
-
const mtimeA = chunkMap.get(a)?.fileMtimeAt ?? 0;
|
|
37120
|
-
const mtimeB = chunkMap.get(b)?.fileMtimeAt ?? 0;
|
|
37121
|
-
if (mtimeB !== mtimeA) return mtimeB - mtimeA;
|
|
37122
|
-
return (unionIndexOf.get(a) ?? 0) - (unionIndexOf.get(b) ?? 0);
|
|
37123
|
-
});
|
|
37124
|
-
const scoreMap = /* @__PURE__ */ new Map();
|
|
37125
|
-
denseResults.forEach((r, rank) => {
|
|
37126
|
-
scoreMap.set(r.id, { dense: 1 / (RRF_K + rank + 1), bm25: 0, recency: 0 });
|
|
37127
|
-
});
|
|
37128
|
-
bm25Results.forEach((r, rank) => {
|
|
37129
|
-
const entry = scoreMap.get(r.id) ?? { dense: 0, bm25: 0, recency: 0 };
|
|
37130
|
-
entry.bm25 = 1 / (RRF_K + rank + 1);
|
|
37131
|
-
scoreMap.set(r.id, entry);
|
|
37117
|
+
const rows = this.store.searchHybrid({
|
|
37118
|
+
query,
|
|
37119
|
+
embedding,
|
|
37120
|
+
topK,
|
|
37121
|
+
fetchLimit,
|
|
37122
|
+
denseScoreFloor: DENSE_SCORE_FLOOR,
|
|
37123
|
+
recencyWeight: RECENCY_WEIGHT,
|
|
37124
|
+
rrfK: RRF_K
|
|
37132
37125
|
});
|
|
37133
|
-
|
|
37134
|
-
const entry = scoreMap.get(id);
|
|
37135
|
-
if (!entry) return;
|
|
37136
|
-
entry.recency = RECENCY_WEIGHT * (1 / (RRF_K + rank + 1));
|
|
37137
|
-
});
|
|
37138
|
-
const denseActive = denseResults.length > 0;
|
|
37139
|
-
const bm25Active = bm25Results.length > 0;
|
|
37140
|
-
const recencyActive = unionIds.size > 0;
|
|
37141
|
-
const maxScore = ((denseActive ? 1 : 0) + (bm25Active ? 1 : 0) + (recencyActive ? RECENCY_WEIGHT : 0)) / (RRF_K + 1);
|
|
37142
|
-
const ranked = [...scoreMap.entries()].map(([id, scores]) => ({
|
|
37143
|
-
id,
|
|
37144
|
-
totalScore: scores.dense + scores.bm25 + scores.recency,
|
|
37145
|
-
hasDense: scores.dense > 0,
|
|
37146
|
-
hasBm25: scores.bm25 > 0,
|
|
37147
|
-
isDenseOnly: denseIds.has(id) && !bm25Ids.has(id)
|
|
37148
|
-
})).filter((candidate) => {
|
|
37149
|
-
if (!candidate.isDenseOnly) return true;
|
|
37150
|
-
const retrieval = denseRetrievalScore.get(candidate.id);
|
|
37151
|
-
if (retrieval === void 0) return true;
|
|
37152
|
-
return retrieval >= DENSE_SCORE_FLOOR;
|
|
37153
|
-
}).sort((a, b) => b.totalScore - a.totalScore);
|
|
37126
|
+
if (rows.length === 0) return [];
|
|
37154
37127
|
const results = [];
|
|
37155
|
-
const
|
|
37156
|
-
for (const r of ranked) {
|
|
37157
|
-
if (results.length >= topK) break;
|
|
37158
|
-
const chunk = chunkMap.get(r.id);
|
|
37159
|
-
if (!chunk) continue;
|
|
37160
|
-
if (seenSources.has(chunk.source)) continue;
|
|
37161
|
-
seenSources.add(chunk.source);
|
|
37162
|
-
const normalizedScore = maxScore > 0 ? r.totalScore / maxScore : 0;
|
|
37163
|
-
const retriever = r.hasDense && r.hasBm25 ? "both" : r.hasDense ? "dense" : "bm25";
|
|
37128
|
+
for (const row of rows) {
|
|
37164
37129
|
let contentSource = "file";
|
|
37165
37130
|
try {
|
|
37166
|
-
chunk.content = fs6.
|
|
37131
|
+
row.chunk.content = await fs6.promises.readFile(path8.join(this.config.wikiDir, row.chunk.source), "utf8");
|
|
37167
37132
|
} catch {
|
|
37168
37133
|
contentSource = "fallback";
|
|
37169
37134
|
}
|
|
37170
|
-
results.push({ chunk, score:
|
|
37135
|
+
results.push({ chunk: row.chunk, score: row.score, retriever: row.retriever, contentSource });
|
|
37171
37136
|
}
|
|
37172
37137
|
return results;
|
|
37173
37138
|
}
|
|
@@ -37176,13 +37141,13 @@ var MemoryIndexer = class {
|
|
|
37176
37141
|
const rawDir = path8.join(this.config.wikiDir, "raw");
|
|
37177
37142
|
const savePath = path8.join(rawDir, `conv_save_${datePart}.md`);
|
|
37178
37143
|
const lockPath = `${savePath}.lock`;
|
|
37179
|
-
fs6.
|
|
37144
|
+
await fs6.promises.mkdir(rawDir, { recursive: true });
|
|
37180
37145
|
const acquired = await acquireLock(lockPath, LOCK_TIMEOUT_MS, LOCK_RETRY_MS);
|
|
37181
37146
|
if (!acquired) {
|
|
37182
37147
|
console.warn("[memory-indexer] Could not acquire write lock, writing anyway (fail-open)");
|
|
37183
37148
|
}
|
|
37184
37149
|
try {
|
|
37185
|
-
fs6.
|
|
37150
|
+
await fs6.promises.appendFile(savePath, `
|
|
37186
37151
|
${content}
|
|
37187
37152
|
`, "utf8");
|
|
37188
37153
|
} finally {
|
|
@@ -37389,10 +37354,6 @@ var MemoryStore = class {
|
|
|
37389
37354
|
}
|
|
37390
37355
|
return Buffer.from(embedding.buffer, embedding.byteOffset, embedding.byteLength);
|
|
37391
37356
|
}
|
|
37392
|
-
normalizeVectorDistance(distance) {
|
|
37393
|
-
if (distance <= 0) return 1;
|
|
37394
|
-
return Math.max(0, Math.min(1, 1 - distance));
|
|
37395
|
-
}
|
|
37396
37357
|
hashesBySource(source) {
|
|
37397
37358
|
const rows = this.db.prepare(`SELECT id FROM memory_chunks WHERE source = ?`).all(source);
|
|
37398
37359
|
return new Set(rows.map((r) => r.id));
|
|
@@ -37429,61 +37390,12 @@ var MemoryStore = class {
|
|
|
37429
37390
|
});
|
|
37430
37391
|
deleteAll();
|
|
37431
37392
|
}
|
|
37432
|
-
|
|
37433
|
-
try {
|
|
37434
|
-
const queryVector = this.toVectorBinding(embedding);
|
|
37435
|
-
const rows = this.db.prepare(
|
|
37436
|
-
`SELECT mc.id, vector_distance_cos(mc.embedding, vector32(?)) AS distance
|
|
37437
|
-
FROM vector_top_k('idx_memory_chunks_embedding', vector32(?), ?) AS vector_matches
|
|
37438
|
-
JOIN memory_chunks mc ON mc.rowid = vector_matches.id
|
|
37439
|
-
ORDER BY distance
|
|
37440
|
-
LIMIT ?`
|
|
37441
|
-
).all(queryVector, queryVector, limit, limit);
|
|
37442
|
-
return rows.map((r) => ({
|
|
37443
|
-
id: r.id,
|
|
37444
|
-
score: this.normalizeVectorDistance(r.distance)
|
|
37445
|
-
}));
|
|
37446
|
-
} catch (err) {
|
|
37447
|
-
console.warn("[memory-store] Dense search failed:", err);
|
|
37448
|
-
return [];
|
|
37449
|
-
}
|
|
37450
|
-
}
|
|
37451
|
-
searchBm25(query, limit) {
|
|
37452
|
-
if (!query.trim()) return [];
|
|
37393
|
+
buildFtsQuery(query) {
|
|
37453
37394
|
const tokens = query.trim().split(/\s+/).map((t) => t.replace(/[^a-zA-Z0-9_]/g, "").toLowerCase()).filter((t) => t.length > 1);
|
|
37454
|
-
|
|
37455
|
-
if (!ftsQuery) return [];
|
|
37456
|
-
try {
|
|
37457
|
-
const rows = this.db.prepare(
|
|
37458
|
-
`SELECT mc.id, bm25(memory_fts, 0.2, 2.0, 5.0) AS rank
|
|
37459
|
-
FROM memory_fts
|
|
37460
|
-
JOIN memory_chunks mc ON mc.rowid = memory_fts.rowid
|
|
37461
|
-
WHERE memory_fts MATCH ?
|
|
37462
|
-
ORDER BY rank
|
|
37463
|
-
LIMIT ?`
|
|
37464
|
-
).all(ftsQuery, limit);
|
|
37465
|
-
if (rows.length === 0) return [];
|
|
37466
|
-
const ranks = rows.map((r) => r.rank);
|
|
37467
|
-
const minRank = Math.min(...ranks);
|
|
37468
|
-
const maxRank = Math.max(...ranks);
|
|
37469
|
-
const range = maxRank - minRank;
|
|
37470
|
-
return rows.map((r) => ({
|
|
37471
|
-
id: r.id,
|
|
37472
|
-
score: range > 0 ? (maxRank - r.rank) / range : 1
|
|
37473
|
-
}));
|
|
37474
|
-
} catch (err) {
|
|
37475
|
-
console.warn("[memory-store] BM25 search failed:", err);
|
|
37476
|
-
return [];
|
|
37477
|
-
}
|
|
37395
|
+
return (0, import_stopword.removeStopwords)(tokens, import_stopword.eng).map((t) => `${t.replace(/s$/, "")}*`).join(" OR ");
|
|
37478
37396
|
}
|
|
37479
|
-
|
|
37480
|
-
|
|
37481
|
-
const placeholders = ids.map(() => "?").join(", ");
|
|
37482
|
-
const rows = this.db.prepare(
|
|
37483
|
-
`SELECT id, source, source_type, heading, heading_level, content, line_start, line_end, file_mtime_at
|
|
37484
|
-
FROM memory_chunks WHERE id IN (${placeholders})`
|
|
37485
|
-
).all(...ids);
|
|
37486
|
-
return rows.map((r) => ({
|
|
37397
|
+
mapMemoryChunkRow(r) {
|
|
37398
|
+
return {
|
|
37487
37399
|
id: r.id,
|
|
37488
37400
|
source: r.source,
|
|
37489
37401
|
sourceType: r.source_type,
|
|
@@ -37493,7 +37405,201 @@ var MemoryStore = class {
|
|
|
37493
37405
|
lineStart: r.line_start,
|
|
37494
37406
|
lineEnd: r.line_end,
|
|
37495
37407
|
fileMtimeAt: r.file_mtime_at
|
|
37496
|
-
}
|
|
37408
|
+
};
|
|
37409
|
+
}
|
|
37410
|
+
searchHybrid(options) {
|
|
37411
|
+
const {
|
|
37412
|
+
query,
|
|
37413
|
+
embedding,
|
|
37414
|
+
topK,
|
|
37415
|
+
fetchLimit,
|
|
37416
|
+
denseScoreFloor,
|
|
37417
|
+
recencyWeight,
|
|
37418
|
+
rrfK,
|
|
37419
|
+
sourceType,
|
|
37420
|
+
sinceMtimeAt,
|
|
37421
|
+
includeDebug
|
|
37422
|
+
} = options;
|
|
37423
|
+
const ftsQuery = this.buildFtsQuery(query);
|
|
37424
|
+
const hasDense = embedding !== void 0;
|
|
37425
|
+
const hasBm25 = ftsQuery.length > 0;
|
|
37426
|
+
if (!hasDense && !hasBm25) return [];
|
|
37427
|
+
const hasFilters = sourceType !== void 0 || sinceMtimeAt !== void 0;
|
|
37428
|
+
try {
|
|
37429
|
+
const params = [];
|
|
37430
|
+
let denseCte = "";
|
|
37431
|
+
let denseRankedCte = "";
|
|
37432
|
+
if (hasDense) {
|
|
37433
|
+
const queryBuf = this.toVectorBinding(embedding);
|
|
37434
|
+
if (hasFilters) {
|
|
37435
|
+
const filterClauses = [];
|
|
37436
|
+
if (sourceType) {
|
|
37437
|
+
filterClauses.push("mc.source_type = ?");
|
|
37438
|
+
params.push(sourceType);
|
|
37439
|
+
}
|
|
37440
|
+
if (sinceMtimeAt !== void 0) {
|
|
37441
|
+
filterClauses.push("mc.file_mtime_at >= ?");
|
|
37442
|
+
params.push(sinceMtimeAt);
|
|
37443
|
+
}
|
|
37444
|
+
const whereClause = filterClauses.length > 0 ? `WHERE ${filterClauses.join(" AND ")}` : "";
|
|
37445
|
+
params.push(queryBuf);
|
|
37446
|
+
params.push(fetchLimit);
|
|
37447
|
+
denseCte = `dense_data AS (
|
|
37448
|
+
SELECT mc.id, mc.source, mc.source_type, mc.file_mtime_at,
|
|
37449
|
+
mc.heading, mc.heading_level, mc.content, mc.line_start, mc.line_end,
|
|
37450
|
+
vector_distance_cos(mc.embedding, vector32(?)) AS dense_dist
|
|
37451
|
+
FROM memory_chunks mc
|
|
37452
|
+
${whereClause}
|
|
37453
|
+
ORDER BY dense_dist
|
|
37454
|
+
LIMIT ?
|
|
37455
|
+
)`;
|
|
37456
|
+
} else {
|
|
37457
|
+
params.push(queryBuf);
|
|
37458
|
+
params.push(queryBuf);
|
|
37459
|
+
params.push(fetchLimit);
|
|
37460
|
+
denseCte = `dense_data AS (
|
|
37461
|
+
SELECT mc.id, mc.source, mc.source_type, mc.file_mtime_at,
|
|
37462
|
+
mc.heading, mc.heading_level, mc.content, mc.line_start, mc.line_end,
|
|
37463
|
+
vector_distance_cos(mc.embedding, vector32(?)) AS dense_dist
|
|
37464
|
+
FROM vector_top_k('idx_memory_chunks_embedding', vector32(?), ?) AS vtk
|
|
37465
|
+
JOIN memory_chunks mc ON mc.rowid = vtk.id
|
|
37466
|
+
)`;
|
|
37467
|
+
}
|
|
37468
|
+
denseRankedCte = `dense_ranked AS (
|
|
37469
|
+
SELECT *, ROW_NUMBER() OVER (ORDER BY dense_dist, id) AS dense_rank
|
|
37470
|
+
FROM dense_data
|
|
37471
|
+
)`;
|
|
37472
|
+
}
|
|
37473
|
+
let bm25Cte = "";
|
|
37474
|
+
let bm25RankedCte = "";
|
|
37475
|
+
if (hasBm25) {
|
|
37476
|
+
const bm25FilterClauses = [];
|
|
37477
|
+
if (sourceType) bm25FilterClauses.push("mc.source_type = ?");
|
|
37478
|
+
if (sinceMtimeAt !== void 0) bm25FilterClauses.push("mc.file_mtime_at >= ?");
|
|
37479
|
+
const bm25WhereExtra = bm25FilterClauses.length > 0 ? ` AND ${bm25FilterClauses.join(" AND ")}` : "";
|
|
37480
|
+
params.push(ftsQuery);
|
|
37481
|
+
if (sourceType) params.push(sourceType);
|
|
37482
|
+
if (sinceMtimeAt !== void 0) params.push(sinceMtimeAt);
|
|
37483
|
+
params.push(fetchLimit);
|
|
37484
|
+
bm25Cte = `bm25_data AS (
|
|
37485
|
+
SELECT mc.id, mc.source, mc.source_type, mc.file_mtime_at,
|
|
37486
|
+
mc.heading, mc.heading_level, mc.content, mc.line_start, mc.line_end,
|
|
37487
|
+
bm25(memory_fts, 0.2, 2.0, 5.0) AS bm25_neg
|
|
37488
|
+
FROM memory_fts
|
|
37489
|
+
JOIN memory_chunks mc ON mc.rowid = memory_fts.rowid
|
|
37490
|
+
WHERE memory_fts MATCH ?${bm25WhereExtra}
|
|
37491
|
+
LIMIT ?
|
|
37492
|
+
)`;
|
|
37493
|
+
bm25RankedCte = `bm25_ranked AS (
|
|
37494
|
+
SELECT *, ROW_NUMBER() OVER (ORDER BY bm25_neg, id) AS bm25_rank
|
|
37495
|
+
FROM bm25_data
|
|
37496
|
+
)`;
|
|
37497
|
+
}
|
|
37498
|
+
let candidatesCte = "";
|
|
37499
|
+
if (hasDense && hasBm25) {
|
|
37500
|
+
candidatesCte = `all_candidates AS (
|
|
37501
|
+
SELECT d.id, d.source, d.source_type, d.file_mtime_at,
|
|
37502
|
+
d.heading, d.heading_level, d.content, d.line_start, d.line_end,
|
|
37503
|
+
CAST(d.dense_rank AS INTEGER) AS dense_rank, d.dense_dist,
|
|
37504
|
+
b.bm25_rank,
|
|
37505
|
+
CASE WHEN b.id IS NOT NULL THEN 'both' ELSE 'dense' END AS retriever
|
|
37506
|
+
FROM dense_ranked d
|
|
37507
|
+
LEFT JOIN bm25_ranked b ON b.id = d.id
|
|
37508
|
+
UNION ALL
|
|
37509
|
+
SELECT b.id, b.source, b.source_type, b.file_mtime_at,
|
|
37510
|
+
b.heading, b.heading_level, b.content, b.line_start, b.line_end,
|
|
37511
|
+
NULL AS dense_rank, NULL AS dense_dist,
|
|
37512
|
+
CAST(b.bm25_rank AS INTEGER) AS bm25_rank,
|
|
37513
|
+
'bm25' AS retriever
|
|
37514
|
+
FROM bm25_ranked b
|
|
37515
|
+
LEFT JOIN dense_ranked d ON d.id = b.id
|
|
37516
|
+
WHERE d.id IS NULL
|
|
37517
|
+
)`;
|
|
37518
|
+
} else if (hasDense) {
|
|
37519
|
+
candidatesCte = `all_candidates AS (
|
|
37520
|
+
SELECT id, source, source_type, file_mtime_at,
|
|
37521
|
+
heading, heading_level, content, line_start, line_end,
|
|
37522
|
+
CAST(dense_rank AS INTEGER) AS dense_rank, dense_dist,
|
|
37523
|
+
NULL AS bm25_rank, 'dense' AS retriever
|
|
37524
|
+
FROM dense_ranked
|
|
37525
|
+
)`;
|
|
37526
|
+
} else {
|
|
37527
|
+
candidatesCte = `all_candidates AS (
|
|
37528
|
+
SELECT id, source, source_type, file_mtime_at,
|
|
37529
|
+
heading, heading_level, content, line_start, line_end,
|
|
37530
|
+
NULL AS dense_rank, NULL AS dense_dist,
|
|
37531
|
+
CAST(bm25_rank AS INTEGER) AS bm25_rank, 'bm25' AS retriever
|
|
37532
|
+
FROM bm25_ranked
|
|
37533
|
+
)`;
|
|
37534
|
+
}
|
|
37535
|
+
params.push(rrfK, rrfK, recencyWeight, rrfK, denseScoreFloor, topK);
|
|
37536
|
+
const sql = `
|
|
37537
|
+
WITH
|
|
37538
|
+
${[denseCte, denseRankedCte, bm25Cte, bm25RankedCte, candidatesCte].filter(Boolean).join(",\n")}
|
|
37539
|
+
,
|
|
37540
|
+
recency_ranked AS (
|
|
37541
|
+
SELECT id, ROW_NUMBER() OVER (ORDER BY file_mtime_at DESC, source, id) AS recency_rank
|
|
37542
|
+
FROM all_candidates
|
|
37543
|
+
),
|
|
37544
|
+
scored AS (
|
|
37545
|
+
SELECT c.id, c.source, c.source_type, c.file_mtime_at,
|
|
37546
|
+
c.heading, c.heading_level, c.content, c.line_start, c.line_end,
|
|
37547
|
+
c.dense_rank, c.dense_dist, c.bm25_rank, r.recency_rank,
|
|
37548
|
+
c.retriever,
|
|
37549
|
+
COALESCE(1.0 / (? + c.dense_rank), 0.0) +
|
|
37550
|
+
COALESCE(1.0 / (? + c.bm25_rank), 0.0) +
|
|
37551
|
+
? * (1.0 / (? + r.recency_rank)) AS total_score
|
|
37552
|
+
FROM all_candidates c
|
|
37553
|
+
JOIN recency_ranked r ON r.id = c.id
|
|
37554
|
+
WHERE c.retriever != 'dense' OR (1.0 - c.dense_dist) >= ?
|
|
37555
|
+
),
|
|
37556
|
+
best_per_source AS (
|
|
37557
|
+
SELECT *,
|
|
37558
|
+
ROW_NUMBER() OVER (PARTITION BY source ORDER BY total_score DESC, source, id) AS source_rank
|
|
37559
|
+
FROM scored
|
|
37560
|
+
)
|
|
37561
|
+
SELECT id, source, source_type, file_mtime_at, heading, heading_level, content, line_start, line_end,
|
|
37562
|
+
dense_rank, dense_dist, bm25_rank, recency_rank, total_score, retriever
|
|
37563
|
+
FROM best_per_source
|
|
37564
|
+
WHERE source_rank = 1
|
|
37565
|
+
ORDER BY total_score DESC, source, id
|
|
37566
|
+
LIMIT ?
|
|
37567
|
+
`;
|
|
37568
|
+
const rows = this.db.prepare(sql).all(...params);
|
|
37569
|
+
if (rows.length === 0) return [];
|
|
37570
|
+
const denseActive = hasDense;
|
|
37571
|
+
const bm25Active = hasBm25;
|
|
37572
|
+
const recencyActive = rows.length > 0;
|
|
37573
|
+
const maxScore = ((denseActive ? 1 : 0) + (bm25Active ? 1 : 0) + (recencyActive ? recencyWeight : 0)) / (rrfK + 1);
|
|
37574
|
+
return rows.map((r) => {
|
|
37575
|
+
const chunk = this.mapMemoryChunkRow(r);
|
|
37576
|
+
const normalizedScore = maxScore > 0 ? r.total_score / maxScore : 0;
|
|
37577
|
+
const row = {
|
|
37578
|
+
chunk,
|
|
37579
|
+
score: normalizedScore,
|
|
37580
|
+
retriever: r.retriever
|
|
37581
|
+
};
|
|
37582
|
+
if (includeDebug) {
|
|
37583
|
+
const denseScore = r.dense_dist !== null ? Math.max(0, Math.min(1, 1 - r.dense_dist)) : void 0;
|
|
37584
|
+
const bm25Score = r.bm25_rank !== null ? 1 / (rrfK + r.bm25_rank) : void 0;
|
|
37585
|
+
const recencyScore = recencyWeight * (1 / (rrfK + r.recency_rank));
|
|
37586
|
+
const debug = {
|
|
37587
|
+
denseRank: r.dense_rank ?? void 0,
|
|
37588
|
+
denseScore,
|
|
37589
|
+
bm25Rank: r.bm25_rank ?? void 0,
|
|
37590
|
+
bm25Score,
|
|
37591
|
+
recencyRank: r.recency_rank,
|
|
37592
|
+
recencyScore,
|
|
37593
|
+
totalScore: r.total_score
|
|
37594
|
+
};
|
|
37595
|
+
row.debug = debug;
|
|
37596
|
+
}
|
|
37597
|
+
return row;
|
|
37598
|
+
});
|
|
37599
|
+
} catch (err) {
|
|
37600
|
+
console.warn("[memory-store] Hybrid search failed:", err);
|
|
37601
|
+
return [];
|
|
37602
|
+
}
|
|
37497
37603
|
}
|
|
37498
37604
|
get vecAvailable() {
|
|
37499
37605
|
return true;
|
|
@@ -37978,7 +38084,7 @@ ${content}`;
|
|
|
37978
38084
|
const blocks = [];
|
|
37979
38085
|
for (const row of rows) {
|
|
37980
38086
|
try {
|
|
37981
|
-
const content = fs11.
|
|
38087
|
+
const content = await fs11.promises.readFile(path14.join(config2.wikiDir, row.source), "utf8");
|
|
37982
38088
|
const displaySource = row.source.replace(/^compiled\//, "");
|
|
37983
38089
|
blocks.push(`### ${displaySource}
|
|
37984
38090
|
${content}`);
|
|
@@ -72,10 +72,8 @@ describe('digestConversationFile', () => {
|
|
|
72
72
|
});
|
|
73
73
|
assert.equal(result.indexed, true);
|
|
74
74
|
assert.equal(result.markdown, markdownPath);
|
|
75
|
-
const denseResults = store.searchDense(new Float32Array(384).fill(0.05), 5);
|
|
76
75
|
const indexedSource = path.relative(config.wikiDir, markdownPath);
|
|
77
|
-
|
|
78
|
-
assert.ok(indexedChunks.some((chunk) => chunk.source === indexedSource), `Expected dense search to return a chunk for ${indexedSource}`);
|
|
76
|
+
assert.ok(store.hashesBySource(indexedSource).size > 0, `Expected source ${indexedSource} to be indexed in store`);
|
|
79
77
|
}
|
|
80
78
|
finally {
|
|
81
79
|
store.close();
|
|
@@ -32,7 +32,7 @@ export class MemoryIndexer {
|
|
|
32
32
|
const existingChunkIds = this.store.hashesBySource(source);
|
|
33
33
|
let text;
|
|
34
34
|
try {
|
|
35
|
-
text = fs.
|
|
35
|
+
text = await fs.promises.readFile(absolutePath, 'utf8');
|
|
36
36
|
}
|
|
37
37
|
catch (err) {
|
|
38
38
|
console.warn(`[memory-indexer] Cannot read file ${absolutePath}:`, err);
|
|
@@ -40,7 +40,7 @@ export class MemoryIndexer {
|
|
|
40
40
|
}
|
|
41
41
|
let fileMtimeAt;
|
|
42
42
|
try {
|
|
43
|
-
fileMtimeAt = fs.
|
|
43
|
+
fileMtimeAt = (await fs.promises.stat(absolutePath)).mtimeMs;
|
|
44
44
|
}
|
|
45
45
|
catch (err) {
|
|
46
46
|
console.warn(`[memory-indexer] Cannot stat file ${absolutePath}:`, err);
|
|
@@ -75,24 +75,34 @@ export class MemoryIndexer {
|
|
|
75
75
|
}
|
|
76
76
|
async indexDirectory(rootDir, opts = {}) {
|
|
77
77
|
const totals = { indexed: 0, deleted: 0, skipped: 0 };
|
|
78
|
-
|
|
78
|
+
let rootExists;
|
|
79
|
+
try {
|
|
80
|
+
await fs.promises.stat(rootDir);
|
|
81
|
+
rootExists = true;
|
|
82
|
+
}
|
|
83
|
+
catch {
|
|
84
|
+
rootExists = false;
|
|
85
|
+
}
|
|
86
|
+
if (!rootExists)
|
|
79
87
|
return totals;
|
|
80
88
|
const relativeBase = opts.relativeBase ?? this.config.wikiDir;
|
|
81
89
|
const excludeFiles = new Set(opts.excludeFiles ?? []);
|
|
82
90
|
const files = [];
|
|
83
|
-
const walk = (dirPath) => {
|
|
91
|
+
const walk = async (dirPath) => {
|
|
84
92
|
let entries;
|
|
85
93
|
try {
|
|
86
|
-
entries = fs.
|
|
94
|
+
entries = await fs.promises.readdir(dirPath, { withFileTypes: true });
|
|
87
95
|
}
|
|
88
96
|
catch (err) {
|
|
89
97
|
console.warn(`[memory-indexer] Cannot scan directory ${dirPath}:`, err);
|
|
90
98
|
return;
|
|
91
99
|
}
|
|
100
|
+
// Sort for deterministic traversal order
|
|
101
|
+
entries.sort((a, b) => a.name.localeCompare(b.name));
|
|
92
102
|
for (const entry of entries) {
|
|
93
103
|
const entryPath = path.join(dirPath, entry.name);
|
|
94
104
|
if (entry.isDirectory()) {
|
|
95
|
-
walk(entryPath);
|
|
105
|
+
await walk(entryPath);
|
|
96
106
|
continue;
|
|
97
107
|
}
|
|
98
108
|
if (entry.isFile() && /\.md$/i.test(entry.name) && !excludeFiles.has(entry.name)) {
|
|
@@ -100,7 +110,7 @@ export class MemoryIndexer {
|
|
|
100
110
|
}
|
|
101
111
|
}
|
|
102
112
|
};
|
|
103
|
-
walk(rootDir);
|
|
113
|
+
await walk(rootDir);
|
|
104
114
|
const currentSources = new Set(files.map((file) => path.relative(relativeBase, file)));
|
|
105
115
|
// Remove stale sources (deleted files or pre-migration daily-file sources)
|
|
106
116
|
for (const stale of this.store.indexedSources()) {
|
|
@@ -118,99 +128,38 @@ export class MemoryIndexer {
|
|
|
118
128
|
return totals;
|
|
119
129
|
}
|
|
120
130
|
async search(query, topK) {
|
|
121
|
-
// Task 2.2: overfetch to supply enough distinct sources after per-source dedup
|
|
122
131
|
const fetchLimit = topK * FETCH_MULTIPLIER;
|
|
123
|
-
let
|
|
132
|
+
let embedding;
|
|
124
133
|
if (this.store.vecAvailable) {
|
|
125
134
|
try {
|
|
126
|
-
const
|
|
127
|
-
|
|
135
|
+
const embeddings = await this.embedder.embed([query]);
|
|
136
|
+
embedding = embeddings[0];
|
|
128
137
|
}
|
|
129
138
|
catch (err) {
|
|
130
139
|
console.warn('[memory-indexer] Dense search embedding failed:', err);
|
|
131
140
|
}
|
|
132
141
|
}
|
|
133
|
-
const
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
const unionArray = [...unionIds];
|
|
142
|
-
const chunks = this.store.getChunksByIds(unionArray);
|
|
143
|
-
const chunkMap = new Map(chunks.map((c) => [c.id, c]));
|
|
144
|
-
// Task 2.3: recency channel — rank union by fileMtimeAt DESC; stable tiebreak on union insertion order
|
|
145
|
-
const unionIndexOf = new Map(unionArray.map((id, idx) => [id, idx]));
|
|
146
|
-
const recencyOrdered = [...unionArray].sort((a, b) => {
|
|
147
|
-
const mtimeA = chunkMap.get(a)?.fileMtimeAt ?? 0;
|
|
148
|
-
const mtimeB = chunkMap.get(b)?.fileMtimeAt ?? 0;
|
|
149
|
-
if (mtimeB !== mtimeA)
|
|
150
|
-
return mtimeB - mtimeA;
|
|
151
|
-
return (unionIndexOf.get(a) ?? 0) - (unionIndexOf.get(b) ?? 0);
|
|
152
|
-
});
|
|
153
|
-
// RRF score accumulation: dense + bm25 + weighted recency channels
|
|
154
|
-
const scoreMap = new Map();
|
|
155
|
-
denseResults.forEach((r, rank) => {
|
|
156
|
-
scoreMap.set(r.id, { dense: 1 / (RRF_K + rank + 1), bm25: 0, recency: 0 });
|
|
157
|
-
});
|
|
158
|
-
bm25Results.forEach((r, rank) => {
|
|
159
|
-
const entry = scoreMap.get(r.id) ?? { dense: 0, bm25: 0, recency: 0 };
|
|
160
|
-
entry.bm25 = 1 / (RRF_K + rank + 1);
|
|
161
|
-
scoreMap.set(r.id, entry);
|
|
162
|
-
});
|
|
163
|
-
recencyOrdered.forEach((id, rank) => {
|
|
164
|
-
const entry = scoreMap.get(id);
|
|
165
|
-
if (!entry)
|
|
166
|
-
return;
|
|
167
|
-
entry.recency = RECENCY_WEIGHT * (1 / (RRF_K + rank + 1));
|
|
142
|
+
const rows = this.store.searchHybrid({
|
|
143
|
+
query,
|
|
144
|
+
embedding,
|
|
145
|
+
topK,
|
|
146
|
+
fetchLimit,
|
|
147
|
+
denseScoreFloor: DENSE_SCORE_FLOOR,
|
|
148
|
+
recencyWeight: RECENCY_WEIGHT,
|
|
149
|
+
rrfK: RRF_K,
|
|
168
150
|
});
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
const bm25Active = bm25Results.length > 0;
|
|
172
|
-
const recencyActive = unionIds.size > 0;
|
|
173
|
-
const maxScore = ((denseActive ? 1 : 0) + (bm25Active ? 1 : 0) + (recencyActive ? RECENCY_WEIGHT : 0)) / (RRF_K + 1);
|
|
174
|
-
// Task 2.5: build ranked list, dropping dense-only candidates below the precision floor
|
|
175
|
-
const ranked = [...scoreMap.entries()]
|
|
176
|
-
.map(([id, scores]) => ({
|
|
177
|
-
id,
|
|
178
|
-
totalScore: scores.dense + scores.bm25 + scores.recency,
|
|
179
|
-
hasDense: scores.dense > 0,
|
|
180
|
-
hasBm25: scores.bm25 > 0,
|
|
181
|
-
isDenseOnly: denseIds.has(id) && !bm25Ids.has(id),
|
|
182
|
-
}))
|
|
183
|
-
.filter((candidate) => {
|
|
184
|
-
if (!candidate.isDenseOnly)
|
|
185
|
-
return true;
|
|
186
|
-
const retrieval = denseRetrievalScore.get(candidate.id);
|
|
187
|
-
// Only drop when a dense retrieval score exists and is below the floor
|
|
188
|
-
if (retrieval === undefined)
|
|
189
|
-
return true;
|
|
190
|
-
return retrieval >= DENSE_SCORE_FLOOR;
|
|
191
|
-
})
|
|
192
|
-
.sort((a, b) => b.totalScore - a.totalScore);
|
|
151
|
+
if (rows.length === 0)
|
|
152
|
+
return [];
|
|
193
153
|
const results = [];
|
|
194
|
-
const
|
|
195
|
-
for (const r of ranked) {
|
|
196
|
-
if (results.length >= topK)
|
|
197
|
-
break;
|
|
198
|
-
const chunk = chunkMap.get(r.id);
|
|
199
|
-
if (!chunk)
|
|
200
|
-
continue;
|
|
201
|
-
if (seenSources.has(chunk.source))
|
|
202
|
-
continue;
|
|
203
|
-
seenSources.add(chunk.source);
|
|
204
|
-
const normalizedScore = maxScore > 0 ? r.totalScore / maxScore : 0;
|
|
205
|
-
const retriever = r.hasDense && r.hasBm25 ? 'both' : r.hasDense ? 'dense' : 'bm25';
|
|
154
|
+
for (const row of rows) {
|
|
206
155
|
let contentSource = 'file';
|
|
207
156
|
try {
|
|
208
|
-
chunk.content = fs.
|
|
157
|
+
row.chunk.content = await fs.promises.readFile(path.join(this.config.wikiDir, row.chunk.source), 'utf8');
|
|
209
158
|
}
|
|
210
159
|
catch {
|
|
211
160
|
contentSource = 'fallback';
|
|
212
161
|
}
|
|
213
|
-
results.push({ chunk, score:
|
|
162
|
+
results.push({ chunk: row.chunk, score: row.score, retriever: row.retriever, contentSource });
|
|
214
163
|
}
|
|
215
164
|
return results;
|
|
216
165
|
}
|
|
@@ -219,13 +168,13 @@ export class MemoryIndexer {
|
|
|
219
168
|
const rawDir = path.join(this.config.wikiDir, 'raw');
|
|
220
169
|
const savePath = path.join(rawDir, `conv_save_${datePart}.md`);
|
|
221
170
|
const lockPath = `${savePath}.lock`;
|
|
222
|
-
fs.
|
|
171
|
+
await fs.promises.mkdir(rawDir, { recursive: true });
|
|
223
172
|
const acquired = await acquireLock(lockPath, LOCK_TIMEOUT_MS, LOCK_RETRY_MS);
|
|
224
173
|
if (!acquired) {
|
|
225
174
|
console.warn('[memory-indexer] Could not acquire write lock, writing anyway (fail-open)');
|
|
226
175
|
}
|
|
227
176
|
try {
|
|
228
|
-
fs.
|
|
177
|
+
await fs.promises.appendFile(savePath, `\n${content}\n`, 'utf8');
|
|
229
178
|
}
|
|
230
179
|
finally {
|
|
231
180
|
if (acquired)
|