@hanhnd/agent-kit 1.0.38 → 1.0.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js CHANGED
@@ -25930,8 +25930,8 @@ function getElementAtPath(obj, path15) {
25930
25930
  }
25931
25931
  function promiseAllObject(promisesObj) {
25932
25932
  const keys = Object.keys(promisesObj);
25933
- const promises = keys.map((key) => promisesObj[key]);
25934
- return Promise.all(promises).then((results) => {
25933
+ const promises3 = keys.map((key) => promisesObj[key]);
25934
+ return Promise.all(promises3).then((results) => {
25935
25935
  const resolvedObj = {};
25936
25936
  for (let i = 0; i < keys.length; i++) {
25937
25937
  resolvedObj[keys[i]] = results[i];
@@ -37016,14 +37016,14 @@ var MemoryIndexer = class {
37016
37016
  const existingChunkIds = this.store.hashesBySource(source);
37017
37017
  let text;
37018
37018
  try {
37019
- text = fs6.readFileSync(absolutePath, "utf8");
37019
+ text = await fs6.promises.readFile(absolutePath, "utf8");
37020
37020
  } catch (err) {
37021
37021
  console.warn(`[memory-indexer] Cannot read file ${absolutePath}:`, err);
37022
37022
  return { indexed: 0, deleted: 0, skipped: existingChunkIds.size };
37023
37023
  }
37024
37024
  let fileMtimeAt;
37025
37025
  try {
37026
- fileMtimeAt = fs6.statSync(absolutePath).mtimeMs;
37026
+ fileMtimeAt = (await fs6.promises.stat(absolutePath)).mtimeMs;
37027
37027
  } catch (err) {
37028
37028
  console.warn(`[memory-indexer] Cannot stat file ${absolutePath}:`, err);
37029
37029
  fileMtimeAt = Date.now();
@@ -37056,22 +37056,30 @@ var MemoryIndexer = class {
37056
37056
  }
37057
37057
  async indexDirectory(rootDir, opts = {}) {
37058
37058
  const totals = { indexed: 0, deleted: 0, skipped: 0 };
37059
- if (!fs6.existsSync(rootDir)) return totals;
37059
+ let rootExists;
37060
+ try {
37061
+ await fs6.promises.stat(rootDir);
37062
+ rootExists = true;
37063
+ } catch {
37064
+ rootExists = false;
37065
+ }
37066
+ if (!rootExists) return totals;
37060
37067
  const relativeBase = opts.relativeBase ?? this.config.wikiDir;
37061
37068
  const excludeFiles = new Set(opts.excludeFiles ?? []);
37062
37069
  const files = [];
37063
- const walk = (dirPath) => {
37070
+ const walk = async (dirPath) => {
37064
37071
  let entries;
37065
37072
  try {
37066
- entries = fs6.readdirSync(dirPath, { withFileTypes: true });
37073
+ entries = await fs6.promises.readdir(dirPath, { withFileTypes: true });
37067
37074
  } catch (err) {
37068
37075
  console.warn(`[memory-indexer] Cannot scan directory ${dirPath}:`, err);
37069
37076
  return;
37070
37077
  }
37078
+ entries.sort((a, b) => a.name.localeCompare(b.name));
37071
37079
  for (const entry of entries) {
37072
37080
  const entryPath = path8.join(dirPath, entry.name);
37073
37081
  if (entry.isDirectory()) {
37074
- walk(entryPath);
37082
+ await walk(entryPath);
37075
37083
  continue;
37076
37084
  }
37077
37085
  if (entry.isFile() && /\.md$/i.test(entry.name) && !excludeFiles.has(entry.name)) {
@@ -37079,7 +37087,7 @@ var MemoryIndexer = class {
37079
37087
  }
37080
37088
  }
37081
37089
  };
37082
- walk(rootDir);
37090
+ await walk(rootDir);
37083
37091
  const currentSources = new Set(files.map((file) => path8.relative(relativeBase, file)));
37084
37092
  for (const stale of this.store.indexedSources()) {
37085
37093
  if (!currentSources.has(stale)) {
@@ -37097,77 +37105,34 @@ var MemoryIndexer = class {
37097
37105
  }
37098
37106
  async search(query, topK) {
37099
37107
  const fetchLimit = topK * FETCH_MULTIPLIER;
37100
- let denseResults = [];
37108
+ let embedding;
37101
37109
  if (this.store.vecAvailable) {
37102
37110
  try {
37103
- const queryEmbedding = await this.embedder.embed([query]);
37104
- denseResults = this.store.searchDense(queryEmbedding[0], fetchLimit);
37111
+ const embeddings = await this.embedder.embed([query]);
37112
+ embedding = embeddings[0];
37105
37113
  } catch (err) {
37106
37114
  console.warn("[memory-indexer] Dense search embedding failed:", err);
37107
37115
  }
37108
37116
  }
37109
- const bm25Results = this.store.searchBm25(query, fetchLimit);
37110
- const denseIds = new Set(denseResults.map((r) => r.id));
37111
- const bm25Ids = new Set(bm25Results.map((r) => r.id));
37112
- const unionIds = /* @__PURE__ */ new Set([...denseIds, ...bm25Ids]);
37113
- const denseRetrievalScore = new Map(denseResults.map((r) => [r.id, r.score]));
37114
- const unionArray = [...unionIds];
37115
- const chunks = this.store.getChunksByIds(unionArray);
37116
- const chunkMap = new Map(chunks.map((c) => [c.id, c]));
37117
- const unionIndexOf = new Map(unionArray.map((id, idx) => [id, idx]));
37118
- const recencyOrdered = [...unionArray].sort((a, b) => {
37119
- const mtimeA = chunkMap.get(a)?.fileMtimeAt ?? 0;
37120
- const mtimeB = chunkMap.get(b)?.fileMtimeAt ?? 0;
37121
- if (mtimeB !== mtimeA) return mtimeB - mtimeA;
37122
- return (unionIndexOf.get(a) ?? 0) - (unionIndexOf.get(b) ?? 0);
37123
- });
37124
- const scoreMap = /* @__PURE__ */ new Map();
37125
- denseResults.forEach((r, rank) => {
37126
- scoreMap.set(r.id, { dense: 1 / (RRF_K + rank + 1), bm25: 0, recency: 0 });
37127
- });
37128
- bm25Results.forEach((r, rank) => {
37129
- const entry = scoreMap.get(r.id) ?? { dense: 0, bm25: 0, recency: 0 };
37130
- entry.bm25 = 1 / (RRF_K + rank + 1);
37131
- scoreMap.set(r.id, entry);
37117
+ const rows = this.store.searchHybrid({
37118
+ query,
37119
+ embedding,
37120
+ topK,
37121
+ fetchLimit,
37122
+ denseScoreFloor: DENSE_SCORE_FLOOR,
37123
+ recencyWeight: RECENCY_WEIGHT,
37124
+ rrfK: RRF_K
37132
37125
  });
37133
- recencyOrdered.forEach((id, rank) => {
37134
- const entry = scoreMap.get(id);
37135
- if (!entry) return;
37136
- entry.recency = RECENCY_WEIGHT * (1 / (RRF_K + rank + 1));
37137
- });
37138
- const denseActive = denseResults.length > 0;
37139
- const bm25Active = bm25Results.length > 0;
37140
- const recencyActive = unionIds.size > 0;
37141
- const maxScore = ((denseActive ? 1 : 0) + (bm25Active ? 1 : 0) + (recencyActive ? RECENCY_WEIGHT : 0)) / (RRF_K + 1);
37142
- const ranked = [...scoreMap.entries()].map(([id, scores]) => ({
37143
- id,
37144
- totalScore: scores.dense + scores.bm25 + scores.recency,
37145
- hasDense: scores.dense > 0,
37146
- hasBm25: scores.bm25 > 0,
37147
- isDenseOnly: denseIds.has(id) && !bm25Ids.has(id)
37148
- })).filter((candidate) => {
37149
- if (!candidate.isDenseOnly) return true;
37150
- const retrieval = denseRetrievalScore.get(candidate.id);
37151
- if (retrieval === void 0) return true;
37152
- return retrieval >= DENSE_SCORE_FLOOR;
37153
- }).sort((a, b) => b.totalScore - a.totalScore);
37126
+ if (rows.length === 0) return [];
37154
37127
  const results = [];
37155
- const seenSources = /* @__PURE__ */ new Set();
37156
- for (const r of ranked) {
37157
- if (results.length >= topK) break;
37158
- const chunk = chunkMap.get(r.id);
37159
- if (!chunk) continue;
37160
- if (seenSources.has(chunk.source)) continue;
37161
- seenSources.add(chunk.source);
37162
- const normalizedScore = maxScore > 0 ? r.totalScore / maxScore : 0;
37163
- const retriever = r.hasDense && r.hasBm25 ? "both" : r.hasDense ? "dense" : "bm25";
37128
+ for (const row of rows) {
37164
37129
  let contentSource = "file";
37165
37130
  try {
37166
- chunk.content = fs6.readFileSync(path8.join(this.config.wikiDir, chunk.source), "utf8");
37131
+ row.chunk.content = await fs6.promises.readFile(path8.join(this.config.wikiDir, row.chunk.source), "utf8");
37167
37132
  } catch {
37168
37133
  contentSource = "fallback";
37169
37134
  }
37170
- results.push({ chunk, score: normalizedScore, retriever, contentSource });
37135
+ results.push({ chunk: row.chunk, score: row.score, retriever: row.retriever, contentSource });
37171
37136
  }
37172
37137
  return results;
37173
37138
  }
@@ -37176,13 +37141,13 @@ var MemoryIndexer = class {
37176
37141
  const rawDir = path8.join(this.config.wikiDir, "raw");
37177
37142
  const savePath = path8.join(rawDir, `conv_save_${datePart}.md`);
37178
37143
  const lockPath = `${savePath}.lock`;
37179
- fs6.mkdirSync(rawDir, { recursive: true });
37144
+ await fs6.promises.mkdir(rawDir, { recursive: true });
37180
37145
  const acquired = await acquireLock(lockPath, LOCK_TIMEOUT_MS, LOCK_RETRY_MS);
37181
37146
  if (!acquired) {
37182
37147
  console.warn("[memory-indexer] Could not acquire write lock, writing anyway (fail-open)");
37183
37148
  }
37184
37149
  try {
37185
- fs6.appendFileSync(savePath, `
37150
+ await fs6.promises.appendFile(savePath, `
37186
37151
  ${content}
37187
37152
  `, "utf8");
37188
37153
  } finally {
@@ -37389,10 +37354,6 @@ var MemoryStore = class {
37389
37354
  }
37390
37355
  return Buffer.from(embedding.buffer, embedding.byteOffset, embedding.byteLength);
37391
37356
  }
37392
- normalizeVectorDistance(distance) {
37393
- if (distance <= 0) return 1;
37394
- return Math.max(0, Math.min(1, 1 - distance));
37395
- }
37396
37357
  hashesBySource(source) {
37397
37358
  const rows = this.db.prepare(`SELECT id FROM memory_chunks WHERE source = ?`).all(source);
37398
37359
  return new Set(rows.map((r) => r.id));
@@ -37429,61 +37390,12 @@ var MemoryStore = class {
37429
37390
  });
37430
37391
  deleteAll();
37431
37392
  }
37432
- searchDense(embedding, limit) {
37433
- try {
37434
- const queryVector = this.toVectorBinding(embedding);
37435
- const rows = this.db.prepare(
37436
- `SELECT mc.id, vector_distance_cos(mc.embedding, vector32(?)) AS distance
37437
- FROM vector_top_k('idx_memory_chunks_embedding', vector32(?), ?) AS vector_matches
37438
- JOIN memory_chunks mc ON mc.rowid = vector_matches.id
37439
- ORDER BY distance
37440
- LIMIT ?`
37441
- ).all(queryVector, queryVector, limit, limit);
37442
- return rows.map((r) => ({
37443
- id: r.id,
37444
- score: this.normalizeVectorDistance(r.distance)
37445
- }));
37446
- } catch (err) {
37447
- console.warn("[memory-store] Dense search failed:", err);
37448
- return [];
37449
- }
37450
- }
37451
- searchBm25(query, limit) {
37452
- if (!query.trim()) return [];
37393
+ buildFtsQuery(query) {
37453
37394
  const tokens = query.trim().split(/\s+/).map((t) => t.replace(/[^a-zA-Z0-9_]/g, "").toLowerCase()).filter((t) => t.length > 1);
37454
- const ftsQuery = (0, import_stopword.removeStopwords)(tokens, import_stopword.eng).map((t) => `${t.replace(/s$/, "")}*`).join(" OR ");
37455
- if (!ftsQuery) return [];
37456
- try {
37457
- const rows = this.db.prepare(
37458
- `SELECT mc.id, bm25(memory_fts, 0.2, 2.0, 5.0) AS rank
37459
- FROM memory_fts
37460
- JOIN memory_chunks mc ON mc.rowid = memory_fts.rowid
37461
- WHERE memory_fts MATCH ?
37462
- ORDER BY rank
37463
- LIMIT ?`
37464
- ).all(ftsQuery, limit);
37465
- if (rows.length === 0) return [];
37466
- const ranks = rows.map((r) => r.rank);
37467
- const minRank = Math.min(...ranks);
37468
- const maxRank = Math.max(...ranks);
37469
- const range = maxRank - minRank;
37470
- return rows.map((r) => ({
37471
- id: r.id,
37472
- score: range > 0 ? (maxRank - r.rank) / range : 1
37473
- }));
37474
- } catch (err) {
37475
- console.warn("[memory-store] BM25 search failed:", err);
37476
- return [];
37477
- }
37395
+ return (0, import_stopword.removeStopwords)(tokens, import_stopword.eng).map((t) => `${t.replace(/s$/, "")}*`).join(" OR ");
37478
37396
  }
37479
- getChunksByIds(ids) {
37480
- if (ids.length === 0) return [];
37481
- const placeholders = ids.map(() => "?").join(", ");
37482
- const rows = this.db.prepare(
37483
- `SELECT id, source, source_type, heading, heading_level, content, line_start, line_end, file_mtime_at
37484
- FROM memory_chunks WHERE id IN (${placeholders})`
37485
- ).all(...ids);
37486
- return rows.map((r) => ({
37397
+ mapMemoryChunkRow(r) {
37398
+ return {
37487
37399
  id: r.id,
37488
37400
  source: r.source,
37489
37401
  sourceType: r.source_type,
@@ -37493,7 +37405,201 @@ var MemoryStore = class {
37493
37405
  lineStart: r.line_start,
37494
37406
  lineEnd: r.line_end,
37495
37407
  fileMtimeAt: r.file_mtime_at
37496
- }));
37408
+ };
37409
+ }
37410
+ searchHybrid(options) {
37411
+ const {
37412
+ query,
37413
+ embedding,
37414
+ topK,
37415
+ fetchLimit,
37416
+ denseScoreFloor,
37417
+ recencyWeight,
37418
+ rrfK,
37419
+ sourceType,
37420
+ sinceMtimeAt,
37421
+ includeDebug
37422
+ } = options;
37423
+ const ftsQuery = this.buildFtsQuery(query);
37424
+ const hasDense = embedding !== void 0;
37425
+ const hasBm25 = ftsQuery.length > 0;
37426
+ if (!hasDense && !hasBm25) return [];
37427
+ const hasFilters = sourceType !== void 0 || sinceMtimeAt !== void 0;
37428
+ try {
37429
+ const params = [];
37430
+ let denseCte = "";
37431
+ let denseRankedCte = "";
37432
+ if (hasDense) {
37433
+ const queryBuf = this.toVectorBinding(embedding);
37434
+ if (hasFilters) {
37435
+ const filterClauses = [];
37436
+ if (sourceType) {
37437
+ filterClauses.push("mc.source_type = ?");
37438
+ params.push(sourceType);
37439
+ }
37440
+ if (sinceMtimeAt !== void 0) {
37441
+ filterClauses.push("mc.file_mtime_at >= ?");
37442
+ params.push(sinceMtimeAt);
37443
+ }
37444
+ const whereClause = filterClauses.length > 0 ? `WHERE ${filterClauses.join(" AND ")}` : "";
37445
+ params.push(queryBuf);
37446
+ params.push(fetchLimit);
37447
+ denseCte = `dense_data AS (
37448
+ SELECT mc.id, mc.source, mc.source_type, mc.file_mtime_at,
37449
+ mc.heading, mc.heading_level, mc.content, mc.line_start, mc.line_end,
37450
+ vector_distance_cos(mc.embedding, vector32(?)) AS dense_dist
37451
+ FROM memory_chunks mc
37452
+ ${whereClause}
37453
+ ORDER BY dense_dist
37454
+ LIMIT ?
37455
+ )`;
37456
+ } else {
37457
+ params.push(queryBuf);
37458
+ params.push(queryBuf);
37459
+ params.push(fetchLimit);
37460
+ denseCte = `dense_data AS (
37461
+ SELECT mc.id, mc.source, mc.source_type, mc.file_mtime_at,
37462
+ mc.heading, mc.heading_level, mc.content, mc.line_start, mc.line_end,
37463
+ vector_distance_cos(mc.embedding, vector32(?)) AS dense_dist
37464
+ FROM vector_top_k('idx_memory_chunks_embedding', vector32(?), ?) AS vtk
37465
+ JOIN memory_chunks mc ON mc.rowid = vtk.id
37466
+ )`;
37467
+ }
37468
+ denseRankedCte = `dense_ranked AS (
37469
+ SELECT *, ROW_NUMBER() OVER (ORDER BY dense_dist, id) AS dense_rank
37470
+ FROM dense_data
37471
+ )`;
37472
+ }
37473
+ let bm25Cte = "";
37474
+ let bm25RankedCte = "";
37475
+ if (hasBm25) {
37476
+ const bm25FilterClauses = [];
37477
+ if (sourceType) bm25FilterClauses.push("mc.source_type = ?");
37478
+ if (sinceMtimeAt !== void 0) bm25FilterClauses.push("mc.file_mtime_at >= ?");
37479
+ const bm25WhereExtra = bm25FilterClauses.length > 0 ? ` AND ${bm25FilterClauses.join(" AND ")}` : "";
37480
+ params.push(ftsQuery);
37481
+ if (sourceType) params.push(sourceType);
37482
+ if (sinceMtimeAt !== void 0) params.push(sinceMtimeAt);
37483
+ params.push(fetchLimit);
37484
+ bm25Cte = `bm25_data AS (
37485
+ SELECT mc.id, mc.source, mc.source_type, mc.file_mtime_at,
37486
+ mc.heading, mc.heading_level, mc.content, mc.line_start, mc.line_end,
37487
+ bm25(memory_fts, 0.2, 2.0, 5.0) AS bm25_neg
37488
+ FROM memory_fts
37489
+ JOIN memory_chunks mc ON mc.rowid = memory_fts.rowid
37490
+ WHERE memory_fts MATCH ?${bm25WhereExtra}
37491
+ LIMIT ?
37492
+ )`;
37493
+ bm25RankedCte = `bm25_ranked AS (
37494
+ SELECT *, ROW_NUMBER() OVER (ORDER BY bm25_neg, id) AS bm25_rank
37495
+ FROM bm25_data
37496
+ )`;
37497
+ }
37498
+ let candidatesCte = "";
37499
+ if (hasDense && hasBm25) {
37500
+ candidatesCte = `all_candidates AS (
37501
+ SELECT d.id, d.source, d.source_type, d.file_mtime_at,
37502
+ d.heading, d.heading_level, d.content, d.line_start, d.line_end,
37503
+ CAST(d.dense_rank AS INTEGER) AS dense_rank, d.dense_dist,
37504
+ b.bm25_rank,
37505
+ CASE WHEN b.id IS NOT NULL THEN 'both' ELSE 'dense' END AS retriever
37506
+ FROM dense_ranked d
37507
+ LEFT JOIN bm25_ranked b ON b.id = d.id
37508
+ UNION ALL
37509
+ SELECT b.id, b.source, b.source_type, b.file_mtime_at,
37510
+ b.heading, b.heading_level, b.content, b.line_start, b.line_end,
37511
+ NULL AS dense_rank, NULL AS dense_dist,
37512
+ CAST(b.bm25_rank AS INTEGER) AS bm25_rank,
37513
+ 'bm25' AS retriever
37514
+ FROM bm25_ranked b
37515
+ LEFT JOIN dense_ranked d ON d.id = b.id
37516
+ WHERE d.id IS NULL
37517
+ )`;
37518
+ } else if (hasDense) {
37519
+ candidatesCte = `all_candidates AS (
37520
+ SELECT id, source, source_type, file_mtime_at,
37521
+ heading, heading_level, content, line_start, line_end,
37522
+ CAST(dense_rank AS INTEGER) AS dense_rank, dense_dist,
37523
+ NULL AS bm25_rank, 'dense' AS retriever
37524
+ FROM dense_ranked
37525
+ )`;
37526
+ } else {
37527
+ candidatesCte = `all_candidates AS (
37528
+ SELECT id, source, source_type, file_mtime_at,
37529
+ heading, heading_level, content, line_start, line_end,
37530
+ NULL AS dense_rank, NULL AS dense_dist,
37531
+ CAST(bm25_rank AS INTEGER) AS bm25_rank, 'bm25' AS retriever
37532
+ FROM bm25_ranked
37533
+ )`;
37534
+ }
37535
+ params.push(rrfK, rrfK, recencyWeight, rrfK, denseScoreFloor, topK);
37536
+ const sql = `
37537
+ WITH
37538
+ ${[denseCte, denseRankedCte, bm25Cte, bm25RankedCte, candidatesCte].filter(Boolean).join(",\n")}
37539
+ ,
37540
+ recency_ranked AS (
37541
+ SELECT id, ROW_NUMBER() OVER (ORDER BY file_mtime_at DESC, source, id) AS recency_rank
37542
+ FROM all_candidates
37543
+ ),
37544
+ scored AS (
37545
+ SELECT c.id, c.source, c.source_type, c.file_mtime_at,
37546
+ c.heading, c.heading_level, c.content, c.line_start, c.line_end,
37547
+ c.dense_rank, c.dense_dist, c.bm25_rank, r.recency_rank,
37548
+ c.retriever,
37549
+ COALESCE(1.0 / (? + c.dense_rank), 0.0) +
37550
+ COALESCE(1.0 / (? + c.bm25_rank), 0.0) +
37551
+ ? * (1.0 / (? + r.recency_rank)) AS total_score
37552
+ FROM all_candidates c
37553
+ JOIN recency_ranked r ON r.id = c.id
37554
+ WHERE c.retriever != 'dense' OR (1.0 - c.dense_dist) >= ?
37555
+ ),
37556
+ best_per_source AS (
37557
+ SELECT *,
37558
+ ROW_NUMBER() OVER (PARTITION BY source ORDER BY total_score DESC, source, id) AS source_rank
37559
+ FROM scored
37560
+ )
37561
+ SELECT id, source, source_type, file_mtime_at, heading, heading_level, content, line_start, line_end,
37562
+ dense_rank, dense_dist, bm25_rank, recency_rank, total_score, retriever
37563
+ FROM best_per_source
37564
+ WHERE source_rank = 1
37565
+ ORDER BY total_score DESC, source, id
37566
+ LIMIT ?
37567
+ `;
37568
+ const rows = this.db.prepare(sql).all(...params);
37569
+ if (rows.length === 0) return [];
37570
+ const denseActive = hasDense;
37571
+ const bm25Active = hasBm25;
37572
+ const recencyActive = rows.length > 0;
37573
+ const maxScore = ((denseActive ? 1 : 0) + (bm25Active ? 1 : 0) + (recencyActive ? recencyWeight : 0)) / (rrfK + 1);
37574
+ return rows.map((r) => {
37575
+ const chunk = this.mapMemoryChunkRow(r);
37576
+ const normalizedScore = maxScore > 0 ? r.total_score / maxScore : 0;
37577
+ const row = {
37578
+ chunk,
37579
+ score: normalizedScore,
37580
+ retriever: r.retriever
37581
+ };
37582
+ if (includeDebug) {
37583
+ const denseScore = r.dense_dist !== null ? Math.max(0, Math.min(1, 1 - r.dense_dist)) : void 0;
37584
+ const bm25Score = r.bm25_rank !== null ? 1 / (rrfK + r.bm25_rank) : void 0;
37585
+ const recencyScore = recencyWeight * (1 / (rrfK + r.recency_rank));
37586
+ const debug = {
37587
+ denseRank: r.dense_rank ?? void 0,
37588
+ denseScore,
37589
+ bm25Rank: r.bm25_rank ?? void 0,
37590
+ bm25Score,
37591
+ recencyRank: r.recency_rank,
37592
+ recencyScore,
37593
+ totalScore: r.total_score
37594
+ };
37595
+ row.debug = debug;
37596
+ }
37597
+ return row;
37598
+ });
37599
+ } catch (err) {
37600
+ console.warn("[memory-store] Hybrid search failed:", err);
37601
+ return [];
37602
+ }
37497
37603
  }
37498
37604
  get vecAvailable() {
37499
37605
  return true;
@@ -37978,7 +38084,7 @@ ${content}`;
37978
38084
  const blocks = [];
37979
38085
  for (const row of rows) {
37980
38086
  try {
37981
- const content = fs11.readFileSync(path14.join(config2.wikiDir, row.source), "utf8");
38087
+ const content = await fs11.promises.readFile(path14.join(config2.wikiDir, row.source), "utf8");
37982
38088
  const displaySource = row.source.replace(/^compiled\//, "");
37983
38089
  blocks.push(`### ${displaySource}
37984
38090
  ${content}`);
@@ -72,10 +72,8 @@ describe('digestConversationFile', () => {
72
72
  });
73
73
  assert.equal(result.indexed, true);
74
74
  assert.equal(result.markdown, markdownPath);
75
- const denseResults = store.searchDense(new Float32Array(384).fill(0.05), 5);
76
75
  const indexedSource = path.relative(config.wikiDir, markdownPath);
77
- const indexedChunks = store.getChunksByIds(denseResults.map((denseResult) => denseResult.id));
78
- assert.ok(indexedChunks.some((chunk) => chunk.source === indexedSource), `Expected dense search to return a chunk for ${indexedSource}`);
76
+ assert.ok(store.hashesBySource(indexedSource).size > 0, `Expected source ${indexedSource} to be indexed in store`);
79
77
  }
80
78
  finally {
81
79
  store.close();
@@ -32,7 +32,7 @@ export class MemoryIndexer {
32
32
  const existingChunkIds = this.store.hashesBySource(source);
33
33
  let text;
34
34
  try {
35
- text = fs.readFileSync(absolutePath, 'utf8');
35
+ text = await fs.promises.readFile(absolutePath, 'utf8');
36
36
  }
37
37
  catch (err) {
38
38
  console.warn(`[memory-indexer] Cannot read file ${absolutePath}:`, err);
@@ -40,7 +40,7 @@ export class MemoryIndexer {
40
40
  }
41
41
  let fileMtimeAt;
42
42
  try {
43
- fileMtimeAt = fs.statSync(absolutePath).mtimeMs;
43
+ fileMtimeAt = (await fs.promises.stat(absolutePath)).mtimeMs;
44
44
  }
45
45
  catch (err) {
46
46
  console.warn(`[memory-indexer] Cannot stat file ${absolutePath}:`, err);
@@ -75,24 +75,34 @@ export class MemoryIndexer {
75
75
  }
76
76
  async indexDirectory(rootDir, opts = {}) {
77
77
  const totals = { indexed: 0, deleted: 0, skipped: 0 };
78
- if (!fs.existsSync(rootDir))
78
+ let rootExists;
79
+ try {
80
+ await fs.promises.stat(rootDir);
81
+ rootExists = true;
82
+ }
83
+ catch {
84
+ rootExists = false;
85
+ }
86
+ if (!rootExists)
79
87
  return totals;
80
88
  const relativeBase = opts.relativeBase ?? this.config.wikiDir;
81
89
  const excludeFiles = new Set(opts.excludeFiles ?? []);
82
90
  const files = [];
83
- const walk = (dirPath) => {
91
+ const walk = async (dirPath) => {
84
92
  let entries;
85
93
  try {
86
- entries = fs.readdirSync(dirPath, { withFileTypes: true });
94
+ entries = await fs.promises.readdir(dirPath, { withFileTypes: true });
87
95
  }
88
96
  catch (err) {
89
97
  console.warn(`[memory-indexer] Cannot scan directory ${dirPath}:`, err);
90
98
  return;
91
99
  }
100
+ // Sort for deterministic traversal order
101
+ entries.sort((a, b) => a.name.localeCompare(b.name));
92
102
  for (const entry of entries) {
93
103
  const entryPath = path.join(dirPath, entry.name);
94
104
  if (entry.isDirectory()) {
95
- walk(entryPath);
105
+ await walk(entryPath);
96
106
  continue;
97
107
  }
98
108
  if (entry.isFile() && /\.md$/i.test(entry.name) && !excludeFiles.has(entry.name)) {
@@ -100,7 +110,7 @@ export class MemoryIndexer {
100
110
  }
101
111
  }
102
112
  };
103
- walk(rootDir);
113
+ await walk(rootDir);
104
114
  const currentSources = new Set(files.map((file) => path.relative(relativeBase, file)));
105
115
  // Remove stale sources (deleted files or pre-migration daily-file sources)
106
116
  for (const stale of this.store.indexedSources()) {
@@ -118,99 +128,38 @@ export class MemoryIndexer {
118
128
  return totals;
119
129
  }
120
130
  async search(query, topK) {
121
- // Task 2.2: overfetch to supply enough distinct sources after per-source dedup
122
131
  const fetchLimit = topK * FETCH_MULTIPLIER;
123
- let denseResults = [];
132
+ let embedding;
124
133
  if (this.store.vecAvailable) {
125
134
  try {
126
- const queryEmbedding = await this.embedder.embed([query]);
127
- denseResults = this.store.searchDense(queryEmbedding[0], fetchLimit);
135
+ const embeddings = await this.embedder.embed([query]);
136
+ embedding = embeddings[0];
128
137
  }
129
138
  catch (err) {
130
139
  console.warn('[memory-indexer] Dense search embedding failed:', err);
131
140
  }
132
141
  }
133
- const bm25Results = this.store.searchBm25(query, fetchLimit);
134
- // Task 2.1: candidate set = dense ∪ bm25 — no intersection filter
135
- const denseIds = new Set(denseResults.map((r) => r.id));
136
- const bm25Ids = new Set(bm25Results.map((r) => r.id));
137
- const unionIds = new Set([...denseIds, ...bm25Ids]);
138
- // Task 2.5: preserve raw dense retrieval scores [0,1] for the precision floor check
139
- const denseRetrievalScore = new Map(denseResults.map((r) => [r.id, r.score]));
140
- // Task 2.3: fetch all union candidates once — supplies recency mtime AND final content (no second DB call)
141
- const unionArray = [...unionIds];
142
- const chunks = this.store.getChunksByIds(unionArray);
143
- const chunkMap = new Map(chunks.map((c) => [c.id, c]));
144
- // Task 2.3: recency channel — rank union by fileMtimeAt DESC; stable tiebreak on union insertion order
145
- const unionIndexOf = new Map(unionArray.map((id, idx) => [id, idx]));
146
- const recencyOrdered = [...unionArray].sort((a, b) => {
147
- const mtimeA = chunkMap.get(a)?.fileMtimeAt ?? 0;
148
- const mtimeB = chunkMap.get(b)?.fileMtimeAt ?? 0;
149
- if (mtimeB !== mtimeA)
150
- return mtimeB - mtimeA;
151
- return (unionIndexOf.get(a) ?? 0) - (unionIndexOf.get(b) ?? 0);
152
- });
153
- // RRF score accumulation: dense + bm25 + weighted recency channels
154
- const scoreMap = new Map();
155
- denseResults.forEach((r, rank) => {
156
- scoreMap.set(r.id, { dense: 1 / (RRF_K + rank + 1), bm25: 0, recency: 0 });
157
- });
158
- bm25Results.forEach((r, rank) => {
159
- const entry = scoreMap.get(r.id) ?? { dense: 0, bm25: 0, recency: 0 };
160
- entry.bm25 = 1 / (RRF_K + rank + 1);
161
- scoreMap.set(r.id, entry);
162
- });
163
- recencyOrdered.forEach((id, rank) => {
164
- const entry = scoreMap.get(id);
165
- if (!entry)
166
- return;
167
- entry.recency = RECENCY_WEIGHT * (1 / (RRF_K + rank + 1));
142
+ const rows = this.store.searchHybrid({
143
+ query,
144
+ embedding,
145
+ topK,
146
+ fetchLimit,
147
+ denseScoreFloor: DENSE_SCORE_FLOOR,
148
+ recencyWeight: RECENCY_WEIGHT,
149
+ rrfK: RRF_K,
168
150
  });
169
- // Task 2.4: active-channel normalization — maxScore sums weights of channels that produced ≥1 result
170
- const denseActive = denseResults.length > 0;
171
- const bm25Active = bm25Results.length > 0;
172
- const recencyActive = unionIds.size > 0;
173
- const maxScore = ((denseActive ? 1 : 0) + (bm25Active ? 1 : 0) + (recencyActive ? RECENCY_WEIGHT : 0)) / (RRF_K + 1);
174
- // Task 2.5: build ranked list, dropping dense-only candidates below the precision floor
175
- const ranked = [...scoreMap.entries()]
176
- .map(([id, scores]) => ({
177
- id,
178
- totalScore: scores.dense + scores.bm25 + scores.recency,
179
- hasDense: scores.dense > 0,
180
- hasBm25: scores.bm25 > 0,
181
- isDenseOnly: denseIds.has(id) && !bm25Ids.has(id),
182
- }))
183
- .filter((candidate) => {
184
- if (!candidate.isDenseOnly)
185
- return true;
186
- const retrieval = denseRetrievalScore.get(candidate.id);
187
- // Only drop when a dense retrieval score exists and is below the floor
188
- if (retrieval === undefined)
189
- return true;
190
- return retrieval >= DENSE_SCORE_FLOOR;
191
- })
192
- .sort((a, b) => b.totalScore - a.totalScore);
151
+ if (rows.length === 0)
152
+ return [];
193
153
  const results = [];
194
- const seenSources = new Set();
195
- for (const r of ranked) {
196
- if (results.length >= topK)
197
- break;
198
- const chunk = chunkMap.get(r.id);
199
- if (!chunk)
200
- continue;
201
- if (seenSources.has(chunk.source))
202
- continue;
203
- seenSources.add(chunk.source);
204
- const normalizedScore = maxScore > 0 ? r.totalScore / maxScore : 0;
205
- const retriever = r.hasDense && r.hasBm25 ? 'both' : r.hasDense ? 'dense' : 'bm25';
154
+ for (const row of rows) {
206
155
  let contentSource = 'file';
207
156
  try {
208
- chunk.content = fs.readFileSync(path.join(this.config.wikiDir, chunk.source), 'utf8');
157
+ row.chunk.content = await fs.promises.readFile(path.join(this.config.wikiDir, row.chunk.source), 'utf8');
209
158
  }
210
159
  catch {
211
160
  contentSource = 'fallback';
212
161
  }
213
- results.push({ chunk, score: normalizedScore, retriever, contentSource });
162
+ results.push({ chunk: row.chunk, score: row.score, retriever: row.retriever, contentSource });
214
163
  }
215
164
  return results;
216
165
  }
@@ -219,13 +168,13 @@ export class MemoryIndexer {
219
168
  const rawDir = path.join(this.config.wikiDir, 'raw');
220
169
  const savePath = path.join(rawDir, `conv_save_${datePart}.md`);
221
170
  const lockPath = `${savePath}.lock`;
222
- fs.mkdirSync(rawDir, { recursive: true });
171
+ await fs.promises.mkdir(rawDir, { recursive: true });
223
172
  const acquired = await acquireLock(lockPath, LOCK_TIMEOUT_MS, LOCK_RETRY_MS);
224
173
  if (!acquired) {
225
174
  console.warn('[memory-indexer] Could not acquire write lock, writing anyway (fail-open)');
226
175
  }
227
176
  try {
228
- fs.appendFileSync(savePath, `\n${content}\n`, 'utf8');
177
+ await fs.promises.appendFile(savePath, `\n${content}\n`, 'utf8');
229
178
  }
230
179
  finally {
231
180
  if (acquired)