akm-cli 0.9.0-beta.2 → 0.9.0-beta.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/CHANGELOG.md +614 -0
  2. package/dist/assets/prompts/consolidate-system.md +23 -0
  3. package/dist/assets/prompts/contradiction-judge.md +33 -0
  4. package/dist/assets/prompts/distill-knowledge-system.md +22 -0
  5. package/dist/assets/prompts/distill-lesson-system.md +36 -0
  6. package/dist/assets/prompts/extract-session.md +5 -1
  7. package/dist/assets/prompts/graph-extract-system.md +1 -0
  8. package/dist/assets/prompts/memory-infer-system.md +1 -0
  9. package/dist/assets/prompts/memory-infer-user.md +5 -0
  10. package/dist/assets/prompts/metadata-enhance-system.md +1 -0
  11. package/dist/assets/prompts/procedural-system.md +44 -0
  12. package/dist/assets/prompts/recombine-system.md +40 -0
  13. package/dist/assets/prompts/staleness-detect-system.md +6 -0
  14. package/dist/assets/prompts/validate-summary-judge.md +1 -0
  15. package/dist/assets/templates/html/default.html +78 -0
  16. package/dist/assets/templates/html/health.html +730 -0
  17. package/dist/assets/templates/html/vendor/echarts.min.js +45 -0
  18. package/dist/cli/shared.js +21 -5
  19. package/dist/cli.js +47 -5
  20. package/dist/commands/agent/contribute-cli.js +16 -3
  21. package/dist/commands/feedback-cli.js +15 -6
  22. package/dist/commands/graph/graph.js +75 -71
  23. package/dist/commands/health/checks.js +48 -0
  24. package/dist/commands/health/html-report.js +790 -0
  25. package/dist/commands/health.js +478 -15
  26. package/dist/commands/improve/calibration.js +161 -0
  27. package/dist/commands/improve/consolidate.js +634 -111
  28. package/dist/commands/improve/dedup.js +482 -0
  29. package/dist/commands/improve/distill.js +145 -69
  30. package/dist/commands/improve/encoding-salience.js +205 -0
  31. package/dist/commands/improve/extract-cli.js +115 -1
  32. package/dist/commands/improve/extract-prompt.js +33 -2
  33. package/dist/commands/improve/extract-watch.js +140 -0
  34. package/dist/commands/improve/extract.js +280 -35
  35. package/dist/commands/improve/feedback-valence.js +54 -0
  36. package/dist/commands/improve/homeostatic.js +467 -0
  37. package/dist/commands/improve/improve-auto-accept.js +139 -6
  38. package/dist/commands/improve/improve-profiles.js +12 -0
  39. package/dist/commands/improve/improve.js +1851 -515
  40. package/dist/commands/improve/memory/memory-contradiction-detect.js +23 -28
  41. package/dist/commands/improve/outcome-loop.js +256 -0
  42. package/dist/commands/improve/proactive-maintenance.js +87 -0
  43. package/dist/commands/improve/procedural.js +409 -0
  44. package/dist/commands/improve/recombine.js +488 -0
  45. package/dist/commands/improve/reflect-noise.js +0 -0
  46. package/dist/commands/improve/reflect.js +51 -1
  47. package/dist/commands/improve/related-sessions.js +120 -0
  48. package/dist/commands/improve/salience.js +386 -0
  49. package/dist/commands/improve/triage.js +95 -0
  50. package/dist/commands/lint/agent-linter.js +19 -24
  51. package/dist/commands/lint/base-linter.js +173 -60
  52. package/dist/commands/lint/command-linter.js +19 -24
  53. package/dist/commands/lint/env-key-rules.js +34 -1
  54. package/dist/commands/lint/index.js +30 -13
  55. package/dist/commands/lint/memory-linter.js +1 -1
  56. package/dist/commands/lint/registry.js +5 -2
  57. package/dist/commands/lint/task-linter.js +3 -3
  58. package/dist/commands/lint/workflow-linter.js +26 -1
  59. package/dist/commands/proposal/drain.js +73 -6
  60. package/dist/commands/proposal/proposal-cli.js +22 -10
  61. package/dist/commands/proposal/proposal.js +17 -1
  62. package/dist/commands/proposal/validators/proposals.js +369 -329
  63. package/dist/commands/read/curate.js +294 -79
  64. package/dist/commands/read/search-cli.js +7 -0
  65. package/dist/commands/read/search.js +1 -0
  66. package/dist/commands/remember.js +6 -2
  67. package/dist/commands/sources/installed-stashes.js +5 -1
  68. package/dist/commands/sources/stash-cli.js +10 -2
  69. package/dist/core/asset/frontmatter.js +166 -167
  70. package/dist/core/asset/markdown.js +8 -0
  71. package/dist/core/config/config-schema.js +241 -0
  72. package/dist/core/config/config.js +2 -2
  73. package/dist/core/logs-db.js +305 -0
  74. package/dist/core/paths.js +3 -0
  75. package/dist/core/state-db.js +706 -42
  76. package/dist/indexer/db/db.js +347 -38
  77. package/dist/indexer/db/graph-db.js +81 -86
  78. package/dist/indexer/ensure-index.js +152 -17
  79. package/dist/indexer/graph/graph-boost.js +51 -41
  80. package/dist/indexer/index-writer-lock.js +99 -0
  81. package/dist/indexer/indexer.js +114 -111
  82. package/dist/indexer/passes/memory-inference.js +71 -25
  83. package/dist/indexer/passes/staleness-detect.js +2 -5
  84. package/dist/indexer/search/db-search.js +15 -4
  85. package/dist/indexer/search/ranking.js +4 -0
  86. package/dist/integrations/harnesses/claude/session-log.js +27 -5
  87. package/dist/integrations/harnesses/opencode/session-log.js +9 -0
  88. package/dist/integrations/session-logs/index.js +16 -0
  89. package/dist/llm/client.js +38 -4
  90. package/dist/llm/embedder.js +27 -3
  91. package/dist/llm/embedders/local.js +66 -2
  92. package/dist/llm/graph-extract.js +2 -1
  93. package/dist/llm/memory-infer.js +4 -8
  94. package/dist/llm/metadata-enhance.js +9 -1
  95. package/dist/llm/usage-persist.js +77 -0
  96. package/dist/llm/usage-telemetry.js +103 -0
  97. package/dist/output/context.js +3 -2
  98. package/dist/output/html-render.js +73 -0
  99. package/dist/output/shapes/curate.js +14 -2
  100. package/dist/output/shapes/helpers.js +17 -1
  101. package/dist/output/text/helpers.js +78 -1
  102. package/dist/runtime.js +25 -1
  103. package/dist/scripts/migrate-storage.js +1194 -607
  104. package/dist/scripts/migrations/import-fs-improve-runs-to-db.js +455 -270
  105. package/dist/sources/providers/tar-utils.js +16 -8
  106. package/dist/storage/sqlite-pragmas.js +146 -0
  107. package/dist/tasks/runner.js +99 -16
  108. package/dist/workflows/db.js +5 -2
  109. package/dist/workflows/validate-summary.js +2 -7
  110. package/docs/data-and-telemetry.md +1 -0
  111. package/package.json +7 -5
@@ -4,7 +4,6 @@
4
4
  import fs from "node:fs";
5
5
  import { rethrowIfTestIsolationError } from "../../core/errors.js";
6
6
  import { getDbPath } from "../../core/paths.js";
7
- import { warn } from "../../core/warn.js";
8
7
  import { closeDatabase, openExistingDatabase } from "./db.js";
9
8
  function withReadableGraphDb(db, fn) {
10
9
  if (db)
@@ -26,38 +25,16 @@ function uniqueSorted(values) {
26
25
  function normalizeEntity(value) {
27
26
  return value.trim().toLowerCase();
28
27
  }
29
- /**
30
- * Resolve a file_path within a stash to its entries.id. Returns null when the
31
- * path has no indexed entry (orphan graph row).
32
- */
33
- export function resolveEntryIdForPath(db, stashRoot, filePath) {
34
- try {
35
- const row = db
36
- .prepare("SELECT id FROM entries WHERE stash_dir = ? AND file_path = ? LIMIT 1")
37
- .get(stashRoot, filePath);
38
- if (row)
39
- return row.id;
40
- // Fall back to file_path-only match (legacy callers may pass a stash root
41
- // that doesn't exactly match entries.stash_dir, e.g. trailing-slash diffs).
42
- const fallback = db.prepare("SELECT id FROM entries WHERE file_path = ? LIMIT 1").get(filePath);
43
- return fallback?.id ?? null;
44
- }
45
- catch {
46
- return null;
47
- }
48
- }
49
28
  /**
50
29
  * Persist (or update) a graph snapshot for a stash root.
51
30
  *
52
- * Implementation: incremental upsert keyed on entries.id. Unchanged files
53
- * (matching body_hash) are skipped; changed files have their child rows
54
- * deleted (CASCADE) and re-inserted; files in DB but absent from the new
55
- * snapshot are deleted. The old behaviour wiped every row for the stash on
56
- * each write, which produced ~22k row writes per re-index even when one
57
- * asset changed.
58
- *
59
- * Orphan files (no entries row resolvable) are skipped and counted in a
60
- * single warn() so the caller sees the magnitude without log spam.
31
+ * #624-P1: keyed on (stash_root, file_path, body_hash) — NOT entries.id. Graph
32
+ * rows are self-keyed by path, so they survive an entries delete + reinsert
33
+ * (a reindex) when body_hash is unchanged. Unchanged files (matching body_hash)
34
+ * only have their file-meta refreshed; files whose body_hash changed have their
35
+ * old row + child rows deleted and the new content inserted; files in DB but
36
+ * absent from the new snapshot are deleted. There is no entry_id resolution and
37
+ * no orphan-skip — a graph file no longer needs a matching entries row.
61
38
  */
62
39
  export function replaceStoredGraph(db, graph) {
63
40
  const upsertMeta = db.prepare(`INSERT INTO graph_meta (
@@ -98,87 +75,98 @@ export function replaceStoredGraph(db, graph) {
98
75
  cache_misses = excluded.cache_misses,
99
76
  truncation_count = excluded.truncation_count,
100
77
  failure_count = excluded.failure_count`);
101
- const selectExisting = db.prepare("SELECT entry_id, file_path, body_hash FROM graph_files WHERE stash_root = ?");
102
- const deleteFile = db.prepare("DELETE FROM graph_files WHERE entry_id = ?");
103
- const deleteEntities = db.prepare("DELETE FROM graph_file_entities WHERE entry_id = ?");
104
- const deleteRelations = db.prepare("DELETE FROM graph_file_relations WHERE entry_id = ?");
78
+ const selectExisting = db.prepare("SELECT file_path, body_hash, file_order FROM graph_files WHERE stash_root = ?");
79
+ const deleteFile = db.prepare("DELETE FROM graph_files WHERE stash_root = ? AND file_path = ? AND body_hash = ?");
80
+ const deleteEntities = db.prepare("DELETE FROM graph_file_entities WHERE stash_root = ? AND file_path = ? AND body_hash = ?");
81
+ const deleteRelations = db.prepare("DELETE FROM graph_file_relations WHERE stash_root = ? AND file_path = ? AND body_hash = ?");
105
82
  const insertFile = db.prepare(`INSERT INTO graph_files (
106
- entry_id, stash_root, file_path, file_order, file_type, body_hash, confidence, status, reason, extraction_run_id
107
- ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`);
83
+ stash_root, file_path, file_order, file_type, body_hash, confidence, status, reason, extraction_run_id
84
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)`);
108
85
  const updateFileMeta = db.prepare(`UPDATE graph_files
109
86
  SET file_order = ?, file_type = ?, confidence = ?, status = ?, reason = ?, extraction_run_id = ?
110
- WHERE entry_id = ?`);
111
- const insertEntity = db.prepare(`INSERT INTO graph_file_entities (entry_id, entity_order, stash_root, entity_norm, entity)
112
- VALUES (?, ?, ?, ?, ?)`);
87
+ WHERE stash_root = ? AND file_path = ? AND body_hash = ?`);
88
+ const insertEntity = db.prepare(`INSERT INTO graph_file_entities (stash_root, file_path, body_hash, entity_order, entity_norm, entity)
89
+ VALUES (?, ?, ?, ?, ?, ?)`);
113
90
  const insertRelation = db.prepare(`INSERT INTO graph_file_relations (
114
- entry_id, relation_order, from_entity_norm, from_entity, to_entity_norm, to_entity, relation_type, confidence
115
- ) VALUES (?, ?, ?, ?, ?, ?, ?, ?)`);
91
+ stash_root, file_path, body_hash, relation_order, from_entity_norm, from_entity, to_entity_norm, to_entity, relation_type, confidence
92
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`);
116
93
  const quality = graph.quality;
117
94
  const telemetry = graph.telemetry;
118
95
  db.transaction(() => {
119
96
  upsertMeta.run(graph.stashRoot, graph.schemaVersion, graph.generatedAt, quality?.consideredFiles ?? graph.files.length, quality?.extractedFiles ?? graph.files.length, quality?.entityCount ?? graph.entities?.length ?? 0, quality?.relationCount ?? graph.relations?.length ?? 0, quality?.extractionCoverage ?? 0, quality?.density ?? 0, telemetry?.extractorId ?? null, telemetry?.extractionRunId ?? null, telemetry?.model ?? null, telemetry?.promptVersion ?? null, telemetry?.batchSize ?? null, telemetry?.cacheHits ?? 0, telemetry?.cacheMisses ?? 0, telemetry?.truncationCount ?? 0, telemetry?.failureCount ?? 0);
120
- // Build a snapshot of existing rows for incremental compare.
97
+ // Build a snapshot of existing rows for incremental compare. The unique
98
+ // index idx_graph_files_path guarantees at most one row per file_path.
121
99
  const existingRows = selectExisting.all(graph.stashRoot);
122
100
  const existingByPath = new Map();
123
101
  for (const row of existingRows)
124
102
  existingByPath.set(row.file_path, row);
125
- let orphanCount = 0;
126
- const presentEntryIds = new Set();
103
+ const presentPaths = new Set();
127
104
  for (const [fileOrder, node] of graph.files.entries()) {
128
- // body_hash is NOT NULL in schema v2; default to a sentinel for inputs
129
- // (test fixtures, legacy imports) that don't supply one. The sentinel
130
- // never equals a real hash so subsequent staleness checks always
131
- // re-extract — correct behaviour for "unknown" bodies.
105
+ // body_hash is part of the PK; default to a sentinel for inputs (test
106
+ // fixtures, legacy imports) that don't supply one. The sentinel never
107
+ // equals a real hash so subsequent staleness checks always re-extract —
108
+ // correct behaviour for "unknown" bodies. Distinct files in one stash
109
+ // are still keyed apart by file_path, so the empty sentinel is safe.
132
110
  const bodyHash = node.bodyHash && node.bodyHash.length > 0 ? node.bodyHash : "";
133
- const entryId = resolveEntryIdForPath(db, graph.stashRoot, node.path);
134
- if (entryId == null) {
135
- orphanCount += 1;
136
- continue;
137
- }
138
- presentEntryIds.add(entryId);
111
+ presentPaths.add(node.path);
139
112
  const existing = existingByPath.get(node.path);
140
- if (existing && existing.entry_id === entryId && existing.body_hash === bodyHash) {
113
+ if (existing && existing.body_hash === bodyHash) {
141
114
  // Body unchanged — only fix up file_order/confidence in case they drifted.
142
- updateFileMeta.run(fileOrder, node.type, node.confidence ?? null, node.status ?? (node.entities.length > 0 ? "extracted" : "empty"), node.reason ?? (node.entities.length > 0 ? "none" : "no_graph_content"), node.extractionRunId ?? telemetry?.extractionRunId ?? null, entryId);
115
+ updateFileMeta.run(fileOrder, node.type, node.confidence ?? null, node.status ?? (node.entities.length > 0 ? "extracted" : "empty"), node.reason ?? (node.entities.length > 0 ? "none" : "no_graph_content"), node.extractionRunId ?? telemetry?.extractionRunId ?? null, graph.stashRoot, node.path, bodyHash);
143
116
  continue;
144
117
  }
145
118
  if (existing) {
146
- // Stale row (different body_hash, or entry_id moved to a different
147
- // path under the same file_path). Wipe child rows; CASCADE would do
148
- // it but explicit DELETE keeps the order deterministic.
149
- deleteEntities.run(existing.entry_id);
150
- deleteRelations.run(existing.entry_id);
151
- deleteFile.run(existing.entry_id);
119
+ // Stale row (different body_hash for this path). Delete the old row by
120
+ // its OLD body_hash; child rows cascade, but explicit DELETE keeps the
121
+ // order deterministic and is safe regardless of the FK pragma.
122
+ deleteEntities.run(graph.stashRoot, existing.file_path, existing.body_hash);
123
+ deleteRelations.run(graph.stashRoot, existing.file_path, existing.body_hash);
124
+ deleteFile.run(graph.stashRoot, existing.file_path, existing.body_hash);
152
125
  }
153
- insertFile.run(entryId, graph.stashRoot, node.path, fileOrder, node.type, bodyHash, node.confidence ?? null, node.status ?? (node.entities.length > 0 ? "extracted" : "empty"), node.reason ?? (node.entities.length > 0 ? "none" : "no_graph_content"), node.extractionRunId ?? telemetry?.extractionRunId ?? null);
126
+ insertFile.run(graph.stashRoot, node.path, fileOrder, node.type, bodyHash, node.confidence ?? null, node.status ?? (node.entities.length > 0 ? "extracted" : "empty"), node.reason ?? (node.entities.length > 0 ? "none" : "no_graph_content"), node.extractionRunId ?? telemetry?.extractionRunId ?? null);
154
127
  for (const [entityOrder, entity] of node.entities.entries()) {
155
- insertEntity.run(entryId, entityOrder, graph.stashRoot, normalizeEntity(entity), entity);
128
+ insertEntity.run(graph.stashRoot, node.path, bodyHash, entityOrder, normalizeEntity(entity), entity);
156
129
  }
157
130
  for (const [relationOrder, relation] of node.relations.entries()) {
158
- insertRelation.run(entryId, relationOrder, normalizeEntity(relation.from), relation.from, normalizeEntity(relation.to), relation.to, relation.type ?? null, relation.confidence ?? null);
131
+ insertRelation.run(graph.stashRoot, node.path, bodyHash, relationOrder, normalizeEntity(relation.from), relation.from, normalizeEntity(relation.to), relation.to, relation.type ?? null, relation.confidence ?? null);
159
132
  }
160
133
  }
161
134
  // Delete files present in DB but absent from the new snapshot. Child
162
- // tables CASCADE on entry_id.
135
+ // tables CASCADE on the composite key; explicit DELETE keeps it determinstic.
163
136
  for (const row of existingRows) {
164
- if (!presentEntryIds.has(row.entry_id)) {
165
- deleteEntities.run(row.entry_id);
166
- deleteRelations.run(row.entry_id);
167
- deleteFile.run(row.entry_id);
137
+ if (!presentPaths.has(row.file_path)) {
138
+ deleteEntities.run(graph.stashRoot, row.file_path, row.body_hash);
139
+ deleteRelations.run(graph.stashRoot, row.file_path, row.body_hash);
140
+ deleteFile.run(graph.stashRoot, row.file_path, row.body_hash);
168
141
  }
169
142
  }
170
- if (orphanCount > 0) {
171
- warn(`[graph] replaceStoredGraph: skipped ${orphanCount} file(s) with no resolvable entry under ${graph.stashRoot}.`);
172
- }
173
143
  })();
174
144
  }
175
145
  export function deleteStoredGraph(db, stashPath) {
176
146
  db.transaction(() => {
177
- // Child rows cascade via entry_id; deleting graph_files clears them.
147
+ // Child rows cascade via the composite (stash_root, file_path, body_hash)
148
+ // FK; deleting graph_files clears them. This is the explicit full-clear
149
+ // path for a stash (entries-delete no longer wipes graph data — see #624-P1).
178
150
  db.prepare("DELETE FROM graph_files WHERE stash_root = ?").run(stashPath);
179
151
  db.prepare("DELETE FROM graph_meta WHERE stash_root = ?").run(stashPath);
180
152
  })();
181
153
  }
154
+ /**
155
+ * #624-P1 — does any graph data exist for a file_path under a stash root?
156
+ * Consumed by show/curate flows (P3) but defined here so the schema and its
157
+ * accessors land together.
158
+ */
159
+ export function hasGraphData(db, stashRoot, filePath) {
160
+ try {
161
+ const row = db
162
+ .prepare("SELECT 1 AS present FROM graph_files WHERE stash_root = ? AND file_path = ? LIMIT 1")
163
+ .get(stashRoot, filePath);
164
+ return row !== undefined;
165
+ }
166
+ catch {
167
+ return false;
168
+ }
169
+ }
182
170
  /**
183
171
  * Scoped loader — only the graph_meta row for a stash. Used by callers that
184
172
  * only need summary numbers (e.g. `akm graph summary`).
@@ -195,13 +183,12 @@ export function loadGraphFilesOnly(stashPath, db) {
195
183
  return withReadableGraphDb(db, (readDb) => {
196
184
  try {
197
185
  const rows = readDb
198
- .prepare(`SELECT entry_id, file_path, file_type, body_hash, confidence, status, reason
186
+ .prepare(`SELECT file_path, file_type, body_hash, confidence, status, reason
199
187
  FROM graph_files
200
188
  WHERE stash_root = ?
201
189
  ORDER BY file_order`)
202
190
  .all(stashPath);
203
191
  return rows.map((row) => ({
204
- entryId: row.entry_id,
205
192
  path: row.file_path,
206
193
  type: row.file_type,
207
194
  bodyHash: row.body_hash,
@@ -222,13 +209,16 @@ export function loadGraphFilesOnly(stashPath, db) {
222
209
  }
223
210
  }
224
211
  /**
225
- * Scoped loader — entities for a single entry_id. Used by per-asset lookups.
212
+ * Scoped loader — entities for a single file, keyed on the #624-P1 composite
213
+ * (stash_root, file_path, body_hash). Used by per-asset show/curate lookups.
226
214
  */
227
- export function loadGraphEntitiesByEntry(db, entryId) {
215
+ export function loadGraphEntitiesByPath(db, stashRoot, filePath, bodyHash) {
228
216
  try {
229
217
  const rows = db
230
- .prepare("SELECT entity FROM graph_file_entities WHERE entry_id = ? ORDER BY entity_order")
231
- .all(entryId);
218
+ .prepare(`SELECT entity FROM graph_file_entities
219
+ WHERE stash_root = ? AND file_path = ? AND body_hash = ?
220
+ ORDER BY entity_order`)
221
+ .all(stashRoot, filePath, bodyHash);
232
222
  return rows.map((r) => r.entity);
233
223
  }
234
224
  catch {
@@ -314,27 +304,32 @@ export function loadStoredGraphSnapshot(stashPath, db) {
314
304
  return null;
315
305
  try {
316
306
  const fileRows = readDb
317
- .prepare(`SELECT entry_id, file_path, file_type, body_hash, confidence, status, reason, extraction_run_id
307
+ .prepare(`SELECT file_path, file_type, body_hash, confidence, status, reason, extraction_run_id
318
308
  FROM graph_files
319
309
  WHERE stash_root = ?
320
310
  ORDER BY file_order`)
321
311
  .all(stashPath);
322
312
  const entityRows = readDb
323
- .prepare(`SELECT gfe.entry_id AS entry_id, gf.file_path AS file_path, gfe.entity AS entity
313
+ .prepare(`SELECT gf.file_path AS file_path, gfe.entity AS entity
324
314
  FROM graph_file_entities gfe
325
- JOIN graph_files gf ON gf.entry_id = gfe.entry_id
315
+ JOIN graph_files gf
316
+ ON gf.stash_root = gfe.stash_root
317
+ AND gf.file_path = gfe.file_path
318
+ AND gf.body_hash = gfe.body_hash
326
319
  WHERE gf.stash_root = ?
327
320
  ORDER BY gf.file_order, gfe.entity_order`)
328
321
  .all(stashPath);
329
322
  const relationRows = readDb
330
- .prepare(`SELECT gfr.entry_id AS entry_id,
331
- gf.file_path AS file_path,
323
+ .prepare(`SELECT gf.file_path AS file_path,
332
324
  gfr.from_entity AS from_entity,
333
325
  gfr.to_entity AS to_entity,
334
326
  gfr.relation_type AS relation_type,
335
327
  gfr.confidence AS confidence
336
328
  FROM graph_file_relations gfr
337
- JOIN graph_files gf ON gf.entry_id = gfr.entry_id
329
+ JOIN graph_files gf
330
+ ON gf.stash_root = gfr.stash_root
331
+ AND gf.file_path = gfr.file_path
332
+ AND gf.body_hash = gfr.body_hash
338
333
  WHERE gf.stash_root = ?
339
334
  ORDER BY gf.file_order, gfr.relation_order`)
340
335
  .all(stashPath);
@@ -11,12 +11,14 @@
11
11
  * `searchLocal()` and `show.ts`, centralizing the "indexed yet?" gap handling
12
12
  * behind a single entry point.
13
13
  */
14
+ import { spawn } from "node:child_process";
14
15
  import fs from "node:fs";
15
16
  import path from "node:path";
16
17
  import { ASSET_SPECS, TYPE_DIRS } from "../core/asset/asset-spec.js";
17
- import { getDbPath } from "../core/paths.js";
18
+ import { getDataDir, getDbPath } from "../core/paths.js";
18
19
  import { warn } from "../core/warn.js";
19
- import { closeDatabase, getEntryCount, getMeta, openExistingDatabase } from "./db/db.js";
20
+ import { closeDatabase, getEntryCount, getIndexedFilePaths, getMeta, openExistingDatabase } from "./db/db.js";
21
+ import { acquireIndexWriterLease, handoffIndexWriterLeaseToPid } from "./index-writer-lock.js";
20
22
  function getIndexableFiles(root, spec) {
21
23
  if (!fs.existsSync(root))
22
24
  return [];
@@ -52,16 +54,34 @@ function getIndexableFiles(root, spec) {
52
54
  }
53
55
  return files;
54
56
  }
55
- function hasNewerIndexableFiles(stashDir, builtAt) {
56
- if (!builtAt)
57
- return true;
58
- const builtAtMs = new Date(builtAt).getTime();
59
- if (!Number.isFinite(builtAtMs))
60
- return true;
57
+ /**
58
+ * Whether any indexable file under `stashDir` is newer than the last build, or
59
+ * has never been indexed at all.
60
+ *
61
+ * Two independent signals, because neither alone is sufficient:
62
+ * 1. **mtime > builtAt** — catches in-place *edits* of already-indexed files.
63
+ * 2. **path not in `indexedPaths`** — catches *newly added* files. This is
64
+ * clock-independent on purpose: a freshly-written file can have a
65
+ * filesystem mtime that compares as *older* than the wall-clock `builtAt`
66
+ * (the two clocks are not perfectly synchronized and `builtAt` is
67
+ * millisecond-truncated), so the mtime test alone silently misses
68
+ * additions made within ~a millisecond of the previous build.
69
+ *
70
+ * `getIndexableFiles` applies each asset type's own relevance filter, so
71
+ * non-indexed companion files (e.g. `package.json` next to a knowledge doc) are
72
+ * never considered and do not produce false "new file" positives.
73
+ */
74
+ function hasNewerIndexableFiles(stashDir, builtAt, indexedPaths) {
75
+ const builtAtMs = builtAt ? new Date(builtAt).getTime() : Number.NaN;
76
+ const builtAtUsable = Number.isFinite(builtAtMs);
61
77
  for (const [type, spec] of Object.entries(ASSET_SPECS)) {
62
78
  const typeRoot = path.join(stashDir, TYPE_DIRS[type] ?? spec.stashDir);
63
79
  const files = getIndexableFiles(typeRoot, spec);
64
80
  for (const file of files) {
81
+ if (!indexedPaths.has(file))
82
+ return true;
83
+ if (!builtAtUsable)
84
+ return true;
65
85
  try {
66
86
  if (fs.statSync(file).mtimeMs > builtAtMs)
67
87
  return true;
@@ -89,7 +109,7 @@ export function isIndexStale(stashDir) {
89
109
  if (entryCount === 0)
90
110
  return true;
91
111
  const builtAt = getMeta(db, "builtAt");
92
- if (hasNewerIndexableFiles(stashDir, builtAt))
112
+ if (hasNewerIndexableFiles(stashDir, builtAt, getIndexedFilePaths(db)))
93
113
  return true;
94
114
  const storedStashDir = getMeta(db, "stashDir");
95
115
  if (storedStashDir !== stashDir) {
@@ -114,16 +134,84 @@ export function isIndexStale(stashDir) {
114
134
  }
115
135
  }
116
136
  /**
117
- * Run an incremental index when the local index is stale. Best-effort
118
- * failures are logged as warnings but never thrown, so the caller can
119
- * proceed (and surface a proper "not in index" error if the index is
120
- * still unusable).
121
- *
122
- * Returns `true` if an index run was attempted.
137
+ * Whether the existing index can serve queries for `stashDir` *right now*
138
+ * i.e. the DB file exists, the `entries` table holds rows, and those rows were
139
+ * built for this stash (it is the stored primary stash or appears in the
140
+ * stored `stashDirs` set). When this is true the index is at worst
141
+ * content-stale, so the `#607` background-reindex optimization is safe: the
142
+ * caller gets slightly-stale-but-relevant results immediately. When it is
143
+ * false the existing index has nothing relevant to return (no DB, no `entries`
144
+ * table, zero rows, or built for a different stash), so a background reindex
145
+ * would leave the caller empty until the next read — those cases must rebuild
146
+ * inline.
123
147
  */
124
- export async function ensureIndex(stashDir) {
125
- if (!isIndexStale(stashDir))
148
+ function indexCanServeStash(stashDir) {
149
+ const dbPath = getDbPath();
150
+ if (!fs.existsSync(dbPath))
151
+ return false;
152
+ let db;
153
+ try {
154
+ db = openExistingDatabase(dbPath);
155
+ if (getEntryCount(db) === 0)
156
+ return false;
157
+ const storedStashDir = getMeta(db, "stashDir");
158
+ if (storedStashDir === stashDir)
159
+ return true;
160
+ try {
161
+ const storedDirs = JSON.parse(getMeta(db, "stashDirs") ?? "[]");
162
+ return storedDirs.includes(stashDir);
163
+ }
164
+ catch {
165
+ return false;
166
+ }
167
+ }
168
+ catch {
169
+ // No `entries` table (or otherwise unreadable) — cannot serve.
126
170
  return false;
171
+ }
172
+ finally {
173
+ if (db)
174
+ closeDatabase(db);
175
+ }
176
+ }
177
+ /**
178
+ * Spawn a background `akm index` process. Non-blocking — returns immediately.
179
+ * Background callers share the same global index-writer lease as foreground
180
+ * writers, so stale-read-triggered auto-index attempts coalesce safely.
181
+ */
182
+ async function spawnBackgroundReindex(_stashDir) {
183
+ const dataDir = getDataDir();
184
+ const logFile = path.join(dataDir, "logs", "index-background.log");
185
+ fs.mkdirSync(path.dirname(logFile), { recursive: true });
186
+ const lease = await acquireIndexWriterLease({ mode: "try", purpose: "background-reindex-spawn" });
187
+ if (!lease)
188
+ return;
189
+ const akmBin = process.argv[0];
190
+ const akmScript = process.argv[1];
191
+ try {
192
+ const child = spawn(akmBin, [akmScript, "index", "--background"], {
193
+ detached: true,
194
+ stdio: ["ignore", fs.openSync(logFile, "a"), fs.openSync(logFile, "a")],
195
+ env: { ...process.env },
196
+ });
197
+ if (!child.pid) {
198
+ lease.release();
199
+ return;
200
+ }
201
+ handoffIndexWriterLeaseToPid(lease, child.pid, "background-reindex");
202
+ try {
203
+ child.unref();
204
+ }
205
+ catch {
206
+ // ignore
207
+ }
208
+ }
209
+ catch (error) {
210
+ lease.release();
211
+ throw error;
212
+ }
213
+ }
214
+ async function runInlineReindex(stashDir) {
127
215
  try {
128
216
  const { akmIndex } = await import("./indexer.js");
129
217
  await akmIndex({ stashDir });
@@ -134,3 +222,50 @@ export async function ensureIndex(stashDir) {
134
222
  return true;
135
223
  }
136
224
  }
225
+ /**
226
+ * Ensure the local index exists and is fresh enough for the caller's needs.
227
+ *
228
+ * Default mode is `background`, which preserves the low-latency behavior used
229
+ * by read paths (`search`, `show`, `feedback`): when a populated index is
230
+ * merely stale, spawn a detached reindex and proceed against the existing
231
+ * index. When the index is entirely absent (no DB / no `entries` table / zero
232
+ * rows) the rebuild runs inline regardless of mode, since there is nothing to
233
+ * proceed against.
234
+ *
235
+ * `mode: "blocking"` waits for the rebuild to finish before returning. Use
236
+ * this for callers like `improve` whose planning logic depends on a populated
237
+ * `entries` table in the same process.
238
+ *
239
+ * Returns `true` if an index run was attempted.
240
+ */
241
+ export async function ensureIndex(stashDir, options = {}) {
242
+ if (!isIndexStale(stashDir))
243
+ return false;
244
+ // Blocking when explicitly requested, or whenever the existing index cannot
245
+ // serve this stash (absent DB, no `entries` table, zero rows, or built for a
246
+ // different stash): a background reindex returns immediately and would leave
247
+ // a first-time caller (search, curate, wiki, show, feedback) with empty
248
+ // results. Building inline is a one-off cost; a populated index for this
249
+ // stash that is merely content-stale still refreshes in the background.
250
+ if (options.mode === "blocking" || !indexCanServeStash(stashDir)) {
251
+ return runInlineReindex(stashDir);
252
+ }
253
+ // The background path re-invokes the akm CLI as a detached child via
254
+ // `process.argv[1]`. That is only the akm entrypoint when THIS process is the
255
+ // akm CLI itself — which the CLI startup block signals with AKM_CLI_ENTRY=1.
256
+ // In any other host (the in-process test runner, a library embedding akm),
257
+ // argv[1] points at the host (e.g. the test runner), so spawning it would
258
+ // launch the wrong program and orphan it. Build inline there instead — same
259
+ // resulting index, no detached process.
260
+ if (process.env.AKM_CLI_ENTRY !== "1") {
261
+ return runInlineReindex(stashDir);
262
+ }
263
+ try {
264
+ await spawnBackgroundReindex(stashDir);
265
+ return true;
266
+ }
267
+ catch (error) {
268
+ warn("Background reindex spawn failed, proceeding with existing index:", error instanceof Error ? error.message : String(error));
269
+ return true;
270
+ }
271
+ }