akm-cli 0.9.17-alpha.8 → 0.9.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/CHANGELOG.md +277 -1694
  2. package/STABILITY.md +9 -8
  3. package/dist/assets/hints/cli-hints-full.md +6 -7
  4. package/dist/assets/improve-strategies/catchup.json +0 -3
  5. package/dist/assets/improve-strategies/consolidate.json +0 -1
  6. package/dist/assets/improve-strategies/default.json +1 -2
  7. package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
  8. package/dist/assets/improve-strategies/quick.json +1 -2
  9. package/dist/assets/improve-strategies/reflect-distill.json +1 -2
  10. package/dist/assets/improve-strategies/thorough.json +0 -3
  11. package/dist/assets/prompts/consolidate-pair.md +20 -0
  12. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +17 -19
  13. package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
  14. package/dist/assets/templates/html/health.html +3 -5
  15. package/dist/cli/retired-commands.js +1 -1
  16. package/dist/cli/unknown-flags.js +24 -1
  17. package/dist/cli.js +46 -1
  18. package/dist/commands/health/archive-usage.js +92 -0
  19. package/dist/commands/health/data-dir-usage.js +25 -13
  20. package/dist/commands/health/html-report.js +1 -4
  21. package/dist/commands/health/improve-metrics.js +25 -37
  22. package/dist/commands/health/md-report.js +1 -6
  23. package/dist/commands/health/report-view-model.js +4 -14
  24. package/dist/commands/health/windows.js +0 -1
  25. package/dist/commands/health.js +13 -0
  26. package/dist/commands/improve/consolidate/continuity-check.js +137 -0
  27. package/dist/commands/improve/consolidate/pair-pass.js +791 -0
  28. package/dist/commands/improve/consolidate.js +38 -63
  29. package/dist/commands/improve/extract-prompt.js +1 -2
  30. package/dist/commands/improve/improve-cli.js +1 -1
  31. package/dist/commands/improve/improve-strategies.js +23 -5
  32. package/dist/commands/improve/improve.js +19 -30
  33. package/dist/commands/improve/ledger.js +3 -2
  34. package/dist/commands/improve/loop-stages.js +5 -85
  35. package/dist/commands/improve/memory/memory-belief.js +3 -1
  36. package/dist/commands/improve/memory/memory-improve.js +262 -11
  37. package/dist/commands/improve/planner.js +0 -5
  38. package/dist/commands/improve/preparation.js +20 -135
  39. package/dist/commands/improve/retrieval-scope.js +19 -4
  40. package/dist/commands/improve/salience.js +1 -14
  41. package/dist/commands/improve/stage.js +0 -1
  42. package/dist/commands/lint/base-linter.js +19 -11
  43. package/dist/commands/proposal/drain.js +8 -1
  44. package/dist/commands/proposal/proposal-cli.js +16 -2
  45. package/dist/commands/proposal/proposal-types.js +7 -0
  46. package/dist/commands/proposal/proposal.js +37 -6
  47. package/dist/commands/proposal/repository.js +613 -4
  48. package/dist/commands/proposal/validators/proposals.js +9 -0
  49. package/dist/commands/read/knowledge.js +3 -2
  50. package/dist/commands/read/show.js +0 -14
  51. package/dist/commands/sources/info.js +122 -18
  52. package/dist/commands/sources/stash-cli.js +23 -3
  53. package/dist/core/bundle-rename.js +1 -7
  54. package/dist/core/config/config-schema.js +8 -1
  55. package/dist/core/config/config.js +23 -48
  56. package/dist/core/config/engine-semantics.js +0 -2
  57. package/dist/core/config/schema/improve-processes.js +17 -42
  58. package/dist/core/config/schema/index-config.js +5 -25
  59. package/dist/core/file-change.js +13 -5
  60. package/dist/core/improve-result.js +22 -6
  61. package/dist/core/improve-types.js +0 -1
  62. package/dist/core/loopback.js +7 -12
  63. package/dist/core/parse.js +13 -16
  64. package/dist/core/state/migrations.js +15 -0
  65. package/dist/core/time.js +0 -20
  66. package/dist/indexer/db/llm-cache.js +2 -2
  67. package/dist/indexer/ensure-index.js +2 -2
  68. package/dist/indexer/index-written-assets.js +2 -3
  69. package/dist/indexer/indexer.js +18 -418
  70. package/dist/indexer/passes/metadata.js +0 -19
  71. package/dist/indexer/walk/walker.js +3 -4
  72. package/dist/llm/client.js +8 -10
  73. package/dist/llm/embedders/remote.js +1 -2
  74. package/dist/llm/feature-gate.js +0 -5
  75. package/dist/output/shapes/helpers.js +20 -4
  76. package/dist/output/text/command-format.js +9 -8
  77. package/dist/output/text/proposal-format.js +47 -1
  78. package/dist/output/text/show-format.js +0 -20
  79. package/dist/scripts/akm-migrate-node.js +923 -950
  80. package/dist/scripts/akm-migrate.js +923 -950
  81. package/dist/setup/steps/connection.js +5 -6
  82. package/dist/setup/steps/platforms.js +2 -2
  83. package/dist/sources/providers/git-stash.js +83 -4
  84. package/dist/storage/repositories/improve-ledger-repository.js +48 -7
  85. package/dist/storage/repositories/index-connection.js +5 -2
  86. package/dist/storage/repositories/index-entries-repository.js +4 -7
  87. package/dist/storage/repositories/index-entry-schema.js +4 -2
  88. package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
  89. package/dist/storage/repositories/index-schema.js +55 -104
  90. package/dist/storage/repositories/proposals-repository.js +61 -0
  91. package/dist/storage/repositories/salience-repository.js +1 -19
  92. package/docs/migration/README.md +1 -1
  93. package/docs/migration/release-notes/0.9.17.md +130 -41
  94. package/docs/migration/release-notes/README.md +7 -0
  95. package/docs/reference/cli.md +27 -21
  96. package/docs/reference/configuration.md +21 -12
  97. package/docs/reference/data-and-telemetry.md +0 -1
  98. package/package.json +1 -1
  99. package/schemas/akm-config.json +0 -342
  100. package/dist/assets/improve-strategies/graph-refresh.json +0 -15
  101. package/dist/assets/prompts/contradiction-judge.md +0 -33
  102. package/dist/assets/prompts/graph-extract-system.md +0 -1
  103. package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
  104. package/dist/assets/prompts/metadata-enhance-system.md +0 -1
  105. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
  106. package/dist/indexer/db/graph-db.js +0 -399
  107. package/dist/indexer/graph/graph-extraction.js +0 -809
  108. package/dist/indexer/graph/graph-related.js +0 -131
  109. package/dist/indexer/graph/graph-types.js +0 -4
  110. package/dist/llm/graph-extract.js +0 -892
  111. package/dist/llm/metadata-enhance.js +0 -95
@@ -122,7 +122,8 @@ const LLM_PRESETS = [
122
122
  },
123
123
  ];
124
124
  /**
125
- * Step 3a: pick an LLM provider. Used for indexing-time metadata enhancement.
125
+ * Step 3a: pick an LLM provider. Used for LLM-gated background features
126
+ * (e.g. memory inference).
126
127
  *
127
128
  * @internal Exported for testing only.
128
129
  */
@@ -151,14 +152,14 @@ export async function stepLlm(current, ollamaEndpoint, ollamaChatModels, lmStudi
151
152
  }
152
153
  options.push({ value: "lmstudio", label: "LM Studio / local server", hint: lmStudioOptionHint(lmStudio) });
153
154
  options.push({ value: "custom", label: "Custom OpenAI-compatible endpoint" });
154
- options.push({ value: "none", label: "Skip LLM", hint: "no metadata enhancement during indexing" });
155
+ options.push({ value: "none", label: "Skip LLM", hint: "LLM-gated background features stay disabled" });
155
156
  const currentLlm = readCurrentLlmEngine(current);
156
157
  if (currentLlm) {
157
158
  options.push(keepCurrentOption(currentLlm));
158
159
  }
159
160
  const initialValue = currentLlm ? "keep" : ollamaAvailable ? "ollama" : (LLM_PRESETS[0]?.value ?? "none");
160
161
  const choice = await prompt(() => p.select({
161
- message: "Configure an LLM for richer metadata during indexing:",
162
+ message: "Configure an LLM for background features:",
162
163
  options,
163
164
  initialValue,
164
165
  }));
@@ -251,7 +252,7 @@ export async function stepLlm(current, ollamaEndpoint, ollamaChatModels, lmStudi
251
252
  return llm;
252
253
  }
253
254
  /**
254
- * Step 1/2: Configure the small model connection used for metadata and bounded LLM features.
255
+ * Step 1/2: Configure the small model connection used for bounded LLM features.
255
256
  *
256
257
  * Detects Ollama automatically and pre-selects it. The user may also choose
257
258
  * OpenAI, LM Studio, a custom endpoint, or skip the step entirely.
@@ -260,7 +261,6 @@ export async function stepSmallModelConnection(current) {
260
261
  p.log.step("Step 1/2: Configure your small model connection");
261
262
  p.note([
262
263
  "This connection is used for background processing:",
263
- " • akm index (metadata enhancement)",
264
264
  " • akm improve (lesson distillation)",
265
265
  " • akm remember --enrich (memory compression)",
266
266
  ].join("\n"));
@@ -301,7 +301,6 @@ export async function stepSmallModelConnection(current) {
301
301
  if (providerChoice === "skip") {
302
302
  p.note([
303
303
  "Enrichment features disabled:",
304
- " • akm index — metadata enhancement disabled",
305
304
  " • akm improve — lesson generation",
306
305
  " • akm remember --enrich",
307
306
  "",
@@ -53,10 +53,10 @@ export function printCapabilitySummary(smallModelSkipped, agentConfigured) {
53
53
  const lines = ["Setup complete. Here's what's enabled:", ""];
54
54
  lines.push(" ✓ akm search, akm curate, akm show, akm index, akm remember — always available");
55
55
  if (!smallModelSkipped) {
56
- lines.push(" ✓ index metadata enhancement, akm improve, akm remember --enrich — small model configured");
56
+ lines.push(" ✓ akm improve, akm remember --enrich — small model configured");
57
57
  }
58
58
  else {
59
- lines.push(" ✗ index metadata enhancement, akm improve, akm remember --enrich — run `akm setup` to enable");
59
+ lines.push(" ✗ akm improve, akm remember --enrich — run `akm setup` to enable");
60
60
  }
61
61
  if (agentConfigured) {
62
62
  lines.push(" ✓ akm proposal new, akm improve, akm task — agent configured");
@@ -25,11 +25,18 @@ import { getCachePaths, parseGitRepoUrl } from "./git-provider.js";
25
25
  export function isGitBackedStash(stashDir) {
26
26
  return fs.existsSync(path.join(stashDir, ".git"));
27
27
  }
28
- /** Return repo-relative dirty/staged paths without changing the index. */
29
- export function listGitChangedPaths(repoDir) {
28
+ /**
29
+ * Return repo-relative dirty/staged paths without changing the index, and
30
+ * whether `git status` itself succeeded. A broken submodule, a detached
31
+ * `GIT_DIR`, or git simply not being on `PATH` all exit nonzero here — a
32
+ * caller that would otherwise read the empty `paths` as "nothing is dirty"
33
+ * must check `ok` first (see {@link listGitChangedPaths}'s callers that
34
+ * cannot, and the archive purge sweep, which can and does).
35
+ */
36
+ export function tryListGitChangedPaths(repoDir) {
30
37
  const result = runGit(["-C", repoDir, "status", "--porcelain", "-z", "--untracked-files=all"]);
31
38
  if (result.status !== 0)
32
- return [];
39
+ return { paths: [], ok: false };
33
40
  const records = result.stdout.split("\0");
34
41
  const paths = [];
35
42
  for (let i = 0; i < records.length; i++) {
@@ -44,7 +51,79 @@ export function listGitChangedPaths(repoDir) {
44
51
  paths.push(previousPath);
45
52
  }
46
53
  }
47
- return paths;
54
+ return { paths, ok: true };
55
+ }
56
+ /** Return repo-relative dirty/staged paths without changing the index. `[]` on any git failure — see {@link tryListGitChangedPaths} for a caller that must tell that apart from "nothing is dirty". */
57
+ export function listGitChangedPaths(repoDir) {
58
+ return tryListGitChangedPaths(repoDir).paths;
59
+ }
60
+ /**
61
+ * Return repo-relative paths git tracks at HEAD/index under `pathspec` (or
62
+ * the whole repo when omitted), and whether `git ls-files` itself
63
+ * succeeded — see {@link tryListGitChangedPaths}, the same contract.
64
+ */
65
+ export function tryListGitTrackedPaths(repoDir, pathspec) {
66
+ const args = ["-C", repoDir, "ls-files", "-z"];
67
+ if (pathspec)
68
+ args.push("--", pathspec);
69
+ const result = runGit(args);
70
+ if (result.status !== 0)
71
+ return { paths: [], ok: false };
72
+ return { paths: result.stdout.split("\0").filter((record) => record.length > 0), ok: true };
73
+ }
74
+ /**
75
+ * Return repo-relative paths under `pathspec` that `git ls-files -v` tags as
76
+ * NOT verifiable against the worktree: `assume-unchanged` (a lowercase tag —
77
+ * `ls-files -v` lowercases a file's normal tag letter when that bit is set)
78
+ * or `skip-worktree` (the literal `S`). `git status` silently omits an edit
79
+ * to either kind of file — the index is telling git not to compare it — so a
80
+ * caller that trusts {@link tryListGitChangedPaths} alone would read a
81
+ * genuinely modified file as clean.
82
+ */
83
+ export function tryListGitUnverifiablePaths(repoDir, pathspec) {
84
+ const args = ["-C", repoDir, "ls-files", "-v", "-z"];
85
+ if (pathspec)
86
+ args.push("--", pathspec);
87
+ const result = runGit(args);
88
+ if (result.status !== 0)
89
+ return { paths: [], ok: false };
90
+ const paths = [];
91
+ for (const record of result.stdout.split("\0")) {
92
+ if (!record)
93
+ continue;
94
+ const tag = record.slice(0, 1);
95
+ if (tag === "S" || (tag >= "a" && tag <= "z"))
96
+ paths.push(record.slice(2));
97
+ }
98
+ return { paths, ok: true };
99
+ }
100
+ /**
101
+ * The three-part git cleanliness check shared by the archive purge sweep
102
+ * (`purgeGracedArchive` in `commands/improve/memory/memory-improve.ts`) and
103
+ * the `memory-cleanup-archive` health advisory (`collectArchiveUsageAdvisory`
104
+ * in `commands/health/archive-usage.ts`): a path is safe to treat as
105
+ * committed only if it is tracked ({@link tryListGitTrackedPaths}), not dirty
106
+ * ({@link tryListGitChangedPaths}), and not assume-unchanged/skip-worktree
107
+ * ({@link tryListGitUnverifiablePaths} — those hide their own edits from
108
+ * `git status`, so an unverifiable file is never trusted as clean either).
109
+ *
110
+ * Each of the three git calls can fail independently (a broken submodule, or
111
+ * git missing from `PATH`); `ok` is `false` if any one does, and `isSafe`
112
+ * then returns `false` for every path rather than guessing — callers that
113
+ * need to short-circuit before doing other work still check `ok` themselves.
114
+ */
115
+ export function checkGitPathSafety(repoDir, pathspec) {
116
+ const dirtyQuery = tryListGitChangedPaths(repoDir);
117
+ const trackedQuery = tryListGitTrackedPaths(repoDir, pathspec);
118
+ const unverifiableQuery = tryListGitUnverifiablePaths(repoDir, pathspec);
119
+ const ok = dirtyQuery.ok && trackedQuery.ok && unverifiableQuery.ok;
120
+ const dirty = new Set(dirtyQuery.paths);
121
+ const tracked = new Set(trackedQuery.paths);
122
+ const unverifiable = new Set(unverifiableQuery.paths);
123
+ return {
124
+ ok,
125
+ isSafe: (repoRelativePath) => ok && tracked.has(repoRelativePath) && !dirty.has(repoRelativePath) && !unverifiable.has(repoRelativePath),
126
+ };
48
127
  }
49
128
  export class GitStashPushError extends Error {
50
129
  commit;
@@ -27,6 +27,17 @@ export const LEDGER_DEFAULT_REJECTION_WINDOW_DAYS = 7;
27
27
  export const LEDGER_EXPIRED_GRACE_DAYS = 1;
28
28
  /** Revisit cadence for a ref the stage looked at and had nothing to do (or is still pending). */
29
29
  export const LEDGER_REVISIT_CADENCE_DAYS = 7;
30
+ /**
31
+ * The consolidate pair pass's own ledger source (alpha.9): kept apart from
32
+ * the promote pass's `consolidate` rows so the two candidate-selection
33
+ * cadences never collide on the same `(stash, ref, source)` key. Its
34
+ * eligibility is entirely content-driven (`content_hash` above, compared by
35
+ * `selectInitiators` in `src/commands/improve/consolidate/pair-pass.ts`) —
36
+ * {@link windowDays} below gives it no `next_eligible_at` timer at all, so a
37
+ * row never "expires" on its own; only a content change makes the ref
38
+ * eligible again.
39
+ */
40
+ export const PAIR_PASS_LEDGER_SOURCE = "consolidate-pair";
30
41
  /**
31
42
  * Outcomes whose window a fresh signal on the asset (new feedback, a content
32
43
  * change) cannot lift. Every other window is a revisit cadence that a signal
@@ -38,6 +49,12 @@ export const LEDGER_HARD_OUTCOMES = new Set([
38
49
  "expired",
39
50
  ]);
40
51
  function windowDays(source, outcome) {
52
+ // The pair pass's own eligibility never reads next_eligible_at (it compares
53
+ // content_hash instead — selectInitiators in pair-pass.ts) — recording a
54
+ // window here would be a number nothing enforces, so every row it writes
55
+ // stays "eligible now" regardless of outcome.
56
+ if (source === PAIR_PASS_LEDGER_SOURCE)
57
+ return null;
41
58
  switch (outcome) {
42
59
  case "rejected":
43
60
  case "quality_rejected":
@@ -91,6 +108,7 @@ function toRow(row) {
91
108
  nextEligibleAt: row.next_eligible_at,
92
109
  proposalId: row.proposal_id,
93
110
  detail: row.detail,
111
+ contentHash: row.content_hash,
94
112
  };
95
113
  }
96
114
  function trimDetail(detail) {
@@ -112,16 +130,18 @@ export function recordImproveLedger(db, input) {
112
130
  nextEligibleAt: nextEligibleAt(input.source, input.outcome, input.at),
113
131
  proposalId: input.proposalId ?? null,
114
132
  detail: trimDetail(input.detail),
133
+ contentHash: input.contentHash ?? null,
115
134
  };
116
135
  db.prepare(`INSERT INTO improve_ledger
117
- (stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail)
118
- VALUES (?, ?, ?, ?, ?, ?, ?, ?)
136
+ (stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail, content_hash)
137
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
119
138
  ON CONFLICT(stash_dir, ref, source) DO UPDATE SET
120
139
  last_attempt_at = excluded.last_attempt_at,
121
140
  outcome = excluded.outcome,
122
141
  next_eligible_at = excluded.next_eligible_at,
123
142
  proposal_id = excluded.proposal_id,
124
- detail = excluded.detail`).run(row.stashDir, row.ref, row.source, row.lastAttemptAt, row.outcome, row.nextEligibleAt, row.proposalId, row.detail);
143
+ detail = excluded.detail,
144
+ content_hash = excluded.content_hash`).run(row.stashDir, row.ref, row.source, row.lastAttemptAt, row.outcome, row.nextEligibleAt, row.proposalId, row.detail, row.contentHash);
125
145
  return row;
126
146
  }
127
147
  /**
@@ -150,19 +170,40 @@ export function recordImproveLedgerDecision(db, input) {
150
170
  ...(input.detail !== undefined ? { detail: input.detail } : {}),
151
171
  });
152
172
  }
173
+ /**
174
+ * Should-fix 6 (second review round): a read-only or dry-run open never
175
+ * migrates, so it can land on a state.db from before migration 029 added
176
+ * `content_hash` — reading it there threw "no such column", which
177
+ * `loadRetrievalScope`'s own catch then reported as "usage history
178
+ * unreadable", making the WHOLE scope `undefined` (every asset eligible) on
179
+ * every read-only/dry-run call against an as-yet-unmigrated database. A
180
+ * per-connection cache, since a real `Database` handle's schema does not
181
+ * change mid-lifetime and this is checked on every ledger read.
182
+ */
183
+ const hasContentHashColumnCache = new WeakMap();
184
+ function hasContentHashColumn(db) {
185
+ const cached = hasContentHashColumnCache.get(db);
186
+ if (cached !== undefined)
187
+ return cached;
188
+ const has = db.prepare("PRAGMA table_info(improve_ledger)").all().some((c) => c.name === "content_hash");
189
+ hasContentHashColumnCache.set(db, has);
190
+ return has;
191
+ }
153
192
  export function getImproveLedgerRow(db, stashDir, ref, source) {
193
+ const withHash = hasContentHashColumn(db);
154
194
  const row = db
155
- .prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail
195
+ .prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail${withHash ? ", content_hash" : ""}
156
196
  FROM improve_ledger WHERE stash_dir = ? AND ref = ? AND source = ?`)
157
197
  .get(stashDir, ref, source);
158
- return row ? toRow(row) : undefined;
198
+ return row ? toRow(withHash ? row : { ...row, content_hash: null }) : undefined;
159
199
  }
160
200
  /** Every row for one stash, optionally narrowed to `sources`. */
161
201
  export function listImproveLedgerRows(db, stashDir, sources) {
202
+ const withHash = hasContentHashColumn(db);
162
203
  const sourceFilter = sources && sources.length > 0 ? ` AND source IN (${sources.map(() => "?").join(", ")})` : "";
163
204
  const rows = db
164
- .prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail
205
+ .prepare(`SELECT stash_dir, ref, source, last_attempt_at, outcome, next_eligible_at, proposal_id, detail${withHash ? ", content_hash" : ""}
165
206
  FROM improve_ledger WHERE stash_dir = ?${sourceFilter} ORDER BY ref ASC, source ASC`)
166
207
  .all(stashDir, ...(sources && sources.length > 0 ? sources : []));
167
- return rows.map(toRow);
208
+ return rows.map((row) => toRow(withHash ? row : { ...row, content_hash: null }));
168
209
  }
@@ -180,9 +180,12 @@ export function openReadonlyExistingDatabase(dbPath, options) {
180
180
  // never block — but in the DELETE/TRUNCATE modes the network-FS fallback and
181
181
  // AKM_SQLITE_JOURNAL_MODE can select, a concurrent writer makes every read
182
182
  // fail instantly with SQLITE_BUSY. busy_timeout is legal on a read-only
183
- // connection, so apply just that one.
183
+ // connection, so apply just that one. `busyTimeoutMs` defaults to the
184
+ // shared 30s constant; a caller that must never sit behind another akm
185
+ // process's write lock for long (e.g. `akm info`) can pass a much shorter
186
+ // bound instead.
184
187
  try {
185
- db.exec(`PRAGMA busy_timeout = ${SQLITE_BUSY_TIMEOUT_MS}`);
188
+ db.exec(`PRAGMA busy_timeout = ${options?.busyTimeoutMs ?? SQLITE_BUSY_TIMEOUT_MS}`);
186
189
  checkIndexLayout(db, resolvedPath);
187
190
  return db;
188
191
  }
@@ -50,8 +50,8 @@ export function upsertEntry(db, filePath, entry, provenance, contentHash) {
50
50
  // does not have to scan + JSON-decode every memory row.
51
51
  const derivedFrom = typeof entry.derivedFrom === "string" && entry.derivedFrom.trim() ? entry.derivedFrom.trim() : null;
52
52
  const hash = embedHash(entry);
53
- // `content_hash` is optional on the LLM-enrichment re-upsert; a missing hash
54
- // preserves the scan writer's current value.
53
+ // `content_hash` is optional; an upsert that omits it preserves the scan
54
+ // writer's current value.
55
55
  const apply = () => {
56
56
  const previous = stmts.findByItemRef.get(provenance.itemRef);
57
57
  const result = stmts.upsert.get(provenance.itemRef, provenance.bundleId, provenance.componentId, provenance.conceptId, provenance.adapterId, entry.type, filePath, contentHash ?? null, JSON.stringify(entry), derivedFrom, hash);
@@ -70,8 +70,8 @@ export function upsertEntry(db, filePath, entry, provenance, contentHash) {
70
70
  return db.transaction(apply)();
71
71
  }
72
72
  const upsertStmtsByDb = new WeakMap();
73
- // item_ref is the sole durable conflict target. `content_hash` COALESCEs so a
74
- // metadata-only enrichment pass cannot wipe a scan hash.
73
+ // item_ref is the sole durable conflict target. `content_hash` COALESCEs so
74
+ // an upsert that omits it (`contentHash` is optional) cannot wipe a scan hash.
75
75
  const UPSERT_SET_CLAUSE = `SET
76
76
  bundle_id = excluded.bundle_id,
77
77
  component_id = excluded.component_id,
@@ -418,9 +418,6 @@ function deleteRelatedRows(db, ids, options = {}) {
418
418
  // commits; standalone delete callers retain the immediate behavior.
419
419
  if (options.cleanupUsageEvents !== false)
420
420
  deleteUsageEventsByEntryIds(numericIds);
421
- // graph_files is keyed by its own stash_root/file_path/body_hash identity,
422
- // so deleting an entry row intentionally leaves extracted graph data intact,
423
- // and with it the graph_meta counts the graph writer derives from those rows.
424
421
  }
425
422
  export function deleteUsageEventsByEntryIds(entryIds) {
426
423
  if (entryIds.length === 0 || !fs.existsSync(getStateDbPath()))
@@ -8,8 +8,10 @@
8
8
  * readers serve what is there, and the writable opener (`ensureSchema`,
9
9
  * `index-schema.ts`) brings it up to date in place — added and dropped
10
10
  * columns, retired tables dropped, a one-time full-text rebuild — without
11
- * touching embeddings, utility scores, graph rows, or the LLM enrichment
12
- * cache. A newer layout is refused, naming the upgrade.
11
+ * touching embeddings, utility scores, or the LLM enrichment cache (the one
12
+ * exception is the LLM entity graph, unconditionally dropped since its
13
+ * retirement in 0.9.17-alpha.9). A newer layout is refused, naming the
14
+ * upgrade.
13
15
  */
14
16
  // 26: declared links (#935) live in `asset_links`, one row per link, owned by
15
17
  // the entry that declares it. The writable opener derives them in place from
@@ -4,11 +4,11 @@
4
4
  /**
5
5
  * `index.db` LLM enrichment-cache repository.
6
6
  *
7
- * Owns the raw SQL for `llm_enrichment_cache` — the body-hash-keyed cache that
8
- * lets `akm index --enrich` skip the LLM call when a file's body is unchanged.
7
+ * Owns the raw SQL for `llm_enrichment_cache` — the body-hash-keyed cache the
8
+ * graph-extraction and memory-inference passes use to skip an LLM call when
9
+ * a file's body is unchanged.
9
10
  */
10
11
  import { sha256Hex } from "../../runtime.js";
11
- import { escapeLikePattern } from "../like-pattern.js";
12
12
  import { SQLITE_CHUNK_SIZE } from "./index-sql.js";
13
13
  /**
14
14
  * Look up a cached LLM result for the given asset_ref.
@@ -18,7 +18,7 @@ import { SQLITE_CHUNK_SIZE } from "./index-sql.js";
18
18
  * cached). In both cases the caller should invoke the LLM and write a new
19
19
  * cache entry.
20
20
  */
21
- export function getLlmCacheEntry(db, assetRef, currentBodyHash, cacheVariant = "") {
21
+ export function getLlmCacheEntry(db, assetRef, currentBodyHash, cacheVariant) {
22
22
  const row = db
23
23
  .prepare("SELECT asset_ref, cache_variant, body_hash, result_json, updated_at FROM llm_enrichment_cache WHERE asset_ref = ? AND cache_variant = ?")
24
24
  .get(assetRef, cacheVariant);
@@ -44,7 +44,7 @@ export function getLlmCacheEntry(db, assetRef, currentBodyHash, cacheVariant = "
44
44
  * compare `entry.bodyHash` against the current body hash themselves. This lets
45
45
  * the batch path issue one DB query per chunk instead of one per file.
46
46
  */
47
- export function getLlmCacheEntriesByRefs(db, refs, cacheVariant = "") {
47
+ export function getLlmCacheEntriesByRefs(db, refs, cacheVariant) {
48
48
  const result = new Map();
49
49
  if (refs.length === 0)
50
50
  return result;
@@ -70,7 +70,7 @@ export function getLlmCacheEntriesByRefs(db, refs, cacheVariant = "") {
70
70
  /**
71
71
  * Insert or update a cached LLM result for the given asset_ref.
72
72
  */
73
- export function upsertLlmCacheEntry(db, assetRef, bodyHash, resultJson, cacheVariant = "") {
73
+ export function upsertLlmCacheEntry(db, assetRef, bodyHash, resultJson, cacheVariant) {
74
74
  db.prepare(`INSERT INTO llm_enrichment_cache (asset_ref, cache_variant, body_hash, result_json, updated_at)
75
75
  VALUES (?, ?, ?, ?, ?)
76
76
  ON CONFLICT(asset_ref, cache_variant) DO UPDATE SET
@@ -83,33 +83,14 @@ export function upsertLlmCacheEntry(db, assetRef, bodyHash, resultJson, cacheVar
83
83
  * `entries` table. Should be called during the cleanup phase of each index
84
84
  * run to prevent the cache from growing unboundedly as assets are removed.
85
85
  *
86
- * Graph/memory cache refs are absolute file paths, while metadata-enrichment
87
- * refs use canonical `item_ref`; preserve a cache row that matches either
88
- * current identity.
86
+ * Cache refs are absolute file paths (memory inference).
89
87
  */
90
88
  export function clearStaleCacheEntries(db) {
91
89
  db.exec(`
92
90
  DELETE FROM llm_enrichment_cache
93
91
  WHERE asset_ref NOT IN (SELECT file_path FROM entries)
94
- AND asset_ref NOT IN (SELECT item_ref FROM entries)
95
92
  `);
96
93
  }
97
- /**
98
- * Rewrite every `llm_enrichment_cache.asset_ref` naming `oldBundleId` (a
99
- * metadata-enrichment cache key, the canonical `<bundle>//conceptId` form of
100
- * `entries.item_ref`) to `newBundleId` (`akm bundle rename`, D6). Must run in
101
- * the same `index.db` write as `renameEntriesBundleId` — otherwise the next
102
- * `akm index`'s {@link clearStaleCacheEntries} deletes every row still keyed
103
- * to the old bundle, forcing a full LLM re-enrichment. A graph/memory cache
104
- * row (keyed by absolute file path, not `item_ref`) never matches the `//`
105
- * prefix and is left alone. Returns the number of rows rewritten.
106
- */
107
- export function renameLlmCacheAssetRefs(db, oldBundleId, newBundleId) {
108
- const prefix = `${escapeLikePattern(oldBundleId)}//`;
109
- return Number(db
110
- .prepare(`UPDATE llm_enrichment_cache SET asset_ref = ? || substr(asset_ref, ?) WHERE asset_ref LIKE ? ESCAPE '\\'`)
111
- .run(newBundleId, oldBundleId.length + 1, `${prefix}%`).changes);
112
- }
113
94
  /**
114
95
  * Compute a stable SHA-256 hex digest of a UTF-8 string. Used as the body_hash
115
96
  * key in `llm_enrichment_cache`. Routed through the runtime boundary so the
@@ -10,12 +10,15 @@
10
10
  * for columns added after a table first shipped, drops of retired derived
11
11
  * tables and columns, and one in-place rebuild of the (derived, cheap) FTS
12
12
  * table when its layout is older than this release's. It never drops
13
- * `entries`, `embeddings`, `utility_scores`, the extracted graph
14
- * (`graph_meta`, `graph_files`, `graph_file_*`), or `llm_enrichment_cache`
15
- * to cross a version boundary; the only from-scratch
16
- * rebuild is the SQLITE_CORRUPT path in `index-connection.ts`. A layout newer
17
- * than this release's is refused, naming the upgrade
18
- * ({@link newerIndexLayoutError}).
13
+ * `entries`, `embeddings`, `utility_scores`, or `llm_enrichment_cache` to
14
+ * cross a version boundary; the only from-scratch rebuild is the
15
+ * SQLITE_CORRUPT path in `index-connection.ts`. A layout newer than this
16
+ * release's is refused, naming the upgrade ({@link newerIndexLayoutError}).
17
+ * The one exception is the LLM entity graph (`graph_meta`, `graph_files`,
18
+ * `graph_file_*`), retired in 0.9.17-alpha.9: those tables are dropped
19
+ * unconditionally below (index.db is a regenerable cache, and declared links
20
+ * — `asset_links` — now back `akm show`'s `links` field, which replaced the
21
+ * graph's `related` list).
19
22
  */
20
23
  import { createRequire } from "node:module";
21
24
  import path from "node:path";
@@ -30,12 +33,6 @@ import { getMeta, setMeta } from "./index-meta-repository.js";
30
33
  export const DB_VERSION = CANONICAL_INDEX_DB_VERSION;
31
34
  /** `index_meta` key set when the writable opener migrated the layout; cleared once `akm index` VACUUMs. */
32
35
  export const VACUUM_PENDING_META = "vacuumPending";
33
- /**
34
- * The value written to `graph_meta.schema_version`, a NOT NULL column in every
35
- * released layout. Releases up to 0.9.17-alpha.5 write 4 and nothing ever
36
- * compares it; the index layout version gates the graph tables' shape.
37
- */
38
- export const GRAPH_SCHEMA_VERSION = 4;
39
36
  /** The layout that added declared links (`asset_links`, #935). */
40
37
  const DECLARED_LINKS_LAYOUT = 26;
41
38
  /**
@@ -69,97 +66,15 @@ const REGISTRY_INDEX_CACHE_DDL = `
69
66
  CREATE INDEX IF NOT EXISTS idx_registry_cache_fetched
70
67
  ON registry_index_cache(fetched_at);
71
68
  `;
72
- /**
73
- * Create the graph-extraction tables (`graph_meta`/`graph_files`/`graph_file_entities`/
74
- * `graph_file_relations`).
75
- *
76
- * graph_files is self-keyed on (stash_root, file_path, body_hash) and is not
77
- * tied to entries.id (#624-P1): re-upserting an entries row never disturbs the
78
- * extracted graph, and a content change yields a distinct key. A UNIQUE index
79
- * on (stash_root, file_path) still enforces one graph_files row per path.
80
- */
81
- function ensureGraphTables(db) {
82
- db.exec(`
83
- CREATE TABLE IF NOT EXISTS graph_meta (
84
- stash_root TEXT PRIMARY KEY,
85
- schema_version INTEGER NOT NULL,
86
- generated_at TEXT NOT NULL,
87
- considered_files INTEGER NOT NULL DEFAULT 0,
88
- extracted_files INTEGER NOT NULL DEFAULT 0,
89
- entity_count INTEGER NOT NULL DEFAULT 0,
90
- relation_count INTEGER NOT NULL DEFAULT 0,
91
- extraction_coverage REAL NOT NULL DEFAULT 0,
92
- density REAL NOT NULL DEFAULT 0,
93
- extractor_id TEXT,
94
- extraction_run_id TEXT,
95
- model TEXT,
96
- prompt_version TEXT,
97
- batch_size INTEGER,
98
- cache_hits INTEGER NOT NULL DEFAULT 0,
99
- cache_misses INTEGER NOT NULL DEFAULT 0,
100
- truncation_count INTEGER NOT NULL DEFAULT 0,
101
- failure_count INTEGER NOT NULL DEFAULT 0
102
- );
103
-
104
- CREATE TABLE IF NOT EXISTS graph_files (
105
- stash_root TEXT NOT NULL,
106
- file_path TEXT NOT NULL,
107
- file_order INTEGER NOT NULL,
108
- file_type TEXT NOT NULL,
109
- body_hash TEXT NOT NULL,
110
- confidence REAL,
111
- status TEXT NOT NULL DEFAULT 'extracted',
112
- reason TEXT,
113
- extraction_run_id TEXT,
114
- PRIMARY KEY (stash_root, file_path, body_hash)
115
- );
116
-
117
- CREATE UNIQUE INDEX IF NOT EXISTS idx_graph_files_path
118
- ON graph_files(stash_root, file_path);
119
-
120
- CREATE INDEX IF NOT EXISTS idx_graph_files_stash_order
121
- ON graph_files(stash_root, file_order);
122
-
123
- CREATE TABLE IF NOT EXISTS graph_file_entities (
124
- stash_root TEXT NOT NULL,
125
- file_path TEXT NOT NULL,
126
- body_hash TEXT NOT NULL,
127
- entity_order INTEGER NOT NULL,
128
- entity_norm TEXT NOT NULL,
129
- entity TEXT NOT NULL,
130
- PRIMARY KEY (stash_root, file_path, body_hash, entity_order),
131
- FOREIGN KEY (stash_root, file_path, body_hash)
132
- REFERENCES graph_files(stash_root, file_path, body_hash) ON DELETE CASCADE
133
- );
134
-
135
- CREATE INDEX IF NOT EXISTS idx_graph_file_entities_entity_norm
136
- ON graph_file_entities(stash_root, entity_norm);
137
-
138
- CREATE TABLE IF NOT EXISTS graph_file_relations (
139
- stash_root TEXT NOT NULL,
140
- file_path TEXT NOT NULL,
141
- body_hash TEXT NOT NULL,
142
- relation_order INTEGER NOT NULL,
143
- from_entity_norm TEXT NOT NULL,
144
- from_entity TEXT NOT NULL,
145
- to_entity_norm TEXT NOT NULL,
146
- to_entity TEXT NOT NULL,
147
- relation_type TEXT,
148
- confidence REAL,
149
- PRIMARY KEY (stash_root, file_path, body_hash, relation_order),
150
- FOREIGN KEY (stash_root, file_path, body_hash)
151
- REFERENCES graph_files(stash_root, file_path, body_hash) ON DELETE CASCADE
152
- );
153
- `);
154
- }
155
69
  /**
156
70
  * An `entries` table missing a required column cannot be read or written by
157
71
  * this release (the last such change was v20→v21, which removed the
158
72
  * transitional `entry_key`/`dir_path`/... columns and made `item_ref` the
159
73
  * key). Recreate only the tables keyed by `entries.id` — their ids are about
160
- * to be re-minted, so the rows would dangle anyway. Graph rows (keyed by
161
- * path) and the LLM enrichment cache (keyed by ref) are kept. The next index
162
- * run re-walks every source.
74
+ * to be re-minted, so the rows would dangle anyway. The LLM enrichment cache
75
+ * (keyed by ref) is kept. The LLM entity-graph tables are unconditionally
76
+ * dropped elsewhere in this file regardless of this recreation (retired
77
+ * 0.9.17-alpha.9), not kept. The next index run re-walks every source.
163
78
  */
164
79
  function ensureEntriesLayout(db) {
165
80
  if (!tableExists(db, "entries"))
@@ -168,8 +83,8 @@ function ensureEntriesLayout(db) {
168
83
  if (missing.length === 0)
169
84
  return;
170
85
  warn(`Index database entries table predates the ${missing.join(", ")} column${missing.length === 1 ? "" : "s"} — ` +
171
- "recreating the entries-keyed tables (entries, full-text, embeddings, utility scores); graph data and the " +
172
- "LLM enrichment cache are kept. The next index run re-walks every source.");
86
+ "recreating the entries-keyed tables (entries, full-text, embeddings, utility scores); the " +
87
+ "LLM enrichment cache is kept. The next index run re-walks every source.");
173
88
  db.transaction(() => {
174
89
  for (const table of [
175
90
  "entries_fts",
@@ -246,7 +161,7 @@ function ensureFtsLayout(db) {
246
161
  const entryCount = Number(db.prepare("SELECT COUNT(*) AS n FROM entries").get().n);
247
162
  if (entryCount > 0) {
248
163
  warn(`Rebuilding the full-text index for ${entryCount} entr${entryCount === 1 ? "y" : "ies"} ` +
249
- "(embeddings, utility scores, graph data and the LLM enrichment cache are kept).");
164
+ "(embeddings, utility scores, and the LLM enrichment cache are kept).");
250
165
  }
251
166
  db.transaction(() => {
252
167
  db.exec("DROP TABLE IF EXISTS entries_fts");
@@ -311,6 +226,25 @@ export function ensureSchema(db) {
311
226
  db.exec("DROP TABLE IF EXISTS entry_fragments_fts");
312
227
  db.exec("DROP TABLE IF EXISTS utility_scores_scoped");
313
228
  db.exec("DROP TABLE IF EXISTS graph_extraction_queue");
229
+ // The LLM entity graph, retired in 0.9.17-alpha.9: declared links
230
+ // (`asset_links`) now back `akm show`'s `links` field (which replaced the
231
+ // graph's `related` list) and curate's support refs (#935), and the
232
+ // navigation eval measured vector kNN beating the graph's `related` list
233
+ // by 0.157 P@5. `graph_files` stands in for the whole set — all four
234
+ // tables are only ever created and dropped together. Gated on it (rather
235
+ // than the unconditional `DROP TABLE IF EXISTS` pattern used above) so
236
+ // this reclaim runs once: after the first writable open drops these
237
+ // tables, every later open finds `graph_files` already gone and skips the
238
+ // no-op DROPs and the repeat VACUUM flag below. An older release's
239
+ // `CREATE TABLE IF NOT EXISTS` still recreates them (empty) if it ever
240
+ // opens this index again — a later open here would then drop them again.
241
+ const hadGraphTables = tableExists(db, "graph_files");
242
+ if (hadGraphTables) {
243
+ db.exec("DROP TABLE IF EXISTS graph_meta");
244
+ db.exec("DROP TABLE IF EXISTS graph_files");
245
+ db.exec("DROP TABLE IF EXISTS graph_file_entities");
246
+ db.exec("DROP TABLE IF EXISTS graph_file_relations");
247
+ }
314
248
  // One float32 BLOB per entry, searched by an exact scan
315
249
  // (index-vec-repository.ts). `model` is the provider fingerprint the vector was generated under
316
250
  // (`deriveSemanticProviderFingerprint`); the embedding pass re-embeds only
@@ -363,8 +297,8 @@ export function ensureSchema(db) {
363
297
  ensureColumn(db, "index_dir_state", "row_count", "INTEGER");
364
298
  ensureColumn(db, "index_dir_state", "index_variant", "TEXT");
365
299
  // LLM enrichment result cache, keyed by a stable asset_ref string (the
366
- // absolute file path for graph/memory passes, `item_ref` for the
367
- // metadata-enhance pass) plus the body hash the result was produced for.
300
+ // absolute file path of the memory-inference pass) plus the body hash the
301
+ // result was produced for.
368
302
  db.exec(`
369
303
  CREATE TABLE IF NOT EXISTS llm_enrichment_cache (
370
304
  asset_ref TEXT NOT NULL,
@@ -378,7 +312,24 @@ export function ensureSchema(db) {
378
312
  CREATE INDEX IF NOT EXISTS idx_llm_cache_updated
379
313
  ON llm_enrichment_cache(updated_at);
380
314
  `);
381
- ensureGraphTables(db);
315
+ // Metadata-enhance retired (RS-D, 0.9.17-alpha.9): its rows were the only
316
+ // ones keyed by the default empty cache_variant (memory inference writes
317
+ // `memory-inference-v2`), so this is safe to run unconditionally on every
318
+ // writable open. The table
319
+ // itself stays — memory inference still reads it.
320
+ db.exec("DELETE FROM llm_enrichment_cache WHERE cache_variant = ''");
321
+ // The graph-extraction cache variant is retired along with the tables
322
+ // above; its rows would otherwise sit unread forever. Gated the same way,
323
+ // on the same one-time flag, so a rerun does not re-scan the cache table
324
+ // for rows that are already gone.
325
+ if (hadGraphTables) {
326
+ db.exec("DELETE FROM llm_enrichment_cache WHERE cache_variant LIKE 'graph-extraction:%'");
327
+ // The drops and delete above freed real space (measured ~68MB on a
328
+ // representative index): flag it the same way a version-gated layout
329
+ // migration does, since this reclaim is unconditional-on-version but
330
+ // still one-time-per-index (guarded by hadGraphTables above).
331
+ setMeta(db, VACUUM_PENDING_META, "1");
332
+ }
382
333
  dropVecMirror(db);
383
334
  // Meta keys only the sqlite-vec mirror read.
384
335
  db.exec("DELETE FROM index_meta WHERE key IN ('embeddingDim', 'vecFastPathReady')");