akm-cli 0.9.17-alpha.8 → 0.9.17-alpha.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/CHANGELOG.md +334 -0
  2. package/STABILITY.md +9 -8
  3. package/dist/assets/hints/cli-hints-full.md +6 -7
  4. package/dist/assets/improve-strategies/catchup.json +0 -3
  5. package/dist/assets/improve-strategies/consolidate.json +0 -1
  6. package/dist/assets/improve-strategies/default.json +1 -2
  7. package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
  8. package/dist/assets/improve-strategies/quick.json +1 -2
  9. package/dist/assets/improve-strategies/reflect-distill.json +1 -2
  10. package/dist/assets/improve-strategies/thorough.json +0 -3
  11. package/dist/assets/prompts/consolidate-pair.md +20 -0
  12. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +17 -19
  13. package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
  14. package/dist/assets/templates/html/health.html +3 -5
  15. package/dist/cli/retired-commands.js +1 -1
  16. package/dist/commands/health/archive-usage.js +98 -0
  17. package/dist/commands/health/data-dir-usage.js +25 -13
  18. package/dist/commands/health/html-report.js +1 -4
  19. package/dist/commands/health/improve-metrics.js +0 -25
  20. package/dist/commands/health/md-report.js +1 -6
  21. package/dist/commands/health/report-view-model.js +4 -14
  22. package/dist/commands/health/windows.js +0 -1
  23. package/dist/commands/health.js +13 -0
  24. package/dist/commands/improve/consolidate/continuity-check.js +137 -0
  25. package/dist/commands/improve/consolidate/pair-pass.js +791 -0
  26. package/dist/commands/improve/consolidate.js +38 -63
  27. package/dist/commands/improve/extract-prompt.js +1 -2
  28. package/dist/commands/improve/improve-cli.js +1 -1
  29. package/dist/commands/improve/improve-strategies.js +23 -5
  30. package/dist/commands/improve/improve.js +19 -30
  31. package/dist/commands/improve/ledger.js +3 -2
  32. package/dist/commands/improve/loop-stages.js +5 -85
  33. package/dist/commands/improve/memory/memory-belief.js +3 -1
  34. package/dist/commands/improve/memory/memory-improve.js +269 -11
  35. package/dist/commands/improve/planner.js +0 -5
  36. package/dist/commands/improve/preparation.js +20 -135
  37. package/dist/commands/improve/retrieval-scope.js +19 -4
  38. package/dist/commands/improve/salience.js +1 -14
  39. package/dist/commands/improve/stage.js +0 -1
  40. package/dist/commands/lint/base-linter.js +19 -11
  41. package/dist/commands/proposal/drain.js +8 -1
  42. package/dist/commands/proposal/proposal-cli.js +16 -2
  43. package/dist/commands/proposal/proposal-types.js +7 -0
  44. package/dist/commands/proposal/proposal.js +37 -6
  45. package/dist/commands/proposal/repository.js +613 -4
  46. package/dist/commands/proposal/validators/proposals.js +9 -0
  47. package/dist/commands/read/knowledge.js +3 -2
  48. package/dist/commands/read/show.js +0 -14
  49. package/dist/commands/sources/stash-cli.js +2 -2
  50. package/dist/core/bundle-rename.js +1 -7
  51. package/dist/core/config/config-schema.js +8 -1
  52. package/dist/core/config/config.js +23 -48
  53. package/dist/core/config/engine-semantics.js +0 -2
  54. package/dist/core/config/schema/improve-processes.js +17 -42
  55. package/dist/core/config/schema/index-config.js +5 -25
  56. package/dist/core/file-change.js +13 -5
  57. package/dist/core/improve-result.js +16 -5
  58. package/dist/core/improve-types.js +0 -1
  59. package/dist/core/loopback.js +7 -12
  60. package/dist/core/parse.js +13 -16
  61. package/dist/core/state/migrations.js +15 -0
  62. package/dist/core/time.js +0 -20
  63. package/dist/indexer/db/llm-cache.js +2 -2
  64. package/dist/indexer/ensure-index.js +2 -2
  65. package/dist/indexer/index-written-assets.js +2 -3
  66. package/dist/indexer/indexer.js +18 -418
  67. package/dist/indexer/passes/metadata.js +0 -19
  68. package/dist/indexer/walk/walker.js +3 -4
  69. package/dist/llm/client.js +8 -10
  70. package/dist/llm/embedders/remote.js +1 -2
  71. package/dist/llm/feature-gate.js +0 -5
  72. package/dist/output/shapes/helpers.js +20 -4
  73. package/dist/output/text/command-format.js +0 -8
  74. package/dist/output/text/proposal-format.js +47 -1
  75. package/dist/output/text/show-format.js +0 -20
  76. package/dist/scripts/akm-migrate-node.js +917 -950
  77. package/dist/scripts/akm-migrate.js +917 -950
  78. package/dist/setup/steps/connection.js +5 -6
  79. package/dist/setup/steps/platforms.js +2 -2
  80. package/dist/sources/providers/git-stash.js +55 -4
  81. package/dist/storage/repositories/improve-ledger-repository.js +48 -7
  82. package/dist/storage/repositories/index-entries-repository.js +4 -7
  83. package/dist/storage/repositories/index-entry-schema.js +4 -2
  84. package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
  85. package/dist/storage/repositories/index-schema.js +55 -104
  86. package/dist/storage/repositories/proposals-repository.js +61 -0
  87. package/dist/storage/repositories/salience-repository.js +1 -19
  88. package/docs/reference/cli.md +16 -19
  89. package/docs/reference/configuration.md +21 -12
  90. package/docs/reference/data-and-telemetry.md +0 -1
  91. package/package.json +1 -1
  92. package/schemas/akm-config.json +0 -342
  93. package/dist/assets/improve-strategies/graph-refresh.json +0 -15
  94. package/dist/assets/prompts/contradiction-judge.md +0 -33
  95. package/dist/assets/prompts/graph-extract-system.md +0 -1
  96. package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
  97. package/dist/assets/prompts/metadata-enhance-system.md +0 -1
  98. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
  99. package/dist/indexer/db/graph-db.js +0 -399
  100. package/dist/indexer/graph/graph-extraction.js +0 -809
  101. package/dist/indexer/graph/graph-related.js +0 -131
  102. package/dist/indexer/graph/graph-types.js +0 -4
  103. package/dist/llm/graph-extract.js +0 -892
  104. package/dist/llm/metadata-enhance.js +0 -95
@@ -325,7 +325,7 @@
325
325
 
326
326
  <div class="chart-panel full-width">
327
327
  <h3>Per-Phase Wall Time Decomposition (stacked, min)</h3>
328
- <div id="chartPhases" class="ec tall" role="img" aria-label="Stacked area chart decomposing wall time per run into consolidation, memory inference, graph extraction, and unattributed phases"></div>
328
+ <div id="chartPhases" class="ec tall" role="img" aria-label="Stacked area chart decomposing wall time per run into consolidation, memory inference, and unattributed phases"></div>
329
329
  <div class="empty-overlay" id="emptyPhases">No runs in the selected slice.</div>
330
330
  </div>
331
331
 
@@ -390,7 +390,7 @@
390
390
  <div class="table-wrap">
391
391
  <table>
392
392
  <thead>
393
- <tr><th>Started</th><th>Task</th><th>Strategy</th><th>Wall</th><th>Promoted</th><th>Merged</th><th>Contradicted</th><th>MI Written</th><th>Entities</th><th>Lint Fixed</th><th>Status</th></tr>
393
+ <tr><th>Started</th><th>Task</th><th>Strategy</th><th>Wall</th><th>Promoted</th><th>Merged</th><th>Contradicted</th><th>MI Written</th><th>Lint Fixed</th><th>Status</th></tr>
394
394
  </thead>
395
395
  <tbody id="lastRunsTable"></tbody>
396
396
  </table>
@@ -539,7 +539,7 @@ mkChart('chartWallTime', {
539
539
  // ── 2. Per-phase wall time decomposition (stacked area) ──────────────────────
540
540
  mkChart('chartPhases', {
541
541
  tooltip: baseTooltip,
542
- legend: { ...baseLegend, data: ['Consolidation', 'Memory inference', 'Graph extraction', 'Unattributed'] },
542
+ legend: { ...baseLegend, data: ['Consolidation', 'Memory inference', 'Unattributed'] },
543
543
  grid: denseGrid,
544
544
  dataZoom: denseZoom,
545
545
  xAxis: { type: 'category', data: xLabels, ...baseAxis },
@@ -547,7 +547,6 @@ mkChart('chartPhases', {
547
547
  series: [
548
548
  { name: 'Consolidation', type: 'line', stack: 'p', areaStyle: { color: 'rgba(88,166,255,0.55)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.consDurationMs)) },
549
549
  { name: 'Memory inference', type: 'line', stack: 'p', areaStyle: { color: 'rgba(63,185,80,0.55)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.miDurationMs)) },
550
- { name: 'Graph extraction', type: 'line', stack: 'p', areaStyle: { color: 'rgba(188,140,255,0.55)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.geDurationMs)) },
551
550
  { name: 'Unattributed', type: 'line', stack: 'p', areaStyle: { color: 'rgba(210,153,34,0.45)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.otherMs)) },
552
551
  ],
553
552
  });
@@ -695,7 +694,6 @@ tbody.innerHTML = '';
695
694
  <td>${r.merged || 0}</td>
696
695
  <td style="color:${(r.contradicted||0)>5?'var(--yellow)':'var(--text)'};">${r.contradicted || 0}</td>
697
696
  <td style="color:var(--green);">${r.miWritten || 0}</td>
698
- <td>${r.geEntities || 0}</td>
699
697
  <td>${r.lintFixed || 0}</td>
700
698
  <td>${badge}</td>
701
699
  `;
@@ -38,7 +38,7 @@ const RETIRED_COMMAND_HINTS = {
38
38
  events: "`akm events` moved in 0.9 — use `akm log`.",
39
39
  // Removed observability surfaces.
40
40
  history: "`akm history` was removed in 0.9 — use `akm log --ref <ref>` for an asset's event trail.",
41
- graph: "`akm graph` was removed in 0.9 — graph counts appear in `akm health`; refresh extraction with `akm improve --strategy graph-refresh`.",
41
+ graph: "`akm graph` was removed in 0.9. The LLM entity graph it inspected was itself retired in 0.9.17-alpha.9 — use `akm show <ref>` for an asset's declared links.",
42
42
  lessons: "`akm lessons` was removed in 0.9 — lesson strength is indexed; use `akm search --type lesson`.",
43
43
  lesson: "`akm lesson` was removed in 0.9 — lesson strength is indexed; use `akm search --type lesson`.",
44
44
  // Relocated guidance / removed asset verbs.
@@ -0,0 +1,98 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * `memory-cleanup-archive` advisory for `akm health` (item 4, 0.9.17-alpha.9
6
+ * plan §5.4/§8 step 8).
7
+ *
8
+ * Reports the archive's size and file count for every bundle, git-backed or
9
+ * not. A bundle with no `.git` of its own keeps every retirement's archived
10
+ * bytes forever (the purge sweep never runs there at all), so that alone is
11
+ * reported. A git-backed bundle can ALSO have bytes the purge sweep will
12
+ * never remove: `.git` presence alone does not prove a retirement was ever
13
+ * committed (`proposal accept` only commits for a `kind: "git"` write
14
+ * target, and a `kind: "filesystem"` bundle that merely happens to have a
15
+ * `.git` directory never gets one) — those bytes sit there indefinitely,
16
+ * however old they get, with nothing else to say so. This checks the SAME
17
+ * git state `purgeGracedArchive` checks (tracked, clean, verifiable) and
18
+ * reports how much of the archive fails it.
19
+ *
20
+ * Silent whenever there is nothing to say: the archive does not exist or is
21
+ * empty, or (for a git-backed bundle) every byte in it is purgeable once it
22
+ * ages out.
23
+ */
24
+ import fs from "node:fs";
25
+ import path from "node:path";
26
+ import { MEMORY_ARCHIVE_REL } from "../../core/asset/memory-archive.js";
27
+ import { toPosix } from "../../core/common.js";
28
+ import { isGitBackedStash, tryListGitChangedPaths, tryListGitTrackedPaths, tryListGitUnverifiablePaths, } from "../../sources/providers/git-stash.js";
29
+ import { MAX_WALK_ENTRIES, sizeOfPath } from "./data-dir-usage.js";
30
+ /**
31
+ * Build the `memory-cleanup-archive` advisory, or `undefined` when there is
32
+ * nothing to report.
33
+ */
34
+ export function collectArchiveUsageAdvisory(stashDir) {
35
+ const archiveRoot = path.join(stashDir, MEMORY_ARCHIVE_REL);
36
+ if (!fs.existsSync(archiveRoot))
37
+ return undefined;
38
+ if (!isGitBackedStash(stashDir)) {
39
+ const usage = sizeOfPath(archiveRoot, { remaining: MAX_WALK_ENTRIES });
40
+ if (usage.files === 0)
41
+ return undefined;
42
+ const lowerBound = usage.truncated ? ` (a lower bound — the walk stopped after ${MAX_WALK_ENTRIES} entries)` : "";
43
+ return {
44
+ name: "memory-cleanup-archive",
45
+ kind: "deterministic",
46
+ status: "pass",
47
+ confidence: "high",
48
+ message: `${usage.files} archived file(s), ${usage.bytes} byte(s)${lowerBound} in .akm/memory-cleanup/archive — ` +
49
+ "this bundle has no git history, so the purge sweep leaves it untouched.",
50
+ evidence: { files: usage.files, bytes: usage.bytes, truncated: usage.truncated },
51
+ };
52
+ }
53
+ // Git-backed: the SAME three checks purgeGracedArchive runs (B1, G10) —
54
+ // computed once here, not per file, and reused via `onFile` below instead
55
+ // of a second walk of the same tree.
56
+ const dirtyQuery = tryListGitChangedPaths(stashDir);
57
+ const trackedQuery = tryListGitTrackedPaths(stashDir, MEMORY_ARCHIVE_REL);
58
+ const unverifiableQuery = tryListGitUnverifiablePaths(stashDir, MEMORY_ARCHIVE_REL);
59
+ // A failed git check here fails the same way purgeGracedArchive's own
60
+ // sweep would: nothing in the archive can be proven purgeable, so every
61
+ // byte counts as unpurgeable rather than guessing.
62
+ const gitStateKnown = dirtyQuery.ok && trackedQuery.ok && unverifiableQuery.ok;
63
+ const dirty = new Set(dirtyQuery.paths);
64
+ const tracked = new Set(trackedQuery.paths);
65
+ const unverifiable = new Set(unverifiableQuery.paths);
66
+ let unpurgeableFiles = 0;
67
+ let unpurgeableBytes = 0;
68
+ const usage = sizeOfPath(archiveRoot, { remaining: MAX_WALK_ENTRIES }, (filePath, bytes) => {
69
+ const key = toPosix(path.relative(stashDir, filePath));
70
+ const safe = gitStateKnown && tracked.has(key) && !dirty.has(key) && !unverifiable.has(key);
71
+ if (!safe) {
72
+ unpurgeableFiles++;
73
+ unpurgeableBytes += bytes;
74
+ }
75
+ });
76
+ if (usage.files === 0)
77
+ return undefined;
78
+ if (unpurgeableFiles === 0)
79
+ return undefined; // everything here is purgeable once it ages out — nothing to say
80
+ const lowerBound = usage.truncated ? ` (a lower bound — the walk stopped after ${MAX_WALK_ENTRIES} entries)` : "";
81
+ return {
82
+ name: "memory-cleanup-archive",
83
+ kind: "deterministic",
84
+ status: "warn",
85
+ confidence: "high",
86
+ message: `${usage.files} archived file(s), ${usage.bytes} byte(s)${lowerBound} in .akm/memory-cleanup/archive; ` +
87
+ `${unpurgeableFiles} file(s), ${unpurgeableBytes} byte(s) of that cannot be purged (untracked, modified, ` +
88
+ "or unverifiable in git) — commit them so the purge sweep can remove them once they age out.",
89
+ evidence: {
90
+ files: usage.files,
91
+ bytes: usage.bytes,
92
+ truncated: usage.truncated,
93
+ unpurgeableFiles,
94
+ unpurgeableBytes,
95
+ gitStateKnown,
96
+ },
97
+ };
98
+ }
@@ -41,7 +41,7 @@ const DOMINANT_SUBDIR_PERCENT_THRESHOLD = 50;
41
41
  * cap the walk stops descending further and the advisory says its size
42
42
  * figures are a lower bound.
43
43
  */
44
- const MAX_WALK_ENTRIES = 100_000;
44
+ export const MAX_WALK_ENTRIES = 100_000;
45
45
  /**
46
46
  * Below this the data dir is not worth an opinion. The advisory exists for
47
47
  * disk blowups (74 GB in the incident); on a small directory a ratio is
@@ -59,31 +59,42 @@ function liveDbBytesFor(name, sizes) {
59
59
  return ((sizes.get(name)?.bytes ?? 0) + (sizes.get(`${name}-wal`)?.bytes ?? 0) + (sizes.get(`${name}-shm`)?.bytes ?? 0));
60
60
  }
61
61
  /**
62
- * Recursively sum file sizes under `root` (stat-only, symlinks not
63
- * followed so a cyclic or huge-target symlink can't blow up the walk).
64
- * `budget` is a shared mutable counter across the whole tree so the
65
- * `MAX_WALK_ENTRIES` cap applies to the walk as a whole, not per-branch.
62
+ * Recursively sum file sizes (and count files) under `root` (stat-only,
63
+ * symlinks not followed so a cyclic or huge-target symlink can't blow up the
64
+ * walk). `budget` is a shared mutable counter across the whole tree so the
65
+ * entry cap applies to the walk as a whole, not per-branch — callers
66
+ * typically pass {@link MAX_WALK_ENTRIES}, sized for this module's own data
67
+ * dir walk, but a smaller/larger budget is fine for a different tree.
68
+ * Shared with the `archive-usage` advisory (N2) — the same "don't let a
69
+ * pathological tree hang a health check" concern applies to both.
70
+ *
71
+ * `onFile`, when given, is called once per leaf file (path, bytes) as the
72
+ * walk visits it — `archive-usage` uses this to classify each file's git
73
+ * state without a second, separate walk of the same tree.
66
74
  */
67
- function sizeOfPath(root, budget) {
75
+ export function sizeOfPath(root, budget, onFile) {
68
76
  let stat;
69
77
  try {
70
78
  stat = fs.lstatSync(root);
71
79
  }
72
80
  catch {
73
- return { bytes: 0, truncated: false };
81
+ return { bytes: 0, files: 0, truncated: false };
74
82
  }
75
83
  if (stat.isSymbolicLink())
76
- return { bytes: 0, truncated: false };
77
- if (!stat.isDirectory())
78
- return { bytes: stat.size, truncated: false };
84
+ return { bytes: 0, files: 0, truncated: false };
85
+ if (!stat.isDirectory()) {
86
+ onFile?.(root, stat.size);
87
+ return { bytes: stat.size, files: 1, truncated: false };
88
+ }
79
89
  let entries;
80
90
  try {
81
91
  entries = fs.readdirSync(root, { withFileTypes: true });
82
92
  }
83
93
  catch {
84
- return { bytes: 0, truncated: false };
94
+ return { bytes: 0, files: 0, truncated: false };
85
95
  }
86
96
  let bytes = 0;
97
+ let files = 0;
87
98
  let truncated = false;
88
99
  for (const entry of entries) {
89
100
  if (budget.remaining <= 0) {
@@ -91,12 +102,13 @@ function sizeOfPath(root, budget) {
91
102
  break;
92
103
  }
93
104
  budget.remaining--;
94
- const sub = sizeOfPath(path.join(root, entry.name), budget);
105
+ const sub = sizeOfPath(path.join(root, entry.name), budget, onFile);
95
106
  bytes += sub.bytes;
107
+ files += sub.files;
96
108
  if (sub.truncated)
97
109
  truncated = true;
98
110
  }
99
- return { bytes, truncated };
111
+ return { bytes, files, truncated };
100
112
  }
101
113
  /** `1610612736` -> `"1.5G"`. Values under 10 in a unit keep one decimal; 10+ round to an integer. */
102
114
  function formatBytes(bytes) {
@@ -131,7 +131,6 @@ function renderExecSummary(vm) {
131
131
  const trendRows = [
132
132
  trendLi("Decision quality", vm.trend.decisionQuality),
133
133
  trendLi("Output volume", vm.trend.outputVolume),
134
- trendLi("Failures", vm.trend.failures),
135
134
  trendLi("Latency", vm.trend.latency),
136
135
  ].join("");
137
136
  const deltaRows = [
@@ -151,7 +150,6 @@ function renderExecSummary(vm) {
151
150
  li("Promoted", String(vm.latest.promoted)),
152
151
  li("Judged: no action", `<abbr title="Candidates reviewed but intentionally left unchanged on this run.">${vm.latest.judgedNoAction}</abbr>`),
153
152
  li("MI written", String(vm.latest.miWritten)),
154
- li("Graph entities/relations", `${vm.latest.geEntities} / ${vm.latest.geRelations}`),
155
153
  ].join("")
156
154
  : '<li><span class="k">No runs in window</span><span class="v">—</span></li>';
157
155
  const windowRows = [
@@ -197,7 +195,7 @@ function renderExecSummary(vm) {
197
195
  </div>
198
196
  </div>
199
197
  <div class="overall">Overall trend: <b>${esc(vm.trend.overall)}</b> ${overallEmoji}
200
- &nbsp;·&nbsp; based on decision quality, output volume, failures, and latency ${vm.comparisonMode === "custom" ? "across the selected windows" : "vs the prior window"}.</div>`.trim();
198
+ &nbsp;·&nbsp; based on decision quality, output volume, and latency ${vm.comparisonMode === "custom" ? "across the selected windows" : "vs the prior window"}.</div>`.trim();
201
199
  }
202
200
  /**
203
201
  * Color is a health SIGNAL, not decoration: green/yellow/red where a card has
@@ -221,7 +219,6 @@ function renderKpiCards(vm) {
221
219
  kpiCard("neutral", "Median Duration", `${vm.medianDurMin}m`, `p95 = ${vm.p95DurMin}m`),
222
220
  kpiCard("blue", "Total Promoted", num(vm.consolidation.promoted), `avg ${vm.avgPromoted} / run`),
223
221
  kpiCard("blue", "MI Written", num(vm.miWritten), `${vm.miYieldRate} yield rate`),
224
- kpiCard("purple", "Graph Entities", num(vm.graphExtraction.entities), `+${num(vm.graphExtraction.relations)} relations`),
225
222
  kpiCard("neutral", "Stash Derived", num(vm.memorySummary.derived), `of ${num(vm.memorySummary.eligible)} eligible (whole-stash)`),
226
223
  // #576: real LLM work — duration leads, tokens compact, not a GPU proxy.
227
224
  kpiCard(vm.llm.calls > 0 ? "purple" : "neutral", "🧠 LLM Work", `${vm.llmTokensCompact} tok`, `${fmtMs(vm.llm.totalDurationMs)} · ${num(vm.llm.calls)} calls · ${compact(vm.llm.reasoningTokens)} reasoning`),
@@ -74,7 +74,6 @@ export function emptyImproveMetrics() {
74
74
  },
75
75
  memoryPrune: 0,
76
76
  memoryInference: 0,
77
- graphExtraction: 0,
78
77
  error: 0,
79
78
  },
80
79
  autoAccept: { promoted: 0, validationFailed: 0 },
@@ -91,7 +90,6 @@ export function emptyImproveMetrics() {
91
90
  durationMs: 0,
92
91
  },
93
92
  memoryInference: { considered: 0, freshAttempts: 0, written: 0, skippedNoFacts: 0, yieldRate: 0, durationMs: 0 },
94
- graphExtraction: { extractedFiles: 0, entities: 0, relations: 0, failures: 0, durationMs: 0 },
95
93
  wallTime: { medianMs: 0, p95Ms: 0 },
96
94
  coverage: { acceptedProposals: 0, distinctRefs: 0 },
97
95
  };
@@ -151,9 +149,6 @@ function applyAction(metrics, action) {
151
149
  case "memory-inference":
152
150
  metrics.actions.memoryInference += 1;
153
151
  break;
154
- case "graph-extraction":
155
- metrics.actions.graphExtraction += 1;
156
- break;
157
152
  case "error":
158
153
  metrics.actions.error += 1;
159
154
  break;
@@ -209,19 +204,6 @@ function projectRunMetrics(result) {
209
204
  mi.skippedNoFacts += toFiniteNumber(memoryInference.skippedNoFacts);
210
205
  }
211
206
  metrics.memoryInference.durationMs += toFiniteNumber(result.memoryInferenceDurationMs);
212
- const graphExtraction = result.graphExtraction;
213
- if (graphExtraction) {
214
- const ge = metrics.graphExtraction;
215
- // This run's counts, like entities and relations. `quality` describes the
216
- // whole stored graph after the run; summing it would count each stored file
217
- // once per run in the window.
218
- ge.extractedFiles += toFiniteNumber(graphExtraction.extracted);
219
- ge.entities += toFiniteNumber(graphExtraction.totalEntities);
220
- ge.relations += toFiniteNumber(graphExtraction.totalRelations);
221
- const telemetry = graphExtraction.telemetry;
222
- ge.failures += toFiniteNumber(telemetry?.failureCount);
223
- }
224
- metrics.graphExtraction.durationMs += toFiniteNumber(result.graphExtractionDurationMs);
225
207
  return metrics;
226
208
  }
227
209
  /** Derived rates on an accumulator (window aggregate or single run). */
@@ -251,7 +233,6 @@ function mergeImproveMetrics(dst, src) {
251
233
  }
252
234
  dst.actions.memoryPrune += src.actions.memoryPrune;
253
235
  dst.actions.memoryInference += src.actions.memoryInference;
254
- dst.actions.graphExtraction += src.actions.graphExtraction;
255
236
  dst.actions.error += src.actions.error;
256
237
  dst.autoAccept.promoted += src.autoAccept.promoted;
257
238
  dst.autoAccept.validationFailed += src.autoAccept.validationFailed;
@@ -269,11 +250,6 @@ function mergeImproveMetrics(dst, src) {
269
250
  dst.memoryInference.written += src.memoryInference.written;
270
251
  dst.memoryInference.skippedNoFacts += src.memoryInference.skippedNoFacts;
271
252
  dst.memoryInference.durationMs += src.memoryInference.durationMs;
272
- dst.graphExtraction.extractedFiles += src.graphExtraction.extractedFiles;
273
- dst.graphExtraction.entities += src.graphExtraction.entities;
274
- dst.graphExtraction.relations += src.graphExtraction.relations;
275
- dst.graphExtraction.failures += src.graphExtraction.failures;
276
- dst.graphExtraction.durationMs += src.graphExtraction.durationMs;
277
253
  }
278
254
  function compareImproveRunRecency(a, b) {
279
255
  const started = a.started_at.localeCompare(b.started_at);
@@ -354,7 +330,6 @@ export function projectImproveRunSummary(row, wallTimeMs, taskId) {
354
330
  memorySummary: perRow.memorySummary,
355
331
  consolidation: perRow.consolidation,
356
332
  memoryInference: perRow.memoryInference,
357
- graphExtraction: perRow.graphExtraction,
358
333
  orphansPurged: toFiniteNumber(result.orphansPurged),
359
334
  lintFixed: lintSummary ? toFiniteNumber(lintSummary.fixed) : 0,
360
335
  lintFlagged: lintSummary ? toFiniteNumber(lintSummary.flagged) : 0,
@@ -20,7 +20,7 @@ function renderTable(headers, rows) {
20
20
  *
21
21
  * Columns: ts | ok | actions | refl_ok/fail/cd/skip |
22
22
  * distill_q/llm-fail/qrej/cfg/skip | cons_proc/promo/merge/del |
23
- * mem_cons/written/skip | graph_f/e/r | orphans | lint_f/fl
23
+ * mem_cons/written/skip | orphans | lint_f/fl
24
24
  */
25
25
  export function renderRunsDetailMd(runs) {
26
26
  const headers = [
@@ -32,7 +32,6 @@ export function renderRunsDetailMd(runs) {
32
32
  "distill_q/llm-fail/judge/validator/cfg/skip",
33
33
  "cons_proc/promo/merge/del",
34
34
  "mem_cons/written/skip",
35
- "graph_f/e/r",
36
35
  "orphans",
37
36
  "lint_f/fl",
38
37
  "result_status",
@@ -50,7 +49,6 @@ export function renderRunsDetailMd(runs) {
50
49
  r.actions.distill.skipped +
51
50
  r.actions.memoryPrune +
52
51
  r.actions.memoryInference +
53
- r.actions.graphExtraction +
54
52
  r.actions.error;
55
53
  return [
56
54
  r.startedAt,
@@ -61,7 +59,6 @@ export function renderRunsDetailMd(runs) {
61
59
  `${r.actions.distill.queued}/${r.actions.distill.llmFailed}/${r.actions.distill.judgeRejected}/${r.actions.distill.validatorRejected}/${r.actions.distill.configDisabled}/${r.actions.distill.skipped}`,
62
60
  `${r.consolidation.processed}/${r.consolidation.promoted}/${r.consolidation.merged}/${r.consolidation.deleted}`,
63
61
  `${r.memoryInference.considered}/${r.memoryInference.written}/${r.memoryInference.skippedNoFacts}`,
64
- `${r.graphExtraction.extractedFiles}/${r.graphExtraction.entities}/${r.graphExtraction.relations}`,
65
62
  String(r.orphansPurged),
66
63
  `${r.lintFixed}/${r.lintFlagged}`,
67
64
  r.resultStatus ?? "valid",
@@ -81,8 +78,6 @@ export function renderWindowCompareMd(windows, deltas) {
81
78
  const badIfPositive = new Set([
82
79
  "improve.actions.reflect.failed",
83
80
  "improve.actions.distill.llmFailed",
84
- "improve.graphExtraction.failures",
85
- "improve.graphExtraction.nonArrayBatchFailures",
86
81
  "improve.wallTime.medianMs",
87
82
  "improve.wallTime.p95Ms",
88
83
  "improve.memoryInference.skippedNoFacts",
@@ -64,11 +64,9 @@ function coercePct(raw) {
64
64
  function reshapeRun(r) {
65
65
  const cons = r.consolidation;
66
66
  const mi = r.memoryInference;
67
- const ge = r.graphExtraction;
68
67
  const wall = r.wallTimeMs || 0;
69
68
  const consMs = cons.durationMs || 0;
70
69
  const miMs = mi.durationMs || 0;
71
- const geMs = ge.durationMs || 0;
72
70
  return {
73
71
  id: r.id,
74
72
  resultStatus: r.resultStatus ?? "valid",
@@ -81,16 +79,13 @@ function reshapeRun(r) {
81
79
  ok: r.ok,
82
80
  consDurationMs: consMs,
83
81
  miDurationMs: miMs,
84
- geDurationMs: geMs,
85
- otherMs: Math.max(0, wall - consMs - miMs - geMs),
82
+ otherMs: Math.max(0, wall - consMs - miMs),
86
83
  promoted: cons.promoted,
87
84
  merged: cons.merged,
88
85
  deleted: cons.deleted,
89
86
  contradicted: cons.contradicted,
90
87
  judgedNoAction: cons.judgedNoAction,
91
88
  miWritten: mi.written,
92
- geEntities: ge.entities,
93
- geRelations: ge.relations,
94
89
  distillByReason: r.actions.distill.skippedByReason,
95
90
  reflectOk: r.actions.reflect.ok,
96
91
  reflectFailed: r.actions.reflect.failed,
@@ -121,11 +116,10 @@ function classify(deltas, metricKeys, lowerIsBetter = false) {
121
116
  function buildTrend(deltas) {
122
117
  const decisionQuality = classify(deltas, ["improve.memoryInference.yieldRate", "improve.consolidation.promoted"]);
123
118
  const outputVolume = classify(deltas, ["improve.consolidation.promoted", "improve.memoryInference.written"]);
124
- const failures = classify(deltas, ["improve.graphExtraction.failures"], true);
125
119
  const latency = classify(deltas, ["improve.wallTime.medianMs", "improve.wallTime.p95Ms"], true);
126
- const score = [decisionQuality, outputVolume, failures, latency].reduce((acc, d) => acc + (d === "up" ? 1 : d === "down" ? -1 : 0), 0);
120
+ const score = [decisionQuality, outputVolume, latency].reduce((acc, d) => acc + (d === "up" ? 1 : d === "down" ? -1 : 0), 0);
127
121
  const overall = score >= 1 ? "improving" : score <= -1 ? "degrading" : "mixed";
128
- return { decisionQuality, outputVolume, failures, latency, overall };
122
+ return { decisionQuality, outputVolume, latency, overall };
129
123
  }
130
124
  function readSemSearch(advisories) {
131
125
  const check = advisories.find((a) => a.name === "semantic-search-runtime");
@@ -184,7 +178,6 @@ function buildAggregatesPhase(result, runsPhase) {
184
178
  const improve = result.improve;
185
179
  const cons = improve.consolidation;
186
180
  const mi = improve.memoryInference;
187
- const ge = improve.graphExtraction;
188
181
  const wallTime = improve.wallTime;
189
182
  const coverage = improve.coverage;
190
183
  // #576: real per-stage LLM token/time accounting (replaces the GPU-time
@@ -209,7 +202,6 @@ function buildAggregatesPhase(result, runsPhase) {
209
202
  return {
210
203
  consolidation: cons,
211
204
  memoryInference: mi,
212
- graphExtraction: ge,
213
205
  wallTime,
214
206
  coverage,
215
207
  llm,
@@ -309,7 +301,7 @@ function groupProposalsBySource(proposals) {
309
301
  }
310
302
  /** Summary table rows: the base metric set + the WS-5 coverage/minting/perf/degradation extensions, when present. */
311
303
  function buildSummaryRows(aggregates, trend) {
312
- const { consolidation: cons, graphExtraction: ge, wallTime, llm, coverage } = aggregates;
304
+ const { consolidation: cons, wallTime, llm, coverage } = aggregates;
313
305
  const summaryRows = [
314
306
  ["Task fail rate", aggregates.taskFailRate, "flat"],
315
307
  ["Agent fail rate", aggregates.agentFailRate, "flat"],
@@ -332,8 +324,6 @@ function buildSummaryRows(aggregates, trend) {
332
324
  "Candidates reviewed but intentionally left unchanged (the 'judgedNoAction' field).",
333
325
  ],
334
326
  ["Chunk failure", aggregates.chunkFail, "flat"],
335
- ["Graph entities", num(ge.entities), "up"],
336
- ["Graph relations", num(ge.relations), "up"],
337
327
  [
338
328
  "Stash derived",
339
329
  num(aggregates.memorySummary.derived),
@@ -81,7 +81,6 @@ export const INTERESTING_DELTA_PATHS = [
81
81
  "improve.memoryInference.written",
82
82
  "improve.memoryInference.yieldRate",
83
83
  "improve.memoryInference.skippedNoFacts",
84
- "improve.graphExtraction.failures",
85
84
  "improve.autoAccept.promoted",
86
85
  "improve.autoAccept.validationFailed",
87
86
  "improve.coverage.acceptedProposals",
@@ -19,6 +19,7 @@ import { getAllEntries } from "../storage/repositories/index-entries-repository.
19
19
  import { queryTaskHistory } from "../storage/repositories/task-history-repository.js";
20
20
  import { getStateDbFreelistInfo, runStateDbQuickCheck } from "../storage/state-db-integrity.js";
21
21
  import { pkgVersion } from "../version.js";
22
+ import { collectArchiveUsageAdvisory } from "./health/archive-usage.js";
22
23
  import { HEALTH_CHECKS, probeActiveImproveStrategy, runHealthEngineProbes, runPendingStateMigrationsCheck, SESSION_EXTRACTION_LEDGER_WINDOW_DAYS, } from "./health/checks.js";
23
24
  import { collectConfigSkewAdvisory } from "./health/config-skew.js";
24
25
  import { collectDataDirUsageAdvisory } from "./health/data-dir-usage.js";
@@ -291,6 +292,18 @@ function gatherAncillaryAdvisories(db, options, egressConfigView) {
291
292
  catch {
292
293
  // Non-fatal.
293
294
  }
295
+ // Item 4 (alpha.9 plan §5.4/§8 step 8): a bundle with no git history of its
296
+ // own never gets the purge sweep, so its memory-cleanup archive only ever
297
+ // grows — report its size and file count instead. Best-effort — an
298
+ // unreadable archive must not abort the health report.
299
+ try {
300
+ const archiveUsage = collectArchiveUsageAdvisory(options.stashDir ?? resolveStashDir());
301
+ if (archiveUsage)
302
+ advisories.push(archiveUsage);
303
+ }
304
+ catch {
305
+ // Non-fatal.
306
+ }
294
307
  // #896: report the data dir's total size and its largest top-level
295
308
  // subdirectory, so a disk-usage blowup (e.g. unpruned migration snapshot
296
309
  // backups, #897) is self-diagnosing instead of requiring `du` archaeology.
@@ -0,0 +1,137 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ import { searchLocal } from "../../../indexer/search/db-search.js";
5
+ import { stripFrontmatterBody } from "../content-hash.js";
6
+ import { stripBundle } from "../ledger.js";
7
+ import { loadRetrievalQueries } from "../retrieval-gate.js";
8
+ /** At most this many of the retired asset's most recent queries are replayed (plan §5.4, §7). */
9
+ export const CONTINUITY_MAX_QUERIES = 5;
10
+ /** "Top 10" per the plan's rule — both the pass/fail cutoff and the search `limit`. */
11
+ export const CONTINUITY_TOP_N = 10;
12
+ /**
13
+ * Builds the real search call the continuity check uses, stateful across
14
+ * every query asked of ONE instance (S2): the first time a query falls back
15
+ * to keyword-only ranking (`mode: "fts-fallback"` — most often a down or
16
+ * unreachable embedding endpoint), every later call through THIS instance
17
+ * forces `semanticSearchMode: "off"` instead of attempting semantic search
18
+ * again, so a dead endpoint costs one failed attempt per run, not one per
19
+ * remaining query (a hanging endpoint at ~3s/query, 300 proposals x 5
20
+ * queries, would otherwise cost on the order of an hour). Construct exactly
21
+ * one instance per pair-pass run and reuse it for every proposal judged, so
22
+ * the throttle covers the whole run, not just one proposal's own queries —
23
+ * and every call made once it has switched, across every remaining proposal
24
+ * in the run, reports `forcedKeywordOnly: true`, not just the one call that
25
+ * discovered the fallback (round-3 review: the first fix only marked THAT
26
+ * call unverified, so a second proposal checked while the endpoint was
27
+ * still down came back with a clean, `mode: "keyword"` — and therefore
28
+ * bulk-acceptable — result).
29
+ */
30
+ export function createContinuitySearch(stashDir, config) {
31
+ const base = {
32
+ searchType: "any",
33
+ limit: CONTINUITY_TOP_N,
34
+ stashDir,
35
+ sources: [{ path: stashDir, isDefault: true }],
36
+ };
37
+ let keywordOnly = false;
38
+ return async (query) => {
39
+ const forcedKeywordOnly = keywordOnly;
40
+ const callConfig = keywordOnly ? { ...config, semanticSearchMode: "off" } : config;
41
+ const result = await searchLocal({ ...base, query, config: callConfig });
42
+ if (result.mode === "fts-fallback")
43
+ keywordOnly = true;
44
+ return { hits: result.hits, mode: result.mode, forcedKeywordOnly };
45
+ };
46
+ }
47
+ /** 1-indexed position of `conceptId` in `hits`, or `undefined` if it is not among them. */
48
+ function rankOf(hits, conceptId) {
49
+ const index = hits.findIndex((hit) => stripBundle(hit.ref) === conceptId);
50
+ return index === -1 ? undefined : index + 1;
51
+ }
52
+ /** Body only, whitespace collapsed — the same shape `db-search.ts`'s own content-dedupe compares (S3b). */
53
+ function normalizedBody(raw) {
54
+ return stripFrontmatterBody(raw).replace(/\s+/g, " ").trim();
55
+ }
56
+ /**
57
+ * Replay `retiredRef`'s own past queries and check that `successorRef` ranks
58
+ * in the top {@link CONTINUITY_TOP_N} for every one where the retired asset
59
+ * did. Returns `undefined` when there is nothing to flag: the two bodies are
60
+ * content-identical, no recorded queries, or every query ran on the real
61
+ * ranking and either the retired asset never ranked top 10 for it, or the
62
+ * survivor always did too. Never throws.
63
+ *
64
+ * S3: two fixes against false flags measured on a real night-1 admission
65
+ * (300 pairs, 6 flags, half spurious):
66
+ * - queries are the SAME cleaned set `loadRetrievalQueries` replays for the
67
+ * retrieval regression gate (`../retrieval-gate.ts`) — stash-README
68
+ * boilerplate, harness/tool envelopes, pastes over 2,000 characters, and
69
+ * duplicates are dropped before replay, not just capped at 5 raw entries;
70
+ * - when the retired and successor bodies normalize identical, the check
71
+ * never runs at all: search's own content-dedupe (`db-search.ts`) already
72
+ * hides the successor behind the retired asset for every such query, so a
73
+ * "successor missing from the top 10" finding here would not be a real
74
+ * risk, just that dedupe working as designed.
75
+ *
76
+ * S2: a query that never ran (the search call threw), ran on the
77
+ * keyword-only fallback (`mode: "fts-fallback"` — the real ranking was
78
+ * attempted and failed, most often a down or unreachable embedding
79
+ * endpoint), or ran after the shared search instance had already switched
80
+ * to forced keyword-only because an EARLIER query in the same run fell back
81
+ * (`forcedKeywordOnly`) is "unverified" — it is dropped from the rank
82
+ * comparison below (its hits cannot be trusted as "the ranking a user
83
+ * actually gets"), but unlike a query the retired asset simply did not rank
84
+ * for, it can never by itself lead to a silent `undefined` — at least one
85
+ * unverified query always produces a `continuityRisk`, so a dead endpoint
86
+ * reads as "risk unknown" for every proposal it touches that run, never as
87
+ * "no risk found" for the ones checked after the first failure.
88
+ */
89
+ export async function checkRetirementContinuity(args) {
90
+ // S3b: identical bodies — search's own content-dedupe already hides the
91
+ // successor for every query that would rank the retired asset, so there is
92
+ // no real risk here to check for.
93
+ if (normalizedBody(args.retiredRaw) === normalizedBody(args.successorRaw))
94
+ return undefined;
95
+ // S3a: the SAME cleaned queries the retrieval regression gate replays —
96
+ // boilerplate, envelopes, pastes and duplicates dropped before replay.
97
+ const queries = loadRetrievalQueries(args.ledgerAccess, args.retiredRef).slice(0, CONTINUITY_MAX_QUERIES);
98
+ if (queries.length === 0)
99
+ return undefined; // no queries recorded: no check, no flag
100
+ const search = args.search ?? createContinuitySearch(args.stashDir, args.config);
101
+ // N2: no rank-change-report abstraction — a query only ever needs "did the
102
+ // retired asset rank top 10, and if so, did the successor too?", and
103
+ // `rankOf` (search itself returning at most CONTINUITY_TOP_N hits) already
104
+ // answers both directly.
105
+ const ranks = [];
106
+ let unverifiedQueries = 0;
107
+ for (const query of queries) {
108
+ let hits;
109
+ try {
110
+ const result = await search(query);
111
+ if (result.mode === "fts-fallback" || result.forcedKeywordOnly) {
112
+ unverifiedQueries++; // S2: never silently compare keyword-only ranks
113
+ continue;
114
+ }
115
+ hits = result.hits;
116
+ }
117
+ catch {
118
+ unverifiedQueries++; // S2: a query that never ran cannot be "no risk"
119
+ continue;
120
+ }
121
+ const retiredRank = rankOf(hits, args.retiredRef);
122
+ if (retiredRank === undefined)
123
+ continue; // the retired asset itself did not rank top 10 here — nothing to protect
124
+ const successorRank = rankOf(hits, args.successorRef);
125
+ if (successorRank === undefined)
126
+ ranks.push({ query, retiredRank, successorRank: null });
127
+ }
128
+ // Every query verified, and either the retired asset never ranked top 10
129
+ // for any of them, or the survivor always did too: nothing to flag.
130
+ if (unverifiedQueries === 0 && ranks.length === 0)
131
+ return undefined;
132
+ return {
133
+ failingQueries: ranks.length,
134
+ ranks,
135
+ ...(unverifiedQueries > 0 ? { unverifiedQueries } : {}),
136
+ };
137
+ }