akm-cli 0.9.17-alpha.8 → 0.9.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/CHANGELOG.md +277 -1694
  2. package/STABILITY.md +9 -8
  3. package/dist/assets/hints/cli-hints-full.md +6 -7
  4. package/dist/assets/improve-strategies/catchup.json +0 -3
  5. package/dist/assets/improve-strategies/consolidate.json +0 -1
  6. package/dist/assets/improve-strategies/default.json +1 -2
  7. package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
  8. package/dist/assets/improve-strategies/quick.json +1 -2
  9. package/dist/assets/improve-strategies/reflect-distill.json +1 -2
  10. package/dist/assets/improve-strategies/thorough.json +0 -3
  11. package/dist/assets/prompts/consolidate-pair.md +20 -0
  12. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +17 -19
  13. package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
  14. package/dist/assets/templates/html/health.html +3 -5
  15. package/dist/cli/retired-commands.js +1 -1
  16. package/dist/cli/unknown-flags.js +24 -1
  17. package/dist/cli.js +46 -1
  18. package/dist/commands/health/archive-usage.js +92 -0
  19. package/dist/commands/health/data-dir-usage.js +25 -13
  20. package/dist/commands/health/html-report.js +1 -4
  21. package/dist/commands/health/improve-metrics.js +25 -37
  22. package/dist/commands/health/md-report.js +1 -6
  23. package/dist/commands/health/report-view-model.js +4 -14
  24. package/dist/commands/health/windows.js +0 -1
  25. package/dist/commands/health.js +13 -0
  26. package/dist/commands/improve/consolidate/continuity-check.js +137 -0
  27. package/dist/commands/improve/consolidate/pair-pass.js +791 -0
  28. package/dist/commands/improve/consolidate.js +38 -63
  29. package/dist/commands/improve/extract-prompt.js +1 -2
  30. package/dist/commands/improve/improve-cli.js +1 -1
  31. package/dist/commands/improve/improve-strategies.js +23 -5
  32. package/dist/commands/improve/improve.js +19 -30
  33. package/dist/commands/improve/ledger.js +3 -2
  34. package/dist/commands/improve/loop-stages.js +5 -85
  35. package/dist/commands/improve/memory/memory-belief.js +3 -1
  36. package/dist/commands/improve/memory/memory-improve.js +262 -11
  37. package/dist/commands/improve/planner.js +0 -5
  38. package/dist/commands/improve/preparation.js +20 -135
  39. package/dist/commands/improve/retrieval-scope.js +19 -4
  40. package/dist/commands/improve/salience.js +1 -14
  41. package/dist/commands/improve/stage.js +0 -1
  42. package/dist/commands/lint/base-linter.js +19 -11
  43. package/dist/commands/proposal/drain.js +8 -1
  44. package/dist/commands/proposal/proposal-cli.js +16 -2
  45. package/dist/commands/proposal/proposal-types.js +7 -0
  46. package/dist/commands/proposal/proposal.js +37 -6
  47. package/dist/commands/proposal/repository.js +613 -4
  48. package/dist/commands/proposal/validators/proposals.js +9 -0
  49. package/dist/commands/read/knowledge.js +3 -2
  50. package/dist/commands/read/show.js +0 -14
  51. package/dist/commands/sources/info.js +122 -18
  52. package/dist/commands/sources/stash-cli.js +23 -3
  53. package/dist/core/bundle-rename.js +1 -7
  54. package/dist/core/config/config-schema.js +8 -1
  55. package/dist/core/config/config.js +23 -48
  56. package/dist/core/config/engine-semantics.js +0 -2
  57. package/dist/core/config/schema/improve-processes.js +17 -42
  58. package/dist/core/config/schema/index-config.js +5 -25
  59. package/dist/core/file-change.js +13 -5
  60. package/dist/core/improve-result.js +22 -6
  61. package/dist/core/improve-types.js +0 -1
  62. package/dist/core/loopback.js +7 -12
  63. package/dist/core/parse.js +13 -16
  64. package/dist/core/state/migrations.js +15 -0
  65. package/dist/core/time.js +0 -20
  66. package/dist/indexer/db/llm-cache.js +2 -2
  67. package/dist/indexer/ensure-index.js +2 -2
  68. package/dist/indexer/index-written-assets.js +2 -3
  69. package/dist/indexer/indexer.js +18 -418
  70. package/dist/indexer/passes/metadata.js +0 -19
  71. package/dist/indexer/walk/walker.js +3 -4
  72. package/dist/llm/client.js +8 -10
  73. package/dist/llm/embedders/remote.js +1 -2
  74. package/dist/llm/feature-gate.js +0 -5
  75. package/dist/output/shapes/helpers.js +20 -4
  76. package/dist/output/text/command-format.js +9 -8
  77. package/dist/output/text/proposal-format.js +47 -1
  78. package/dist/output/text/show-format.js +0 -20
  79. package/dist/scripts/akm-migrate-node.js +923 -950
  80. package/dist/scripts/akm-migrate.js +923 -950
  81. package/dist/setup/steps/connection.js +5 -6
  82. package/dist/setup/steps/platforms.js +2 -2
  83. package/dist/sources/providers/git-stash.js +83 -4
  84. package/dist/storage/repositories/improve-ledger-repository.js +48 -7
  85. package/dist/storage/repositories/index-connection.js +5 -2
  86. package/dist/storage/repositories/index-entries-repository.js +4 -7
  87. package/dist/storage/repositories/index-entry-schema.js +4 -2
  88. package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
  89. package/dist/storage/repositories/index-schema.js +55 -104
  90. package/dist/storage/repositories/proposals-repository.js +61 -0
  91. package/dist/storage/repositories/salience-repository.js +1 -19
  92. package/docs/migration/README.md +1 -1
  93. package/docs/migration/release-notes/0.9.17.md +130 -41
  94. package/docs/migration/release-notes/README.md +7 -0
  95. package/docs/reference/cli.md +27 -21
  96. package/docs/reference/configuration.md +21 -12
  97. package/docs/reference/data-and-telemetry.md +0 -1
  98. package/package.json +1 -1
  99. package/schemas/akm-config.json +0 -342
  100. package/dist/assets/improve-strategies/graph-refresh.json +0 -15
  101. package/dist/assets/prompts/contradiction-judge.md +0 -33
  102. package/dist/assets/prompts/graph-extract-system.md +0 -1
  103. package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
  104. package/dist/assets/prompts/metadata-enhance-system.md +0 -1
  105. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
  106. package/dist/indexer/db/graph-db.js +0 -399
  107. package/dist/indexer/graph/graph-extraction.js +0 -809
  108. package/dist/indexer/graph/graph-related.js +0 -131
  109. package/dist/indexer/graph/graph-types.js +0 -4
  110. package/dist/llm/graph-extract.js +0 -892
  111. package/dist/llm/metadata-enhance.js +0 -95
@@ -131,7 +131,6 @@ function renderExecSummary(vm) {
131
131
  const trendRows = [
132
132
  trendLi("Decision quality", vm.trend.decisionQuality),
133
133
  trendLi("Output volume", vm.trend.outputVolume),
134
- trendLi("Failures", vm.trend.failures),
135
134
  trendLi("Latency", vm.trend.latency),
136
135
  ].join("");
137
136
  const deltaRows = [
@@ -151,7 +150,6 @@ function renderExecSummary(vm) {
151
150
  li("Promoted", String(vm.latest.promoted)),
152
151
  li("Judged: no action", `<abbr title="Candidates reviewed but intentionally left unchanged on this run.">${vm.latest.judgedNoAction}</abbr>`),
153
152
  li("MI written", String(vm.latest.miWritten)),
154
- li("Graph entities/relations", `${vm.latest.geEntities} / ${vm.latest.geRelations}`),
155
153
  ].join("")
156
154
  : '<li><span class="k">No runs in window</span><span class="v">—</span></li>';
157
155
  const windowRows = [
@@ -197,7 +195,7 @@ function renderExecSummary(vm) {
197
195
  </div>
198
196
  </div>
199
197
  <div class="overall">Overall trend: <b>${esc(vm.trend.overall)}</b> ${overallEmoji}
200
- &nbsp;·&nbsp; based on decision quality, output volume, failures, and latency ${vm.comparisonMode === "custom" ? "across the selected windows" : "vs the prior window"}.</div>`.trim();
198
+ &nbsp;·&nbsp; based on decision quality, output volume, and latency ${vm.comparisonMode === "custom" ? "across the selected windows" : "vs the prior window"}.</div>`.trim();
201
199
  }
202
200
  /**
203
201
  * Color is a health SIGNAL, not decoration: green/yellow/red where a card has
@@ -221,7 +219,6 @@ function renderKpiCards(vm) {
221
219
  kpiCard("neutral", "Median Duration", `${vm.medianDurMin}m`, `p95 = ${vm.p95DurMin}m`),
222
220
  kpiCard("blue", "Total Promoted", num(vm.consolidation.promoted), `avg ${vm.avgPromoted} / run`),
223
221
  kpiCard("blue", "MI Written", num(vm.miWritten), `${vm.miYieldRate} yield rate`),
224
- kpiCard("purple", "Graph Entities", num(vm.graphExtraction.entities), `+${num(vm.graphExtraction.relations)} relations`),
225
222
  kpiCard("neutral", "Stash Derived", num(vm.memorySummary.derived), `of ${num(vm.memorySummary.eligible)} eligible (whole-stash)`),
226
223
  // #576: real LLM work — duration leads, tokens compact, not a GPU proxy.
227
224
  kpiCard(vm.llm.calls > 0 ? "purple" : "neutral", "🧠 LLM Work", `${vm.llmTokensCompact} tok`, `${fmtMs(vm.llm.totalDurationMs)} · ${num(vm.llm.calls)} calls · ${compact(vm.llm.reasoningTokens)} reasoning`),
@@ -53,6 +53,25 @@ export function countAgentFailureReasons(agentFailures) {
53
53
  }
54
54
  return counts;
55
55
  }
56
+ /**
57
+ * Decode one `improve_runs.result_json` envelope, warning once per row on
58
+ * failure (mirrors {@link taskFailureDetail}'s handling of the analogous
59
+ * `task_history` case) and returning `undefined` instead of throwing.
60
+ * Callers count the `undefined` case themselves (`resultRows.skipped.invalid`
61
+ * / `resultStatus: "invalid"`) so a decode failure is never silent — the
62
+ * warning names *why* (corrupt data, or a decoder too strict for a shape an
63
+ * older release legitimately wrote), the counters say *how many*.
64
+ */
65
+ function decodeImproveResultRow(row) {
66
+ try {
67
+ return decodeImproveResult(row.result_json).envelope;
68
+ }
69
+ catch (error) {
70
+ const message = error instanceof Error ? error.message : String(error);
71
+ console.warn(`[akm] Skipping unparseable improve_runs row in health metrics (id=${row.id}, started_at=${row.started_at}): ${message}`);
72
+ return undefined;
73
+ }
74
+ }
56
75
  /** A zeroed accumulator — also what health reports when it could not read state.db at all (#791). */
57
76
  export function emptyImproveMetrics() {
58
77
  return {
@@ -74,7 +93,6 @@ export function emptyImproveMetrics() {
74
93
  },
75
94
  memoryPrune: 0,
76
95
  memoryInference: 0,
77
- graphExtraction: 0,
78
96
  error: 0,
79
97
  },
80
98
  autoAccept: { promoted: 0, validationFailed: 0 },
@@ -91,7 +109,6 @@ export function emptyImproveMetrics() {
91
109
  durationMs: 0,
92
110
  },
93
111
  memoryInference: { considered: 0, freshAttempts: 0, written: 0, skippedNoFacts: 0, yieldRate: 0, durationMs: 0 },
94
- graphExtraction: { extractedFiles: 0, entities: 0, relations: 0, failures: 0, durationMs: 0 },
95
112
  wallTime: { medianMs: 0, p95Ms: 0 },
96
113
  coverage: { acceptedProposals: 0, distinctRefs: 0 },
97
114
  };
@@ -151,9 +168,6 @@ function applyAction(metrics, action) {
151
168
  case "memory-inference":
152
169
  metrics.actions.memoryInference += 1;
153
170
  break;
154
- case "graph-extraction":
155
- metrics.actions.graphExtraction += 1;
156
- break;
157
171
  case "error":
158
172
  metrics.actions.error += 1;
159
173
  break;
@@ -209,19 +223,6 @@ function projectRunMetrics(result) {
209
223
  mi.skippedNoFacts += toFiniteNumber(memoryInference.skippedNoFacts);
210
224
  }
211
225
  metrics.memoryInference.durationMs += toFiniteNumber(result.memoryInferenceDurationMs);
212
- const graphExtraction = result.graphExtraction;
213
- if (graphExtraction) {
214
- const ge = metrics.graphExtraction;
215
- // This run's counts, like entities and relations. `quality` describes the
216
- // whole stored graph after the run; summing it would count each stored file
217
- // once per run in the window.
218
- ge.extractedFiles += toFiniteNumber(graphExtraction.extracted);
219
- ge.entities += toFiniteNumber(graphExtraction.totalEntities);
220
- ge.relations += toFiniteNumber(graphExtraction.totalRelations);
221
- const telemetry = graphExtraction.telemetry;
222
- ge.failures += toFiniteNumber(telemetry?.failureCount);
223
- }
224
- metrics.graphExtraction.durationMs += toFiniteNumber(result.graphExtractionDurationMs);
225
226
  return metrics;
226
227
  }
227
228
  /** Derived rates on an accumulator (window aggregate or single run). */
@@ -251,7 +252,6 @@ function mergeImproveMetrics(dst, src) {
251
252
  }
252
253
  dst.actions.memoryPrune += src.actions.memoryPrune;
253
254
  dst.actions.memoryInference += src.actions.memoryInference;
254
- dst.actions.graphExtraction += src.actions.graphExtraction;
255
255
  dst.actions.error += src.actions.error;
256
256
  dst.autoAccept.promoted += src.autoAccept.promoted;
257
257
  dst.autoAccept.validationFailed += src.autoAccept.validationFailed;
@@ -269,11 +269,6 @@ function mergeImproveMetrics(dst, src) {
269
269
  dst.memoryInference.written += src.memoryInference.written;
270
270
  dst.memoryInference.skippedNoFacts += src.memoryInference.skippedNoFacts;
271
271
  dst.memoryInference.durationMs += src.memoryInference.durationMs;
272
- dst.graphExtraction.extractedFiles += src.graphExtraction.extractedFiles;
273
- dst.graphExtraction.entities += src.graphExtraction.entities;
274
- dst.graphExtraction.relations += src.graphExtraction.relations;
275
- dst.graphExtraction.failures += src.graphExtraction.failures;
276
- dst.graphExtraction.durationMs += src.graphExtraction.durationMs;
277
272
  }
278
273
  function compareImproveRunRecency(a, b) {
279
274
  const started = a.started_at.localeCompare(b.started_at);
@@ -297,11 +292,8 @@ export function summarizeImproveRuns(db, since, until) {
297
292
  // newest complete run's snapshot (current state) — not a sum across runs.
298
293
  let latest;
299
294
  for (const row of rows) {
300
- let result;
301
- try {
302
- result = decodeImproveResult(row.result_json).envelope;
303
- }
304
- catch {
295
+ const result = decodeImproveResultRow(row);
296
+ if (!result) {
305
297
  resultRows.skipped.invalid += 1;
306
298
  continue;
307
299
  }
@@ -322,13 +314,10 @@ export function summarizeImproveRuns(db, since, until) {
322
314
  }
323
315
  /** Project an improve_runs row + wall time + task attribution into one {@link ImproveRunSummary}. */
324
316
  export function projectImproveRunSummary(row, wallTimeMs, taskId) {
325
- let result = {};
326
- let resultStatus = "invalid";
327
- try {
328
- result = decodeImproveResult(row.result_json).envelope;
329
- resultStatus = "valid";
330
- }
331
- catch {
317
+ const decoded = decodeImproveResultRow(row);
318
+ const result = decoded ?? {};
319
+ const resultStatus = decoded ? "valid" : "invalid";
320
+ if (!decoded) {
332
321
  // Keep the persisted row visible in per-run output, but do not project its
333
322
  // unknown payload or admit its duration to result-derived denominators.
334
323
  wallTimeMs = 0;
@@ -354,7 +343,6 @@ export function projectImproveRunSummary(row, wallTimeMs, taskId) {
354
343
  memorySummary: perRow.memorySummary,
355
344
  consolidation: perRow.consolidation,
356
345
  memoryInference: perRow.memoryInference,
357
- graphExtraction: perRow.graphExtraction,
358
346
  orphansPurged: toFiniteNumber(result.orphansPurged),
359
347
  lintFixed: lintSummary ? toFiniteNumber(lintSummary.fixed) : 0,
360
348
  lintFlagged: lintSummary ? toFiniteNumber(lintSummary.flagged) : 0,
@@ -20,7 +20,7 @@ function renderTable(headers, rows) {
20
20
  *
21
21
  * Columns: ts | ok | actions | refl_ok/fail/cd/skip |
22
22
  * distill_q/llm-fail/qrej/cfg/skip | cons_proc/promo/merge/del |
23
- * mem_cons/written/skip | graph_f/e/r | orphans | lint_f/fl
23
+ * mem_cons/written/skip | orphans | lint_f/fl
24
24
  */
25
25
  export function renderRunsDetailMd(runs) {
26
26
  const headers = [
@@ -32,7 +32,6 @@ export function renderRunsDetailMd(runs) {
32
32
  "distill_q/llm-fail/judge/validator/cfg/skip",
33
33
  "cons_proc/promo/merge/del",
34
34
  "mem_cons/written/skip",
35
- "graph_f/e/r",
36
35
  "orphans",
37
36
  "lint_f/fl",
38
37
  "result_status",
@@ -50,7 +49,6 @@ export function renderRunsDetailMd(runs) {
50
49
  r.actions.distill.skipped +
51
50
  r.actions.memoryPrune +
52
51
  r.actions.memoryInference +
53
- r.actions.graphExtraction +
54
52
  r.actions.error;
55
53
  return [
56
54
  r.startedAt,
@@ -61,7 +59,6 @@ export function renderRunsDetailMd(runs) {
61
59
  `${r.actions.distill.queued}/${r.actions.distill.llmFailed}/${r.actions.distill.judgeRejected}/${r.actions.distill.validatorRejected}/${r.actions.distill.configDisabled}/${r.actions.distill.skipped}`,
62
60
  `${r.consolidation.processed}/${r.consolidation.promoted}/${r.consolidation.merged}/${r.consolidation.deleted}`,
63
61
  `${r.memoryInference.considered}/${r.memoryInference.written}/${r.memoryInference.skippedNoFacts}`,
64
- `${r.graphExtraction.extractedFiles}/${r.graphExtraction.entities}/${r.graphExtraction.relations}`,
65
62
  String(r.orphansPurged),
66
63
  `${r.lintFixed}/${r.lintFlagged}`,
67
64
  r.resultStatus ?? "valid",
@@ -81,8 +78,6 @@ export function renderWindowCompareMd(windows, deltas) {
81
78
  const badIfPositive = new Set([
82
79
  "improve.actions.reflect.failed",
83
80
  "improve.actions.distill.llmFailed",
84
- "improve.graphExtraction.failures",
85
- "improve.graphExtraction.nonArrayBatchFailures",
86
81
  "improve.wallTime.medianMs",
87
82
  "improve.wallTime.p95Ms",
88
83
  "improve.memoryInference.skippedNoFacts",
@@ -64,11 +64,9 @@ function coercePct(raw) {
64
64
  function reshapeRun(r) {
65
65
  const cons = r.consolidation;
66
66
  const mi = r.memoryInference;
67
- const ge = r.graphExtraction;
68
67
  const wall = r.wallTimeMs || 0;
69
68
  const consMs = cons.durationMs || 0;
70
69
  const miMs = mi.durationMs || 0;
71
- const geMs = ge.durationMs || 0;
72
70
  return {
73
71
  id: r.id,
74
72
  resultStatus: r.resultStatus ?? "valid",
@@ -81,16 +79,13 @@ function reshapeRun(r) {
81
79
  ok: r.ok,
82
80
  consDurationMs: consMs,
83
81
  miDurationMs: miMs,
84
- geDurationMs: geMs,
85
- otherMs: Math.max(0, wall - consMs - miMs - geMs),
82
+ otherMs: Math.max(0, wall - consMs - miMs),
86
83
  promoted: cons.promoted,
87
84
  merged: cons.merged,
88
85
  deleted: cons.deleted,
89
86
  contradicted: cons.contradicted,
90
87
  judgedNoAction: cons.judgedNoAction,
91
88
  miWritten: mi.written,
92
- geEntities: ge.entities,
93
- geRelations: ge.relations,
94
89
  distillByReason: r.actions.distill.skippedByReason,
95
90
  reflectOk: r.actions.reflect.ok,
96
91
  reflectFailed: r.actions.reflect.failed,
@@ -121,11 +116,10 @@ function classify(deltas, metricKeys, lowerIsBetter = false) {
121
116
  function buildTrend(deltas) {
122
117
  const decisionQuality = classify(deltas, ["improve.memoryInference.yieldRate", "improve.consolidation.promoted"]);
123
118
  const outputVolume = classify(deltas, ["improve.consolidation.promoted", "improve.memoryInference.written"]);
124
- const failures = classify(deltas, ["improve.graphExtraction.failures"], true);
125
119
  const latency = classify(deltas, ["improve.wallTime.medianMs", "improve.wallTime.p95Ms"], true);
126
- const score = [decisionQuality, outputVolume, failures, latency].reduce((acc, d) => acc + (d === "up" ? 1 : d === "down" ? -1 : 0), 0);
120
+ const score = [decisionQuality, outputVolume, latency].reduce((acc, d) => acc + (d === "up" ? 1 : d === "down" ? -1 : 0), 0);
127
121
  const overall = score >= 1 ? "improving" : score <= -1 ? "degrading" : "mixed";
128
- return { decisionQuality, outputVolume, failures, latency, overall };
122
+ return { decisionQuality, outputVolume, latency, overall };
129
123
  }
130
124
  function readSemSearch(advisories) {
131
125
  const check = advisories.find((a) => a.name === "semantic-search-runtime");
@@ -184,7 +178,6 @@ function buildAggregatesPhase(result, runsPhase) {
184
178
  const improve = result.improve;
185
179
  const cons = improve.consolidation;
186
180
  const mi = improve.memoryInference;
187
- const ge = improve.graphExtraction;
188
181
  const wallTime = improve.wallTime;
189
182
  const coverage = improve.coverage;
190
183
  // #576: real per-stage LLM token/time accounting (replaces the GPU-time
@@ -209,7 +202,6 @@ function buildAggregatesPhase(result, runsPhase) {
209
202
  return {
210
203
  consolidation: cons,
211
204
  memoryInference: mi,
212
- graphExtraction: ge,
213
205
  wallTime,
214
206
  coverage,
215
207
  llm,
@@ -309,7 +301,7 @@ function groupProposalsBySource(proposals) {
309
301
  }
310
302
  /** Summary table rows: the base metric set + the WS-5 coverage/minting/perf/degradation extensions, when present. */
311
303
  function buildSummaryRows(aggregates, trend) {
312
- const { consolidation: cons, graphExtraction: ge, wallTime, llm, coverage } = aggregates;
304
+ const { consolidation: cons, wallTime, llm, coverage } = aggregates;
313
305
  const summaryRows = [
314
306
  ["Task fail rate", aggregates.taskFailRate, "flat"],
315
307
  ["Agent fail rate", aggregates.agentFailRate, "flat"],
@@ -332,8 +324,6 @@ function buildSummaryRows(aggregates, trend) {
332
324
  "Candidates reviewed but intentionally left unchanged (the 'judgedNoAction' field).",
333
325
  ],
334
326
  ["Chunk failure", aggregates.chunkFail, "flat"],
335
- ["Graph entities", num(ge.entities), "up"],
336
- ["Graph relations", num(ge.relations), "up"],
337
327
  [
338
328
  "Stash derived",
339
329
  num(aggregates.memorySummary.derived),
@@ -81,7 +81,6 @@ export const INTERESTING_DELTA_PATHS = [
81
81
  "improve.memoryInference.written",
82
82
  "improve.memoryInference.yieldRate",
83
83
  "improve.memoryInference.skippedNoFacts",
84
- "improve.graphExtraction.failures",
85
84
  "improve.autoAccept.promoted",
86
85
  "improve.autoAccept.validationFailed",
87
86
  "improve.coverage.acceptedProposals",
@@ -19,6 +19,7 @@ import { getAllEntries } from "../storage/repositories/index-entries-repository.
19
19
  import { queryTaskHistory } from "../storage/repositories/task-history-repository.js";
20
20
  import { getStateDbFreelistInfo, runStateDbQuickCheck } from "../storage/state-db-integrity.js";
21
21
  import { pkgVersion } from "../version.js";
22
+ import { collectArchiveUsageAdvisory } from "./health/archive-usage.js";
22
23
  import { HEALTH_CHECKS, probeActiveImproveStrategy, runHealthEngineProbes, runPendingStateMigrationsCheck, SESSION_EXTRACTION_LEDGER_WINDOW_DAYS, } from "./health/checks.js";
23
24
  import { collectConfigSkewAdvisory } from "./health/config-skew.js";
24
25
  import { collectDataDirUsageAdvisory } from "./health/data-dir-usage.js";
@@ -291,6 +292,18 @@ function gatherAncillaryAdvisories(db, options, egressConfigView) {
291
292
  catch {
292
293
  // Non-fatal.
293
294
  }
295
+ // Item 4 (alpha.9 plan §5.4/§8 step 8): a bundle with no git history of its
296
+ // own never gets the purge sweep, so its memory-cleanup archive only ever
297
+ // grows — report its size and file count instead. Best-effort — an
298
+ // unreadable archive must not abort the health report.
299
+ try {
300
+ const archiveUsage = collectArchiveUsageAdvisory(options.stashDir ?? resolveStashDir());
301
+ if (archiveUsage)
302
+ advisories.push(archiveUsage);
303
+ }
304
+ catch {
305
+ // Non-fatal.
306
+ }
294
307
  // #896: report the data dir's total size and its largest top-level
295
308
  // subdirectory, so a disk-usage blowup (e.g. unpruned migration snapshot
296
309
  // backups, #897) is self-diagnosing instead of requiring `du` archaeology.
@@ -0,0 +1,137 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ import { searchLocal } from "../../../indexer/search/db-search.js";
5
+ import { stripFrontmatterBody } from "../content-hash.js";
6
+ import { stripBundle } from "../ledger.js";
7
+ import { loadRetrievalQueries } from "../retrieval-gate.js";
8
+ /** At most this many of the retired asset's most recent queries are replayed (plan §5.4, §7). */
9
+ export const CONTINUITY_MAX_QUERIES = 5;
10
+ /** "Top 10" per the plan's rule — both the pass/fail cutoff and the search `limit`. */
11
+ export const CONTINUITY_TOP_N = 10;
12
+ /**
13
+ * Builds the real search call the continuity check uses, stateful across
14
+ * every query asked of ONE instance (S2): the first time a query falls back
15
+ * to keyword-only ranking (`mode: "fts-fallback"` — most often a down or
16
+ * unreachable embedding endpoint), every later call through THIS instance
17
+ * forces `semanticSearchMode: "off"` instead of attempting semantic search
18
+ * again, so a dead endpoint costs one failed attempt per run, not one per
19
+ * remaining query (a hanging endpoint at ~3s/query, 300 proposals x 5
20
+ * queries, would otherwise cost on the order of an hour). Construct exactly
21
+ * one instance per pair-pass run and reuse it for every proposal judged, so
22
+ * the throttle covers the whole run, not just one proposal's own queries —
23
+ * and every call made once it has switched, across every remaining proposal
24
+ * in the run, reports `forcedKeywordOnly: true`, not just the one call that
25
+ * discovered the fallback (round-3 review: the first fix only marked THAT
26
+ * call unverified, so a second proposal checked while the endpoint was
27
+ * still down came back with a clean, `mode: "keyword"` — and therefore
28
+ * bulk-acceptable — result).
29
+ */
30
+ export function createContinuitySearch(stashDir, config) {
31
+ const base = {
32
+ searchType: "any",
33
+ limit: CONTINUITY_TOP_N,
34
+ stashDir,
35
+ sources: [{ path: stashDir, isDefault: true }],
36
+ };
37
+ let keywordOnly = false;
38
+ return async (query) => {
39
+ const forcedKeywordOnly = keywordOnly;
40
+ const callConfig = keywordOnly ? { ...config, semanticSearchMode: "off" } : config;
41
+ const result = await searchLocal({ ...base, query, config: callConfig });
42
+ if (result.mode === "fts-fallback")
43
+ keywordOnly = true;
44
+ return { hits: result.hits, mode: result.mode, forcedKeywordOnly };
45
+ };
46
+ }
47
+ /** 1-indexed position of `conceptId` in `hits`, or `undefined` if it is not among them. */
48
+ function rankOf(hits, conceptId) {
49
+ const index = hits.findIndex((hit) => stripBundle(hit.ref) === conceptId);
50
+ return index === -1 ? undefined : index + 1;
51
+ }
52
+ /** Body only, whitespace collapsed — the same shape `db-search.ts`'s own content-dedupe compares (S3b). */
53
+ function normalizedBody(raw) {
54
+ return stripFrontmatterBody(raw).replace(/\s+/g, " ").trim();
55
+ }
56
+ /**
57
+ * Replay `retiredRef`'s own past queries and check that `successorRef` ranks
58
+ * in the top {@link CONTINUITY_TOP_N} for every one where the retired asset
59
+ * did. Returns `undefined` when there is nothing to flag: the two bodies are
60
+ * content-identical, no recorded queries, or every query ran on the real
61
+ * ranking and either the retired asset never ranked top 10 for it, or the
62
+ * survivor always did too. Never throws.
63
+ *
64
+ * S3: two fixes against false flags measured on a real night-1 admission
65
+ * (300 pairs, 6 flags, half spurious):
66
+ * - queries are the SAME cleaned set `loadRetrievalQueries` replays for the
67
+ * retrieval regression gate (`../retrieval-gate.ts`) — stash-README
68
+ * boilerplate, harness/tool envelopes, pastes over 2,000 characters, and
69
+ * duplicates are dropped before replay, not just capped at 5 raw entries;
70
+ * - when the retired and successor bodies normalize identical, the check
71
+ * never runs at all: search's own content-dedupe (`db-search.ts`) already
72
+ * hides the successor behind the retired asset for every such query, so a
73
+ * "successor missing from the top 10" finding here would not be a real
74
+ * risk, just that dedupe working as designed.
75
+ *
76
+ * S2: a query that never ran (the search call threw), ran on the
77
+ * keyword-only fallback (`mode: "fts-fallback"` — the real ranking was
78
+ * attempted and failed, most often a down or unreachable embedding
79
+ * endpoint), or ran after the shared search instance had already switched
80
+ * to forced keyword-only because an EARLIER query in the same run fell back
81
+ * (`forcedKeywordOnly`) is "unverified" — it is dropped from the rank
82
+ * comparison below (its hits cannot be trusted as "the ranking a user
83
+ * actually gets"), but unlike a query the retired asset simply did not rank
84
+ * for, it can never by itself lead to a silent `undefined` — at least one
85
+ * unverified query always produces a `continuityRisk`, so a dead endpoint
86
+ * reads as "risk unknown" for every proposal it touches that run, never as
87
+ * "no risk found" for the ones checked after the first failure.
88
+ */
89
+ export async function checkRetirementContinuity(args) {
90
+ // S3b: identical bodies — search's own content-dedupe already hides the
91
+ // successor for every query that would rank the retired asset, so there is
92
+ // no real risk here to check for.
93
+ if (normalizedBody(args.retiredRaw) === normalizedBody(args.successorRaw))
94
+ return undefined;
95
+ // S3a: the SAME cleaned queries the retrieval regression gate replays —
96
+ // boilerplate, envelopes, pastes and duplicates dropped before replay.
97
+ const queries = loadRetrievalQueries(args.ledgerAccess, args.retiredRef).slice(0, CONTINUITY_MAX_QUERIES);
98
+ if (queries.length === 0)
99
+ return undefined; // no queries recorded: no check, no flag
100
+ const search = args.search ?? createContinuitySearch(args.stashDir, args.config);
101
+ // N2: no rank-change-report abstraction — a query only ever needs "did the
102
+ // retired asset rank top 10, and if so, did the successor too?", and
103
+ // `rankOf` (search itself returning at most CONTINUITY_TOP_N hits) already
104
+ // answers both directly.
105
+ const ranks = [];
106
+ let unverifiedQueries = 0;
107
+ for (const query of queries) {
108
+ let hits;
109
+ try {
110
+ const result = await search(query);
111
+ if (result.mode === "fts-fallback" || result.forcedKeywordOnly) {
112
+ unverifiedQueries++; // S2: never silently compare keyword-only ranks
113
+ continue;
114
+ }
115
+ hits = result.hits;
116
+ }
117
+ catch {
118
+ unverifiedQueries++; // S2: a query that never ran cannot be "no risk"
119
+ continue;
120
+ }
121
+ const retiredRank = rankOf(hits, args.retiredRef);
122
+ if (retiredRank === undefined)
123
+ continue; // the retired asset itself did not rank top 10 here — nothing to protect
124
+ const successorRank = rankOf(hits, args.successorRef);
125
+ if (successorRank === undefined)
126
+ ranks.push({ query, retiredRank, successorRank: null });
127
+ }
128
+ // Every query verified, and either the retired asset never ranked top 10
129
+ // for any of them, or the survivor always did too: nothing to flag.
130
+ if (unverifiedQueries === 0 && ranks.length === 0)
131
+ return undefined;
132
+ return {
133
+ failingQueries: ranks.length,
134
+ ranks,
135
+ ...(unverifiedQueries > 0 ? { unverifiedQueries } : {}),
136
+ };
137
+ }