akm-cli 0.9.17-alpha.8 → 0.9.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +277 -1694
- package/STABILITY.md +9 -8
- package/dist/assets/hints/cli-hints-full.md +6 -7
- package/dist/assets/improve-strategies/catchup.json +0 -3
- package/dist/assets/improve-strategies/consolidate.json +0 -1
- package/dist/assets/improve-strategies/default.json +1 -2
- package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
- package/dist/assets/improve-strategies/quick.json +1 -2
- package/dist/assets/improve-strategies/reflect-distill.json +1 -2
- package/dist/assets/improve-strategies/thorough.json +0 -3
- package/dist/assets/prompts/consolidate-pair.md +20 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +17 -19
- package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
- package/dist/assets/templates/html/health.html +3 -5
- package/dist/cli/retired-commands.js +1 -1
- package/dist/cli/unknown-flags.js +24 -1
- package/dist/cli.js +46 -1
- package/dist/commands/health/archive-usage.js +92 -0
- package/dist/commands/health/data-dir-usage.js +25 -13
- package/dist/commands/health/html-report.js +1 -4
- package/dist/commands/health/improve-metrics.js +25 -37
- package/dist/commands/health/md-report.js +1 -6
- package/dist/commands/health/report-view-model.js +4 -14
- package/dist/commands/health/windows.js +0 -1
- package/dist/commands/health.js +13 -0
- package/dist/commands/improve/consolidate/continuity-check.js +137 -0
- package/dist/commands/improve/consolidate/pair-pass.js +791 -0
- package/dist/commands/improve/consolidate.js +38 -63
- package/dist/commands/improve/extract-prompt.js +1 -2
- package/dist/commands/improve/improve-cli.js +1 -1
- package/dist/commands/improve/improve-strategies.js +23 -5
- package/dist/commands/improve/improve.js +19 -30
- package/dist/commands/improve/ledger.js +3 -2
- package/dist/commands/improve/loop-stages.js +5 -85
- package/dist/commands/improve/memory/memory-belief.js +3 -1
- package/dist/commands/improve/memory/memory-improve.js +262 -11
- package/dist/commands/improve/planner.js +0 -5
- package/dist/commands/improve/preparation.js +20 -135
- package/dist/commands/improve/retrieval-scope.js +19 -4
- package/dist/commands/improve/salience.js +1 -14
- package/dist/commands/improve/stage.js +0 -1
- package/dist/commands/lint/base-linter.js +19 -11
- package/dist/commands/proposal/drain.js +8 -1
- package/dist/commands/proposal/proposal-cli.js +16 -2
- package/dist/commands/proposal/proposal-types.js +7 -0
- package/dist/commands/proposal/proposal.js +37 -6
- package/dist/commands/proposal/repository.js +613 -4
- package/dist/commands/proposal/validators/proposals.js +9 -0
- package/dist/commands/read/knowledge.js +3 -2
- package/dist/commands/read/show.js +0 -14
- package/dist/commands/sources/info.js +122 -18
- package/dist/commands/sources/stash-cli.js +23 -3
- package/dist/core/bundle-rename.js +1 -7
- package/dist/core/config/config-schema.js +8 -1
- package/dist/core/config/config.js +23 -48
- package/dist/core/config/engine-semantics.js +0 -2
- package/dist/core/config/schema/improve-processes.js +17 -42
- package/dist/core/config/schema/index-config.js +5 -25
- package/dist/core/file-change.js +13 -5
- package/dist/core/improve-result.js +22 -6
- package/dist/core/improve-types.js +0 -1
- package/dist/core/loopback.js +7 -12
- package/dist/core/parse.js +13 -16
- package/dist/core/state/migrations.js +15 -0
- package/dist/core/time.js +0 -20
- package/dist/indexer/db/llm-cache.js +2 -2
- package/dist/indexer/ensure-index.js +2 -2
- package/dist/indexer/index-written-assets.js +2 -3
- package/dist/indexer/indexer.js +18 -418
- package/dist/indexer/passes/metadata.js +0 -19
- package/dist/indexer/walk/walker.js +3 -4
- package/dist/llm/client.js +8 -10
- package/dist/llm/embedders/remote.js +1 -2
- package/dist/llm/feature-gate.js +0 -5
- package/dist/output/shapes/helpers.js +20 -4
- package/dist/output/text/command-format.js +9 -8
- package/dist/output/text/proposal-format.js +47 -1
- package/dist/output/text/show-format.js +0 -20
- package/dist/scripts/akm-migrate-node.js +923 -950
- package/dist/scripts/akm-migrate.js +923 -950
- package/dist/setup/steps/connection.js +5 -6
- package/dist/setup/steps/platforms.js +2 -2
- package/dist/sources/providers/git-stash.js +83 -4
- package/dist/storage/repositories/improve-ledger-repository.js +48 -7
- package/dist/storage/repositories/index-connection.js +5 -2
- package/dist/storage/repositories/index-entries-repository.js +4 -7
- package/dist/storage/repositories/index-entry-schema.js +4 -2
- package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
- package/dist/storage/repositories/index-schema.js +55 -104
- package/dist/storage/repositories/proposals-repository.js +61 -0
- package/dist/storage/repositories/salience-repository.js +1 -19
- package/docs/migration/README.md +1 -1
- package/docs/migration/release-notes/0.9.17.md +130 -41
- package/docs/migration/release-notes/README.md +7 -0
- package/docs/reference/cli.md +27 -21
- package/docs/reference/configuration.md +21 -12
- package/docs/reference/data-and-telemetry.md +0 -1
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -342
- package/dist/assets/improve-strategies/graph-refresh.json +0 -15
- package/dist/assets/prompts/contradiction-judge.md +0 -33
- package/dist/assets/prompts/graph-extract-system.md +0 -1
- package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
- package/dist/assets/prompts/metadata-enhance-system.md +0 -1
- package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
- package/dist/indexer/db/graph-db.js +0 -399
- package/dist/indexer/graph/graph-extraction.js +0 -809
- package/dist/indexer/graph/graph-related.js +0 -131
- package/dist/indexer/graph/graph-types.js +0 -4
- package/dist/llm/graph-extract.js +0 -892
- package/dist/llm/metadata-enhance.js +0 -95
|
@@ -131,7 +131,6 @@ function renderExecSummary(vm) {
|
|
|
131
131
|
const trendRows = [
|
|
132
132
|
trendLi("Decision quality", vm.trend.decisionQuality),
|
|
133
133
|
trendLi("Output volume", vm.trend.outputVolume),
|
|
134
|
-
trendLi("Failures", vm.trend.failures),
|
|
135
134
|
trendLi("Latency", vm.trend.latency),
|
|
136
135
|
].join("");
|
|
137
136
|
const deltaRows = [
|
|
@@ -151,7 +150,6 @@ function renderExecSummary(vm) {
|
|
|
151
150
|
li("Promoted", String(vm.latest.promoted)),
|
|
152
151
|
li("Judged: no action", `<abbr title="Candidates reviewed but intentionally left unchanged on this run.">${vm.latest.judgedNoAction}</abbr>`),
|
|
153
152
|
li("MI written", String(vm.latest.miWritten)),
|
|
154
|
-
li("Graph entities/relations", `${vm.latest.geEntities} / ${vm.latest.geRelations}`),
|
|
155
153
|
].join("")
|
|
156
154
|
: '<li><span class="k">No runs in window</span><span class="v">—</span></li>';
|
|
157
155
|
const windowRows = [
|
|
@@ -197,7 +195,7 @@ function renderExecSummary(vm) {
|
|
|
197
195
|
</div>
|
|
198
196
|
</div>
|
|
199
197
|
<div class="overall">Overall trend: <b>${esc(vm.trend.overall)}</b> ${overallEmoji}
|
|
200
|
-
· based on decision quality, output volume,
|
|
198
|
+
· based on decision quality, output volume, and latency ${vm.comparisonMode === "custom" ? "across the selected windows" : "vs the prior window"}.</div>`.trim();
|
|
201
199
|
}
|
|
202
200
|
/**
|
|
203
201
|
* Color is a health SIGNAL, not decoration: green/yellow/red where a card has
|
|
@@ -221,7 +219,6 @@ function renderKpiCards(vm) {
|
|
|
221
219
|
kpiCard("neutral", "Median Duration", `${vm.medianDurMin}m`, `p95 = ${vm.p95DurMin}m`),
|
|
222
220
|
kpiCard("blue", "Total Promoted", num(vm.consolidation.promoted), `avg ${vm.avgPromoted} / run`),
|
|
223
221
|
kpiCard("blue", "MI Written", num(vm.miWritten), `${vm.miYieldRate} yield rate`),
|
|
224
|
-
kpiCard("purple", "Graph Entities", num(vm.graphExtraction.entities), `+${num(vm.graphExtraction.relations)} relations`),
|
|
225
222
|
kpiCard("neutral", "Stash Derived", num(vm.memorySummary.derived), `of ${num(vm.memorySummary.eligible)} eligible (whole-stash)`),
|
|
226
223
|
// #576: real LLM work — duration leads, tokens compact, not a GPU proxy.
|
|
227
224
|
kpiCard(vm.llm.calls > 0 ? "purple" : "neutral", "🧠 LLM Work", `${vm.llmTokensCompact} tok`, `${fmtMs(vm.llm.totalDurationMs)} · ${num(vm.llm.calls)} calls · ${compact(vm.llm.reasoningTokens)} reasoning`),
|
|
@@ -53,6 +53,25 @@ export function countAgentFailureReasons(agentFailures) {
|
|
|
53
53
|
}
|
|
54
54
|
return counts;
|
|
55
55
|
}
|
|
56
|
+
/**
|
|
57
|
+
* Decode one `improve_runs.result_json` envelope, warning once per row on
|
|
58
|
+
* failure (mirrors {@link taskFailureDetail}'s handling of the analogous
|
|
59
|
+
* `task_history` case) and returning `undefined` instead of throwing.
|
|
60
|
+
* Callers count the `undefined` case themselves (`resultRows.skipped.invalid`
|
|
61
|
+
* / `resultStatus: "invalid"`) so a decode failure is never silent — the
|
|
62
|
+
* warning names *why* (corrupt data, or a decoder too strict for a shape an
|
|
63
|
+
* older release legitimately wrote), the counters say *how many*.
|
|
64
|
+
*/
|
|
65
|
+
function decodeImproveResultRow(row) {
|
|
66
|
+
try {
|
|
67
|
+
return decodeImproveResult(row.result_json).envelope;
|
|
68
|
+
}
|
|
69
|
+
catch (error) {
|
|
70
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
71
|
+
console.warn(`[akm] Skipping unparseable improve_runs row in health metrics (id=${row.id}, started_at=${row.started_at}): ${message}`);
|
|
72
|
+
return undefined;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
56
75
|
/** A zeroed accumulator — also what health reports when it could not read state.db at all (#791). */
|
|
57
76
|
export function emptyImproveMetrics() {
|
|
58
77
|
return {
|
|
@@ -74,7 +93,6 @@ export function emptyImproveMetrics() {
|
|
|
74
93
|
},
|
|
75
94
|
memoryPrune: 0,
|
|
76
95
|
memoryInference: 0,
|
|
77
|
-
graphExtraction: 0,
|
|
78
96
|
error: 0,
|
|
79
97
|
},
|
|
80
98
|
autoAccept: { promoted: 0, validationFailed: 0 },
|
|
@@ -91,7 +109,6 @@ export function emptyImproveMetrics() {
|
|
|
91
109
|
durationMs: 0,
|
|
92
110
|
},
|
|
93
111
|
memoryInference: { considered: 0, freshAttempts: 0, written: 0, skippedNoFacts: 0, yieldRate: 0, durationMs: 0 },
|
|
94
|
-
graphExtraction: { extractedFiles: 0, entities: 0, relations: 0, failures: 0, durationMs: 0 },
|
|
95
112
|
wallTime: { medianMs: 0, p95Ms: 0 },
|
|
96
113
|
coverage: { acceptedProposals: 0, distinctRefs: 0 },
|
|
97
114
|
};
|
|
@@ -151,9 +168,6 @@ function applyAction(metrics, action) {
|
|
|
151
168
|
case "memory-inference":
|
|
152
169
|
metrics.actions.memoryInference += 1;
|
|
153
170
|
break;
|
|
154
|
-
case "graph-extraction":
|
|
155
|
-
metrics.actions.graphExtraction += 1;
|
|
156
|
-
break;
|
|
157
171
|
case "error":
|
|
158
172
|
metrics.actions.error += 1;
|
|
159
173
|
break;
|
|
@@ -209,19 +223,6 @@ function projectRunMetrics(result) {
|
|
|
209
223
|
mi.skippedNoFacts += toFiniteNumber(memoryInference.skippedNoFacts);
|
|
210
224
|
}
|
|
211
225
|
metrics.memoryInference.durationMs += toFiniteNumber(result.memoryInferenceDurationMs);
|
|
212
|
-
const graphExtraction = result.graphExtraction;
|
|
213
|
-
if (graphExtraction) {
|
|
214
|
-
const ge = metrics.graphExtraction;
|
|
215
|
-
// This run's counts, like entities and relations. `quality` describes the
|
|
216
|
-
// whole stored graph after the run; summing it would count each stored file
|
|
217
|
-
// once per run in the window.
|
|
218
|
-
ge.extractedFiles += toFiniteNumber(graphExtraction.extracted);
|
|
219
|
-
ge.entities += toFiniteNumber(graphExtraction.totalEntities);
|
|
220
|
-
ge.relations += toFiniteNumber(graphExtraction.totalRelations);
|
|
221
|
-
const telemetry = graphExtraction.telemetry;
|
|
222
|
-
ge.failures += toFiniteNumber(telemetry?.failureCount);
|
|
223
|
-
}
|
|
224
|
-
metrics.graphExtraction.durationMs += toFiniteNumber(result.graphExtractionDurationMs);
|
|
225
226
|
return metrics;
|
|
226
227
|
}
|
|
227
228
|
/** Derived rates on an accumulator (window aggregate or single run). */
|
|
@@ -251,7 +252,6 @@ function mergeImproveMetrics(dst, src) {
|
|
|
251
252
|
}
|
|
252
253
|
dst.actions.memoryPrune += src.actions.memoryPrune;
|
|
253
254
|
dst.actions.memoryInference += src.actions.memoryInference;
|
|
254
|
-
dst.actions.graphExtraction += src.actions.graphExtraction;
|
|
255
255
|
dst.actions.error += src.actions.error;
|
|
256
256
|
dst.autoAccept.promoted += src.autoAccept.promoted;
|
|
257
257
|
dst.autoAccept.validationFailed += src.autoAccept.validationFailed;
|
|
@@ -269,11 +269,6 @@ function mergeImproveMetrics(dst, src) {
|
|
|
269
269
|
dst.memoryInference.written += src.memoryInference.written;
|
|
270
270
|
dst.memoryInference.skippedNoFacts += src.memoryInference.skippedNoFacts;
|
|
271
271
|
dst.memoryInference.durationMs += src.memoryInference.durationMs;
|
|
272
|
-
dst.graphExtraction.extractedFiles += src.graphExtraction.extractedFiles;
|
|
273
|
-
dst.graphExtraction.entities += src.graphExtraction.entities;
|
|
274
|
-
dst.graphExtraction.relations += src.graphExtraction.relations;
|
|
275
|
-
dst.graphExtraction.failures += src.graphExtraction.failures;
|
|
276
|
-
dst.graphExtraction.durationMs += src.graphExtraction.durationMs;
|
|
277
272
|
}
|
|
278
273
|
function compareImproveRunRecency(a, b) {
|
|
279
274
|
const started = a.started_at.localeCompare(b.started_at);
|
|
@@ -297,11 +292,8 @@ export function summarizeImproveRuns(db, since, until) {
|
|
|
297
292
|
// newest complete run's snapshot (current state) — not a sum across runs.
|
|
298
293
|
let latest;
|
|
299
294
|
for (const row of rows) {
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
result = decodeImproveResult(row.result_json).envelope;
|
|
303
|
-
}
|
|
304
|
-
catch {
|
|
295
|
+
const result = decodeImproveResultRow(row);
|
|
296
|
+
if (!result) {
|
|
305
297
|
resultRows.skipped.invalid += 1;
|
|
306
298
|
continue;
|
|
307
299
|
}
|
|
@@ -322,13 +314,10 @@ export function summarizeImproveRuns(db, since, until) {
|
|
|
322
314
|
}
|
|
323
315
|
/** Project an improve_runs row + wall time + task attribution into one {@link ImproveRunSummary}. */
|
|
324
316
|
export function projectImproveRunSummary(row, wallTimeMs, taskId) {
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
resultStatus = "valid";
|
|
330
|
-
}
|
|
331
|
-
catch {
|
|
317
|
+
const decoded = decodeImproveResultRow(row);
|
|
318
|
+
const result = decoded ?? {};
|
|
319
|
+
const resultStatus = decoded ? "valid" : "invalid";
|
|
320
|
+
if (!decoded) {
|
|
332
321
|
// Keep the persisted row visible in per-run output, but do not project its
|
|
333
322
|
// unknown payload or admit its duration to result-derived denominators.
|
|
334
323
|
wallTimeMs = 0;
|
|
@@ -354,7 +343,6 @@ export function projectImproveRunSummary(row, wallTimeMs, taskId) {
|
|
|
354
343
|
memorySummary: perRow.memorySummary,
|
|
355
344
|
consolidation: perRow.consolidation,
|
|
356
345
|
memoryInference: perRow.memoryInference,
|
|
357
|
-
graphExtraction: perRow.graphExtraction,
|
|
358
346
|
orphansPurged: toFiniteNumber(result.orphansPurged),
|
|
359
347
|
lintFixed: lintSummary ? toFiniteNumber(lintSummary.fixed) : 0,
|
|
360
348
|
lintFlagged: lintSummary ? toFiniteNumber(lintSummary.flagged) : 0,
|
|
@@ -20,7 +20,7 @@ function renderTable(headers, rows) {
|
|
|
20
20
|
*
|
|
21
21
|
* Columns: ts | ok | actions | refl_ok/fail/cd/skip |
|
|
22
22
|
* distill_q/llm-fail/qrej/cfg/skip | cons_proc/promo/merge/del |
|
|
23
|
-
* mem_cons/written/skip |
|
|
23
|
+
* mem_cons/written/skip | orphans | lint_f/fl
|
|
24
24
|
*/
|
|
25
25
|
export function renderRunsDetailMd(runs) {
|
|
26
26
|
const headers = [
|
|
@@ -32,7 +32,6 @@ export function renderRunsDetailMd(runs) {
|
|
|
32
32
|
"distill_q/llm-fail/judge/validator/cfg/skip",
|
|
33
33
|
"cons_proc/promo/merge/del",
|
|
34
34
|
"mem_cons/written/skip",
|
|
35
|
-
"graph_f/e/r",
|
|
36
35
|
"orphans",
|
|
37
36
|
"lint_f/fl",
|
|
38
37
|
"result_status",
|
|
@@ -50,7 +49,6 @@ export function renderRunsDetailMd(runs) {
|
|
|
50
49
|
r.actions.distill.skipped +
|
|
51
50
|
r.actions.memoryPrune +
|
|
52
51
|
r.actions.memoryInference +
|
|
53
|
-
r.actions.graphExtraction +
|
|
54
52
|
r.actions.error;
|
|
55
53
|
return [
|
|
56
54
|
r.startedAt,
|
|
@@ -61,7 +59,6 @@ export function renderRunsDetailMd(runs) {
|
|
|
61
59
|
`${r.actions.distill.queued}/${r.actions.distill.llmFailed}/${r.actions.distill.judgeRejected}/${r.actions.distill.validatorRejected}/${r.actions.distill.configDisabled}/${r.actions.distill.skipped}`,
|
|
62
60
|
`${r.consolidation.processed}/${r.consolidation.promoted}/${r.consolidation.merged}/${r.consolidation.deleted}`,
|
|
63
61
|
`${r.memoryInference.considered}/${r.memoryInference.written}/${r.memoryInference.skippedNoFacts}`,
|
|
64
|
-
`${r.graphExtraction.extractedFiles}/${r.graphExtraction.entities}/${r.graphExtraction.relations}`,
|
|
65
62
|
String(r.orphansPurged),
|
|
66
63
|
`${r.lintFixed}/${r.lintFlagged}`,
|
|
67
64
|
r.resultStatus ?? "valid",
|
|
@@ -81,8 +78,6 @@ export function renderWindowCompareMd(windows, deltas) {
|
|
|
81
78
|
const badIfPositive = new Set([
|
|
82
79
|
"improve.actions.reflect.failed",
|
|
83
80
|
"improve.actions.distill.llmFailed",
|
|
84
|
-
"improve.graphExtraction.failures",
|
|
85
|
-
"improve.graphExtraction.nonArrayBatchFailures",
|
|
86
81
|
"improve.wallTime.medianMs",
|
|
87
82
|
"improve.wallTime.p95Ms",
|
|
88
83
|
"improve.memoryInference.skippedNoFacts",
|
|
@@ -64,11 +64,9 @@ function coercePct(raw) {
|
|
|
64
64
|
function reshapeRun(r) {
|
|
65
65
|
const cons = r.consolidation;
|
|
66
66
|
const mi = r.memoryInference;
|
|
67
|
-
const ge = r.graphExtraction;
|
|
68
67
|
const wall = r.wallTimeMs || 0;
|
|
69
68
|
const consMs = cons.durationMs || 0;
|
|
70
69
|
const miMs = mi.durationMs || 0;
|
|
71
|
-
const geMs = ge.durationMs || 0;
|
|
72
70
|
return {
|
|
73
71
|
id: r.id,
|
|
74
72
|
resultStatus: r.resultStatus ?? "valid",
|
|
@@ -81,16 +79,13 @@ function reshapeRun(r) {
|
|
|
81
79
|
ok: r.ok,
|
|
82
80
|
consDurationMs: consMs,
|
|
83
81
|
miDurationMs: miMs,
|
|
84
|
-
|
|
85
|
-
otherMs: Math.max(0, wall - consMs - miMs - geMs),
|
|
82
|
+
otherMs: Math.max(0, wall - consMs - miMs),
|
|
86
83
|
promoted: cons.promoted,
|
|
87
84
|
merged: cons.merged,
|
|
88
85
|
deleted: cons.deleted,
|
|
89
86
|
contradicted: cons.contradicted,
|
|
90
87
|
judgedNoAction: cons.judgedNoAction,
|
|
91
88
|
miWritten: mi.written,
|
|
92
|
-
geEntities: ge.entities,
|
|
93
|
-
geRelations: ge.relations,
|
|
94
89
|
distillByReason: r.actions.distill.skippedByReason,
|
|
95
90
|
reflectOk: r.actions.reflect.ok,
|
|
96
91
|
reflectFailed: r.actions.reflect.failed,
|
|
@@ -121,11 +116,10 @@ function classify(deltas, metricKeys, lowerIsBetter = false) {
|
|
|
121
116
|
function buildTrend(deltas) {
|
|
122
117
|
const decisionQuality = classify(deltas, ["improve.memoryInference.yieldRate", "improve.consolidation.promoted"]);
|
|
123
118
|
const outputVolume = classify(deltas, ["improve.consolidation.promoted", "improve.memoryInference.written"]);
|
|
124
|
-
const failures = classify(deltas, ["improve.graphExtraction.failures"], true);
|
|
125
119
|
const latency = classify(deltas, ["improve.wallTime.medianMs", "improve.wallTime.p95Ms"], true);
|
|
126
|
-
const score = [decisionQuality, outputVolume,
|
|
120
|
+
const score = [decisionQuality, outputVolume, latency].reduce((acc, d) => acc + (d === "up" ? 1 : d === "down" ? -1 : 0), 0);
|
|
127
121
|
const overall = score >= 1 ? "improving" : score <= -1 ? "degrading" : "mixed";
|
|
128
|
-
return { decisionQuality, outputVolume,
|
|
122
|
+
return { decisionQuality, outputVolume, latency, overall };
|
|
129
123
|
}
|
|
130
124
|
function readSemSearch(advisories) {
|
|
131
125
|
const check = advisories.find((a) => a.name === "semantic-search-runtime");
|
|
@@ -184,7 +178,6 @@ function buildAggregatesPhase(result, runsPhase) {
|
|
|
184
178
|
const improve = result.improve;
|
|
185
179
|
const cons = improve.consolidation;
|
|
186
180
|
const mi = improve.memoryInference;
|
|
187
|
-
const ge = improve.graphExtraction;
|
|
188
181
|
const wallTime = improve.wallTime;
|
|
189
182
|
const coverage = improve.coverage;
|
|
190
183
|
// #576: real per-stage LLM token/time accounting (replaces the GPU-time
|
|
@@ -209,7 +202,6 @@ function buildAggregatesPhase(result, runsPhase) {
|
|
|
209
202
|
return {
|
|
210
203
|
consolidation: cons,
|
|
211
204
|
memoryInference: mi,
|
|
212
|
-
graphExtraction: ge,
|
|
213
205
|
wallTime,
|
|
214
206
|
coverage,
|
|
215
207
|
llm,
|
|
@@ -309,7 +301,7 @@ function groupProposalsBySource(proposals) {
|
|
|
309
301
|
}
|
|
310
302
|
/** Summary table rows: the base metric set + the WS-5 coverage/minting/perf/degradation extensions, when present. */
|
|
311
303
|
function buildSummaryRows(aggregates, trend) {
|
|
312
|
-
const { consolidation: cons,
|
|
304
|
+
const { consolidation: cons, wallTime, llm, coverage } = aggregates;
|
|
313
305
|
const summaryRows = [
|
|
314
306
|
["Task fail rate", aggregates.taskFailRate, "flat"],
|
|
315
307
|
["Agent fail rate", aggregates.agentFailRate, "flat"],
|
|
@@ -332,8 +324,6 @@ function buildSummaryRows(aggregates, trend) {
|
|
|
332
324
|
"Candidates reviewed but intentionally left unchanged (the 'judgedNoAction' field).",
|
|
333
325
|
],
|
|
334
326
|
["Chunk failure", aggregates.chunkFail, "flat"],
|
|
335
|
-
["Graph entities", num(ge.entities), "up"],
|
|
336
|
-
["Graph relations", num(ge.relations), "up"],
|
|
337
327
|
[
|
|
338
328
|
"Stash derived",
|
|
339
329
|
num(aggregates.memorySummary.derived),
|
|
@@ -81,7 +81,6 @@ export const INTERESTING_DELTA_PATHS = [
|
|
|
81
81
|
"improve.memoryInference.written",
|
|
82
82
|
"improve.memoryInference.yieldRate",
|
|
83
83
|
"improve.memoryInference.skippedNoFacts",
|
|
84
|
-
"improve.graphExtraction.failures",
|
|
85
84
|
"improve.autoAccept.promoted",
|
|
86
85
|
"improve.autoAccept.validationFailed",
|
|
87
86
|
"improve.coverage.acceptedProposals",
|
package/dist/commands/health.js
CHANGED
|
@@ -19,6 +19,7 @@ import { getAllEntries } from "../storage/repositories/index-entries-repository.
|
|
|
19
19
|
import { queryTaskHistory } from "../storage/repositories/task-history-repository.js";
|
|
20
20
|
import { getStateDbFreelistInfo, runStateDbQuickCheck } from "../storage/state-db-integrity.js";
|
|
21
21
|
import { pkgVersion } from "../version.js";
|
|
22
|
+
import { collectArchiveUsageAdvisory } from "./health/archive-usage.js";
|
|
22
23
|
import { HEALTH_CHECKS, probeActiveImproveStrategy, runHealthEngineProbes, runPendingStateMigrationsCheck, SESSION_EXTRACTION_LEDGER_WINDOW_DAYS, } from "./health/checks.js";
|
|
23
24
|
import { collectConfigSkewAdvisory } from "./health/config-skew.js";
|
|
24
25
|
import { collectDataDirUsageAdvisory } from "./health/data-dir-usage.js";
|
|
@@ -291,6 +292,18 @@ function gatherAncillaryAdvisories(db, options, egressConfigView) {
|
|
|
291
292
|
catch {
|
|
292
293
|
// Non-fatal.
|
|
293
294
|
}
|
|
295
|
+
// Item 4 (alpha.9 plan §5.4/§8 step 8): a bundle with no git history of its
|
|
296
|
+
// own never gets the purge sweep, so its memory-cleanup archive only ever
|
|
297
|
+
// grows — report its size and file count instead. Best-effort — an
|
|
298
|
+
// unreadable archive must not abort the health report.
|
|
299
|
+
try {
|
|
300
|
+
const archiveUsage = collectArchiveUsageAdvisory(options.stashDir ?? resolveStashDir());
|
|
301
|
+
if (archiveUsage)
|
|
302
|
+
advisories.push(archiveUsage);
|
|
303
|
+
}
|
|
304
|
+
catch {
|
|
305
|
+
// Non-fatal.
|
|
306
|
+
}
|
|
294
307
|
// #896: report the data dir's total size and its largest top-level
|
|
295
308
|
// subdirectory, so a disk-usage blowup (e.g. unpruned migration snapshot
|
|
296
309
|
// backups, #897) is self-diagnosing instead of requiring `du` archaeology.
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import { searchLocal } from "../../../indexer/search/db-search.js";
|
|
5
|
+
import { stripFrontmatterBody } from "../content-hash.js";
|
|
6
|
+
import { stripBundle } from "../ledger.js";
|
|
7
|
+
import { loadRetrievalQueries } from "../retrieval-gate.js";
|
|
8
|
+
/** At most this many of the retired asset's most recent queries are replayed (plan §5.4, §7). */
|
|
9
|
+
export const CONTINUITY_MAX_QUERIES = 5;
|
|
10
|
+
/** "Top 10" per the plan's rule — both the pass/fail cutoff and the search `limit`. */
|
|
11
|
+
export const CONTINUITY_TOP_N = 10;
|
|
12
|
+
/**
|
|
13
|
+
* Builds the real search call the continuity check uses, stateful across
|
|
14
|
+
* every query asked of ONE instance (S2): the first time a query falls back
|
|
15
|
+
* to keyword-only ranking (`mode: "fts-fallback"` — most often a down or
|
|
16
|
+
* unreachable embedding endpoint), every later call through THIS instance
|
|
17
|
+
* forces `semanticSearchMode: "off"` instead of attempting semantic search
|
|
18
|
+
* again, so a dead endpoint costs one failed attempt per run, not one per
|
|
19
|
+
* remaining query (a hanging endpoint at ~3s/query, 300 proposals x 5
|
|
20
|
+
* queries, would otherwise cost on the order of an hour). Construct exactly
|
|
21
|
+
* one instance per pair-pass run and reuse it for every proposal judged, so
|
|
22
|
+
* the throttle covers the whole run, not just one proposal's own queries —
|
|
23
|
+
* and every call made once it has switched, across every remaining proposal
|
|
24
|
+
* in the run, reports `forcedKeywordOnly: true`, not just the one call that
|
|
25
|
+
* discovered the fallback (round-3 review: the first fix only marked THAT
|
|
26
|
+
* call unverified, so a second proposal checked while the endpoint was
|
|
27
|
+
* still down came back with a clean, `mode: "keyword"` — and therefore
|
|
28
|
+
* bulk-acceptable — result).
|
|
29
|
+
*/
|
|
30
|
+
export function createContinuitySearch(stashDir, config) {
|
|
31
|
+
const base = {
|
|
32
|
+
searchType: "any",
|
|
33
|
+
limit: CONTINUITY_TOP_N,
|
|
34
|
+
stashDir,
|
|
35
|
+
sources: [{ path: stashDir, isDefault: true }],
|
|
36
|
+
};
|
|
37
|
+
let keywordOnly = false;
|
|
38
|
+
return async (query) => {
|
|
39
|
+
const forcedKeywordOnly = keywordOnly;
|
|
40
|
+
const callConfig = keywordOnly ? { ...config, semanticSearchMode: "off" } : config;
|
|
41
|
+
const result = await searchLocal({ ...base, query, config: callConfig });
|
|
42
|
+
if (result.mode === "fts-fallback")
|
|
43
|
+
keywordOnly = true;
|
|
44
|
+
return { hits: result.hits, mode: result.mode, forcedKeywordOnly };
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
/** 1-indexed position of `conceptId` in `hits`, or `undefined` if it is not among them. */
|
|
48
|
+
function rankOf(hits, conceptId) {
|
|
49
|
+
const index = hits.findIndex((hit) => stripBundle(hit.ref) === conceptId);
|
|
50
|
+
return index === -1 ? undefined : index + 1;
|
|
51
|
+
}
|
|
52
|
+
/** Body only, whitespace collapsed — the same shape `db-search.ts`'s own content-dedupe compares (S3b). */
|
|
53
|
+
function normalizedBody(raw) {
|
|
54
|
+
return stripFrontmatterBody(raw).replace(/\s+/g, " ").trim();
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* Replay `retiredRef`'s own past queries and check that `successorRef` ranks
|
|
58
|
+
* in the top {@link CONTINUITY_TOP_N} for every one where the retired asset
|
|
59
|
+
* did. Returns `undefined` when there is nothing to flag: the two bodies are
|
|
60
|
+
* content-identical, no recorded queries, or every query ran on the real
|
|
61
|
+
* ranking and either the retired asset never ranked top 10 for it, or the
|
|
62
|
+
* survivor always did too. Never throws.
|
|
63
|
+
*
|
|
64
|
+
* S3: two fixes against false flags measured on a real night-1 admission
|
|
65
|
+
* (300 pairs, 6 flags, half spurious):
|
|
66
|
+
* - queries are the SAME cleaned set `loadRetrievalQueries` replays for the
|
|
67
|
+
* retrieval regression gate (`../retrieval-gate.ts`) — stash-README
|
|
68
|
+
* boilerplate, harness/tool envelopes, pastes over 2,000 characters, and
|
|
69
|
+
* duplicates are dropped before replay, not just capped at 5 raw entries;
|
|
70
|
+
* - when the retired and successor bodies normalize identical, the check
|
|
71
|
+
* never runs at all: search's own content-dedupe (`db-search.ts`) already
|
|
72
|
+
* hides the successor behind the retired asset for every such query, so a
|
|
73
|
+
* "successor missing from the top 10" finding here would not be a real
|
|
74
|
+
* risk, just that dedupe working as designed.
|
|
75
|
+
*
|
|
76
|
+
* S2: a query that never ran (the search call threw), ran on the
|
|
77
|
+
* keyword-only fallback (`mode: "fts-fallback"` — the real ranking was
|
|
78
|
+
* attempted and failed, most often a down or unreachable embedding
|
|
79
|
+
* endpoint), or ran after the shared search instance had already switched
|
|
80
|
+
* to forced keyword-only because an EARLIER query in the same run fell back
|
|
81
|
+
* (`forcedKeywordOnly`) is "unverified" — it is dropped from the rank
|
|
82
|
+
* comparison below (its hits cannot be trusted as "the ranking a user
|
|
83
|
+
* actually gets"), but unlike a query the retired asset simply did not rank
|
|
84
|
+
* for, it can never by itself lead to a silent `undefined` — at least one
|
|
85
|
+
* unverified query always produces a `continuityRisk`, so a dead endpoint
|
|
86
|
+
* reads as "risk unknown" for every proposal it touches that run, never as
|
|
87
|
+
* "no risk found" for the ones checked after the first failure.
|
|
88
|
+
*/
|
|
89
|
+
export async function checkRetirementContinuity(args) {
|
|
90
|
+
// S3b: identical bodies — search's own content-dedupe already hides the
|
|
91
|
+
// successor for every query that would rank the retired asset, so there is
|
|
92
|
+
// no real risk here to check for.
|
|
93
|
+
if (normalizedBody(args.retiredRaw) === normalizedBody(args.successorRaw))
|
|
94
|
+
return undefined;
|
|
95
|
+
// S3a: the SAME cleaned queries the retrieval regression gate replays —
|
|
96
|
+
// boilerplate, envelopes, pastes and duplicates dropped before replay.
|
|
97
|
+
const queries = loadRetrievalQueries(args.ledgerAccess, args.retiredRef).slice(0, CONTINUITY_MAX_QUERIES);
|
|
98
|
+
if (queries.length === 0)
|
|
99
|
+
return undefined; // no queries recorded: no check, no flag
|
|
100
|
+
const search = args.search ?? createContinuitySearch(args.stashDir, args.config);
|
|
101
|
+
// N2: no rank-change-report abstraction — a query only ever needs "did the
|
|
102
|
+
// retired asset rank top 10, and if so, did the successor too?", and
|
|
103
|
+
// `rankOf` (search itself returning at most CONTINUITY_TOP_N hits) already
|
|
104
|
+
// answers both directly.
|
|
105
|
+
const ranks = [];
|
|
106
|
+
let unverifiedQueries = 0;
|
|
107
|
+
for (const query of queries) {
|
|
108
|
+
let hits;
|
|
109
|
+
try {
|
|
110
|
+
const result = await search(query);
|
|
111
|
+
if (result.mode === "fts-fallback" || result.forcedKeywordOnly) {
|
|
112
|
+
unverifiedQueries++; // S2: never silently compare keyword-only ranks
|
|
113
|
+
continue;
|
|
114
|
+
}
|
|
115
|
+
hits = result.hits;
|
|
116
|
+
}
|
|
117
|
+
catch {
|
|
118
|
+
unverifiedQueries++; // S2: a query that never ran cannot be "no risk"
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
const retiredRank = rankOf(hits, args.retiredRef);
|
|
122
|
+
if (retiredRank === undefined)
|
|
123
|
+
continue; // the retired asset itself did not rank top 10 here — nothing to protect
|
|
124
|
+
const successorRank = rankOf(hits, args.successorRef);
|
|
125
|
+
if (successorRank === undefined)
|
|
126
|
+
ranks.push({ query, retiredRank, successorRank: null });
|
|
127
|
+
}
|
|
128
|
+
// Every query verified, and either the retired asset never ranked top 10
|
|
129
|
+
// for any of them, or the survivor always did too: nothing to flag.
|
|
130
|
+
if (unverifiedQueries === 0 && ranks.length === 0)
|
|
131
|
+
return undefined;
|
|
132
|
+
return {
|
|
133
|
+
failingQueries: ranks.length,
|
|
134
|
+
ranks,
|
|
135
|
+
...(unverifiedQueries > 0 ? { unverifiedQueries } : {}),
|
|
136
|
+
};
|
|
137
|
+
}
|