akm-cli 0.9.17-alpha.8 → 0.9.17-alpha.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +334 -0
- package/STABILITY.md +9 -8
- package/dist/assets/hints/cli-hints-full.md +6 -7
- package/dist/assets/improve-strategies/catchup.json +0 -3
- package/dist/assets/improve-strategies/consolidate.json +0 -1
- package/dist/assets/improve-strategies/default.json +1 -2
- package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
- package/dist/assets/improve-strategies/quick.json +1 -2
- package/dist/assets/improve-strategies/reflect-distill.json +1 -2
- package/dist/assets/improve-strategies/thorough.json +0 -3
- package/dist/assets/prompts/consolidate-pair.md +20 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +17 -19
- package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
- package/dist/assets/templates/html/health.html +3 -5
- package/dist/cli/retired-commands.js +1 -1
- package/dist/commands/health/archive-usage.js +98 -0
- package/dist/commands/health/data-dir-usage.js +25 -13
- package/dist/commands/health/html-report.js +1 -4
- package/dist/commands/health/improve-metrics.js +0 -25
- package/dist/commands/health/md-report.js +1 -6
- package/dist/commands/health/report-view-model.js +4 -14
- package/dist/commands/health/windows.js +0 -1
- package/dist/commands/health.js +13 -0
- package/dist/commands/improve/consolidate/continuity-check.js +137 -0
- package/dist/commands/improve/consolidate/pair-pass.js +791 -0
- package/dist/commands/improve/consolidate.js +38 -63
- package/dist/commands/improve/extract-prompt.js +1 -2
- package/dist/commands/improve/improve-cli.js +1 -1
- package/dist/commands/improve/improve-strategies.js +23 -5
- package/dist/commands/improve/improve.js +19 -30
- package/dist/commands/improve/ledger.js +3 -2
- package/dist/commands/improve/loop-stages.js +5 -85
- package/dist/commands/improve/memory/memory-belief.js +3 -1
- package/dist/commands/improve/memory/memory-improve.js +269 -11
- package/dist/commands/improve/planner.js +0 -5
- package/dist/commands/improve/preparation.js +20 -135
- package/dist/commands/improve/retrieval-scope.js +19 -4
- package/dist/commands/improve/salience.js +1 -14
- package/dist/commands/improve/stage.js +0 -1
- package/dist/commands/lint/base-linter.js +19 -11
- package/dist/commands/proposal/drain.js +8 -1
- package/dist/commands/proposal/proposal-cli.js +16 -2
- package/dist/commands/proposal/proposal-types.js +7 -0
- package/dist/commands/proposal/proposal.js +37 -6
- package/dist/commands/proposal/repository.js +613 -4
- package/dist/commands/proposal/validators/proposals.js +9 -0
- package/dist/commands/read/knowledge.js +3 -2
- package/dist/commands/read/show.js +0 -14
- package/dist/commands/sources/stash-cli.js +2 -2
- package/dist/core/bundle-rename.js +1 -7
- package/dist/core/config/config-schema.js +8 -1
- package/dist/core/config/config.js +23 -48
- package/dist/core/config/engine-semantics.js +0 -2
- package/dist/core/config/schema/improve-processes.js +17 -42
- package/dist/core/config/schema/index-config.js +5 -25
- package/dist/core/file-change.js +13 -5
- package/dist/core/improve-result.js +16 -5
- package/dist/core/improve-types.js +0 -1
- package/dist/core/loopback.js +7 -12
- package/dist/core/parse.js +13 -16
- package/dist/core/state/migrations.js +15 -0
- package/dist/core/time.js +0 -20
- package/dist/indexer/db/llm-cache.js +2 -2
- package/dist/indexer/ensure-index.js +2 -2
- package/dist/indexer/index-written-assets.js +2 -3
- package/dist/indexer/indexer.js +18 -418
- package/dist/indexer/passes/metadata.js +0 -19
- package/dist/indexer/walk/walker.js +3 -4
- package/dist/llm/client.js +8 -10
- package/dist/llm/embedders/remote.js +1 -2
- package/dist/llm/feature-gate.js +0 -5
- package/dist/output/shapes/helpers.js +20 -4
- package/dist/output/text/command-format.js +0 -8
- package/dist/output/text/proposal-format.js +47 -1
- package/dist/output/text/show-format.js +0 -20
- package/dist/scripts/akm-migrate-node.js +917 -950
- package/dist/scripts/akm-migrate.js +917 -950
- package/dist/setup/steps/connection.js +5 -6
- package/dist/setup/steps/platforms.js +2 -2
- package/dist/sources/providers/git-stash.js +55 -4
- package/dist/storage/repositories/improve-ledger-repository.js +48 -7
- package/dist/storage/repositories/index-entries-repository.js +4 -7
- package/dist/storage/repositories/index-entry-schema.js +4 -2
- package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
- package/dist/storage/repositories/index-schema.js +55 -104
- package/dist/storage/repositories/proposals-repository.js +61 -0
- package/dist/storage/repositories/salience-repository.js +1 -19
- package/docs/reference/cli.md +16 -19
- package/docs/reference/configuration.md +21 -12
- package/docs/reference/data-and-telemetry.md +0 -1
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -342
- package/dist/assets/improve-strategies/graph-refresh.json +0 -15
- package/dist/assets/prompts/contradiction-judge.md +0 -33
- package/dist/assets/prompts/graph-extract-system.md +0 -1
- package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
- package/dist/assets/prompts/metadata-enhance-system.md +0 -1
- package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
- package/dist/indexer/db/graph-db.js +0 -399
- package/dist/indexer/graph/graph-extraction.js +0 -809
- package/dist/indexer/graph/graph-related.js +0 -131
- package/dist/indexer/graph/graph-types.js +0 -4
- package/dist/llm/graph-extract.js +0 -892
- package/dist/llm/metadata-enhance.js +0 -95
|
@@ -325,7 +325,7 @@
|
|
|
325
325
|
|
|
326
326
|
<div class="chart-panel full-width">
|
|
327
327
|
<h3>Per-Phase Wall Time Decomposition (stacked, min)</h3>
|
|
328
|
-
<div id="chartPhases" class="ec tall" role="img" aria-label="Stacked area chart decomposing wall time per run into consolidation, memory inference,
|
|
328
|
+
<div id="chartPhases" class="ec tall" role="img" aria-label="Stacked area chart decomposing wall time per run into consolidation, memory inference, and unattributed phases"></div>
|
|
329
329
|
<div class="empty-overlay" id="emptyPhases">No runs in the selected slice.</div>
|
|
330
330
|
</div>
|
|
331
331
|
|
|
@@ -390,7 +390,7 @@
|
|
|
390
390
|
<div class="table-wrap">
|
|
391
391
|
<table>
|
|
392
392
|
<thead>
|
|
393
|
-
<tr><th>Started</th><th>Task</th><th>Strategy</th><th>Wall</th><th>Promoted</th><th>Merged</th><th>Contradicted</th><th>MI Written</th><th>
|
|
393
|
+
<tr><th>Started</th><th>Task</th><th>Strategy</th><th>Wall</th><th>Promoted</th><th>Merged</th><th>Contradicted</th><th>MI Written</th><th>Lint Fixed</th><th>Status</th></tr>
|
|
394
394
|
</thead>
|
|
395
395
|
<tbody id="lastRunsTable"></tbody>
|
|
396
396
|
</table>
|
|
@@ -539,7 +539,7 @@ mkChart('chartWallTime', {
|
|
|
539
539
|
// ── 2. Per-phase wall time decomposition (stacked area) ──────────────────────
|
|
540
540
|
mkChart('chartPhases', {
|
|
541
541
|
tooltip: baseTooltip,
|
|
542
|
-
legend: { ...baseLegend, data: ['Consolidation', 'Memory inference', '
|
|
542
|
+
legend: { ...baseLegend, data: ['Consolidation', 'Memory inference', 'Unattributed'] },
|
|
543
543
|
grid: denseGrid,
|
|
544
544
|
dataZoom: denseZoom,
|
|
545
545
|
xAxis: { type: 'category', data: xLabels, ...baseAxis },
|
|
@@ -547,7 +547,6 @@ mkChart('chartPhases', {
|
|
|
547
547
|
series: [
|
|
548
548
|
{ name: 'Consolidation', type: 'line', stack: 'p', areaStyle: { color: 'rgba(88,166,255,0.55)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.consDurationMs)) },
|
|
549
549
|
{ name: 'Memory inference', type: 'line', stack: 'p', areaStyle: { color: 'rgba(63,185,80,0.55)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.miDurationMs)) },
|
|
550
|
-
{ name: 'Graph extraction', type: 'line', stack: 'p', areaStyle: { color: 'rgba(188,140,255,0.55)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.geDurationMs)) },
|
|
551
550
|
{ name: 'Unattributed', type: 'line', stack: 'p', areaStyle: { color: 'rgba(210,153,34,0.45)' }, lineStyle: { width: 0 }, showSymbol: false, data: rs.map(r => toMin(r.otherMs)) },
|
|
552
551
|
],
|
|
553
552
|
});
|
|
@@ -695,7 +694,6 @@ tbody.innerHTML = '';
|
|
|
695
694
|
<td>${r.merged || 0}</td>
|
|
696
695
|
<td style="color:${(r.contradicted||0)>5?'var(--yellow)':'var(--text)'};">${r.contradicted || 0}</td>
|
|
697
696
|
<td style="color:var(--green);">${r.miWritten || 0}</td>
|
|
698
|
-
<td>${r.geEntities || 0}</td>
|
|
699
697
|
<td>${r.lintFixed || 0}</td>
|
|
700
698
|
<td>${badge}</td>
|
|
701
699
|
`;
|
|
@@ -38,7 +38,7 @@ const RETIRED_COMMAND_HINTS = {
|
|
|
38
38
|
events: "`akm events` moved in 0.9 — use `akm log`.",
|
|
39
39
|
// Removed observability surfaces.
|
|
40
40
|
history: "`akm history` was removed in 0.9 — use `akm log --ref <ref>` for an asset's event trail.",
|
|
41
|
-
graph: "`akm graph` was removed in 0.9
|
|
41
|
+
graph: "`akm graph` was removed in 0.9. The LLM entity graph it inspected was itself retired in 0.9.17-alpha.9 — use `akm show <ref>` for an asset's declared links.",
|
|
42
42
|
lessons: "`akm lessons` was removed in 0.9 — lesson strength is indexed; use `akm search --type lesson`.",
|
|
43
43
|
lesson: "`akm lesson` was removed in 0.9 — lesson strength is indexed; use `akm search --type lesson`.",
|
|
44
44
|
// Relocated guidance / removed asset verbs.
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* `memory-cleanup-archive` advisory for `akm health` (item 4, 0.9.17-alpha.9
|
|
6
|
+
* plan §5.4/§8 step 8).
|
|
7
|
+
*
|
|
8
|
+
* Reports the archive's size and file count for every bundle, git-backed or
|
|
9
|
+
* not. A bundle with no `.git` of its own keeps every retirement's archived
|
|
10
|
+
* bytes forever (the purge sweep never runs there at all), so that alone is
|
|
11
|
+
* reported. A git-backed bundle can ALSO have bytes the purge sweep will
|
|
12
|
+
* never remove: `.git` presence alone does not prove a retirement was ever
|
|
13
|
+
* committed (`proposal accept` only commits for a `kind: "git"` write
|
|
14
|
+
* target, and a `kind: "filesystem"` bundle that merely happens to have a
|
|
15
|
+
* `.git` directory never gets one) — those bytes sit there indefinitely,
|
|
16
|
+
* however old they get, with nothing else to say so. This checks the SAME
|
|
17
|
+
* git state `purgeGracedArchive` checks (tracked, clean, verifiable) and
|
|
18
|
+
* reports how much of the archive fails it.
|
|
19
|
+
*
|
|
20
|
+
* Silent whenever there is nothing to say: the archive does not exist or is
|
|
21
|
+
* empty, or (for a git-backed bundle) every byte in it is purgeable once it
|
|
22
|
+
* ages out.
|
|
23
|
+
*/
|
|
24
|
+
import fs from "node:fs";
|
|
25
|
+
import path from "node:path";
|
|
26
|
+
import { MEMORY_ARCHIVE_REL } from "../../core/asset/memory-archive.js";
|
|
27
|
+
import { toPosix } from "../../core/common.js";
|
|
28
|
+
import { isGitBackedStash, tryListGitChangedPaths, tryListGitTrackedPaths, tryListGitUnverifiablePaths, } from "../../sources/providers/git-stash.js";
|
|
29
|
+
import { MAX_WALK_ENTRIES, sizeOfPath } from "./data-dir-usage.js";
|
|
30
|
+
/**
|
|
31
|
+
* Build the `memory-cleanup-archive` advisory, or `undefined` when there is
|
|
32
|
+
* nothing to report.
|
|
33
|
+
*/
|
|
34
|
+
export function collectArchiveUsageAdvisory(stashDir) {
|
|
35
|
+
const archiveRoot = path.join(stashDir, MEMORY_ARCHIVE_REL);
|
|
36
|
+
if (!fs.existsSync(archiveRoot))
|
|
37
|
+
return undefined;
|
|
38
|
+
if (!isGitBackedStash(stashDir)) {
|
|
39
|
+
const usage = sizeOfPath(archiveRoot, { remaining: MAX_WALK_ENTRIES });
|
|
40
|
+
if (usage.files === 0)
|
|
41
|
+
return undefined;
|
|
42
|
+
const lowerBound = usage.truncated ? ` (a lower bound — the walk stopped after ${MAX_WALK_ENTRIES} entries)` : "";
|
|
43
|
+
return {
|
|
44
|
+
name: "memory-cleanup-archive",
|
|
45
|
+
kind: "deterministic",
|
|
46
|
+
status: "pass",
|
|
47
|
+
confidence: "high",
|
|
48
|
+
message: `${usage.files} archived file(s), ${usage.bytes} byte(s)${lowerBound} in .akm/memory-cleanup/archive — ` +
|
|
49
|
+
"this bundle has no git history, so the purge sweep leaves it untouched.",
|
|
50
|
+
evidence: { files: usage.files, bytes: usage.bytes, truncated: usage.truncated },
|
|
51
|
+
};
|
|
52
|
+
}
|
|
53
|
+
// Git-backed: the SAME three checks purgeGracedArchive runs (B1, G10) —
|
|
54
|
+
// computed once here, not per file, and reused via `onFile` below instead
|
|
55
|
+
// of a second walk of the same tree.
|
|
56
|
+
const dirtyQuery = tryListGitChangedPaths(stashDir);
|
|
57
|
+
const trackedQuery = tryListGitTrackedPaths(stashDir, MEMORY_ARCHIVE_REL);
|
|
58
|
+
const unverifiableQuery = tryListGitUnverifiablePaths(stashDir, MEMORY_ARCHIVE_REL);
|
|
59
|
+
// A failed git check here fails the same way purgeGracedArchive's own
|
|
60
|
+
// sweep would: nothing in the archive can be proven purgeable, so every
|
|
61
|
+
// byte counts as unpurgeable rather than guessing.
|
|
62
|
+
const gitStateKnown = dirtyQuery.ok && trackedQuery.ok && unverifiableQuery.ok;
|
|
63
|
+
const dirty = new Set(dirtyQuery.paths);
|
|
64
|
+
const tracked = new Set(trackedQuery.paths);
|
|
65
|
+
const unverifiable = new Set(unverifiableQuery.paths);
|
|
66
|
+
let unpurgeableFiles = 0;
|
|
67
|
+
let unpurgeableBytes = 0;
|
|
68
|
+
const usage = sizeOfPath(archiveRoot, { remaining: MAX_WALK_ENTRIES }, (filePath, bytes) => {
|
|
69
|
+
const key = toPosix(path.relative(stashDir, filePath));
|
|
70
|
+
const safe = gitStateKnown && tracked.has(key) && !dirty.has(key) && !unverifiable.has(key);
|
|
71
|
+
if (!safe) {
|
|
72
|
+
unpurgeableFiles++;
|
|
73
|
+
unpurgeableBytes += bytes;
|
|
74
|
+
}
|
|
75
|
+
});
|
|
76
|
+
if (usage.files === 0)
|
|
77
|
+
return undefined;
|
|
78
|
+
if (unpurgeableFiles === 0)
|
|
79
|
+
return undefined; // everything here is purgeable once it ages out — nothing to say
|
|
80
|
+
const lowerBound = usage.truncated ? ` (a lower bound — the walk stopped after ${MAX_WALK_ENTRIES} entries)` : "";
|
|
81
|
+
return {
|
|
82
|
+
name: "memory-cleanup-archive",
|
|
83
|
+
kind: "deterministic",
|
|
84
|
+
status: "warn",
|
|
85
|
+
confidence: "high",
|
|
86
|
+
message: `${usage.files} archived file(s), ${usage.bytes} byte(s)${lowerBound} in .akm/memory-cleanup/archive; ` +
|
|
87
|
+
`${unpurgeableFiles} file(s), ${unpurgeableBytes} byte(s) of that cannot be purged (untracked, modified, ` +
|
|
88
|
+
"or unverifiable in git) — commit them so the purge sweep can remove them once they age out.",
|
|
89
|
+
evidence: {
|
|
90
|
+
files: usage.files,
|
|
91
|
+
bytes: usage.bytes,
|
|
92
|
+
truncated: usage.truncated,
|
|
93
|
+
unpurgeableFiles,
|
|
94
|
+
unpurgeableBytes,
|
|
95
|
+
gitStateKnown,
|
|
96
|
+
},
|
|
97
|
+
};
|
|
98
|
+
}
|
|
@@ -41,7 +41,7 @@ const DOMINANT_SUBDIR_PERCENT_THRESHOLD = 50;
|
|
|
41
41
|
* cap the walk stops descending further and the advisory says its size
|
|
42
42
|
* figures are a lower bound.
|
|
43
43
|
*/
|
|
44
|
-
const MAX_WALK_ENTRIES = 100_000;
|
|
44
|
+
export const MAX_WALK_ENTRIES = 100_000;
|
|
45
45
|
/**
|
|
46
46
|
* Below this the data dir is not worth an opinion. The advisory exists for
|
|
47
47
|
* disk blowups (74 GB in the incident); on a small directory a ratio is
|
|
@@ -59,31 +59,42 @@ function liveDbBytesFor(name, sizes) {
|
|
|
59
59
|
return ((sizes.get(name)?.bytes ?? 0) + (sizes.get(`${name}-wal`)?.bytes ?? 0) + (sizes.get(`${name}-shm`)?.bytes ?? 0));
|
|
60
60
|
}
|
|
61
61
|
/**
|
|
62
|
-
* Recursively sum file sizes under `root` (stat-only,
|
|
63
|
-
* followed so a cyclic or huge-target symlink can't blow up the
|
|
64
|
-
* `budget` is a shared mutable counter across the whole tree so the
|
|
65
|
-
*
|
|
62
|
+
* Recursively sum file sizes (and count files) under `root` (stat-only,
|
|
63
|
+
* symlinks not followed so a cyclic or huge-target symlink can't blow up the
|
|
64
|
+
* walk). `budget` is a shared mutable counter across the whole tree so the
|
|
65
|
+
* entry cap applies to the walk as a whole, not per-branch — callers
|
|
66
|
+
* typically pass {@link MAX_WALK_ENTRIES}, sized for this module's own data
|
|
67
|
+
* dir walk, but a smaller/larger budget is fine for a different tree.
|
|
68
|
+
* Shared with the `archive-usage` advisory (N2) — the same "don't let a
|
|
69
|
+
* pathological tree hang a health check" concern applies to both.
|
|
70
|
+
*
|
|
71
|
+
* `onFile`, when given, is called once per leaf file (path, bytes) as the
|
|
72
|
+
* walk visits it — `archive-usage` uses this to classify each file's git
|
|
73
|
+
* state without a second, separate walk of the same tree.
|
|
66
74
|
*/
|
|
67
|
-
function sizeOfPath(root, budget) {
|
|
75
|
+
export function sizeOfPath(root, budget, onFile) {
|
|
68
76
|
let stat;
|
|
69
77
|
try {
|
|
70
78
|
stat = fs.lstatSync(root);
|
|
71
79
|
}
|
|
72
80
|
catch {
|
|
73
|
-
return { bytes: 0, truncated: false };
|
|
81
|
+
return { bytes: 0, files: 0, truncated: false };
|
|
74
82
|
}
|
|
75
83
|
if (stat.isSymbolicLink())
|
|
76
|
-
return { bytes: 0, truncated: false };
|
|
77
|
-
if (!stat.isDirectory())
|
|
78
|
-
|
|
84
|
+
return { bytes: 0, files: 0, truncated: false };
|
|
85
|
+
if (!stat.isDirectory()) {
|
|
86
|
+
onFile?.(root, stat.size);
|
|
87
|
+
return { bytes: stat.size, files: 1, truncated: false };
|
|
88
|
+
}
|
|
79
89
|
let entries;
|
|
80
90
|
try {
|
|
81
91
|
entries = fs.readdirSync(root, { withFileTypes: true });
|
|
82
92
|
}
|
|
83
93
|
catch {
|
|
84
|
-
return { bytes: 0, truncated: false };
|
|
94
|
+
return { bytes: 0, files: 0, truncated: false };
|
|
85
95
|
}
|
|
86
96
|
let bytes = 0;
|
|
97
|
+
let files = 0;
|
|
87
98
|
let truncated = false;
|
|
88
99
|
for (const entry of entries) {
|
|
89
100
|
if (budget.remaining <= 0) {
|
|
@@ -91,12 +102,13 @@ function sizeOfPath(root, budget) {
|
|
|
91
102
|
break;
|
|
92
103
|
}
|
|
93
104
|
budget.remaining--;
|
|
94
|
-
const sub = sizeOfPath(path.join(root, entry.name), budget);
|
|
105
|
+
const sub = sizeOfPath(path.join(root, entry.name), budget, onFile);
|
|
95
106
|
bytes += sub.bytes;
|
|
107
|
+
files += sub.files;
|
|
96
108
|
if (sub.truncated)
|
|
97
109
|
truncated = true;
|
|
98
110
|
}
|
|
99
|
-
return { bytes, truncated };
|
|
111
|
+
return { bytes, files, truncated };
|
|
100
112
|
}
|
|
101
113
|
/** `1610612736` -> `"1.5G"`. Values under 10 in a unit keep one decimal; 10+ round to an integer. */
|
|
102
114
|
function formatBytes(bytes) {
|
|
@@ -131,7 +131,6 @@ function renderExecSummary(vm) {
|
|
|
131
131
|
const trendRows = [
|
|
132
132
|
trendLi("Decision quality", vm.trend.decisionQuality),
|
|
133
133
|
trendLi("Output volume", vm.trend.outputVolume),
|
|
134
|
-
trendLi("Failures", vm.trend.failures),
|
|
135
134
|
trendLi("Latency", vm.trend.latency),
|
|
136
135
|
].join("");
|
|
137
136
|
const deltaRows = [
|
|
@@ -151,7 +150,6 @@ function renderExecSummary(vm) {
|
|
|
151
150
|
li("Promoted", String(vm.latest.promoted)),
|
|
152
151
|
li("Judged: no action", `<abbr title="Candidates reviewed but intentionally left unchanged on this run.">${vm.latest.judgedNoAction}</abbr>`),
|
|
153
152
|
li("MI written", String(vm.latest.miWritten)),
|
|
154
|
-
li("Graph entities/relations", `${vm.latest.geEntities} / ${vm.latest.geRelations}`),
|
|
155
153
|
].join("")
|
|
156
154
|
: '<li><span class="k">No runs in window</span><span class="v">—</span></li>';
|
|
157
155
|
const windowRows = [
|
|
@@ -197,7 +195,7 @@ function renderExecSummary(vm) {
|
|
|
197
195
|
</div>
|
|
198
196
|
</div>
|
|
199
197
|
<div class="overall">Overall trend: <b>${esc(vm.trend.overall)}</b> ${overallEmoji}
|
|
200
|
-
· based on decision quality, output volume,
|
|
198
|
+
· based on decision quality, output volume, and latency ${vm.comparisonMode === "custom" ? "across the selected windows" : "vs the prior window"}.</div>`.trim();
|
|
201
199
|
}
|
|
202
200
|
/**
|
|
203
201
|
* Color is a health SIGNAL, not decoration: green/yellow/red where a card has
|
|
@@ -221,7 +219,6 @@ function renderKpiCards(vm) {
|
|
|
221
219
|
kpiCard("neutral", "Median Duration", `${vm.medianDurMin}m`, `p95 = ${vm.p95DurMin}m`),
|
|
222
220
|
kpiCard("blue", "Total Promoted", num(vm.consolidation.promoted), `avg ${vm.avgPromoted} / run`),
|
|
223
221
|
kpiCard("blue", "MI Written", num(vm.miWritten), `${vm.miYieldRate} yield rate`),
|
|
224
|
-
kpiCard("purple", "Graph Entities", num(vm.graphExtraction.entities), `+${num(vm.graphExtraction.relations)} relations`),
|
|
225
222
|
kpiCard("neutral", "Stash Derived", num(vm.memorySummary.derived), `of ${num(vm.memorySummary.eligible)} eligible (whole-stash)`),
|
|
226
223
|
// #576: real LLM work — duration leads, tokens compact, not a GPU proxy.
|
|
227
224
|
kpiCard(vm.llm.calls > 0 ? "purple" : "neutral", "🧠 LLM Work", `${vm.llmTokensCompact} tok`, `${fmtMs(vm.llm.totalDurationMs)} · ${num(vm.llm.calls)} calls · ${compact(vm.llm.reasoningTokens)} reasoning`),
|
|
@@ -74,7 +74,6 @@ export function emptyImproveMetrics() {
|
|
|
74
74
|
},
|
|
75
75
|
memoryPrune: 0,
|
|
76
76
|
memoryInference: 0,
|
|
77
|
-
graphExtraction: 0,
|
|
78
77
|
error: 0,
|
|
79
78
|
},
|
|
80
79
|
autoAccept: { promoted: 0, validationFailed: 0 },
|
|
@@ -91,7 +90,6 @@ export function emptyImproveMetrics() {
|
|
|
91
90
|
durationMs: 0,
|
|
92
91
|
},
|
|
93
92
|
memoryInference: { considered: 0, freshAttempts: 0, written: 0, skippedNoFacts: 0, yieldRate: 0, durationMs: 0 },
|
|
94
|
-
graphExtraction: { extractedFiles: 0, entities: 0, relations: 0, failures: 0, durationMs: 0 },
|
|
95
93
|
wallTime: { medianMs: 0, p95Ms: 0 },
|
|
96
94
|
coverage: { acceptedProposals: 0, distinctRefs: 0 },
|
|
97
95
|
};
|
|
@@ -151,9 +149,6 @@ function applyAction(metrics, action) {
|
|
|
151
149
|
case "memory-inference":
|
|
152
150
|
metrics.actions.memoryInference += 1;
|
|
153
151
|
break;
|
|
154
|
-
case "graph-extraction":
|
|
155
|
-
metrics.actions.graphExtraction += 1;
|
|
156
|
-
break;
|
|
157
152
|
case "error":
|
|
158
153
|
metrics.actions.error += 1;
|
|
159
154
|
break;
|
|
@@ -209,19 +204,6 @@ function projectRunMetrics(result) {
|
|
|
209
204
|
mi.skippedNoFacts += toFiniteNumber(memoryInference.skippedNoFacts);
|
|
210
205
|
}
|
|
211
206
|
metrics.memoryInference.durationMs += toFiniteNumber(result.memoryInferenceDurationMs);
|
|
212
|
-
const graphExtraction = result.graphExtraction;
|
|
213
|
-
if (graphExtraction) {
|
|
214
|
-
const ge = metrics.graphExtraction;
|
|
215
|
-
// This run's counts, like entities and relations. `quality` describes the
|
|
216
|
-
// whole stored graph after the run; summing it would count each stored file
|
|
217
|
-
// once per run in the window.
|
|
218
|
-
ge.extractedFiles += toFiniteNumber(graphExtraction.extracted);
|
|
219
|
-
ge.entities += toFiniteNumber(graphExtraction.totalEntities);
|
|
220
|
-
ge.relations += toFiniteNumber(graphExtraction.totalRelations);
|
|
221
|
-
const telemetry = graphExtraction.telemetry;
|
|
222
|
-
ge.failures += toFiniteNumber(telemetry?.failureCount);
|
|
223
|
-
}
|
|
224
|
-
metrics.graphExtraction.durationMs += toFiniteNumber(result.graphExtractionDurationMs);
|
|
225
207
|
return metrics;
|
|
226
208
|
}
|
|
227
209
|
/** Derived rates on an accumulator (window aggregate or single run). */
|
|
@@ -251,7 +233,6 @@ function mergeImproveMetrics(dst, src) {
|
|
|
251
233
|
}
|
|
252
234
|
dst.actions.memoryPrune += src.actions.memoryPrune;
|
|
253
235
|
dst.actions.memoryInference += src.actions.memoryInference;
|
|
254
|
-
dst.actions.graphExtraction += src.actions.graphExtraction;
|
|
255
236
|
dst.actions.error += src.actions.error;
|
|
256
237
|
dst.autoAccept.promoted += src.autoAccept.promoted;
|
|
257
238
|
dst.autoAccept.validationFailed += src.autoAccept.validationFailed;
|
|
@@ -269,11 +250,6 @@ function mergeImproveMetrics(dst, src) {
|
|
|
269
250
|
dst.memoryInference.written += src.memoryInference.written;
|
|
270
251
|
dst.memoryInference.skippedNoFacts += src.memoryInference.skippedNoFacts;
|
|
271
252
|
dst.memoryInference.durationMs += src.memoryInference.durationMs;
|
|
272
|
-
dst.graphExtraction.extractedFiles += src.graphExtraction.extractedFiles;
|
|
273
|
-
dst.graphExtraction.entities += src.graphExtraction.entities;
|
|
274
|
-
dst.graphExtraction.relations += src.graphExtraction.relations;
|
|
275
|
-
dst.graphExtraction.failures += src.graphExtraction.failures;
|
|
276
|
-
dst.graphExtraction.durationMs += src.graphExtraction.durationMs;
|
|
277
253
|
}
|
|
278
254
|
function compareImproveRunRecency(a, b) {
|
|
279
255
|
const started = a.started_at.localeCompare(b.started_at);
|
|
@@ -354,7 +330,6 @@ export function projectImproveRunSummary(row, wallTimeMs, taskId) {
|
|
|
354
330
|
memorySummary: perRow.memorySummary,
|
|
355
331
|
consolidation: perRow.consolidation,
|
|
356
332
|
memoryInference: perRow.memoryInference,
|
|
357
|
-
graphExtraction: perRow.graphExtraction,
|
|
358
333
|
orphansPurged: toFiniteNumber(result.orphansPurged),
|
|
359
334
|
lintFixed: lintSummary ? toFiniteNumber(lintSummary.fixed) : 0,
|
|
360
335
|
lintFlagged: lintSummary ? toFiniteNumber(lintSummary.flagged) : 0,
|
|
@@ -20,7 +20,7 @@ function renderTable(headers, rows) {
|
|
|
20
20
|
*
|
|
21
21
|
* Columns: ts | ok | actions | refl_ok/fail/cd/skip |
|
|
22
22
|
* distill_q/llm-fail/qrej/cfg/skip | cons_proc/promo/merge/del |
|
|
23
|
-
* mem_cons/written/skip |
|
|
23
|
+
* mem_cons/written/skip | orphans | lint_f/fl
|
|
24
24
|
*/
|
|
25
25
|
export function renderRunsDetailMd(runs) {
|
|
26
26
|
const headers = [
|
|
@@ -32,7 +32,6 @@ export function renderRunsDetailMd(runs) {
|
|
|
32
32
|
"distill_q/llm-fail/judge/validator/cfg/skip",
|
|
33
33
|
"cons_proc/promo/merge/del",
|
|
34
34
|
"mem_cons/written/skip",
|
|
35
|
-
"graph_f/e/r",
|
|
36
35
|
"orphans",
|
|
37
36
|
"lint_f/fl",
|
|
38
37
|
"result_status",
|
|
@@ -50,7 +49,6 @@ export function renderRunsDetailMd(runs) {
|
|
|
50
49
|
r.actions.distill.skipped +
|
|
51
50
|
r.actions.memoryPrune +
|
|
52
51
|
r.actions.memoryInference +
|
|
53
|
-
r.actions.graphExtraction +
|
|
54
52
|
r.actions.error;
|
|
55
53
|
return [
|
|
56
54
|
r.startedAt,
|
|
@@ -61,7 +59,6 @@ export function renderRunsDetailMd(runs) {
|
|
|
61
59
|
`${r.actions.distill.queued}/${r.actions.distill.llmFailed}/${r.actions.distill.judgeRejected}/${r.actions.distill.validatorRejected}/${r.actions.distill.configDisabled}/${r.actions.distill.skipped}`,
|
|
62
60
|
`${r.consolidation.processed}/${r.consolidation.promoted}/${r.consolidation.merged}/${r.consolidation.deleted}`,
|
|
63
61
|
`${r.memoryInference.considered}/${r.memoryInference.written}/${r.memoryInference.skippedNoFacts}`,
|
|
64
|
-
`${r.graphExtraction.extractedFiles}/${r.graphExtraction.entities}/${r.graphExtraction.relations}`,
|
|
65
62
|
String(r.orphansPurged),
|
|
66
63
|
`${r.lintFixed}/${r.lintFlagged}`,
|
|
67
64
|
r.resultStatus ?? "valid",
|
|
@@ -81,8 +78,6 @@ export function renderWindowCompareMd(windows, deltas) {
|
|
|
81
78
|
const badIfPositive = new Set([
|
|
82
79
|
"improve.actions.reflect.failed",
|
|
83
80
|
"improve.actions.distill.llmFailed",
|
|
84
|
-
"improve.graphExtraction.failures",
|
|
85
|
-
"improve.graphExtraction.nonArrayBatchFailures",
|
|
86
81
|
"improve.wallTime.medianMs",
|
|
87
82
|
"improve.wallTime.p95Ms",
|
|
88
83
|
"improve.memoryInference.skippedNoFacts",
|
|
@@ -64,11 +64,9 @@ function coercePct(raw) {
|
|
|
64
64
|
function reshapeRun(r) {
|
|
65
65
|
const cons = r.consolidation;
|
|
66
66
|
const mi = r.memoryInference;
|
|
67
|
-
const ge = r.graphExtraction;
|
|
68
67
|
const wall = r.wallTimeMs || 0;
|
|
69
68
|
const consMs = cons.durationMs || 0;
|
|
70
69
|
const miMs = mi.durationMs || 0;
|
|
71
|
-
const geMs = ge.durationMs || 0;
|
|
72
70
|
return {
|
|
73
71
|
id: r.id,
|
|
74
72
|
resultStatus: r.resultStatus ?? "valid",
|
|
@@ -81,16 +79,13 @@ function reshapeRun(r) {
|
|
|
81
79
|
ok: r.ok,
|
|
82
80
|
consDurationMs: consMs,
|
|
83
81
|
miDurationMs: miMs,
|
|
84
|
-
|
|
85
|
-
otherMs: Math.max(0, wall - consMs - miMs - geMs),
|
|
82
|
+
otherMs: Math.max(0, wall - consMs - miMs),
|
|
86
83
|
promoted: cons.promoted,
|
|
87
84
|
merged: cons.merged,
|
|
88
85
|
deleted: cons.deleted,
|
|
89
86
|
contradicted: cons.contradicted,
|
|
90
87
|
judgedNoAction: cons.judgedNoAction,
|
|
91
88
|
miWritten: mi.written,
|
|
92
|
-
geEntities: ge.entities,
|
|
93
|
-
geRelations: ge.relations,
|
|
94
89
|
distillByReason: r.actions.distill.skippedByReason,
|
|
95
90
|
reflectOk: r.actions.reflect.ok,
|
|
96
91
|
reflectFailed: r.actions.reflect.failed,
|
|
@@ -121,11 +116,10 @@ function classify(deltas, metricKeys, lowerIsBetter = false) {
|
|
|
121
116
|
function buildTrend(deltas) {
|
|
122
117
|
const decisionQuality = classify(deltas, ["improve.memoryInference.yieldRate", "improve.consolidation.promoted"]);
|
|
123
118
|
const outputVolume = classify(deltas, ["improve.consolidation.promoted", "improve.memoryInference.written"]);
|
|
124
|
-
const failures = classify(deltas, ["improve.graphExtraction.failures"], true);
|
|
125
119
|
const latency = classify(deltas, ["improve.wallTime.medianMs", "improve.wallTime.p95Ms"], true);
|
|
126
|
-
const score = [decisionQuality, outputVolume,
|
|
120
|
+
const score = [decisionQuality, outputVolume, latency].reduce((acc, d) => acc + (d === "up" ? 1 : d === "down" ? -1 : 0), 0);
|
|
127
121
|
const overall = score >= 1 ? "improving" : score <= -1 ? "degrading" : "mixed";
|
|
128
|
-
return { decisionQuality, outputVolume,
|
|
122
|
+
return { decisionQuality, outputVolume, latency, overall };
|
|
129
123
|
}
|
|
130
124
|
function readSemSearch(advisories) {
|
|
131
125
|
const check = advisories.find((a) => a.name === "semantic-search-runtime");
|
|
@@ -184,7 +178,6 @@ function buildAggregatesPhase(result, runsPhase) {
|
|
|
184
178
|
const improve = result.improve;
|
|
185
179
|
const cons = improve.consolidation;
|
|
186
180
|
const mi = improve.memoryInference;
|
|
187
|
-
const ge = improve.graphExtraction;
|
|
188
181
|
const wallTime = improve.wallTime;
|
|
189
182
|
const coverage = improve.coverage;
|
|
190
183
|
// #576: real per-stage LLM token/time accounting (replaces the GPU-time
|
|
@@ -209,7 +202,6 @@ function buildAggregatesPhase(result, runsPhase) {
|
|
|
209
202
|
return {
|
|
210
203
|
consolidation: cons,
|
|
211
204
|
memoryInference: mi,
|
|
212
|
-
graphExtraction: ge,
|
|
213
205
|
wallTime,
|
|
214
206
|
coverage,
|
|
215
207
|
llm,
|
|
@@ -309,7 +301,7 @@ function groupProposalsBySource(proposals) {
|
|
|
309
301
|
}
|
|
310
302
|
/** Summary table rows: the base metric set + the WS-5 coverage/minting/perf/degradation extensions, when present. */
|
|
311
303
|
function buildSummaryRows(aggregates, trend) {
|
|
312
|
-
const { consolidation: cons,
|
|
304
|
+
const { consolidation: cons, wallTime, llm, coverage } = aggregates;
|
|
313
305
|
const summaryRows = [
|
|
314
306
|
["Task fail rate", aggregates.taskFailRate, "flat"],
|
|
315
307
|
["Agent fail rate", aggregates.agentFailRate, "flat"],
|
|
@@ -332,8 +324,6 @@ function buildSummaryRows(aggregates, trend) {
|
|
|
332
324
|
"Candidates reviewed but intentionally left unchanged (the 'judgedNoAction' field).",
|
|
333
325
|
],
|
|
334
326
|
["Chunk failure", aggregates.chunkFail, "flat"],
|
|
335
|
-
["Graph entities", num(ge.entities), "up"],
|
|
336
|
-
["Graph relations", num(ge.relations), "up"],
|
|
337
327
|
[
|
|
338
328
|
"Stash derived",
|
|
339
329
|
num(aggregates.memorySummary.derived),
|
|
@@ -81,7 +81,6 @@ export const INTERESTING_DELTA_PATHS = [
|
|
|
81
81
|
"improve.memoryInference.written",
|
|
82
82
|
"improve.memoryInference.yieldRate",
|
|
83
83
|
"improve.memoryInference.skippedNoFacts",
|
|
84
|
-
"improve.graphExtraction.failures",
|
|
85
84
|
"improve.autoAccept.promoted",
|
|
86
85
|
"improve.autoAccept.validationFailed",
|
|
87
86
|
"improve.coverage.acceptedProposals",
|
package/dist/commands/health.js
CHANGED
|
@@ -19,6 +19,7 @@ import { getAllEntries } from "../storage/repositories/index-entries-repository.
|
|
|
19
19
|
import { queryTaskHistory } from "../storage/repositories/task-history-repository.js";
|
|
20
20
|
import { getStateDbFreelistInfo, runStateDbQuickCheck } from "../storage/state-db-integrity.js";
|
|
21
21
|
import { pkgVersion } from "../version.js";
|
|
22
|
+
import { collectArchiveUsageAdvisory } from "./health/archive-usage.js";
|
|
22
23
|
import { HEALTH_CHECKS, probeActiveImproveStrategy, runHealthEngineProbes, runPendingStateMigrationsCheck, SESSION_EXTRACTION_LEDGER_WINDOW_DAYS, } from "./health/checks.js";
|
|
23
24
|
import { collectConfigSkewAdvisory } from "./health/config-skew.js";
|
|
24
25
|
import { collectDataDirUsageAdvisory } from "./health/data-dir-usage.js";
|
|
@@ -291,6 +292,18 @@ function gatherAncillaryAdvisories(db, options, egressConfigView) {
|
|
|
291
292
|
catch {
|
|
292
293
|
// Non-fatal.
|
|
293
294
|
}
|
|
295
|
+
// Item 4 (alpha.9 plan §5.4/§8 step 8): a bundle with no git history of its
|
|
296
|
+
// own never gets the purge sweep, so its memory-cleanup archive only ever
|
|
297
|
+
// grows — report its size and file count instead. Best-effort — an
|
|
298
|
+
// unreadable archive must not abort the health report.
|
|
299
|
+
try {
|
|
300
|
+
const archiveUsage = collectArchiveUsageAdvisory(options.stashDir ?? resolveStashDir());
|
|
301
|
+
if (archiveUsage)
|
|
302
|
+
advisories.push(archiveUsage);
|
|
303
|
+
}
|
|
304
|
+
catch {
|
|
305
|
+
// Non-fatal.
|
|
306
|
+
}
|
|
294
307
|
// #896: report the data dir's total size and its largest top-level
|
|
295
308
|
// subdirectory, so a disk-usage blowup (e.g. unpruned migration snapshot
|
|
296
309
|
// backups, #897) is self-diagnosing instead of requiring `du` archaeology.
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import { searchLocal } from "../../../indexer/search/db-search.js";
|
|
5
|
+
import { stripFrontmatterBody } from "../content-hash.js";
|
|
6
|
+
import { stripBundle } from "../ledger.js";
|
|
7
|
+
import { loadRetrievalQueries } from "../retrieval-gate.js";
|
|
8
|
+
/** At most this many of the retired asset's most recent queries are replayed (plan §5.4, §7). */
|
|
9
|
+
export const CONTINUITY_MAX_QUERIES = 5;
|
|
10
|
+
/** "Top 10" per the plan's rule — both the pass/fail cutoff and the search `limit`. */
|
|
11
|
+
export const CONTINUITY_TOP_N = 10;
|
|
12
|
+
/**
|
|
13
|
+
* Builds the real search call the continuity check uses, stateful across
|
|
14
|
+
* every query asked of ONE instance (S2): the first time a query falls back
|
|
15
|
+
* to keyword-only ranking (`mode: "fts-fallback"` — most often a down or
|
|
16
|
+
* unreachable embedding endpoint), every later call through THIS instance
|
|
17
|
+
* forces `semanticSearchMode: "off"` instead of attempting semantic search
|
|
18
|
+
* again, so a dead endpoint costs one failed attempt per run, not one per
|
|
19
|
+
* remaining query (a hanging endpoint at ~3s/query, 300 proposals x 5
|
|
20
|
+
* queries, would otherwise cost on the order of an hour). Construct exactly
|
|
21
|
+
* one instance per pair-pass run and reuse it for every proposal judged, so
|
|
22
|
+
* the throttle covers the whole run, not just one proposal's own queries —
|
|
23
|
+
* and every call made once it has switched, across every remaining proposal
|
|
24
|
+
* in the run, reports `forcedKeywordOnly: true`, not just the one call that
|
|
25
|
+
* discovered the fallback (round-3 review: the first fix only marked THAT
|
|
26
|
+
* call unverified, so a second proposal checked while the endpoint was
|
|
27
|
+
* still down came back with a clean, `mode: "keyword"` — and therefore
|
|
28
|
+
* bulk-acceptable — result).
|
|
29
|
+
*/
|
|
30
|
+
export function createContinuitySearch(stashDir, config) {
|
|
31
|
+
const base = {
|
|
32
|
+
searchType: "any",
|
|
33
|
+
limit: CONTINUITY_TOP_N,
|
|
34
|
+
stashDir,
|
|
35
|
+
sources: [{ path: stashDir, isDefault: true }],
|
|
36
|
+
};
|
|
37
|
+
let keywordOnly = false;
|
|
38
|
+
return async (query) => {
|
|
39
|
+
const forcedKeywordOnly = keywordOnly;
|
|
40
|
+
const callConfig = keywordOnly ? { ...config, semanticSearchMode: "off" } : config;
|
|
41
|
+
const result = await searchLocal({ ...base, query, config: callConfig });
|
|
42
|
+
if (result.mode === "fts-fallback")
|
|
43
|
+
keywordOnly = true;
|
|
44
|
+
return { hits: result.hits, mode: result.mode, forcedKeywordOnly };
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
/** 1-indexed position of `conceptId` in `hits`, or `undefined` if it is not among them. */
|
|
48
|
+
function rankOf(hits, conceptId) {
|
|
49
|
+
const index = hits.findIndex((hit) => stripBundle(hit.ref) === conceptId);
|
|
50
|
+
return index === -1 ? undefined : index + 1;
|
|
51
|
+
}
|
|
52
|
+
/** Body only, whitespace collapsed — the same shape `db-search.ts`'s own content-dedupe compares (S3b). */
|
|
53
|
+
function normalizedBody(raw) {
|
|
54
|
+
return stripFrontmatterBody(raw).replace(/\s+/g, " ").trim();
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* Replay `retiredRef`'s own past queries and check that `successorRef` ranks
|
|
58
|
+
* in the top {@link CONTINUITY_TOP_N} for every one where the retired asset
|
|
59
|
+
* did. Returns `undefined` when there is nothing to flag: the two bodies are
|
|
60
|
+
* content-identical, no recorded queries, or every query ran on the real
|
|
61
|
+
* ranking and either the retired asset never ranked top 10 for it, or the
|
|
62
|
+
* survivor always did too. Never throws.
|
|
63
|
+
*
|
|
64
|
+
* S3: two fixes against false flags measured on a real night-1 admission
|
|
65
|
+
* (300 pairs, 6 flags, half spurious):
|
|
66
|
+
* - queries are the SAME cleaned set `loadRetrievalQueries` replays for the
|
|
67
|
+
* retrieval regression gate (`../retrieval-gate.ts`) — stash-README
|
|
68
|
+
* boilerplate, harness/tool envelopes, pastes over 2,000 characters, and
|
|
69
|
+
* duplicates are dropped before replay, not just capped at 5 raw entries;
|
|
70
|
+
* - when the retired and successor bodies normalize identical, the check
|
|
71
|
+
* never runs at all: search's own content-dedupe (`db-search.ts`) already
|
|
72
|
+
* hides the successor behind the retired asset for every such query, so a
|
|
73
|
+
* "successor missing from the top 10" finding here would not be a real
|
|
74
|
+
* risk, just that dedupe working as designed.
|
|
75
|
+
*
|
|
76
|
+
* S2: a query that never ran (the search call threw), ran on the
|
|
77
|
+
* keyword-only fallback (`mode: "fts-fallback"` — the real ranking was
|
|
78
|
+
* attempted and failed, most often a down or unreachable embedding
|
|
79
|
+
* endpoint), or ran after the shared search instance had already switched
|
|
80
|
+
* to forced keyword-only because an EARLIER query in the same run fell back
|
|
81
|
+
* (`forcedKeywordOnly`) is "unverified" — it is dropped from the rank
|
|
82
|
+
* comparison below (its hits cannot be trusted as "the ranking a user
|
|
83
|
+
* actually gets"), but unlike a query the retired asset simply did not rank
|
|
84
|
+
* for, it can never by itself lead to a silent `undefined` — at least one
|
|
85
|
+
* unverified query always produces a `continuityRisk`, so a dead endpoint
|
|
86
|
+
* reads as "risk unknown" for every proposal it touches that run, never as
|
|
87
|
+
* "no risk found" for the ones checked after the first failure.
|
|
88
|
+
*/
|
|
89
|
+
export async function checkRetirementContinuity(args) {
|
|
90
|
+
// S3b: identical bodies — search's own content-dedupe already hides the
|
|
91
|
+
// successor for every query that would rank the retired asset, so there is
|
|
92
|
+
// no real risk here to check for.
|
|
93
|
+
if (normalizedBody(args.retiredRaw) === normalizedBody(args.successorRaw))
|
|
94
|
+
return undefined;
|
|
95
|
+
// S3a: the SAME cleaned queries the retrieval regression gate replays —
|
|
96
|
+
// boilerplate, envelopes, pastes and duplicates dropped before replay.
|
|
97
|
+
const queries = loadRetrievalQueries(args.ledgerAccess, args.retiredRef).slice(0, CONTINUITY_MAX_QUERIES);
|
|
98
|
+
if (queries.length === 0)
|
|
99
|
+
return undefined; // no queries recorded: no check, no flag
|
|
100
|
+
const search = args.search ?? createContinuitySearch(args.stashDir, args.config);
|
|
101
|
+
// N2: no rank-change-report abstraction — a query only ever needs "did the
|
|
102
|
+
// retired asset rank top 10, and if so, did the successor too?", and
|
|
103
|
+
// `rankOf` (search itself returning at most CONTINUITY_TOP_N hits) already
|
|
104
|
+
// answers both directly.
|
|
105
|
+
const ranks = [];
|
|
106
|
+
let unverifiedQueries = 0;
|
|
107
|
+
for (const query of queries) {
|
|
108
|
+
let hits;
|
|
109
|
+
try {
|
|
110
|
+
const result = await search(query);
|
|
111
|
+
if (result.mode === "fts-fallback" || result.forcedKeywordOnly) {
|
|
112
|
+
unverifiedQueries++; // S2: never silently compare keyword-only ranks
|
|
113
|
+
continue;
|
|
114
|
+
}
|
|
115
|
+
hits = result.hits;
|
|
116
|
+
}
|
|
117
|
+
catch {
|
|
118
|
+
unverifiedQueries++; // S2: a query that never ran cannot be "no risk"
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
const retiredRank = rankOf(hits, args.retiredRef);
|
|
122
|
+
if (retiredRank === undefined)
|
|
123
|
+
continue; // the retired asset itself did not rank top 10 here — nothing to protect
|
|
124
|
+
const successorRank = rankOf(hits, args.successorRef);
|
|
125
|
+
if (successorRank === undefined)
|
|
126
|
+
ranks.push({ query, retiredRank, successorRank: null });
|
|
127
|
+
}
|
|
128
|
+
// Every query verified, and either the retired asset never ranked top 10
|
|
129
|
+
// for any of them, or the survivor always did too: nothing to flag.
|
|
130
|
+
if (unverifiedQueries === 0 && ranks.length === 0)
|
|
131
|
+
return undefined;
|
|
132
|
+
return {
|
|
133
|
+
failingQueries: ranks.length,
|
|
134
|
+
ranks,
|
|
135
|
+
...(unverifiedQueries > 0 ? { unverifiedQueries } : {}),
|
|
136
|
+
};
|
|
137
|
+
}
|