ruvnet-brain 4.0.1 → 4.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +303 -24
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/commands/whats-new.md +6 -6
- package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +90 -32
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +34 -29
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +503 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/scripts/whats-new.mjs +42 -0
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +98 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/plugin/skills/whats-new/SKILL.md +4 -4
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +531 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/fix-workstream.mjs +291 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +237 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +558 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +326 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +172 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2922 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/protected-release-invocation.mjs +76 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/publication-receipt.mjs +307 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-authority.mjs +93 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +439 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +296 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stabilization-receipt.mjs +108 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +867 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,710 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// scripts/claims-verify.mjs — the CLAIMS LEDGER (ADR-0011 Phase 0, last open item).
|
|
3
|
+
//
|
|
4
|
+
// Every user-facing number must be regenerable from a real artifact on disk, and CI must fail
|
|
5
|
+
// when one can't be. Each ledger entry is { claim, source, verify } where verify() re-derives
|
|
6
|
+
// the number from the artifact — NO network, NO model calls, brain-independent, so it runs on
|
|
7
|
+
// a bare CI runner. If a check needs the 512MB brain and the brain is absent, it SKIPs LOUDLY
|
|
8
|
+
// (printed with a reason) — a skip is never a silent pass.
|
|
9
|
+
//
|
|
10
|
+
// Usage: node scripts/claims-verify.mjs (or: npm run claims:verify) — READ-ONLY gate
|
|
11
|
+
// node scripts/claims-verify.mjs --fix (or: npm run claims:fix) — regenerate the surfaces
|
|
12
|
+
// Exit: 0 = every claim regenerated (skips allowed, printed); 1 = any claim failed.
|
|
13
|
+
//
|
|
14
|
+
// Verify functions take artifact paths as parameters (repo paths as defaults) so tests can
|
|
15
|
+
// point them at tampered copies. Main-guarded like scripts/eval-brain.mjs so tests can import
|
|
16
|
+
// the functions without running the CLI.
|
|
17
|
+
|
|
18
|
+
import fs from 'node:fs';
|
|
19
|
+
import path from 'node:path';
|
|
20
|
+
import readline from 'node:readline';
|
|
21
|
+
import { spawnSync } from 'node:child_process';
|
|
22
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
23
|
+
|
|
24
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
25
|
+
|
|
26
|
+
const PASS = 'PASS';
|
|
27
|
+
const FAIL = 'FAIL';
|
|
28
|
+
const SKIP = 'SKIP';
|
|
29
|
+
|
|
30
|
+
// The invariant name, spelled here rather than imported, so a broken or absent learning-replay.mjs
|
|
31
|
+
// degrades to ONE red row instead of taking the whole ledger down with an import error (the ledger
|
|
32
|
+
// has to run on a bare CI runner and report what it can). tests/unit/learning-replay.test.mjs
|
|
33
|
+
// asserts this literal equals the module's exported INVARIANT, so the two cannot drift silently.
|
|
34
|
+
const LEARNING_REPLAY = 'LEARNING-REPLAY';
|
|
35
|
+
const pass = (evidence) => ({ status: PASS, evidence });
|
|
36
|
+
const fail = (evidence) => ({ status: FAIL, evidence });
|
|
37
|
+
const skip = (evidence) => ({ status: SKIP, evidence });
|
|
38
|
+
|
|
39
|
+
// ── claim 1: "grounded 12/12 → n=120 baseline" ──────────────────────────────────────────────────
|
|
40
|
+
// evals/baseline.json is the recorded truth the eval gate promotes against. Assert the recorded
|
|
41
|
+
// stratum sizes (grounded k=n=100, routed n=80) plus internal consistency for EVERY metric:
|
|
42
|
+
// k ≤ n, p = k/n, lo ≤ p ≤ hi, all probabilities in [0,1]. No other expectations are hardcoded —
|
|
43
|
+
// the point is that the file is self-consistent and matches what we advertise, not that the
|
|
44
|
+
// scores themselves are any particular value.
|
|
45
|
+
export function verifyBaseline(file = path.join(ROOT, 'evals', 'baseline.json')) {
|
|
46
|
+
let b;
|
|
47
|
+
try {
|
|
48
|
+
b = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
49
|
+
} catch (e) {
|
|
50
|
+
return fail(`cannot read/parse ${file}: ${e.message}`);
|
|
51
|
+
}
|
|
52
|
+
if (!b.score || typeof b.score !== 'object') return fail('baseline has no score object');
|
|
53
|
+
|
|
54
|
+
for (const [name, m] of Object.entries(b.score)) {
|
|
55
|
+
if (!m || typeof m !== 'object') return fail(`metric ${name} is not an object`);
|
|
56
|
+
const { k, n, p, lo, hi } = m;
|
|
57
|
+
if (!Number.isInteger(k) || !Number.isInteger(n) || k < 0 || n <= 0)
|
|
58
|
+
return fail(`metric ${name}: k/n must be non-negative integers with n>0 (k=${k}, n=${n})`);
|
|
59
|
+
if (k > n) return fail(`metric ${name}: k=${k} > n=${n} — impossible count`);
|
|
60
|
+
if (Math.abs(p - k / n) > 1e-9) return fail(`metric ${name}: p=${p} ≠ k/n=${k / n}`);
|
|
61
|
+
// EPS absorbs float artifacts of the Wilson formula (k=n yields hi = 0.9999999999999999,
|
|
62
|
+
// one ulp under p=1) without letting a real inconsistency through.
|
|
63
|
+
const EPS = 1e-9;
|
|
64
|
+
if (!(lo >= -EPS && hi <= 1 + EPS && lo <= p + EPS && p <= hi + EPS))
|
|
65
|
+
return fail(`metric ${name}: interval broken — need 0 ≤ lo ≤ p ≤ hi ≤ 1 (lo=${lo}, p=${p}, hi=${hi})`);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const g = b.score.grounded, r = b.score.routed;
|
|
69
|
+
if (!g || g.n !== 100 || g.k !== 100)
|
|
70
|
+
return fail(`recorded truth drifted: expected grounded k=100/n=100, got k=${g?.k}/n=${g?.n} — if the baseline was deliberately re-recorded, update this ledger in the same commit`);
|
|
71
|
+
if (!r || r.n !== 80)
|
|
72
|
+
return fail(`recorded truth drifted: expected routed n=80, got n=${r?.n} — if the baseline was deliberately re-recorded, update this ledger in the same commit`);
|
|
73
|
+
|
|
74
|
+
return pass(`grounded ${g.k}/${g.n}, routed n=${r.n}; all ${Object.keys(b.score).length} metrics satisfy k≤n and lo≤p≤hi`);
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// ── claim 2: "held-out set is frozen at 120 questions across 5 strata" ──────────────────────────
|
|
78
|
+
// Recount the strata from the artifact itself. The expected census is the advertised one.
|
|
79
|
+
export const EXPECTED_STRATA = { named: 28, described: 32, scenario: 20, adversarial: 20, provenance: 20 };
|
|
80
|
+
|
|
81
|
+
export function verifyHeldOutStrata(file = path.join(ROOT, 'evals', 'held-out.json'), expected = EXPECTED_STRATA) {
|
|
82
|
+
let setJson;
|
|
83
|
+
try {
|
|
84
|
+
setJson = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
85
|
+
} catch (e) {
|
|
86
|
+
return fail(`cannot read/parse ${file}: ${e.message}`);
|
|
87
|
+
}
|
|
88
|
+
const qs = setJson.questions;
|
|
89
|
+
if (!Array.isArray(qs)) return fail('held-out set has no questions array');
|
|
90
|
+
|
|
91
|
+
const counts = {};
|
|
92
|
+
for (const q of qs) counts[q.stratum] = (counts[q.stratum] || 0) + 1;
|
|
93
|
+
|
|
94
|
+
const expectedTotal = Object.values(expected).reduce((a, b) => a + b, 0);
|
|
95
|
+
const problems = [];
|
|
96
|
+
for (const [stratum, want] of Object.entries(expected)) {
|
|
97
|
+
if ((counts[stratum] || 0) !== want) problems.push(`${stratum}: ${counts[stratum] || 0} ≠ ${want}`);
|
|
98
|
+
}
|
|
99
|
+
for (const stratum of Object.keys(counts)) {
|
|
100
|
+
if (!(stratum in expected)) problems.push(`unexpected stratum "${stratum}" (${counts[stratum]})`);
|
|
101
|
+
}
|
|
102
|
+
if (qs.length !== expectedTotal) problems.push(`total: ${qs.length} ≠ ${expectedTotal}`);
|
|
103
|
+
if (problems.length) return fail(`strata census drifted: ${problems.join('; ')}`);
|
|
104
|
+
|
|
105
|
+
return pass(`${qs.length} questions: ` + Object.entries(expected).map(([s, c]) => `${s} ${c}`).join(', '));
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// ── claim 3: "~56× cheaper" (explainer + hook) ──────────────────────────────────────────────────
|
|
109
|
+
// Derived from the corpus fact recorded in the metaharness KB: a run cost $0.267
|
|
110
|
+
// (with the 51.33 figure alongside it). ~56× = $15 / $0.267 ≈ 56.2. We re-find both corpus
|
|
111
|
+
// strings with a plain streaming grep (first match wins) and re-do the arithmetic. The passages
|
|
112
|
+
// file ships with the 512MB brain, which CI does not have — absent file is a LOUD SKIP, never
|
|
113
|
+
// a silent pass.
|
|
114
|
+
export async function verifyCheaperFactor(file = path.join(ROOT, 'kb', 'metaharness.passages.jsonl')) {
|
|
115
|
+
if (!fs.existsSync(file)) {
|
|
116
|
+
return skip(`brain not installed — ${path.relative(ROOT, file)} absent, cannot re-derive ~56× from corpus (runs on machines with the brain)`);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
const needles = ['$0.267', '51.33'];
|
|
120
|
+
const found = new Set();
|
|
121
|
+
const stream = fs.createReadStream(file, { encoding: 'utf8' });
|
|
122
|
+
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
|
|
123
|
+
try {
|
|
124
|
+
for await (const line of rl) {
|
|
125
|
+
for (const needle of needles) if (!found.has(needle) && line.includes(needle)) found.add(needle);
|
|
126
|
+
if (found.size === needles.length) break; // first match wins; stop streaming
|
|
127
|
+
}
|
|
128
|
+
} finally {
|
|
129
|
+
rl.close();
|
|
130
|
+
stream.destroy();
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
const missing = needles.filter((n) => !found.has(n));
|
|
134
|
+
if (missing.length) return fail(`corpus fact missing from ${path.relative(ROOT, file)}: ${missing.join(', ')} — the ~56× claim no longer regenerates`);
|
|
135
|
+
|
|
136
|
+
const factor = 15 / 0.267;
|
|
137
|
+
if (Math.abs(factor - 56.2) > 0.1) return fail(`arithmetic drifted: 15 / 0.267 = ${factor.toFixed(2)}, expected ≈ 56.2`);
|
|
138
|
+
|
|
139
|
+
return pass(`"$0.267" and "51.33" found in corpus; 15 / 0.267 = ${factor.toFixed(1)} ≈ 56×`);
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
// ── claim 4: "coverage badge says N% of ALL source" — the % is RE-DERIVED, never string-matched ──
|
|
143
|
+
// (2026-07-18) The old check only asserted the literal string `coverage-10%25%20of%20ALL%20source`
|
|
144
|
+
// was PRESENT and that vitest set `all:true`. It never re-derived the real percentage — so the
|
|
145
|
+
// badge sat at "10%" while the true floor was ~14.55% and this gate reported PASS. A gate that
|
|
146
|
+
// cannot fail on the number it guards is not a gate; it launders a false assurance. The badge now
|
|
147
|
+
// advertises floor(min of the four v8 metrics) over ALL shipped source; this re-reads the real
|
|
148
|
+
// coverage-summary.json produced by `npm run test:cov` and fails if the badge drifts >1pt from it.
|
|
149
|
+
// Mirrors verifyChunkCountSurfaces (re-derive from artifact) rather than string-matching. ADR-0020.
|
|
150
|
+
export const BADGE_RE = /coverage-(\d+)%25%20of%20ALL%20source/;
|
|
151
|
+
export const BADGE_RE_G = new RegExp(BADGE_RE.source, 'g'); // the writer needs the global form
|
|
152
|
+
|
|
153
|
+
/** Extract the integer % the README coverage badge advertises, or null if the badge is gone/malformed. */
|
|
154
|
+
export function readBadgePct(readme) {
|
|
155
|
+
const m = readme.match(BADGE_RE);
|
|
156
|
+
return m ? Number(m[1]) : null;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// Every place the README states the coverage % IN PROSE (the badge has its own regex above). ONE
|
|
160
|
+
// table feeds both the checker (drift ⇒ FAIL) and the writer (--fix rewrites them), so a phrasing
|
|
161
|
+
// can never be guarded by one and missed by the other — which is how a badge and its own caption
|
|
162
|
+
// end up disagreeing.
|
|
163
|
+
export const COVERAGE_PROSE_RES = [
|
|
164
|
+
/(\d+)% of ALL source/g,
|
|
165
|
+
/(\d+)% is the honest number/g,
|
|
166
|
+
];
|
|
167
|
+
|
|
168
|
+
/** Every prose statement of the coverage % in the README, as { pct, text }. */
|
|
169
|
+
export function readCoverageClaims(readme) {
|
|
170
|
+
const out = [];
|
|
171
|
+
for (const re of COVERAGE_PROSE_RES) for (const m of readme.matchAll(re)) out.push({ pct: Number(m[1]), text: m[0] });
|
|
172
|
+
return out;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// ── FRESHNESS IS A PRECONDITION, not a nicety (2026-07-26) ──────────────────────────────────────
|
|
176
|
+
// The defect: coverage/coverage-summary.json sat NINE DAYS stale (mtime Jul 18) while this gate
|
|
177
|
+
// re-derived "the truth" from it and graded the badge against a measurement nobody had taken since;
|
|
178
|
+
// then a coverage run that ended on a failing test deleted it and wrote nothing at all. One
|
|
179
|
+
// unvalidated, rottable input, three possible lies: a false FAIL, a false PASS, a crash. A truth-gate
|
|
180
|
+
// that trusts a rotting artifact is ceremony wearing the costume of substance.
|
|
181
|
+
//
|
|
182
|
+
// So the coverage claim is now UNVERIFIABLE unless the artifact is (1) present, (2) COMPLETE — it
|
|
183
|
+
// contains an entry for every file the coverage config says it measures, so a half-written summary
|
|
184
|
+
// from an aborted run can never be mistaken for a measurement — and (3) NEWER than every source in
|
|
185
|
+
// that set (and newer than the config that defines the set, since a config edit redefines the
|
|
186
|
+
// denominator). Unverifiable ⇒ a LOUD skip naming the exact regenerate command. Never a number.
|
|
187
|
+
//
|
|
188
|
+
// WHY MTIME, NOT A CONTENT HASH STORED BESIDE THE SUMMARY (the alternative I rejected): a hash
|
|
189
|
+
// sidecar is only evidence if it is written by whatever produced the summary. Nothing in this
|
|
190
|
+
// pipeline can do that — vitest writes coverage-summary.json, and any hash THIS script wrote
|
|
191
|
+
// afterwards would be a hash of the tree as it looks at gate time, i.e. a self-certifying stamp
|
|
192
|
+
// that agrees with itself by construction and cannot detect the very drift it exists to catch.
|
|
193
|
+
// mtime is produced by the OS as a side effect of the write we actually care about: evidence, not
|
|
194
|
+
// testimony. Its known weaknesses are both handled: a git checkout resets every source mtime to
|
|
195
|
+
// checkout time, but the coverage run that follows writes the summary later, so a fresh CI clone is
|
|
196
|
+
// always fresh; and `touch` can forge an mtime, but this gate defends against ROT, not against a
|
|
197
|
+
// hostile committer — who could edit the advertised number directly anyway.
|
|
198
|
+
export const COVERAGE_REGEN_CMD = 'npm run test:cov';
|
|
199
|
+
|
|
200
|
+
const DIR_PRUNE = new Set(['node_modules', '.git', 'coverage', 'dist', 'clones']);
|
|
201
|
+
|
|
202
|
+
/** Minimal glob → RegExp for the shapes coverage.include/exclude actually use (`**`, `*`, `?`). */
|
|
203
|
+
function globToRe(glob) {
|
|
204
|
+
let re = '';
|
|
205
|
+
for (let i = 0; i < glob.length; i++) {
|
|
206
|
+
const c = glob[i];
|
|
207
|
+
if (c === '*') {
|
|
208
|
+
if (glob[i + 1] === '*') {
|
|
209
|
+
i++;
|
|
210
|
+
if (glob[i + 1] === '/') { i++; re += '(?:.*/)?'; } else re += '.*';
|
|
211
|
+
} else re += '[^/]*';
|
|
212
|
+
} else if (c === '?') re += '[^/]';
|
|
213
|
+
else re += c.replace(/[.+^${}()|[\]\\]/g, '\\$&');
|
|
214
|
+
}
|
|
215
|
+
return new RegExp(`^${re}$`);
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/**
|
|
219
|
+
* The source set the coverage summary CLAIMS to measure — read from vitest.config.mjs itself
|
|
220
|
+
* (imported, not regex-scraped, so a future edit to the globs is followed automatically instead of
|
|
221
|
+
* silently mis-parsed). Falls back to the summary's own file keys when the config can't be imported
|
|
222
|
+
* (e.g. devDeps absent), and says so, rather than pretending to know the denominator.
|
|
223
|
+
*/
|
|
224
|
+
export async function coverageSourceSet(vitestFile = path.join(ROOT, 'vitest.config.mjs'), root = ROOT) {
|
|
225
|
+
let cfg;
|
|
226
|
+
try {
|
|
227
|
+
cfg = (await import(pathToFileURL(vitestFile).href)).default;
|
|
228
|
+
} catch (e) {
|
|
229
|
+
return { files: null, derivedFrom: null, reason: `cannot import ${path.basename(vitestFile)}: ${e.message}` };
|
|
230
|
+
}
|
|
231
|
+
const cov = cfg?.test?.coverage ?? {};
|
|
232
|
+
const include = Array.isArray(cov.include) ? cov.include : null;
|
|
233
|
+
if (!include || !include.length) {
|
|
234
|
+
return { files: null, derivedFrom: null, reason: `${path.basename(vitestFile)} declares no coverage.include globs` };
|
|
235
|
+
}
|
|
236
|
+
const exclude = (Array.isArray(cov.exclude) ? cov.exclude : []).map(globToRe);
|
|
237
|
+
const incl = include.map(globToRe);
|
|
238
|
+
|
|
239
|
+
const files = [];
|
|
240
|
+
const walk = (dir) => {
|
|
241
|
+
let ents;
|
|
242
|
+
try { ents = fs.readdirSync(dir, { withFileTypes: true }); } catch { return; }
|
|
243
|
+
for (const ent of ents) {
|
|
244
|
+
const abs = path.join(dir, ent.name);
|
|
245
|
+
if (ent.isDirectory()) { if (!DIR_PRUNE.has(ent.name)) walk(abs); continue; }
|
|
246
|
+
if (!ent.isFile()) continue;
|
|
247
|
+
const rel = path.relative(root, abs).split(path.sep).join('/');
|
|
248
|
+
if (incl.some((r) => r.test(rel)) && !exclude.some((r) => r.test(rel))) files.push(abs);
|
|
249
|
+
}
|
|
250
|
+
};
|
|
251
|
+
walk(root);
|
|
252
|
+
return { files, derivedFrom: include, reason: null };
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
/**
|
|
256
|
+
* Is the coverage artifact admissible as evidence?
|
|
257
|
+
* → { fresh, code: 'fresh'|'absent'|'unreadable'|'partial'|'stale', reason }
|
|
258
|
+
* `reason` always names the exact regenerate command, so a skip is actionable, never just sad.
|
|
259
|
+
*/
|
|
260
|
+
export async function coverageFreshness(
|
|
261
|
+
summaryFile = path.join(ROOT, 'coverage', 'coverage-summary.json'),
|
|
262
|
+
vitestFile = path.join(ROOT, 'vitest.config.mjs'),
|
|
263
|
+
root = ROOT,
|
|
264
|
+
) {
|
|
265
|
+
const rel = (p) => path.relative(root, p).split(path.sep).join('/') || p;
|
|
266
|
+
const regen = ` — regenerate with \`${COVERAGE_REGEN_CMD}\``;
|
|
267
|
+
|
|
268
|
+
if (!fs.existsSync(summaryFile)) {
|
|
269
|
+
return { fresh: false, code: 'absent', reason: `${rel(summaryFile)} absent — no coverage run has produced it (a run that ends on a failing test deletes it and writes nothing)${regen}` };
|
|
270
|
+
}
|
|
271
|
+
let summary, summaryMtime;
|
|
272
|
+
try {
|
|
273
|
+
summary = JSON.parse(fs.readFileSync(summaryFile, 'utf8'));
|
|
274
|
+
summaryMtime = fs.statSync(summaryFile).mtimeMs;
|
|
275
|
+
} catch (e) {
|
|
276
|
+
return { fresh: false, code: 'unreadable', reason: `${rel(summaryFile)} is present but unreadable (${e.message}) — a half-written artifact is not a measurement${regen}` };
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
const measured = new Set(Object.keys(summary).filter((k) => k !== 'total').map((k) => path.resolve(root, k)));
|
|
280
|
+
const { files, reason: setReason } = await coverageSourceSet(vitestFile, root);
|
|
281
|
+
|
|
282
|
+
// Compare against the config's source set when we can derive it; otherwise fall back to the files
|
|
283
|
+
// the summary itself names (still catches "source edited after the run", just not "file never measured").
|
|
284
|
+
const sources = files ?? [...measured];
|
|
285
|
+
if (files) {
|
|
286
|
+
const unmeasured = files.filter((f) => !measured.has(path.resolve(f)));
|
|
287
|
+
if (unmeasured.length) {
|
|
288
|
+
const sample = unmeasured.slice(0, 3).map(rel).join(', ');
|
|
289
|
+
return { fresh: false, code: 'partial', reason: `${rel(summaryFile)} is PARTIAL — ${unmeasured.length} file(s) the coverage config measures are missing from it (${sample}${unmeasured.length > 3 ? ', …' : ''}), so its total is a denominator nobody measured${regen}` };
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
let newest = { file: null, mtime: -Infinity };
|
|
294
|
+
for (const f of [...sources, vitestFile]) {
|
|
295
|
+
let m;
|
|
296
|
+
try { m = fs.statSync(f).mtimeMs; } catch { continue; }
|
|
297
|
+
if (m > newest.mtime) newest = { file: f, mtime: m };
|
|
298
|
+
}
|
|
299
|
+
if (newest.file && newest.mtime > summaryMtime) {
|
|
300
|
+
const age = Math.round((newest.mtime - summaryMtime) / 1000);
|
|
301
|
+
const human = age >= 86400 ? `${(age / 86400).toFixed(1)} days` : age >= 3600 ? `${(age / 3600).toFixed(1)} h` : `${age}s`;
|
|
302
|
+
return { fresh: false, code: 'stale', reason: `${rel(summaryFile)} is STALE — ${rel(newest.file)} was modified ${human} AFTER the coverage run that produced it, so the summary measures source that no longer exists${regen}${files ? '' : ` (source set inferred from the summary itself: ${setReason})`}` };
|
|
303
|
+
}
|
|
304
|
+
return { fresh: true, code: 'fresh', reason: `${rel(summaryFile)} covers all ${sources.length} configured source file(s) and is newer than every one of them` };
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
export async function verifyCoverageBadge(
|
|
308
|
+
readmeFile = path.join(ROOT, 'README.md'),
|
|
309
|
+
summaryFile = path.join(ROOT, 'coverage', 'coverage-summary.json'),
|
|
310
|
+
vitestFile = path.join(ROOT, 'vitest.config.mjs'),
|
|
311
|
+
root = ROOT,
|
|
312
|
+
) {
|
|
313
|
+
let readme, vitestCfg;
|
|
314
|
+
try {
|
|
315
|
+
readme = fs.readFileSync(readmeFile, 'utf8');
|
|
316
|
+
vitestCfg = fs.readFileSync(vitestFile, 'utf8');
|
|
317
|
+
} catch (e) {
|
|
318
|
+
return fail(`cannot read artifact: ${e.message}`);
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
const advertised = readBadgePct(readme);
|
|
322
|
+
if (advertised === null) {
|
|
323
|
+
return fail('README no longer carries a "coverage-N%25%20of%20ALL%20source" badge — the number that must stay honest is missing');
|
|
324
|
+
}
|
|
325
|
+
// "ALL source" is only an honest claim while every shipped file is in the denominator.
|
|
326
|
+
const allTrue = /\ball\s*:\s*true\b/.test(vitestCfg);
|
|
327
|
+
if (!allTrue) {
|
|
328
|
+
return fail('vitest.config.mjs no longer sets coverage `all: true` — the badge advertises "ALL source", so the denominator must be every shipped file, not a flattering subset');
|
|
329
|
+
}
|
|
330
|
+
// The badge is not the only place the README states the number; the prose must move with it, or
|
|
331
|
+
// one of the two is lying no matter which one matches the artifact.
|
|
332
|
+
const proseDrift = readCoverageClaims(readme).filter((c) => c.pct !== advertised);
|
|
333
|
+
if (proseDrift.length) {
|
|
334
|
+
return fail(`README states the coverage % ${proseDrift.length + 1} different ways: badge ${advertised}%, but also ${proseDrift.map((c) => `"${c.text}"`).join(', ')} — one number, every surface (\`npm run claims:fix\`)`);
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
// FRESHNESS PRECONDITION — refuse to grade a claim against an artifact that cannot be trusted.
|
|
338
|
+
// Never a derived number, never a silent pass; a loud SKIP that names the command that fixes it.
|
|
339
|
+
const state = await coverageFreshness(summaryFile, vitestFile, root);
|
|
340
|
+
if (!state.fresh) {
|
|
341
|
+
return skip(`coverage claim NOT GRADED (badge says "${advertised}%", unverified): ${state.reason}`);
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
let summary;
|
|
345
|
+
try {
|
|
346
|
+
summary = JSON.parse(fs.readFileSync(summaryFile, 'utf8'));
|
|
347
|
+
} catch (e) {
|
|
348
|
+
return fail(`cannot parse ${path.relative(ROOT, summaryFile)}: ${e.message}`);
|
|
349
|
+
}
|
|
350
|
+
const t = summary.total;
|
|
351
|
+
const metrics = ['statements', 'branches', 'functions', 'lines'];
|
|
352
|
+
for (const k of metrics) {
|
|
353
|
+
if (!t?.[k] || typeof t[k].pct !== 'number') return fail(`coverage summary malformed: missing total.${k}.pct`);
|
|
354
|
+
}
|
|
355
|
+
const min = Math.min(...metrics.map((k) => t[k].pct));
|
|
356
|
+
const realFloor = Math.floor(min);
|
|
357
|
+
const TOL = 1; // floor(min) is stable across a whole integer band; ±1 absorbs boundary jitter.
|
|
358
|
+
const detail = metrics.map((k) => `${k} ${t[k].pct}%`).join(', ');
|
|
359
|
+
if (Math.abs(advertised - realFloor) > TOL) {
|
|
360
|
+
return fail(`coverage badge advertises ${advertised}% but the re-derived floor is ${realFloor}% (${detail}). Update the README badge AND prose to ${realFloor}% — run \`npm run claims:fix\`, never hand-type it.`);
|
|
361
|
+
}
|
|
362
|
+
return pass(`badge ${advertised}% within ${TOL}pt of re-derived floor ${realFloor}% (min metric ${min.toFixed(2)}%; all:true; ${detail}); artifact fresh: ${state.reason}`);
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
// ── claim 6: "32 repos · N source chunks" — the advertised chunk count regenerates ──────────────
|
|
366
|
+
// The public chunk total is re-derived from the per-store idmap sidecars (id count == passages rows
|
|
367
|
+
// == vectors, enforced by corpus-qa S3), summed over PUBLIC stores only (kb/PRIVATE-STORES.json is
|
|
368
|
+
// the fence). Every user-facing surface that quotes the number must quote THIS number — the count
|
|
369
|
+
// sat at a stale 128,994 across 10+ surfaces after a rebuild moved it (2026-07-10 gremlin hunt).
|
|
370
|
+
// The sidecars ship with the brain, which bare CI does not have — absent kb dir is a LOUD SKIP.
|
|
371
|
+
export const CHUNK_SURFACES = ['README.md', 'explainer/index.html', 'explainer/llms.txt', 'explainer/llms-full.txt'];
|
|
372
|
+
|
|
373
|
+
export function computePublicChunkTotal(kbDir = path.join(ROOT, 'kb')) {
|
|
374
|
+
const privFile = path.join(kbDir, 'PRIVATE-STORES.json');
|
|
375
|
+
const priv = new Set(fs.existsSync(privFile) ? JSON.parse(fs.readFileSync(privFile, 'utf8')).privateStores : []);
|
|
376
|
+
let total = 0, stores = 0;
|
|
377
|
+
for (const f of fs.readdirSync(kbDir)) {
|
|
378
|
+
const m = f.match(/^(.+)\.big\.rvf\.idmap\.json$/);
|
|
379
|
+
if (!m || priv.has(m[1])) continue;
|
|
380
|
+
total += Object.keys(JSON.parse(fs.readFileSync(path.join(kbDir, f), 'utf8')).idToLabel).length;
|
|
381
|
+
stores++;
|
|
382
|
+
}
|
|
383
|
+
return { total, stores };
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/** Every built store on disk, private ones included — what "N built stores incl. private" advertises. */
|
|
387
|
+
export function countBuiltStores(kbDir = path.join(ROOT, 'kb')) {
|
|
388
|
+
return fs.readdirSync(kbDir).filter((f) => /\.big\.rvf\.idmap\.json$/.test(f)).length;
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
/** The three brain census numbers every public surface quotes, all from the same artifacts. */
|
|
392
|
+
export function brainCensus(kbDir = path.join(ROOT, 'kb')) {
|
|
393
|
+
const { total, stores } = computePublicChunkTotal(kbDir);
|
|
394
|
+
return { chunks: total, publicStores: stores, builtStores: countBuiltStores(kbDir) };
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
// ONE table of "how a census number appears on a public surface", shared by the checker (drift ⇒
|
|
398
|
+
// FAIL) and the writer (--fix ⇒ restamp). Four surfaces cannot drift four ways when one table decides
|
|
399
|
+
// what the number looks like in all of them. `(?<![\d,])` stops a rule matching the tail of a longer
|
|
400
|
+
// number. Each group is rewritten in the writer with its own format (raw for a data-count attribute,
|
|
401
|
+
// comma-grouped for prose).
|
|
402
|
+
export const SURFACE_CLAIM_RULES = [
|
|
403
|
+
{ name: 'chunk count', key: 'chunks', re: /(?<![\d,])(\d{1,3}(?:,\d{3})+)([^0-9]{0,40}?chunks)/gis, groups: [{ i: 1, fmt: 'comma' }] },
|
|
404
|
+
{ name: 'chunk counter', key: 'chunks', re: /data-count="(\d{4,})"([^>]*>)([\d,]+)(<\/span>\s*chunks)/gi, groups: [{ i: 1, fmt: 'raw' }, { i: 3, fmt: 'comma' }] },
|
|
405
|
+
{ name: 'public-store count', key: 'publicStores', re: /(?<![\d,])(\d{1,3}(?:,\d{3})*)(\s+public\s+stores)/gi, groups: [{ i: 1, fmt: 'comma' }] },
|
|
406
|
+
{ name: 'public-store counter', key: 'publicStores', re: /data-count="(\d+)"([^>]*>)([\d,]+)(<\/span>\s*public\s+stores)/gi, groups: [{ i: 1, fmt: 'raw' }, { i: 3, fmt: 'comma' }] },
|
|
407
|
+
{ name: 'built-store count', key: 'builtStores', re: /(?<![\d,])(\d{1,3}(?:,\d{3})*)(\s+built\s+stores)/gi, groups: [{ i: 1, fmt: 'comma' }] },
|
|
408
|
+
];
|
|
409
|
+
|
|
410
|
+
const fmtNum = (n, fmt) => (fmt === 'comma' ? n.toLocaleString('en-US') : String(n));
|
|
411
|
+
|
|
412
|
+
export function verifyChunkCountSurfaces(kbDir = path.join(ROOT, 'kb'), surfaces = CHUNK_SURFACES, root = ROOT) {
|
|
413
|
+
if (!fs.existsSync(kbDir) || !fs.readdirSync(kbDir).some((f) => f.endsWith('.big.rvf.idmap.json'))) {
|
|
414
|
+
return skip('brain not installed — kb/*.big.rvf.idmap.json absent, cannot re-derive the chunk count (runs on machines with the brain)');
|
|
415
|
+
}
|
|
416
|
+
const census = brainCensus(kbDir);
|
|
417
|
+
const want = census.chunks.toLocaleString('en-US');
|
|
418
|
+
|
|
419
|
+
const problems = [];
|
|
420
|
+
for (const rel of surfaces) {
|
|
421
|
+
const p = path.join(root, rel);
|
|
422
|
+
if (!fs.existsSync(p)) { problems.push(`${rel}: file missing`); continue; }
|
|
423
|
+
const s = fs.readFileSync(p, 'utf8');
|
|
424
|
+
if (!s.includes(want)) problems.push(`${rel}: does not contain "${want}"`);
|
|
425
|
+
// Any census number quoted anywhere on the surface must BE the re-derived one. Store counts are
|
|
426
|
+
// drift-checked but not required to be present — not every surface quotes them.
|
|
427
|
+
for (const rule of SURFACE_CLAIM_RULES) {
|
|
428
|
+
for (const m of s.matchAll(rule.re)) {
|
|
429
|
+
for (const g of rule.groups) {
|
|
430
|
+
if (Number(String(m[g.i]).replace(/,/g, '')) !== census[rule.key]) {
|
|
431
|
+
problems.push(`${rel}: stale ${rule.name} "${m[0].trim()}" (expected ${fmtNum(census[rule.key], g.fmt)})`);
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
}
|
|
437
|
+
if (problems.length) return fail(`brain census drifted: ${problems.join('; ')} — one pass fixes every surface: \`npm run claims:fix\``);
|
|
438
|
+
return pass(`${want} chunks · ${census.publicStores} public stores · ${census.builtStores} built stores re-derived from the idmaps; all ${surfaces.length} surfaces quote them`);
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
// ── claim 5: "version surfaces agree" ───────────────────────────────────────────────────────────
|
|
442
|
+
// Delegated to the existing single-source-of-truth checker; we propagate its exit code.
|
|
443
|
+
export function verifyVersionSurfaces(root = ROOT) {
|
|
444
|
+
const res = spawnSync(process.execPath, [path.join(root, 'scripts', 'sync-version.mjs'), '--check'], {
|
|
445
|
+
cwd: root,
|
|
446
|
+
encoding: 'utf8',
|
|
447
|
+
});
|
|
448
|
+
const out = `${res.stdout || ''}${res.stderr || ''}`.trim().split('\n').pop() || '(no output)';
|
|
449
|
+
if (res.status !== 0) return fail(`sync-version --check exited ${res.status}: ${out}`);
|
|
450
|
+
return pass(out);
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
// ── the WRITER (--fix): regenerate every advertised number in ONE pass ──────────────────────────
|
|
454
|
+
// Four surfaces quoted "149,930" for a fortnight after the brain moved to 150,161 because keeping
|
|
455
|
+
// them in step was a human chore performed four times. The fix is not more diligence, it is one
|
|
456
|
+
// writer: `npm run claims:fix` re-derives the census from the idmaps and the coverage % from the
|
|
457
|
+
// coverage summary, and stamps ALL of them together. sync-version.mjs's idiom exactly — one source
|
|
458
|
+
// of truth, everything else inherits, `--check` (here: the plain gate) proves it.
|
|
459
|
+
//
|
|
460
|
+
// The writer obeys the SAME freshness precondition as the gate: it will not propagate a coverage
|
|
461
|
+
// number derived from an absent/stale/partial artifact. A writer that stamps a rotten number is
|
|
462
|
+
// worse than no writer — it launders the rot into four more places.
|
|
463
|
+
|
|
464
|
+
/**
|
|
465
|
+
* Rewrite the numeric groups of one rule to `value`; returns { out, hits } (hits = groups changed).
|
|
466
|
+
* Offsets are computed from a cursor that only moves forward WITHIN each match, so a rule whose two
|
|
467
|
+
* groups can hold identical text (`data-count="1234">1234</span>`) still rewrites the right one —
|
|
468
|
+
* a plain string replace would have clobbered the first occurrence twice.
|
|
469
|
+
*/
|
|
470
|
+
function restampRule(s, rule, value) {
|
|
471
|
+
let out = '', last = 0, hits = 0;
|
|
472
|
+
for (const m of s.matchAll(rule.re)) {
|
|
473
|
+
let rel = 0;
|
|
474
|
+
for (const g of rule.groups) {
|
|
475
|
+
const cur = m[g.i];
|
|
476
|
+
if (cur === undefined) continue;
|
|
477
|
+
const idx = m[0].indexOf(cur, rel);
|
|
478
|
+
if (idx < 0) continue;
|
|
479
|
+
rel = idx + cur.length;
|
|
480
|
+
const want = fmtNum(value, g.fmt);
|
|
481
|
+
if (cur === want) continue;
|
|
482
|
+
const at = m.index + idx;
|
|
483
|
+
out += s.slice(last, at) + want;
|
|
484
|
+
last = at + cur.length;
|
|
485
|
+
hits++;
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
return { out: out + s.slice(last), hits };
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
export async function applyFix({
|
|
492
|
+
root = ROOT,
|
|
493
|
+
kbDir = path.join(root, 'kb'),
|
|
494
|
+
surfaces = CHUNK_SURFACES,
|
|
495
|
+
readmeFile = path.join(root, 'README.md'),
|
|
496
|
+
summaryFile = path.join(root, 'coverage', 'coverage-summary.json'),
|
|
497
|
+
vitestFile = path.join(root, 'vitest.config.mjs'),
|
|
498
|
+
write = false,
|
|
499
|
+
} = {}) {
|
|
500
|
+
const report = { changed: [], census: null, coverage: { skipped: true, reason: 'not attempted' }, notes: [] };
|
|
501
|
+
const edits = new Map(); // abs path -> content
|
|
502
|
+
|
|
503
|
+
const readSurface = (rel) => {
|
|
504
|
+
const p = path.join(root, rel);
|
|
505
|
+
if (edits.has(p)) return { p, s: edits.get(p) };
|
|
506
|
+
if (!fs.existsSync(p)) return { p, s: null };
|
|
507
|
+
return { p, s: fs.readFileSync(p, 'utf8') };
|
|
508
|
+
};
|
|
509
|
+
|
|
510
|
+
// ---- the brain census (chunks + store counts), re-derived from the idmap sidecars -------------
|
|
511
|
+
const haveBrain = fs.existsSync(kbDir) && fs.readdirSync(kbDir).some((f) => f.endsWith('.big.rvf.idmap.json'));
|
|
512
|
+
if (!haveBrain) {
|
|
513
|
+
report.notes.push('brain not installed — chunk/store counts left untouched (nothing to re-derive them from)');
|
|
514
|
+
} else {
|
|
515
|
+
const census = brainCensus(kbDir);
|
|
516
|
+
report.census = census;
|
|
517
|
+
for (const rel of surfaces) {
|
|
518
|
+
const { p, s } = readSurface(rel);
|
|
519
|
+
if (s === null) { report.notes.push(`${rel}: file missing, skipped`); continue; }
|
|
520
|
+
let next = s, hits = 0;
|
|
521
|
+
for (const rule of SURFACE_CLAIM_RULES) {
|
|
522
|
+
const r = restampRule(next, rule, census[rule.key]);
|
|
523
|
+
next = r.out; hits += r.hits;
|
|
524
|
+
}
|
|
525
|
+
if (!next.includes(census.chunks.toLocaleString('en-US'))) {
|
|
526
|
+
// A surface with no existing claim is REPORTED, never improvised into someone else's prose —
|
|
527
|
+
// the first stamp on a new surface is a deliberate one-time copy edit.
|
|
528
|
+
report.notes.push(`${rel}: no chunk-count claim to restamp — add one by hand once (e.g. "${census.chunks.toLocaleString('en-US')} source chunks"), then this keeps it fresh`);
|
|
529
|
+
}
|
|
530
|
+
if (hits) { edits.set(p, next); report.changed.push(rel); }
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
// ---- the coverage % (badge + every prose copy), re-derived from the coverage summary ----------
|
|
535
|
+
const state = await coverageFreshness(summaryFile, vitestFile, root);
|
|
536
|
+
if (!state.fresh) {
|
|
537
|
+
report.coverage = { skipped: true, reason: state.reason, code: state.code };
|
|
538
|
+
} else {
|
|
539
|
+
const summary = JSON.parse(fs.readFileSync(summaryFile, 'utf8'));
|
|
540
|
+
const metrics = ['statements', 'branches', 'functions', 'lines'];
|
|
541
|
+
const pcts = metrics.map((k) => summary.total?.[k]?.pct);
|
|
542
|
+
if (pcts.some((v) => typeof v !== 'number')) {
|
|
543
|
+
report.coverage = { skipped: true, reason: 'coverage summary has no usable total.*.pct', code: 'malformed' };
|
|
544
|
+
} else {
|
|
545
|
+
const pct = Math.floor(Math.min(...pcts));
|
|
546
|
+
const relReadme = path.relative(root, readmeFile).split(path.sep).join('/');
|
|
547
|
+
const { p, s } = readSurface(relReadme);
|
|
548
|
+
const before = s ?? fs.readFileSync(readmeFile, 'utf8');
|
|
549
|
+
let next = before, hits = 0;
|
|
550
|
+
for (const rule of [{ re: BADGE_RE_G, groups: [{ i: 1, fmt: 'raw' }] }, ...COVERAGE_PROSE_RES.map((re) => ({ re, groups: [{ i: 1, fmt: 'raw' }] }))]) {
|
|
551
|
+
const r = restampRule(next, rule, pct);
|
|
552
|
+
next = r.out; hits += r.hits;
|
|
553
|
+
}
|
|
554
|
+
report.coverage = { skipped: false, pct, metrics: Object.fromEntries(metrics.map((k, i) => [k, pcts[i]])), changed: hits > 0 };
|
|
555
|
+
if (hits) {
|
|
556
|
+
edits.set(p, next);
|
|
557
|
+
if (!report.changed.includes(relReadme)) report.changed.push(relReadme);
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
}
|
|
561
|
+
|
|
562
|
+
if (write) for (const [p, s] of edits) fs.writeFileSync(p, s);
|
|
563
|
+
return report;
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
// ── the CRITICAL-INVARIANT VECTOR (ADR-058, "never an average") ─────────────────────────────────
|
|
567
|
+
//
|
|
568
|
+
// ADR-058's release gate is a VECTOR of named invariants, not a mean of scores — an average lets a
|
|
569
|
+
// dead invariant be paid for by seven live ones, which is precisely how a release ships with its
|
|
570
|
+
// learning wire unproven. Each entry is one NAMED invariant whose state is DERIVED from an artifact
|
|
571
|
+
// on disk that states the SHA it was measured on.
|
|
572
|
+
//
|
|
573
|
+
// The vector ADR-058 §"The release gate" specifies in full:
|
|
574
|
+
// INSTALL-FAILS-LOUD · INTERFACE-CORPUS · LATENCY-DECISION-LANE · COEXIST-BYTE-EQUAL ·
|
|
575
|
+
// LEARNING-REPLAY · SIGNAL-WATCH-FIRES · SCENARIOS-CURRENT · GUARANTEE-RUNS
|
|
576
|
+
//
|
|
577
|
+
// Only the ones whose harness EXISTS are registered here. Registering the other seven as always-SKIP
|
|
578
|
+
// rows would be the ceremony this repo keeps deleting: a name in a table is not a check. They land
|
|
579
|
+
// as their own D-items land.
|
|
580
|
+
export const invariants = [
|
|
581
|
+
{
|
|
582
|
+
name: LEARNING_REPLAY,
|
|
583
|
+
what: 'a lesson recorded in project A changes an agent\'s produced artifact in project B, against a brain-off control, and survives a nightly refresh',
|
|
584
|
+
source: 'data/learning-replay-result.json (written by scripts/learning-replay.mjs)',
|
|
585
|
+
verify: verifyLearningReplay,
|
|
586
|
+
},
|
|
587
|
+
];
|
|
588
|
+
|
|
589
|
+
/**
|
|
590
|
+
* LEARNING-REPLAY — read the trap's own result artifact.
|
|
591
|
+
*
|
|
592
|
+
* The ONE rule that governs the mapping: **UNKNOWN IS NEVER PASS, and neither is INCONCLUSIVE.**
|
|
593
|
+
* A trap that has not run, whose result predates a load-bearing edit, or whose control arm also
|
|
594
|
+
* produced the token, reports SKIP — printed loudly by the runner, never counted as verified. FAIL
|
|
595
|
+
* is a real red: the treated arm did not change its artifact.
|
|
596
|
+
*
|
|
597
|
+
* The verdict is not recomputed here. It is READ from an artifact that states the SHA it was
|
|
598
|
+
* measured on, and `checkArtifact()` re-derives whether that SHA is still current for this tree.
|
|
599
|
+
* A gate that recomputed the verdict from nothing would be asserting, not deriving.
|
|
600
|
+
*/
|
|
601
|
+
export async function verifyLearningReplay() {
|
|
602
|
+
let mod;
|
|
603
|
+
try {
|
|
604
|
+
mod = await import(pathToFileURL(path.join(ROOT, 'scripts', 'learning-replay.mjs')).href);
|
|
605
|
+
} catch (e) {
|
|
606
|
+
return fail(`cannot load scripts/learning-replay.mjs: ${e.message}`);
|
|
607
|
+
}
|
|
608
|
+
const res = mod.checkArtifact();
|
|
609
|
+
if (res.status === 'PASS') return pass(res.why);
|
|
610
|
+
if (res.status === 'FAIL') return fail(res.why);
|
|
611
|
+
// UNKNOWN | INCONCLUSIVE — loud, and explicitly not a pass.
|
|
612
|
+
return skip(`${res.status} (never a pass): ${res.why}`);
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
// ── the ledger ──────────────────────────────────────────────────────────────────────────────────
|
|
616
|
+
export const ledger = [
|
|
617
|
+
{
|
|
618
|
+
claim: 'grounded 12/12 → n=120 baseline',
|
|
619
|
+
source: 'evals/baseline.json',
|
|
620
|
+
verify: verifyBaseline,
|
|
621
|
+
},
|
|
622
|
+
{
|
|
623
|
+
claim: 'held-out set is frozen at 120 questions across 5 strata',
|
|
624
|
+
source: 'evals/held-out.json',
|
|
625
|
+
verify: verifyHeldOutStrata,
|
|
626
|
+
},
|
|
627
|
+
{
|
|
628
|
+
claim: '~56× cheaper (explainer + hook)',
|
|
629
|
+
source: 'kb/metaharness.passages.jsonl',
|
|
630
|
+
verify: verifyCheaperFactor,
|
|
631
|
+
},
|
|
632
|
+
{
|
|
633
|
+
claim: 'coverage badge % re-derives from the real coverage run (ALL source)',
|
|
634
|
+
source: 'README.md + coverage/coverage-summary.json + vitest.config.mjs',
|
|
635
|
+
verify: verifyCoverageBadge,
|
|
636
|
+
},
|
|
637
|
+
{
|
|
638
|
+
claim: 'version surfaces agree',
|
|
639
|
+
source: 'scripts/sync-version.mjs --check',
|
|
640
|
+
verify: verifyVersionSurfaces,
|
|
641
|
+
},
|
|
642
|
+
{
|
|
643
|
+
claim: 'advertised source-chunk count regenerates from the brain (all surfaces agree)',
|
|
644
|
+
source: 'kb/*.big.rvf.idmap.json + kb/PRIVATE-STORES.json',
|
|
645
|
+
verify: verifyChunkCountSurfaces,
|
|
646
|
+
},
|
|
647
|
+
// The invariant vector rides the same runner so it is printed in the same table and obeys the same
|
|
648
|
+
// "a skip is never a silent pass" rule. The vector is ALSO exported separately (`invariants`) so a
|
|
649
|
+
// future release gate can consume the named states without reading prose out of a markdown row.
|
|
650
|
+
...invariants.map((iv) => ({ claim: `invariant ${iv.name}: ${iv.what}`, source: iv.source, verify: iv.verify })),
|
|
651
|
+
];
|
|
652
|
+
|
|
653
|
+
// ── runner ──────────────────────────────────────────────────────────────────────────────────────
|
|
654
|
+
const cell = (s) => String(s).replaceAll('|', '\\|');
|
|
655
|
+
|
|
656
|
+
export async function runLedger(entries = ledger) {
|
|
657
|
+
const rows = [];
|
|
658
|
+
for (const entry of entries) {
|
|
659
|
+
let result;
|
|
660
|
+
try {
|
|
661
|
+
result = await entry.verify();
|
|
662
|
+
} catch (e) {
|
|
663
|
+
result = fail(`verify() threw: ${e.message}`);
|
|
664
|
+
}
|
|
665
|
+
rows.push({ claim: entry.claim, source: entry.source, ...result });
|
|
666
|
+
}
|
|
667
|
+
return rows;
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
async function main() {
|
|
671
|
+
// --fix is the ONLY writing path; the gate itself never mutates a surface.
|
|
672
|
+
if (process.argv.includes('--fix')) {
|
|
673
|
+
const report = await applyFix({ write: true });
|
|
674
|
+
if (report.census) {
|
|
675
|
+
console.log(`[claims:fix] census re-derived: ${report.census.chunks.toLocaleString('en-US')} chunks · ${report.census.publicStores} public stores · ${report.census.builtStores} built stores`);
|
|
676
|
+
}
|
|
677
|
+
if (report.coverage.skipped) {
|
|
678
|
+
console.log(`[claims:fix] coverage % NOT stamped (left exactly as it was): ${report.coverage.reason}`);
|
|
679
|
+
} else {
|
|
680
|
+
console.log(`[claims:fix] coverage % re-derived: ${report.coverage.pct}% = floor(min ${Object.entries(report.coverage.metrics).map(([k, v]) => `${k} ${v}%`).join(', ')})`);
|
|
681
|
+
}
|
|
682
|
+
for (const n of report.notes) console.log(`[claims:fix] ${n}`);
|
|
683
|
+
console.log(report.changed.length
|
|
684
|
+
? `[claims:fix] rewrote ${report.changed.length} surface(s) in one pass: ${report.changed.join(', ')}`
|
|
685
|
+
: '[claims:fix] every surface already agrees with the artifacts — nothing to write');
|
|
686
|
+
console.log('');
|
|
687
|
+
}
|
|
688
|
+
|
|
689
|
+
const rows = await runLedger();
|
|
690
|
+
|
|
691
|
+
console.log('## Claims ledger — every advertised number must regenerate from an artifact\n');
|
|
692
|
+
console.log('| claim | status | evidence |');
|
|
693
|
+
console.log('|---|---|---|');
|
|
694
|
+
for (const r of rows) console.log(`| ${cell(r.claim)} | ${r.status} | ${cell(r.evidence)} |`);
|
|
695
|
+
console.log('');
|
|
696
|
+
|
|
697
|
+
const skipped = rows.filter((r) => r.status === SKIP);
|
|
698
|
+
for (const s of skipped) console.log(`SKIPPED (not a pass): ${s.claim} — ${s.evidence}`);
|
|
699
|
+
|
|
700
|
+
const failed = rows.filter((r) => r.status === FAIL);
|
|
701
|
+
if (failed.length) {
|
|
702
|
+
console.error(`\nclaims:verify FAILED — ${failed.length} claim(s) no longer regenerate from their artifacts.`);
|
|
703
|
+
process.exit(1);
|
|
704
|
+
}
|
|
705
|
+
console.log(`\nclaims:verify OK — ${rows.length - skipped.length} verified, ${skipped.length} skipped (loudly).`);
|
|
706
|
+
}
|
|
707
|
+
|
|
708
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
709
|
+
await main();
|
|
710
|
+
}
|