ruvnet-brain 4.3.36 → 4.3.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +2 -2
  2. package/bin/install.mjs +32 -9
  3. package/kb/forge-update.mjs +1867 -0
  4. package/package.json +3 -1
  5. package/plugin/.claude-plugin/plugin.json +1 -1
  6. package/plugin/.codex-plugin/plugin.json +1 -1
  7. package/plugin/hooks/codex-hooks.json +6 -1
  8. package/plugin/hooks/hook-contracts.json +11 -9
  9. package/plugin/scripts/capability-claim-evidence.mjs +11 -2
  10. package/plugin/scripts/capacity-aware-parallel-work.mjs +71 -15
  11. package/plugin/scripts/completion-claim-evidence.mjs +262 -0
  12. package/plugin/scripts/continuation-gate.mjs +105 -21
  13. package/plugin/scripts/continuation-objective.mjs +15 -0
  14. package/plugin/scripts/continuity-hook-policy.mjs +10 -3
  15. package/plugin/scripts/decision-gate.mjs +22 -2
  16. package/plugin/scripts/duplicate-gate.mjs +503 -0
  17. package/plugin/scripts/grounding-turn-evidence.mjs +339 -0
  18. package/plugin/scripts/grounding-turn-gate.mjs +77 -25
  19. package/plugin/scripts/grounding-turn-mark.mjs +61 -14
  20. package/plugin/scripts/hook-input.mjs +15 -0
  21. package/plugin/scripts/hook-shim.mjs +3 -1
  22. package/plugin/scripts/host-update.mjs +45 -0
  23. package/plugin/scripts/nightly-scheduler.mjs +34 -0
  24. package/plugin/scripts/session-snapshot-hook.mjs +11 -1
  25. package/plugin/scripts/session-start-budget.mjs +1 -0
  26. package/plugin/scripts/session-start-core.mjs +15 -2
  27. package/plugin/scripts/session-start-health.mjs +101 -1
  28. package/plugin/scripts/session-start-update-plane.mjs +71 -1
  29. package/plugin/scripts/turn-outcome-capture.mjs +292 -0
  30. package/scripts/completion-claim-replay.mjs +114 -0
  31. package/scripts/corpus-canary.mjs +399 -0
  32. package/scripts/corpus-dispatch-decision.mjs +2 -2
  33. package/scripts/corpus-promotion.mjs +49 -0
  34. package/scripts/corpus-reconcile.mjs +27 -3
  35. package/scripts/corpus-watchdog.mjs +45 -6
  36. package/scripts/derive-passage-content-map.mjs +71 -0
  37. package/scripts/duplicate-gate-replay.mjs +98 -0
  38. package/scripts/grounding-turn-replay.mjs +131 -0
  39. package/scripts/nightly-watchdog.mjs +3 -3
  40. package/scripts/public-verification-inputs.mjs +24 -2
  41. package/scripts/release-transaction-provider.mjs +29 -6
  42. package/scripts/release.mjs +177 -74
  43. package/scripts/retrieval-canary.mjs +41 -4
  44. package/scripts/retrieval-passage-identity.mjs +58 -0
  45. package/scripts/sync-census.mjs +0 -0
  46. package/scripts/wired-check.mjs +6 -0
@@ -15,6 +15,10 @@
15
15
  // (c) the newest vX.Y.Z code release has no VERIFIED public-verification aggregate 24h after
16
16
  // publish, or carries one that does not verify (the nightly cannot arm without it)
17
17
  // (d) any store has been deferred (carried STALE or MISSING) for more than 7 days
18
+ // (e) the generation customers actually receive (releases/latest) is older than 36h —
19
+ // the server-side half of "nobody's Brain is ever more than 48h old"
20
+ // (f) the most recent customer canary refused its candidate (canary-rejected): a night
21
+ // that built and staged a corpus no clean customer install could apply
18
22
  // WARNING a superseded or degraded night, a stand-down tonight, an unknown outcome, a long run
19
23
  // GREEN none of the above
20
24
  //
@@ -38,6 +42,7 @@ import { fileURLToPath, pathToFileURL } from 'node:url';
38
42
  import { CODE_TAG_PATTERN, isCorpusReleaseTag } from './release-channel-kind.mjs';
39
43
  import { AGGREGATE_ASSET, SIGNING_PUBLIC_KEY_FILE } from './approved-runtime.mjs';
40
44
  import { COVERAGE_ASSET, COVERAGE_RECEIPT_ASSET, verifyCoverageSidecar } from './corpus-coverage-sidecar.mjs';
45
+ import { parseCorpusGeneration } from './corpus-promotion.mjs';
41
46
 
42
47
  const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
43
48
  export const RED = 'RED';
@@ -50,12 +55,14 @@ export const TONIGHT_WINDOW_MS = 24 * HOUR;
50
55
  export const AGGREGATE_GRACE_MS = 24 * HOUR;
51
56
  export const DEFERRAL_LIMIT_MS = 7 * 24 * HOUR;
52
57
  export const LONG_RUN_MS = 8 * HOUR;
58
+ /** A customer Brain must never be >48h old; the promoted generation pages at 36h so a same-day fix lands first. */
59
+ export const PROMOTED_AGE_LIMIT_MS = 36 * HOUR;
53
60
  /** protected-release.yml's run-name for a corpus run: `protected-release corpus <dispatch id>`. */
54
61
  export const CORPUS_RUN_TITLE = /^protected-release corpus (\S+)$/;
55
62
  /** The dispatch id corpus-nightly-dispatch.yml passes: `corpus-<its run id>-<its attempt>`. */
56
63
  const DISPATCH_ID = /^corpus-(\d+)-(\d+)$/;
57
64
  /** Terminal outcomes a corpus-release-outcome.json may declare in an `outcome` field (design D2). */
58
- const DECLARED_OUTCOMES = new Set(['published', 'no-change', 'superseded', 'degraded', 'failed']);
65
+ const DECLARED_OUTCOMES = new Set(['published', 'no-change', 'superseded', 'degraded', 'canary-rejected', 'failed']);
59
66
  const GOOD_NIGHT = new Set(['published', 'no-change']);
60
67
 
61
68
  const ms = (iso) => {
@@ -75,11 +82,17 @@ export function classifyCorpusRun(run) {
75
82
  if (run?.status !== 'completed') return 'in-progress';
76
83
  const declared = String(run.outcomeRecord?.outcome ?? '').replace('no_change', 'no-change');
77
84
  if (DECLARED_OUTCOMES.has(declared)) return declared;
78
- if (run.conclusion !== 'success') return 'failed';
79
85
  const recorded = run.outcomeRecord?.jobs;
80
86
  const byName = new Map((run.jobs || []).map((job) => [job.name, job.conclusion]));
87
+ // The customer canary refusing its candidate is its own outcome: red, and nothing reached customers.
88
+ if ((recorded?.canary ?? byName.get('customer-canary')) === 'failure') return 'canary-rejected';
89
+ if (run.conclusion !== 'success') return 'failed';
81
90
  const noChange = recorded?.no_change ?? byName.get('corpus-no-change-round');
82
- const publish = recorded?.publish ?? byName.get('protected-corpus-publisher');
91
+ // Since the customer canary, `published` means PROMOTED: the promote job, not the (staging) publisher.
92
+ // Runs recorded before the canary existed carry no promote job; their publisher moved latest itself.
93
+ const publish = recorded
94
+ ? (Object.hasOwn(recorded, 'promote') ? recorded.promote : recorded.publish)
95
+ : (byName.has('promote-canaried-corpus') ? byName.get('promote-canaried-corpus') : byName.get('protected-corpus-publisher'));
83
96
  if (noChange === 'success') return 'no-change';
84
97
  if (publish === 'success') return 'published';
85
98
  return 'unknown';
@@ -110,7 +123,8 @@ export function deferralSince(generations) {
110
123
 
111
124
  /**
112
125
  * The pure verdict. `input`:
113
- * corpusReleases [{ tag, publishedAt }] non-draft corpus-sha256-* releases
126
+ * corpusReleases [{ tag, publishedAt }] PROMOTED (non-draft, non-prerelease) corpus-sha256-* releases
127
+ * latestRelease { tag, publishedAt, generation } | null what releases/latest serves customers right now
114
128
  * corpusRuns [{ id, title, status, conclusion, createdAt, updatedAt, outcome }] protected-release corpus runs
115
129
  * dispatcherRuns [{ id, attempt, status, conclusion, createdAt }] corpus-nightly-dispatch runs
116
130
  * codeRelease { tag, publishedAt, aggregate: { state: verified|missing|invalid, reason } } | null
@@ -153,6 +167,7 @@ export function judgeCorpusHealth(input, now) {
153
167
  if (tonight) {
154
168
  const ref = `corpus run ${tonight.id} (${tonight.title})`;
155
169
  if (tonight.outcome === 'failed') add(RED, 'tonight', `${ref} failed (conclusion ${tonight.conclusion})`);
170
+ else if (tonight.outcome === 'canary-rejected') add(RED, 'tonight', `${ref}: the customer canary refused the candidate; it stays an unpromoted prerelease`);
156
171
  else if (tonight.outcome === 'superseded') add(WARNING, 'tonight', `${ref} was superseded by a newer code release before publish`);
157
172
  else if (tonight.outcome === 'degraded') add(WARNING, 'tonight', `${ref} published a degraded generation`);
158
173
  else if (tonight.outcome === 'unknown') add(WARNING, 'tonight', `${ref} succeeded but shows neither a publish nor a no-change round`);
@@ -198,6 +213,24 @@ export function judgeCorpusHealth(input, now) {
198
213
  if (degraded.length) add(WARNING, 'degraded', `${deferral.newestTag} is degraded: ${deferral.degraded.carried.length} carried, ${deferral.degraded.missing.length} missing`);
199
214
  }
200
215
 
216
+ // (e) the generation customers receive. Its own generation stamp when it carries one (the corpus
217
+ // build time), else its publish time. Missing or unreadable is RED: absence of evidence is failure.
218
+ const latest = input?.latestRelease;
219
+ const latestAge = latest ? age(latest.generation || latest.publishedAt) : null;
220
+ if (latestAge === null) {
221
+ add(RED, 'promoted-age', `releases/latest could not be dated (${latest ? latest.tag : 'no latest release observed'})`);
222
+ } else if (latestAge > PROMOTED_AGE_LIMIT_MS) {
223
+ add(RED, 'promoted-age', `customers receive ${latest.tag}, ${hours(latestAge)} old (limit 36h; a customer Brain must never pass 48h)`);
224
+ } else add(GREEN, 'promoted-age', `customers receive ${latest.tag}, ${hours(latestAge)} old`);
225
+
226
+ // (f) the most recent customer canary verdict among the observed runs.
227
+ const lastCanary = runs.filter((run) => run.outcome === 'published' || run.outcome === 'canary-rejected')
228
+ .sort((a, b) => ms(b.createdAt) - ms(a.createdAt))[0];
229
+ if (!lastCanary) add(INFO, 'canary', 'no customer canary verdict among the observed corpus runs');
230
+ else if (lastCanary.outcome === 'canary-rejected') {
231
+ add(RED, 'canary', `the last customer canary (run ${lastCanary.id}) refused its candidate: a clean install could not apply it`);
232
+ } else add(GREEN, 'canary', `the last customer canary (run ${lastCanary.id}) applied its candidate and it was promoted`);
233
+
201
234
  const verdict = findings.some((f) => f.level === RED) ? RED : findings.some((f) => f.level === WARNING) ? WARNING : GREEN;
202
235
  return { verdict, findings, lastGoodNight };
203
236
  }
@@ -277,7 +310,8 @@ export async function gatherCorpusHealthInput({ repo, now, gh = defaultGh, root
277
310
  .map((run) => ({ id: run.databaseId, title: run.displayTitle, status: run.status, conclusion: run.conclusion,
278
311
  createdAt: run.createdAt, updatedAt: run.updatedAt, ...corpusRunEvidence({ gh, repo, run, scratch }) }));
279
312
  const releases = JSON.parse(gh(['release', 'list', '--repo', repo, '--limit', '100', '--json', 'tagName,publishedAt,isDraft,isPrerelease']));
280
- const corpusReleases = releases.filter((row) => !row.isDraft && isCorpusReleaseTag(row.tagName))
313
+ // PROMOTED only: a staged candidate (and a bootstrap seed) is a prerelease no customer receives.
314
+ const corpusReleases = releases.filter((row) => !row.isDraft && !row.isPrerelease && isCorpusReleaseTag(row.tagName))
281
315
  .map((row) => ({ tag: row.tagName, publishedAt: row.publishedAt }))
282
316
  .sort((a, b) => (ms(b.publishedAt) ?? 0) - (ms(a.publishedAt) ?? 0));
283
317
  const newestCode = newestCodeRelease(releases);
@@ -300,7 +334,12 @@ export async function gatherCorpusHealthInput({ repo, now, gh = defaultGh, root
300
334
  // so which stores were carried, and since when, is simply not recorded anywhere a reader can verify.
301
335
  : { observable: false, reason: `${generations[0].tag} predates the D6.2 coverage sidecar (${COVERAGE_ASSET} + ${COVERAGE_RECEIPT_ASSET})` };
302
336
  }
303
- return { corpusReleases, corpusRuns, dispatcherRuns, codeRelease, deferral };
337
+ let latestRelease = null;
338
+ try {
339
+ const latest = JSON.parse(gh(['api', `repos/${repo}/releases/latest`]));
340
+ latestRelease = { tag: latest.tag_name, publishedAt: latest.published_at, generation: parseCorpusGeneration(latest.body)?.value || null };
341
+ } catch { latestRelease = null; } // judged RED as "could not be dated"
342
+ return { corpusReleases, corpusRuns, dispatcherRuns, codeRelease, deferral, latestRelease };
304
343
  }
305
344
 
306
345
  export function renderReport(result, now) {
@@ -0,0 +1,71 @@
1
+ #!/usr/bin/env node
2
+ // Derive data/retrieval-passage-content-digests.json from a corpus archive built with ordinal passage
3
+ // ids. See scripts/retrieval-passage-identity.mjs for why the map exists.
4
+ //
5
+ // node scripts/derive-passage-content-map.mjs --zip <old-format ruvnet-brain.zip> \
6
+ // --source-tag v4.3.36 [--fixture data/retrieval-query-evidence.json] [--out <file>]
7
+ //
8
+ // For every expected passage the frozen fixture pins (primary + alternatives), find the ONE row in the
9
+ // archive's store whose path matches and whose digest equals the pin, and record that row's id-less
10
+ // content digest. A pin that is not found exactly once is listed as `unresolved` (never guessed): it
11
+ // keeps exact-digest matching only. The output is deterministic, so re-running proves the committed map.
12
+ import fs from 'node:fs';
13
+ import path from 'node:path';
14
+ import crypto from 'node:crypto';
15
+ import { spawnSync } from 'node:child_process';
16
+ import { fileURLToPath } from 'node:url';
17
+ import { digest } from './coverage-integrity.mjs';
18
+ import { CONTENT_MAP_FILE, CONTENT_MAP_KIND, passageContentDigest } from './retrieval-passage-identity.mjs';
19
+
20
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
21
+ const sha256File = (file) => crypto.createHash('sha256').update(fs.readFileSync(file)).digest('hex');
22
+
23
+ /** Every pin in the fixture: [{ store, path, pinned }]. */
24
+ export function fixturePins(fixture) {
25
+ const pins = [];
26
+ for (const [store, row] of Object.entries(fixture.queries || {})) {
27
+ const sources = [{ path: row.expected.path, passageSha256: row.expected.passageSha256 }, ...(row.expected.alternatives || [])];
28
+ for (const source of sources) pins.push({ store, path: source.path, pinned: source.passageSha256 });
29
+ }
30
+ return pins;
31
+ }
32
+
33
+ export function deriveContentMap({ fixture, fixtureSha256, readRows, sourceTag, archiveSha256 }) {
34
+ const entries = {};
35
+ const unresolved = [];
36
+ for (const { store, path: expectedPath, pinned } of fixturePins(fixture)) {
37
+ const rows = readRows(store);
38
+ const found = (rows || []).filter((row) => row.path === expectedPath && digest(row) === pinned);
39
+ if (found.length === 1) entries[pinned] = passageContentDigest(found[0]);
40
+ else unresolved.push({ store, path: expectedPath, pinned, reason: rows ? `${found.length} matching rows` : 'store absent from archive' });
41
+ }
42
+ const sorted = Object.fromEntries(Object.entries(entries).sort(([a], [b]) => a.localeCompare(b)));
43
+ return { schemaVersion: 1, kind: CONTENT_MAP_KIND, fixtureSha256,
44
+ derivedFrom: { tag: sourceTag, archiveSha256 },
45
+ entries: sorted,
46
+ unresolved: unresolved.sort((a, b) => a.store.localeCompare(b.store) || a.path.localeCompare(b.path)) };
47
+ }
48
+
49
+ function zipRows(zipFile, store) {
50
+ const result = spawnSync('unzip', ['-p', zipFile, `${store}.passages.jsonl`], { maxBuffer: 1 << 30 });
51
+ if (result.status !== 0) return null;
52
+ // Split on \n only: rows may contain U+2028/U+2029, which are not JSONL separators.
53
+ return result.stdout.toString('utf8').split('\n').filter((line) => line.trim()).map((line) => JSON.parse(line));
54
+ }
55
+
56
+ function main(argv) {
57
+ const arg = (name, fallback = null) => { const i = argv.indexOf(name); return i >= 0 ? argv[i + 1] : fallback; };
58
+ const zip = arg('--zip');
59
+ const sourceTag = arg('--source-tag');
60
+ if (!zip || !sourceTag) { console.error('usage: derive-passage-content-map.mjs --zip <archive> --source-tag <tag> [--fixture f] [--out f]'); process.exit(2); }
61
+ const fixtureFile = path.resolve(arg('--fixture', path.join(ROOT, 'data', 'retrieval-query-evidence.json')));
62
+ const out = path.resolve(arg('--out', CONTENT_MAP_FILE));
63
+ const cache = new Map();
64
+ const readRows = (store) => { if (!cache.has(store)) cache.set(store, zipRows(zip, store)); return cache.get(store); };
65
+ const map = deriveContentMap({ fixture: JSON.parse(fs.readFileSync(fixtureFile, 'utf8')), fixtureSha256: sha256File(fixtureFile),
66
+ readRows, sourceTag, archiveSha256: sha256File(zip) });
67
+ fs.writeFileSync(out, `${JSON.stringify(map, null, 2)}\n`);
68
+ console.log(JSON.stringify({ out, resolved: Object.keys(map.entries).length, unresolved: map.unresolved.length }));
69
+ }
70
+
71
+ if (process.argv[1] && fs.realpathSync(process.argv[1]) === fileURLToPath(import.meta.url)) main(process.argv.slice(2));
@@ -0,0 +1,98 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * duplicate-gate-replay.mjs — replay plugin/scripts/duplicate-gate.mjs over real history, read-only.
4
+ *
5
+ * For each code file ADDED by the last N commits (git log --diff-filter=A), score it exactly as the
6
+ * gate would have at that moment: against the tree at the commit's first parent. Uses the gate's own
7
+ * extract/prepare/rank/exemption — the replay is the gate, pointed at the past, never a second copy.
8
+ * Features are cached per blob sha, so N commits cost one extraction per distinct blob.
9
+ *
10
+ * node scripts/duplicate-gate-replay.mjs [--commits 300] [--threshold 0.5] [--seed <sha>]... [--json out.json]
11
+ */
12
+ import fs from 'node:fs';
13
+ import { spawnSync } from 'node:child_process';
14
+ import { extract, prepare, rank, exemption, isTest, isRelocation, strengthOf, SCOPE, THRESHOLD, MIN_COPIED_LINES } from '../plugin/scripts/duplicate-gate.mjs';
15
+
16
+ const argv = process.argv.slice(2);
17
+ const opt = (k, d) => { const i = argv.indexOf(k); return i >= 0 ? argv[i + 1] : d; };
18
+ const N = Number(opt('--commits', 300));
19
+ const threshold = Number(opt('--threshold', THRESHOLD));
20
+ const minCopied = Number(opt('--min-copied', MIN_COPIED_LINES));
21
+ const seeds = argv.flatMap((a, i) => (a === '--seed' ? [argv[i + 1]] : []));
22
+ const CODE = /\.(mjs|cjs|js|ts|mts|sh|py)$/;
23
+ const FIXTURE = /(^|\/)(__)?fixtures?(__)?\/|node_modules\//;
24
+
25
+ const git = (args, input) => {
26
+ const r = spawnSync('git', args, { encoding: 'utf8', maxBuffer: 1 << 30, input });
27
+ if (r.status !== 0) throw new Error(`git ${args.join(' ')}: ${r.stderr}`);
28
+ return r.stdout;
29
+ };
30
+
31
+ /** Read many objects in one `git cat-file --batch` call; texts in input order ('' when missing). */
32
+ function blobs(specs) {
33
+ if (!specs.length) return [];
34
+ const r = spawnSync('git', ['cat-file', '--batch'], { input: `${specs.join('\n')}\n`, maxBuffer: 1 << 30 });
35
+ const buf = r.stdout; const out = []; let pos = 0;
36
+ while (pos < buf.length) {
37
+ const nl = buf.indexOf(10, pos);
38
+ const head = buf.slice(pos, nl).toString().split(' ');
39
+ if (head.length < 3 || head[1] === 'missing') { out.push(''); pos = nl + 1; continue; }
40
+ const n = Number(head[2]); out.push(buf.slice(nl + 1, nl + 1 + n).toString('utf8')); pos = nl + 1 + n + 1;
41
+ }
42
+ return out;
43
+ }
44
+
45
+ const commits = [];
46
+ let cur = null;
47
+ for (const line of git(['log', '--diff-filter=A', '--name-only', '--format=COMMIT %H %P', `-${N}`]).split('\n')) {
48
+ if (line.startsWith('COMMIT ')) { const [, sha, parent] = line.split(' '); cur = { sha, parent, added: [] }; commits.push(cur); } else if (line.trim() && cur) cur.added.push(line.trim());
49
+ }
50
+ for (const s of seeds) {
51
+ const [sha, parent] = git(['log', '-1', '--format=%H %P', s]).trim().split(' ');
52
+ const added = git(['show', '--diff-filter=A', '--name-only', '--format=', sha]).split('\n').filter(Boolean);
53
+ commits.push({ sha, parent, added, seed: true });
54
+ }
55
+
56
+ const featureCache = new Map(); // `${blob sha}:${path}` -> features (the stem makes them path-dependent)
57
+ const rows = []; const timings = [];
58
+ for (const c of commits) {
59
+ const targets = c.added.filter((p) => SCOPE.some((d) => p.startsWith(d)) && CODE.test(p));
60
+ if (!targets.length || !c.parent) continue;
61
+ const t0 = Date.now();
62
+ const tree = git(['ls-tree', '-r', c.parent]).split('\n').filter(Boolean).map((l) => {
63
+ const [meta, p] = l.split('\t'); const [, type, sha] = meta.split(' '); return { p, sha, type };
64
+ }).filter((x) => x.type === 'blob' && CODE.test(x.p) && !FIXTURE.test(x.p));
65
+ const missing = tree.filter((x) => !featureCache.has(`${x.sha}:${x.p}`));
66
+ const texts = blobs(missing.map((x) => x.sha));
67
+ missing.forEach((x, i) => { if (texts[i].length <= 400_000) featureCache.set(`${x.sha}:${x.p}`, extract(x.p, texts[i])); });
68
+ const entries = tree.filter((x) => featureCache.has(`${x.sha}:${x.p}`)).map((x) => ({ path: x.p, f: featureCache.get(`${x.sha}:${x.p}`) }));
69
+ const model = prepare(entries);
70
+ const stems = new Set([...tree.map((x) => x.p), ...c.added].filter((p) => !isTest(p)).map((p) => p.split('/').pop().replace(/\.[^.]+$/, '')));
71
+ const newTexts = blobs(targets.map((p) => `${c.sha}:${p}`));
72
+ for (const [i, p] of targets.entries()) {
73
+ const text = newTexts[i];
74
+ const ex = exemption(p, text, stems);
75
+ if (ex) { rows.push({ commit: c.sha.slice(0, 8), file: p, skip: ex, seed: !!c.seed }); continue; }
76
+ const t1 = Date.now();
77
+ const ranked = rank(model, extract(p, text), { self: p, testsOnly: isTest(p), limit: 10 }).filter((m) => !isRelocation(p, m.path));
78
+ timings.push(Date.now() - t1);
79
+ // Re-derive strength at the replay's own knobs so thresholds can be swept without editing the gate.
80
+ const knobs = { threshold, minCopied };
81
+ const top = ranked.sort((a, b) => strengthOf(b, knobs) - strengthOf(a, knobs));
82
+ const strength = top[0] ? strengthOf(top[0], knobs) : 0;
83
+ rows.push({ commit: c.sha.slice(0, 8), file: p, seed: !!c.seed, score: top[0]?.score ?? 0, strength, copied: top[0]?.copied.length ?? 0, block: strength >= 1 && !isTest(p), shadow: strength >= 1 && isTest(p),
84
+ top: top.map((m) => ({ path: m.path, score: +m.score.toFixed(3), copied: m.copied.length, copyShare: +m.copyShare.toFixed(3), parts: Object.fromEntries(Object.entries(m.parts).map(([k, v]) => [k, +v.toFixed(2)])), exports: m.sharedExports.slice(0, 4), lits: m.sharedLits })) });
85
+ }
86
+ timings.push(-(Date.now() - t0));
87
+ }
88
+
89
+ const judged = rows.filter((r) => !r.skip);
90
+ const blocks = judged.filter((r) => r.block);
91
+ const rankMs = timings.filter((t) => t >= 0).sort((a, b) => a - b);
92
+ console.log(`commits=${commits.length} candidates=${rows.length} judged=${judged.length} skipped=${rows.length - judged.length} `
93
+ + `blocks=${blocks.length} (${(100 * blocks.length / (judged.length || 1)).toFixed(1)}%) threshold=${threshold} `
94
+ + `rank p50=${rankMs[Math.floor(rankMs.length / 2)] ?? 0}ms max=${rankMs.at(-1) ?? 0}ms`);
95
+ console.log(`shadow would-blocks (tests, never refused): ${judged.filter((r) => r.shadow).length}`);
96
+ for (const r of blocks) console.log(`BLOCK strength=${r.strength.toFixed(2)} score=${r.score.toFixed(3)} copied=${r.copied} ${r.commit} ${r.file} -> ${r.top.slice(0, 3).map((m) => `${m.path}@${m.score}/${m.copied}`).join(' | ')}`);
97
+ const out = opt('--json', '');
98
+ if (out) fs.writeFileSync(out, JSON.stringify({ threshold, rows }, null, 1));
@@ -0,0 +1,131 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * grounding-turn-replay.mjs — measure grounding-turn-gate.mjs's assertion gate (ADR-0030 #1) and its
4
+ * shadow gates (#2 architecture options, #3 relayed numbers) against REAL Claude Code transcripts,
5
+ * read-only. Same turn model and walkers as scripts/completion-claim-replay.mjs (imported, not
6
+ * copied); same pure functions the hooks call (grounding-turn-mark.mjs armFor at the prompt,
7
+ * grounding-turn-evidence.mjs at Stop).
8
+ *
9
+ * node scripts/grounding-turn-replay.mjs [--root ~/.claude/projects] [--limit 60] [--out blocks.jsonl]
10
+ * [--grep <regex over the final answer>]
11
+ *
12
+ * STOP POINTS, not turns: every `stop_hook_summary` record in a turn is one real Stop event, judged on
13
+ * the transcript up to that point with the last assistant text before it as `last_assistant_message`
14
+ * — exactly what the hook saw. A turn with no recorded Stop is judged once at its end. As at runtime,
15
+ * only the FIRST stop of an episode may block (stop_hook_active silences the continued stop).
16
+ * `--at <file>:<line>` replays one Stop point (the incident check). `--subagents` also replays
17
+ * subagent transcripts (<session>/subagents/*.jsonl, their sidechain flag cleared) — NOT a runtime
18
+ * surface (the gate is registered on Stop, not SubagentStop); it only widens the labelling sample.
19
+ */
20
+ import fs from 'node:fs';
21
+ import os from 'node:os';
22
+ import path from 'node:path';
23
+ import { transcripts, turnsOf, finalText } from './completion-claim-replay.mjs';
24
+ import { armFor } from '../plugin/scripts/grounding-turn-mark.mjs';
25
+ import {
26
+ architectureShadow, auditAssertions, loadVocabulary, relayShadow, searchedThisTurn, turnSources,
27
+ } from '../plugin/scripts/grounding-turn-evidence.mjs';
28
+
29
+ const argv = process.argv.slice(2);
30
+ const opt = (flag, fallback) => { const i = argv.indexOf(flag); return i >= 0 && argv[i + 1] ? argv[i + 1] : fallback; };
31
+ const root = path.resolve(opt('--root', path.join(os.homedir(), '.claude', 'projects')));
32
+ const limit = Number(opt('--limit', '60'));
33
+ const out = opt('--out', null);
34
+ const grep = opt('--grep', null);
35
+ const at = opt('--at', null);
36
+ const withSubagents = argv.includes('--subagents');
37
+ // --unarmed: judge EVERY stop point as if armed (subjects from the prompt, or none) — measures the
38
+ // Stop-side evaluator alone, i.e. what prompt-time arming saves. Not the runtime behaviour.
39
+ const unarmed = argv.includes('--unarmed');
40
+ function subagentFiles() {
41
+ const files = [];
42
+ for (const project of fs.readdirSync(root)) {
43
+ let sessions = [];
44
+ try { sessions = fs.readdirSync(path.join(root, project)); } catch { continue; }
45
+ for (const session of sessions) {
46
+ const dir = path.join(root, project, session, 'subagents');
47
+ try { for (const f of fs.readdirSync(dir)) if (f.endsWith('.jsonl') && fs.statSync(path.join(dir, f)).size > 20_000) files.push(path.join(dir, f)); } catch { /* none */ }
48
+ }
49
+ }
50
+ return files;
51
+ }
52
+
53
+ const parse = (l) => { try { return JSON.parse(l); } catch { return null; } };
54
+ const assistantText = (o) => (o?.type === 'assistant' && !o.isSidechain && Array.isArray(o.message?.content)
55
+ ? o.message.content.filter((c) => c?.type === 'text').map((c) => c.text).join('\n').trim() : '');
56
+ /** [lines-up-to-stop, last assistant text] for each Stop in a turn (the first per episode only). */
57
+ function stopPoints(turnLines) {
58
+ const points = [];
59
+ let last = '';
60
+ let episodeOpen = true;
61
+ turnLines.forEach((l, i) => {
62
+ const o = parse(l);
63
+ const t = assistantText(o);
64
+ if (t) last = t;
65
+ if (o?.type === 'system' && o.subtype === 'stop_hook_summary') {
66
+ if (episodeOpen && last) points.push([turnLines.slice(0, i + 1), last]);
67
+ episodeOpen = false; // later stops in this turn are continuations (stop_hook_active)
68
+ }
69
+ });
70
+ if (!points.length) { const m = finalText(turnLines); if (m) points.push([turnLines, m]); }
71
+ return points;
72
+ }
73
+
74
+ const vocab = loadVocabulary();
75
+ if (at) {
76
+ const [file, line] = [at.slice(0, at.lastIndexOf(':')), Number(at.slice(at.lastIndexOf(':') + 1))];
77
+ const lines = fs.readFileSync(file, 'utf8').split('\n').slice(0, line);
78
+ const message = assistantText(parse(lines[lines.length - 1])) || finalText(lines);
79
+ const turn = turnSources(lines);
80
+ const arm = armFor({ hook_event_name: 'UserPromptSubmit', session_id: 'replay', prompt: turn.prompt }, vocab);
81
+ const audit = arm?.assert ? auditAssertions({ message, subjects: arm.subjects, vocab, sources: turn.sources }) : null;
82
+ console.log(JSON.stringify({ at, arm, verdict: audit?.findings.length ? 'BLOCK' : 'PASS', findings: audit?.findings,
83
+ sources: turn.sources.map((x) => `${x.order} ${x.kind} ${x.strength} ${String(x.ref).slice(0, 60)}`) }, null, 2));
84
+ process.exit(0);
85
+ }
86
+ const r = { root, transcripts: 0, turns: 0, armed: 0, armedGate1: 0, blocked: 0, gate1WouldFire: 0,
87
+ shadowArchitecture: 0, shadowRelay: 0, claims: 0, ms: [] };
88
+ const samples = [];
89
+ for (const file of [...transcripts(root, limit), ...(withSubagents ? subagentFiles() : [])]) {
90
+ r.transcripts += 1;
91
+ const sub = file.includes(`${path.sep}subagents${path.sep}`);
92
+ const lines = fs.readFileSync(file, 'utf8').split('\n').filter(Boolean)
93
+ .map((l) => (sub ? l.replace('"isSidechain":true', '"isSidechain":false') : l));
94
+ for (const [turnLines, message] of turnsOf(lines).flatMap(stopPoints)) {
95
+ r.turns += 1;
96
+ const t0 = performance.now();
97
+ const turn = turnSources(turnLines);
98
+ const arm = unarmed ? { gate1: false, assert: true, architecture: false, subjects: [] }
99
+ : armFor({ hook_event_name: 'UserPromptSubmit', session_id: 'replay', prompt: turn.prompt }, vocab);
100
+ if (!arm) { r.ms.push(performance.now() - t0); continue; }
101
+ if (arm.gate1) { r.armedGate1 += 1; if (!searchedThisTurn(turn.sources)) r.gate1WouldFire += 1; }
102
+ let audit = { claims: [], findings: [] };
103
+ if (arm.assert) {
104
+ r.armed += 1;
105
+ audit = auditAssertions({ message, subjects: arm.subjects, vocab, sources: turn.sources });
106
+ r.claims += audit.claims.length;
107
+ if (audit.findings.length) r.blocked += 1;
108
+ if (architectureShadow({ architecture: arm.architecture, message })) r.shadowArchitecture += 1;
109
+ if (relayShadow({ message, sources: turn.sources })) r.shadowRelay += 1;
110
+ }
111
+ r.ms.push(performance.now() - t0);
112
+ const hit = grep && new RegExp(grep, 'i').test(message);
113
+ if (audit.findings.length || hit || (argv.includes('--all-claims') && audit.claims.length)) {
114
+ samples.push({ file: path.basename(file), subagent: sub, grep: !!hit, verdict: audit.findings.length ? 'BLOCK' : 'PASS',
115
+ prompt: turn.prompt.slice(0, 300), subjects: arm.subjects,
116
+ findings: audit.findings.map((f) => ({ claim: f.text.slice(0, 300), subject: f.subject, reason: f.reason, read: f.read })),
117
+ passed: audit.claims.filter((c) => !audit.findings.some((f) => f.text === c.text)).map((c) => ({ claim: c.text.slice(0, 300), subject: c.subject })),
118
+ sources: turn.sources.length });
119
+ }
120
+ }
121
+ }
122
+ const ms = r.ms.sort((a, b) => a - b);
123
+ console.log(JSON.stringify({ ...r, ms: undefined,
124
+ armRate: +(r.armed / (r.turns || 1)).toFixed(3), blockRate: +(r.blocked / (r.turns || 1)).toFixed(4),
125
+ blockRateOfArmed: +(r.blocked / (r.armed || 1)).toFixed(3),
126
+ gate1FireRate: +(r.gate1WouldFire / (r.turns || 1)).toFixed(4),
127
+ shadowArchitectureRate: +(r.shadowArchitecture / (r.turns || 1)).toFixed(4), shadowRelayRate: +(r.shadowRelay / (r.turns || 1)).toFixed(4),
128
+ auditP50Ms: +(ms[Math.floor(ms.length / 2)] || 0).toFixed(2), auditP99Ms: +(ms[Math.floor(ms.length * 0.99)] || 0).toFixed(2),
129
+ unit: 'turns = first Stop point per turn (stop_hook_summary), else turn end', subagents: withSubagents, unarmed }, null, 2));
130
+ if (out) fs.writeFileSync(out, samples.map((s) => JSON.stringify(s)).join('\n') + '\n');
131
+ if (grep) for (const s of samples.filter((x) => x.grep)) console.log(JSON.stringify(s).slice(0, 1500));
@@ -38,7 +38,7 @@ import path from 'node:path';
38
38
  import os from 'node:os';
39
39
  import { spawnSync, execFileSync } from 'node:child_process';
40
40
  import { fileURLToPath, pathToFileURL } from 'node:url';
41
- import { NIGHTLY_LABEL, schedulerStatus } from '../plugin/scripts/nightly-scheduler.mjs';
41
+ import { NIGHTLY_LABEL, schedulerStatus, updateOwnedByAgenticKit } from '../plugin/scripts/nightly-scheduler.mjs';
42
42
  import { oldestIncompleteProgress } from '../kb/shard-progress.mjs';
43
43
 
44
44
  const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
@@ -230,9 +230,9 @@ export function productSchedulerVerdict(status) {
230
230
  /** ONE update owner per machine (CONTRIBUTING.md § The knowledge corpus). agentic-kit removes the
231
231
  * Brain's own scheduler by design and updates through `ak sync`, which the registry watches as
232
232
  * com.stuartkerr.ak-sync — so on such a machine the Brain scheduler is not expected to exist. */
233
+ // One reader of agentic-kit ownership, shared with SessionStart's knowledge-currency line.
233
234
  export function brainUpdateOwnedByAgenticKit(home = os.homedir()) {
234
- try { return JSON.parse(fs.readFileSync(path.join(home, '.config', 'agentic-kit', 'kit.json'), 'utf8')).ruvnetBrain === true; }
235
- catch { return false; }
235
+ return updateOwnedByAgenticKit(home);
236
236
  }
237
237
 
238
238
  const loadState = () => { try { return JSON.parse(fs.readFileSync(STATE, 'utf8')); } catch { return {}; } };
@@ -11,6 +11,7 @@ import { extractZip } from '../kb/zip-extract.mjs';
11
11
  import { canonicalJson, digest, validateCoverageLedger, validateCoverageLink } from './coverage-integrity.mjs';
12
12
  import { validatePublicInventory } from './public-inventory.mjs';
13
13
  import { verifySeedBaseline } from './corpus-candidate.mjs';
14
+ import { loadFixture, readRecallReport } from './oracle/repo-recall.mjs';
14
15
  import {
15
16
  buildRetrievalCanaryPlan,
16
17
  validateRetrievalQueryEvidence,
@@ -430,9 +431,27 @@ function writeExactOutputs(outDir, outputs) {
430
431
  return root;
431
432
  }
432
433
 
434
+ /**
435
+ * The stores a corpus generation's OWN repo-recall measurement retrieved (exact file within top-k). The
436
+ * report is bound to the exact baseline archive (sha256 + bytes) and to the frozen fixture the canary
437
+ * samples from, and is re-derived through the same reader the corpus pipeline uses — a report for another
438
+ * archive or another fixture is refused, never silently accepted.
439
+ */
440
+ export function measuredHitStores({ recallFile, baselineArchive, oracleFile }) {
441
+ const stat = fs.statSync(baselineArchive);
442
+ // A report that claims retired questions must be verified against the coverage it names; the
443
+ // generation's sealed coverage sits beside its report in the seed directory when it exists.
444
+ const siblingCoverage = path.join(path.dirname(path.resolve(recallFile)), 'CORPUS-COVERAGE.json');
445
+ const { report } = readRecallReport({ reportFile: recallFile,
446
+ archive: { sha256: sha256File(baselineArchive), bytes: stat.size },
447
+ expectedFixtureSha256: loadFixture(oracleFile).fixtureSha256,
448
+ coverageBytes: fs.existsSync(siblingCoverage) ? fs.readFileSync(siblingCoverage) : null });
449
+ return new Set(report.rows.filter((row) => Number.isInteger(row.exactFileRank)).map((row) => String(row.store).toLowerCase()));
450
+ }
451
+
433
452
  export async function createPublicVerificationInputs({ baselineBundle, candidateBundle,
434
453
  candidatePackage, oracleFile, repo = process.cwd(), outDir = 'release-evidence', baselineMode = 'verified',
435
- baselineReceipt = null } = {}) {
454
+ baselineReceipt = null, baselineRecall = null } = {}) {
436
455
  const baselineArchive = trustedFile(baselineBundle, 'baseline archive');
437
456
  const candidateArchive = trustedFile(candidateBundle, 'candidate archive');
438
457
  const packageFile = trustedFile(candidatePackage, 'candidate package');
@@ -513,9 +532,11 @@ export async function createPublicVerificationInputs({ baselineBundle, candidate
513
532
  verifyQueryOracleSource(queryEvidence, candidateResult.candidate.sourceSha, {
514
533
  cwd: path.resolve(repo), allowSquashedSource: true,
515
534
  });
535
+ const knownHitStores = baselineRecall
536
+ ? measuredHitStores({ recallFile: baselineRecall, baselineArchive, oracleFile: oraclePath }) : null;
516
537
  const plan = buildRetrievalCanaryPlan({ coverage: candidateResult.coverage, baseline,
517
538
  candidate: candidateResult.candidate, coverageIdentity: candidateResult.coverageIdentity,
518
- queryEvidence, assetsDir: candidateTree.root, allowNoDelta: true });
539
+ queryEvidence, assetsDir: candidateTree.root, allowNoDelta: true, knownHitStores });
519
540
  writeExactOutputs(outDir, {
520
541
  [baselineMode === 'observed' ? 'baseline-observation-receipt.json' : 'baseline-verification-receipt.json']: baselineProof.bytes,
521
542
  'COVERAGE.json': candidateResult.coverageBytes,
@@ -574,6 +595,7 @@ export async function main(argv = process.argv.slice(2)) {
574
595
  outDir: arg(argv, '--out-dir') || 'release-evidence',
575
596
  baselineMode: argv.includes('--receipted-baseline') ? 'receipted' : argv.includes('--observed-baseline') ? 'observed' : 'verified',
576
597
  baselineReceipt: arg(argv, '--baseline-receipt'),
598
+ baselineRecall: arg(argv, '--baseline-recall'),
577
599
  });
578
600
  console.log(JSON.stringify({ ok: true, sourceSha: result.candidate.sourceSha,
579
601
  coverageGeneration: result.coverage.releaseCoverageGeneration, cases: result.plan.cases.length }));
@@ -60,13 +60,36 @@ const tagSha = (tag, root) => {
60
60
  // Assets now stream to a temp file and are hashed incrementally, so peak memory is one 1MB chunk
61
61
  // instead of the whole bundle and there is no ceiling to outgrow. Small assets (receipts) still
62
62
  // come back as bytes, because callers parse them as JSON.
63
- const assetToFile = (asset, destination) => {
64
- const result = spawnSync('gh', ['api', asset.url, '-H', 'Accept: application/octet-stream'], {
65
- stdio: ['ignore', fs.openSync(destination, 'w'), 'pipe'], timeout: ASSET_DOWNLOAD_TIMEOUT_MS,
66
- });
67
- if (result.error || result.signal || result.status !== 0) {
68
- throw new Error(`cannot download transaction asset ${asset.name}: ${result.error?.message || result.signal || `exit ${result.status}`}`);
63
+ //
64
+ // 2026-09-29: ONE FLAKY DOWNLOAD MUST NOT ABORT A RELEASE. discover() reads the receipts of every
65
+ // published release (411 assets, serially, ~3 minutes); a single transient `gh api` failure among
66
+ // them killed the 4.3.36 publish with only "exit 1" -- the stderr that said why was discarded, and
67
+ // re-fetching the same asset a minute later worked. A download is now retried with backoff, and the
68
+ // final error carries the real cause. A genuinely missing/corrupt asset still fails, just not on
69
+ // the first hiccup; the digest checks downstream are unchanged.
70
+ export const ASSET_DOWNLOAD_ATTEMPTS = 4;
71
+ const sleepSync = (ms) => { Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms); };
72
+ export const downloadAsset = (asset, destination, {
73
+ spawn = spawnSync, wait = sleepSync, attempts = ASSET_DOWNLOAD_ATTEMPTS, baseDelayMs = 2000,
74
+ } = {}) => {
75
+ let cause = '';
76
+ for (let attempt = 1; attempt <= attempts; attempt += 1) {
77
+ const fd = fs.openSync(destination, 'w'); // 'w' truncates any partial bytes from the previous attempt
78
+ let result;
79
+ try {
80
+ result = spawn('gh', ['api', asset.url, '-H', 'Accept: application/octet-stream'], {
81
+ stdio: ['ignore', fd, 'pipe'], timeout: ASSET_DOWNLOAD_TIMEOUT_MS,
82
+ });
83
+ } finally { fs.closeSync(fd); }
84
+ if (!result.error && !result.signal && result.status === 0) return destination;
85
+ const stderr = String(result.stderr || '').trim().slice(0, 300);
86
+ cause = `${result.error?.message || result.signal || `exit ${result.status}`}${stderr ? ` (${stderr})` : ''}`;
87
+ if (attempt < attempts) wait(baseDelayMs * 2 ** (attempt - 1));
69
88
  }
89
+ throw new Error(`cannot download transaction asset ${asset.name} after ${attempts} attempts: ${cause}`);
90
+ };
91
+ const assetToFile = (asset, destination) => {
92
+ downloadAsset(asset, destination);
70
93
  return destination;
71
94
  };
72
95
  const withTempAsset = (asset, fn) => {