ruvnet-brain 4.3.27 → 4.3.29

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/README.md +1 -1
  2. package/bin/install.mjs +74 -7
  3. package/console/app.js +7 -2
  4. package/kb/corpus-release-identity.mjs +73 -0
  5. package/package.json +4 -3
  6. package/plugin/.claude-plugin/plugin.json +1 -1
  7. package/plugin/.codex-plugin/plugin.json +1 -1
  8. package/plugin/mcp/managed-cli-interface.mjs +83 -9
  9. package/plugin/scripts/advocacy-route.mjs +4 -1
  10. package/plugin/scripts/capacity-aware-parallel-work.mjs +4 -0
  11. package/plugin/scripts/continuation-gate.mjs +71 -0
  12. package/plugin/scripts/decision-gate.mjs +29 -57
  13. package/plugin/scripts/ground-ruvnet.sh +77 -4
  14. package/plugin/scripts/grounding-stamp.sh +34 -5
  15. package/plugin/scripts/grounding-turn-gate.mjs +8 -2
  16. package/plugin/scripts/grounding-turn-mark.mjs +6 -1
  17. package/plugin/scripts/hook-input.mjs +54 -0
  18. package/plugin/scripts/hook-shim.mjs +10 -0
  19. package/plugin/scripts/memory-doctor.mjs +43 -0
  20. package/plugin/scripts/nightly-controller.mjs +6 -1
  21. package/plugin/scripts/project-progression-contract.mjs +17 -0
  22. package/plugin/scripts/project-progression-hook.mjs +18 -5
  23. package/plugin/scripts/project-progression-producer.mjs +16 -12
  24. package/plugin/scripts/ruvnet-gate1-pattern.mjs +17 -0
  25. package/plugin/scripts/session-start-core.mjs +11 -5
  26. package/plugin/scripts/session-start-update-plane.mjs +5 -1
  27. package/plugin/scripts/unprompted-runtime.mjs +29 -8
  28. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +1 -1
  29. package/scripts/adr-072-completion.mjs +2 -1
  30. package/scripts/calibrate-router.mjs +6 -6
  31. package/scripts/candidate-host-evidence.mjs +44 -16
  32. package/scripts/corpus-reconcile.mjs +32 -3
  33. package/scripts/correction-detect.mjs +10 -11
  34. package/scripts/dispatch-receipt.mjs +2 -2
  35. package/scripts/dual-host-deliberation.mjs +3 -2
  36. package/scripts/execution-policy.mjs +10 -2
  37. package/scripts/gen-console-images.mjs +0 -1
  38. package/scripts/gen-images.mjs +0 -4
  39. package/scripts/host-install-matrix.mjs +60 -14
  40. package/scripts/independent-review-receipt.mjs +7 -8
  41. package/scripts/ingest-repo.mjs +45 -1
  42. package/scripts/learning-replay-cli.mjs +2 -1
  43. package/scripts/learning-replay-fixture.mjs +2 -1
  44. package/scripts/learnings.mjs +19 -5
  45. package/scripts/lesson-migrate-agentdb.mjs +635 -0
  46. package/scripts/loop-checkpoint.mjs +37 -1
  47. package/scripts/metaharness-receipts.mjs +3 -2
  48. package/scripts/nightly-two-run-proof.mjs +6 -6
  49. package/scripts/nightly-watchdog.mjs +10 -2
  50. package/scripts/onboarding-console.mjs +98 -49
  51. package/scripts/oracle/producer-hosts.mjs +2 -1
  52. package/scripts/prepublication-evidence.mjs +8 -0
  53. package/scripts/private-overlay.mjs +31 -9
  54. package/scripts/public-verification-aggregate.mjs +4 -3
  55. package/scripts/publication-receipt.mjs +80 -30
  56. package/scripts/qe/agentic-qe-4.3.mjs +0 -1
  57. package/scripts/qe/ux-suite.mjs +17 -5
  58. package/scripts/rebuild-gists-from-receipts.mjs +1 -1
  59. package/scripts/reconcile-project.mjs +0 -0
  60. package/scripts/release-qualification-contract.mjs +11 -1
  61. package/scripts/release.mjs +34 -82
  62. package/scripts/retrieval-canary.mjs +3 -2
  63. package/scripts/review-model-defaults.mjs +26 -0
  64. package/scripts/route-cheap.mjs +20 -7
  65. package/scripts/router-utilization.mjs +5 -5
  66. package/scripts/rvf-generation.mjs +68 -2
  67. package/scripts/single-source-check.mjs +270 -0
  68. package/scripts/staged-host-verifier.mjs +2 -1
  69. package/scripts/subscription-hosts.mjs +4 -0
  70. package/scripts/sync-version.mjs +22 -0
  71. package/scripts/trismart.mjs +3 -3
  72. package/scripts/wired-check.mjs +40 -4
  73. package/console/assets/memory.webp +0 -0
  74. package/plugin/scripts/version-bump-gate.sh +0 -124
  75. package/scripts/qe/aggregate-4.3.mjs +0 -42
  76. package/scripts/release-convergence-watchdog.mjs +0 -112
  77. package/scripts/stamp-existing-rvf-generations.mjs +0 -53
  78. package/scripts/verify-channels.mjs +0 -196
@@ -27,3 +27,20 @@ export const RUVNET_GATE1_PATTERN =
27
27
  export function ruvnetGate1Matches(text) {
28
28
  return new RegExp(RUVNET_GATE1_PATTERN, 'i').test(String(text ?? ''));
29
29
  }
30
+
31
+ /**
32
+ * H1 / GitHub #316: the plain substring vocabulary mechanically derived from RUVNET_GATE1_PATTERN —
33
+ * one lowercased term per `|`-separated alternative, with the `\b` word-boundary anchors stripped
34
+ * (irrelevant to a substring scan) and the one optional-plural alternative ("swarms?") reduced to
35
+ * its shortest substring-safe form ("swarm", which is a substring of both "swarm" and "swarms").
36
+ *
37
+ * This is the ONE vocabulary grounding-stamp.sh's GATE1_ONLY_TERMS must mirror byte-for-byte
38
+ * (tests/unit/grounding-stamp-terms.test.mjs enforces it, same idiom as
39
+ * tests/unit/ruvnet-gate1-pattern.test.mjs's byte-identity check against ground-ruvnet.sh). Before
40
+ * that fix, grounding-stamp.sh hard-coded its own narrower 9-term list that omitted `ruvnet` itself
41
+ * — so a search literally about "ruvnet" minted no stamp and grounding-turn-gate.mjs's Stop-time
42
+ * check wrongly reported "no successful search_ruvnet call this turn".
43
+ */
44
+ export const RUVNET_GATE1_TERMS = RUVNET_GATE1_PATTERN
45
+ .split('|')
46
+ .map((alt) => alt.replace(/\\b/g, '').replace(/s\?$/, '').toLowerCase());
@@ -283,8 +283,6 @@ export async function runSessionStart({
283
283
  } else if (running) {
284
284
  write(path.join(stateDir, '.dev-version'), `${running}\n`);
285
285
  }
286
- const source = json(path.join(home, '.cache', 'ruvnet-brain', 'kb', 'SOURCE.json'), {});
287
- const kbVersion = typeof source?.releaseTag === 'string' ? source.releaseTag : '';
288
286
  const readiness = mcpReadiness(env, home);
289
287
 
290
288
  const grounding = json(path.join(stateDir, 'install-state.json'));
@@ -315,10 +313,18 @@ export async function runSessionStart({
315
313
  // alarm mechanism as HEALTH ALARM above (never gated by maintainerIssueEntitlement).
316
314
  emit(`[RuvNet Brain v${bannerVersion} — active this session${updated ? ` · updated ${updated}` : ''}]`);
317
315
  bannerEmitted = true;
318
- const bundleTag = String(kbVersion).replace(/^v/, '');
319
- if (bundleTag && bannerVersion !== 'unknown' && bundleTag !== bannerVersion) {
316
+ // S2 (ONE CURRENCY VERDICT): this no longer compares the KB's own SOURCE.json releaseTag to the
317
+ // plugin version — that heuristic fires a FALSE POSITIVE for the entire (normal, expected)
318
+ // window between the plugin auto-updating and the KB's own background --check catching up, and
319
+ // it duplicated, less accurately, a comparison kb/forge-update.mjs already makes properly
320
+ // (against the LIVE canonical release, not a locally-observed string). Instead this reads the
321
+ // recorded verdict from the SessionStart heartbeat's own `--check --result-file` run — the SAME
322
+ // structured verdict --apply and bin/install.mjs read — and only alarms on a genuine CODE-release
323
+ // mismatch (a corpus-only update does not mean the plugin's own code is out of sync).
324
+ const kbCheck = json(path.join(stateDir, '.last-kb-check-result.json'));
325
+ if (kbCheck?.currencyVerdict === 'UPDATE_AVAILABLE' && kbCheck?.candidateKind === 'code') {
320
326
  emit('🚨 [RuvNet Brain — INSTALL ALARM: plugin and knowledge bundle are out of sync] 🚨');
321
- emit(`Your plugin is v${bannerVersion} but the knowledge bundle on this machine is v${bundleTag} — they are meant to ship together, so search results may not match this plugin's behavior yet. Fix: npx ruvnet-brain@latest --update (or reinstall: npx github:stuinfla/ruvnet-brain --force).`);
327
+ emit(`${kbCheck.currencyReason || 'a newer code release is available'} — they are meant to ship together, so search results may not match this plugin's behavior yet. Fix: npx ruvnet-brain@latest --update (or reinstall: npx github:stuinfla/ruvnet-brain --force).`);
322
328
  }
323
329
 
324
330
  const hookContracts = readHookContracts(path.join(pluginRoot, 'hooks', 'hook-contracts.json'));
@@ -77,8 +77,12 @@ export const heartbeat = ({ env, hookDir, stateDir, home, running, seedDispatche
77
77
  if (/\bBEHIND\b/.test(read(kbLog))) {
78
78
  emit('[RuvNet Brain — a newer knowledge bundle is available. It is signed (Ed25519) and the updater verifies that signature before extracting anything. We do NOT auto-apply it: applying replaces executable tool files, which is your call. To update: cd ~/.cache/ruvnet-brain/kb && node forge-update.mjs --apply]');
79
79
  }
80
+ // S2 (ONE CURRENCY VERDICT): --result-file records the SAME structured verdict --check/--apply
81
+ // and bin/install.mjs already read (forge-update.mjs's currencyVerdict()), at the well-known path
82
+ // session-start-core.mjs's banner stage reads it from — so the banner's own "is my code out of
83
+ // sync" alarm reads this recorded verdict instead of re-deriving its own comparison.
80
84
  dispatchDetached(hookDir, 60, kbLog, process.execPath,
81
- [path.join(kbDir, 'forge-update.mjs'), '--check'], env);
85
+ [path.join(kbDir, 'forge-update.mjs'), '--check', '--result-file', path.join(stateDir, '.last-kb-check-result.json')], env);
82
86
  }
83
87
  const versionLog = path.join(stateDir, '.last-version-check.log');
84
88
  const remoteVersion = firstVersion(read(versionLog));
@@ -86,7 +86,7 @@ import fs from 'node:fs';
86
86
  import path from 'node:path';
87
87
  import { spawnSync } from 'node:child_process';
88
88
  import { fileURLToPath, pathToFileURL } from 'node:url';
89
- import { readStdinBounded } from './hook-input.mjs';
89
+ import { readStdinBounded, isHarnessGenerated } from './hook-input.mjs';
90
90
  import { resolveBash } from './hook-shim-bash.mjs';
91
91
 
92
92
  // WHERE THIS FILE'S SIBLINGS LIVE. Resolved from THIS file's own location so it is correct under the
@@ -112,9 +112,16 @@ const SCRIPTS_DIR = path.dirname(SELF); // the payload's scr
112
112
  // The CC event name the shim forwarded. No event → nothing to run; stay silent.
113
113
  const EVENT = process.argv[2] || '';
114
114
 
115
- // Bound the whole runtime well under the 5s hook budget: producers run sequentially and each has its
116
- // own internal watchdog, but a backstop timeout here means a wedged producer can never hang the turn.
117
- const PRODUCER_TIMEOUT_MS = Number(process.env.RUVNET_UNPROMPTED_TIMEOUT_MS) || 4000;
115
+ // Bound the whole runtime well under this hook's DECLARED timeout (hooks.json / codex-hooks.json:
116
+ // 'unprompted-speech' is 3s, not the 5s this comment used to assume) — producers run sequentially and
117
+ // each has its own internal watchdog, but a backstop timeout here means a wedged producer can never
118
+ // hang the turn. Found live 2026-09-27: the old 4000ms default left NO real margin under a 3000ms
119
+ // declared timeout once Node startup + per-producer spawnSync overhead is counted, and slower
120
+ // per-process-spawn hosts (Windows CI) pushed measured wall-clock to 83% of budget — selfcheck.mjs's
121
+ // own 80%-margin rule exists exactly to catch a hook running this close to its declared contract.
122
+ // 2000ms leaves real headroom (Node/import startup + the 80% margin check at 2400ms) on every host,
123
+ // not just a Windows-specific patch.
124
+ const PRODUCER_TIMEOUT_MS = Number(process.env.RUVNET_UNPROMPTED_TIMEOUT_MS) || 2000;
118
125
  const MAX_BUFFER = 1 << 20;
119
126
 
120
127
  const VALID_CHANNELS = new Set(['advocacy', 'promotion', 'lesson', 'alarm']);
@@ -223,6 +230,17 @@ try {
223
230
  } catch { /* not JSON → no occasion → silence, below */ }
224
231
  if (!event) silent();
225
232
 
233
+ // H2: a background task notification, slash-command scaffold, or other harness-authored message
234
+ // arrives on UserPromptSubmit exactly like real user text (wrapped in tags such as
235
+ // <task-notification>, <local-command-caveat>, <system-reminder>, ...). None of that is something a
236
+ // user typed, so no producer here should react to it — an advocacy nudge, a lesson prompt, or a
237
+ // promotion offer fired at the harness's own bookkeeping is noise on every background-task turn.
238
+ // PreToolUse payloads carry no `prompt`/`user_prompt`/`input` field, so this is a no-op for them.
239
+ {
240
+ const promptText = event.prompt ?? event.user_prompt ?? event.input;
241
+ if (typeof promptText === 'string' && isHarnessGenerated(promptText)) silent();
242
+ }
243
+
226
244
  const producers = resolveProducers(EVENT);
227
245
  if (!producers.length) silent(); // unknown event, or nothing wired for it — never speak on a guess
228
246
 
@@ -396,10 +414,13 @@ for (const c of candidates) {
396
414
  let offer = true;
397
415
  try { offer = led.shouldStillOffer(findingId, { severity, stateHash }); } catch { offer = false; }
398
416
  if (!offer) break; // dismissed / budget spent → drop
399
- // Deliver, and record the OFFERED denominator centrally (best-effort; recording never breaks
400
- // the hook it measures). Moving OFFERED here is what makes precision computable at the one place
401
- // that actually decides to show a card.
402
- try { led.record({ id: findingId, action: led.ACTIONS.OFFERED, severity, stateHash }); } catch { /* a lost row costs one denominator, never the turn */ }
417
+ // Persist the OFFERED denominator before delivery. A recommendation whose delivery receipt was
418
+ // not durably written cannot participate in the later applied/dismissed lifecycle; emitting it
419
+ // anyway would create a card the next prompt cannot resolve and would make precision lie.
420
+ let receipt;
421
+ try { receipt = led.record({ id: findingId, action: led.ACTIONS.OFFERED, severity, stateHash }); }
422
+ catch { receipt = null; }
423
+ if (!receipt?.ok) break;
403
424
  advisories.push({ copy, hookEventName });
404
425
  break;
405
426
  }
@@ -13,7 +13,7 @@ not all individually audited and remain diagnostics. A diagnostic pass is not pr
13
13
  Promote only the clean exact candidate SHA and its sealed artifacts after qualification. Public
14
14
  acceptance still requires all nine OS/host-mode leaves plus real native installed-update proof on
15
15
  each platform, ending at `install-verified`. Imported upstream corpus freshness remains UNKNOWN
16
- until separately proven. See `docs/QA-RELEASE-PROCESS.md` and the audit records in `docs/reviews/`.
16
+ until separately proven. See `CONTRIBUTING.md` and the audit records in `docs/reviews/`.
17
17
 
18
18
  ---
19
19
 
@@ -4,9 +4,10 @@ import path from 'node:path';
4
4
  import { spawnSync } from 'node:child_process';
5
5
  import { fileURLToPath } from 'node:url';
6
6
  import { verifyCapabilityClaimAggregate } from '../plugin/scripts/capability-claim-evidence.mjs';
7
+ import { LEGACY_REVIEW_MODEL_IDS } from './review-model-defaults.mjs';
7
8
 
8
9
  const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
9
- const REQUIRED_REVIEWERS = Object.freeze(['claude-fable-5', 'gpt-5.6-sol']);
10
+ const REQUIRED_REVIEWERS = LEGACY_REVIEW_MODEL_IDS;
10
11
 
11
12
  const command = (cwd, bin, args) => {
12
13
  const result = spawnSync(bin, args, { cwd, encoding: 'utf8', maxBuffer: 64 * 1024 * 1024 });
@@ -26,18 +26,18 @@ import fs from 'node:fs';
26
26
  import os from 'node:os';
27
27
  import path from 'node:path';
28
28
  import { recordOutcome } from './metaharness-router.mjs';
29
- import { estimateCosts, estTokens } from './route-cheap.mjs';
29
+ import { CLAUDE_MODEL_IDS, estimateCosts, estTokens } from './route-cheap.mjs';
30
30
 
31
31
  const CLAUDE = path.join(os.homedir(), '.npm-global/bin/claude');
32
32
  const RECEIPTS = process.env.METAHARNESS_RECEIPTS
33
33
  || path.join(os.homedir(), '.claude', 'metaharness', 'routing-receipts.jsonl');
34
34
 
35
35
  const MODELS = [
36
- { alias: 'haiku', name: 'claude-haiku-4.5' },
37
- { alias: 'sonnet', name: 'claude-sonnet-5' },
38
- { alias: 'opus', name: 'claude-opus-4.8' }, // the baseline tier
36
+ { alias: 'haiku', name: CLAUDE_MODEL_IDS.haiku },
37
+ { alias: 'sonnet', name: CLAUDE_MODEL_IDS.sonnet },
38
+ { alias: 'opus', name: CLAUDE_MODEL_IDS.opus }, // the baseline tier
39
39
  ];
40
- const BASELINE = 'claude-opus-4.8';
40
+ const BASELINE = CLAUDE_MODEL_IDS.opus;
41
41
 
42
42
  // Deterministic tasks: known answers, graded by regex — no LLM judge, no judgment calls.
43
43
  const TASKS = [
@@ -90,7 +90,7 @@ for (const [ti, task] of TASKS.entries()) {
90
90
  }
91
91
  }
92
92
 
93
- const cheap = results.flatMap((r) => [r.runs['claude-haiku-4.5'].ms]);
93
+ const cheap = results.flatMap((r) => [r.runs[CLAUDE_MODEL_IDS.haiku].ms]);
94
94
  const base = results.map((r) => r.runs[BASELINE].ms);
95
95
  const sum = (a) => a.reduce((s, x) => s + x, 0);
96
96
  console.log(`\nhaiku total ${(sum(cheap) / 1000).toFixed(1)}s vs opus baseline ${(sum(base) / 1000).toFixed(1)}s on ${TASKS.length} tasks`);
@@ -5,6 +5,7 @@ import { fileURLToPath } from 'node:url';
5
5
  import { stagedHostVerifier, readCandidateRetrieval, verifyCandidateRetrievalAssets } from './staged-host-verifier.mjs';
6
6
  import { payloadIdFor } from './release-payload.mjs';
7
7
  import { validateRetrievalCanaryReceipt } from './retrieval-canary.mjs';
8
+ import { HOST_WARMUP_TIMEOUT_MS, RELEASE_SEARCH_DEADLINE_MS } from './host-install-matrix.mjs';
8
9
 
9
10
  const arg = (name) => {
10
11
  const index = process.argv.indexOf(name);
@@ -12,13 +13,13 @@ const arg = (name) => {
12
13
  };
13
14
 
14
15
  export async function buildCandidateHostEvidence({ manifestFile, packagePath, bundlePath, planFile, coverageFile, failureFile },
15
- { createVerifier = stagedHostVerifier } = {}) {
16
+ { createVerifier = stagedHostVerifier, sequentialSearches = false } = {}) {
16
17
  const manifest = JSON.parse(fs.readFileSync(manifestFile, 'utf8'));
17
18
  const payloadId = payloadIdFor(manifest);
18
19
  const identity = { version: manifest.version, candidateSha: manifest.candidateSha, payloadId };
19
20
  const retrieval = readCandidateRetrieval({ manifest, planFile, coverageFile });
20
21
  verifyCandidateRetrievalAssets({ retrieval, assets: { packagePath, bundlePath } });
21
- const result = await createVerifier({ assets: { packagePath, bundlePath }, identity, retrieval })
22
+ const result = await createVerifier({ assets: { packagePath, bundlePath }, identity, retrieval, sequentialSearches })
22
23
  .verify({ source: 'candidate', assets: { packagePath, bundlePath } });
23
24
  verifyCandidateRetrievalAssets({ retrieval, assets: { packagePath, bundlePath } });
24
25
  if (result.verdict !== 'PASS') {
@@ -32,34 +33,60 @@ export async function buildCandidateHostEvidence({ manifestFile, packagePath, bu
32
33
  }
33
34
 
34
35
  const modeNames = { claude: 'claude-only', codex: 'codex-only', dual: 'dual-host' };
35
- const leaves = Object.entries(modeNames).map(([mode, name]) => {
36
+ const leaves = [];
37
+ const failures = [];
38
+ for (const [mode, name] of Object.entries(modeNames)) {
36
39
  const fixture = result.fixtures?.[mode];
37
40
  const grounding = fixture?.grounding;
38
- const grounded = grounding && ['repo', 'path', 'file', 'storedPath']
39
- .every((field) => typeof grounding[field] === 'string' && grounding[field].trim());
40
- if (fixture?.status !== 'PASS' || fixture?.process?.status !== 0 || !grounded) {
41
- throw new Error(`${name} did not produce a clean installed-search grounding receipt`);
41
+ const warmupGrounding = fixture?.warmupGrounding;
42
+ const hasReceipt = (receipt) => receipt && ['repo', 'path', 'file', 'storedPath']
43
+ .every((field) => typeof receipt[field] === 'string' && receipt[field].trim());
44
+ if (!Number.isFinite(fixture?.warmupMs) || fixture.warmupMs < 0 || fixture.warmupMs > HOST_WARMUP_TIMEOUT_MS
45
+ || !hasReceipt(warmupGrounding) || warmupGrounding.repo !== 'ruvnet-brain') {
46
+ failures.push(`${name} did not produce a grounded worker warmup within ${HOST_WARMUP_TIMEOUT_MS}ms (${fixture?.warmupMs ?? 'unmeasured'}ms)`);
42
47
  }
43
- validateRetrievalCanaryReceipt(fixture.retrieval, { plan: retrieval.plan });
44
- return {
48
+ if (!Number.isFinite(fixture?.searchMs) || fixture.searchMs < 0 || fixture.searchMs > RELEASE_SEARCH_DEADLINE_MS) {
49
+ failures.push(`${name} first measured cited search exceeded the ${RELEASE_SEARCH_DEADLINE_MS}ms candidate deadline (${fixture?.searchMs ?? 'unmeasured'}ms)`);
50
+ }
51
+ if (fixture?.status !== 'PASS' || fixture?.process?.status !== 0 || !hasReceipt(grounding)) {
52
+ failures.push(`${name} did not produce a clean installed-search grounding receipt`);
53
+ }
54
+ try { validateRetrievalCanaryReceipt(fixture?.retrieval, { plan: retrieval.plan }); }
55
+ catch (error) { failures.push(`${name} retrieval canary rejected: ${error.message}`); }
56
+ leaves.push({
45
57
  name,
46
58
  sha: manifest.candidateSha,
47
59
  payloadId,
48
- status: fixture.status === 'PASS' ? 'completed' : 'failed',
49
- conclusion: 'success',
60
+ status: fixture?.status === 'PASS' && fixture?.process?.status === 0 && hasReceipt(grounding) ? 'completed' : 'failed',
61
+ conclusion: failures.length ? 'failure' : 'success',
50
62
  verdict: 'PASS',
51
63
  source: 'candidate-host-evidence',
52
64
  mode,
53
- functionalSearch: true,
54
- searchExit: fixture.process.status,
65
+ functionalSearch: fixture?.status === 'PASS',
66
+ searchExit: fixture?.process?.status ?? null,
67
+ warmupMs: fixture?.warmupMs,
68
+ warmupGrounding,
69
+ searchMs: fixture?.searchMs,
55
70
  grounding,
56
- retrieval: fixture.retrieval,
71
+ retrieval: fixture?.retrieval,
57
72
  artifactSha256: retrieval.artifactSha256,
73
+ });
74
+ }
75
+ if (failures.length) {
76
+ const failure = {
77
+ schemaVersion: 1, kind: 'ruvnet-brain-candidate-host-failure', verdict: 'FAIL',
78
+ sha: manifest.candidateSha, payloadId, artifactSha256: retrieval.artifactSha256,
79
+ candidateArchiveSha256: retrieval.candidateArchiveSha256, planSha256: retrieval.plan.planSha256,
80
+ failures, result,
58
81
  };
59
- });
82
+ if (failureFile) fs.writeFileSync(path.resolve(failureFile), JSON.stringify(failure, null, 2) + '\n',
83
+ { flag: 'wx', mode: 0o600 });
84
+ throw new Error(`candidate host matrix failed: ${failures.join('; ')}`);
85
+ }
60
86
  return {
61
87
  schemaVersion: 1,
62
88
  sha: manifest.candidateSha,
89
+ hostPlatform: process.platform,
63
90
  payloadId,
64
91
  artifactSha256: retrieval.artifactSha256,
65
92
  leaves,
@@ -68,7 +95,8 @@ export async function buildCandidateHostEvidence({ manifestFile, packagePath, bu
68
95
 
69
96
  if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
70
97
  const evidence = await buildCandidateHostEvidence({ manifestFile: arg('--manifest'),
71
- packagePath: arg('--package'), bundlePath: arg('--bundle'), planFile: arg('--plan'), coverageFile: arg('--coverage'), failureFile: arg('--out') ? `${arg('--out')}.failure.json` : null });
98
+ packagePath: arg('--package'), bundlePath: arg('--bundle'), planFile: arg('--plan'), coverageFile: arg('--coverage'), failureFile: arg('--out') ? `${arg('--out')}.failure.json` : null },
99
+ { sequentialSearches: process.argv.includes('--sequential-searches') });
72
100
  fs.writeFileSync(path.resolve(arg('--out')), `${JSON.stringify(evidence, null, 2)}\n`, { flag: 'wx', mode: 0o600 });
73
101
  console.log(JSON.stringify({ verdict: 'PASS', payloadId: evidence.payloadId, leaves: evidence.leaves.map(({ name }) => name) }));
74
102
  }
@@ -17,8 +17,10 @@ import { promoteArtifactSet } from '../kb/incremental-refresh.mjs';
17
17
  import { rebuildCorpusAggregates } from './corpus-aggregates.mjs';
18
18
  import { assertCapabilityOnlyStore, isCapabilityOnly, CAPABILITY_RETIRED_SUFFIXES } from '../kb/capability-only.mjs';
19
19
  import { fileIdentity } from '../plugin/scripts/coverage-integrity.mjs';
20
+ import { readDiagnosticAccuracyReport } from './oracle/retrieval-accuracy.mjs';
20
21
  import { storeRoot } from '../kb/store-root.mjs';
21
22
  import { captureGistSources } from './gist-receipts.mjs';
23
+ import { projectSourceStore } from './rvf-generation.mjs';
22
24
 
23
25
  export { rebuildCorpusAggregates };
24
26
 
@@ -540,7 +542,13 @@ export async function executeReconciliation({
540
542
  promotedFiles.push(file.name);
541
543
  }
542
544
  mergedLedger.stores[result.store] = result.generation;
543
- mergedSource.stores[result.store] = result.source;
545
+ // S4 (ONE PROVENANCE RECORD): re-project the merged SOURCE.json entry FROM the merged ledger
546
+ // row rather than trusting the worker's own already-written SOURCE.json fragment verbatim —
547
+ // the merge boundary is where multiple workers' results combine, so it is the right place to
548
+ // assert "the ledger is the source of truth" rather than assume every worker upheld it.
549
+ // `result.source` (validateWorkerOutput's read of the worker's own output, already checked
550
+ // there to bind the exact upstream SHA) supplies the non-identity updater fields unchanged.
551
+ mergedSource.stores[result.store] = projectSourceStore(result.store, result.generation, result.source);
544
552
  }
545
553
  mergedLedger.stores = Object.fromEntries(Object.entries(mergedLedger.stores).sort(([a], [b]) => a.localeCompare(b)));
546
554
  mergedSource.stores = Object.fromEntries(Object.entries(mergedSource.stores).sort(([a], [b]) => a.localeCompare(b)));
@@ -766,12 +774,33 @@ export function prepareCorpusCandidate({
766
774
  // bounded run (--stores/--sample) still writes a report, but it marks itself incomplete and the
767
775
  // seal refuses it, so a bounded measurement can never be presented as a corpus-wide pass.
768
776
  const accuracyReportFile = `${bundleFile}.accuracy.json`;
769
- checked(run, process.execPath, [accuracyScript, '--bundle', bundleFile,
777
+ // A stale leftover report from a prior run must never be mistaken for a fresh measurement of
778
+ // THIS bundle -- delete it before invoking the script so only a report the script just wrote
779
+ // (or none at all) can be found below.
780
+ fs.rmSync(accuracyReportFile, { force: true });
781
+ const accuracyResult = run(process.execPath, [accuracyScript, '--bundle', bundleFile,
770
782
  '--oracle', accuracyOracle, '--out', accuracyReportFile,
771
783
  ...(accuracyStores != null ? ['--stores', String(accuracyStores)] : []),
772
784
  ...(accuracySample != null ? ['--sample', String(accuracySample)] : []),
773
785
  ...(accuracyTimeoutMs != null ? ['--timeout-ms', String(accuracyTimeoutMs)] : [])],
774
- { stdio: 'inherit' });
786
+ { stdio: 'inherit' }) || {};
787
+ // C3 was demoted to a non-blocking diagnostic on 2026-09-15 (commit a20727b7, ADR-086
788
+ // amendment) -- every other caller (corpus-candidate.mjs, release.mjs, corpus-seed.yml) reads
789
+ // it through readDiagnosticAccuracyReport, which checks the report's integrity/archive binding,
790
+ // never its score. A nonzero exit here is therefore NOT immediately fatal: it may just mean the
791
+ // measured score fell below the (no-longer-enforced) threshold. What stays fatal is a CRASHED
792
+ // measurement -- no valid report bound to this exact archive was produced at all.
793
+ if (accuracyResult.error || accuracyResult.status !== 0) {
794
+ let diagnostic;
795
+ try {
796
+ diagnostic = readDiagnosticAccuracyReport({ reportFile: accuracyReportFile, archive: fileIdentity(bundleFile) });
797
+ } catch (error) {
798
+ fail(`retrieval-accuracy diagnostic (C3) crashed with no valid report to show for it: ${error.message}`);
799
+ }
800
+ console.log(`[corpus-reconcile] C3 retrieval-accuracy diagnostic scored below its (non-blocking) `
801
+ + `threshold: state=${diagnostic.state} classification=${diagnostic.classification} `
802
+ + `totals=${JSON.stringify(diagnostic.totals)} -- continuing, C3 is advisory only.`);
803
+ }
775
804
  // THE BLOCKING RETRIEVAL GATE (ADR-086 amendment 2026-09-15). Same placement and same discipline
776
805
  // as the C3 run above — the EXTRACTED final archive through the customer query path — but this is
777
806
  // the measurement that can refuse a candidate. It asks the 194 frozen human questions, one per
@@ -188,6 +188,8 @@
188
188
  // industrialised. `confidence` orders the ratification queue and does nothing else — a confidence
189
189
  // threshold is just ratification with the human removed and the word "confidence" in front of it.
190
190
 
191
+ import { HARNESS_GENERATED_PATTERNS } from '../plugin/scripts/hook-input.mjs';
192
+
191
193
  /** Corrections are short. The measured tightened detector used this bound; specs and briefs exceed it. */
192
194
  export const MAX_UTTERANCE_CHARS = 800;
193
195
 
@@ -224,18 +226,15 @@ export const ACCEPTED_MISSES = Object.freeze([
224
226
  * speech at all. The detector was right to ignore them; the pool handed them to a human to label
225
227
  * anyway, burning ~29% of the scarcest resource in this whole problem (labelled examples) on rows
226
228
  * whose answer is definitionally "no", and diluting the base rate with them.
229
+ *
230
+ * H2 (2026-09-26): re-exported from plugin/scripts/hook-input.mjs's HARNESS_GENERATED_PATTERNS
231
+ * rather than kept as this file's own literal array. Other UserPromptSubmit consumers
232
+ * (unprompted-runtime.mjs, capacity-aware-parallel-work.mjs, grounding-turn-mark.mjs) needed the
233
+ * exact same recognition and, before this fix, had no shared place to get it from — hook-input.mjs
234
+ * is that ONE owner now; this name stays so correction-detect-measure.mjs and this file's own test
235
+ * do not need to change.
227
236
  */
228
- export const HARNESS_TEMPLATES = [
229
- /\[Your previous response/i,
230
- /\[Request interrupted/i,
231
- /<\/?system-reminder>/i,
232
- /<\/?(?:command-name|command-message|command-args|local-command-stdout|local-command-stderr|local-command-caveat|task-notification|function_results|function_calls|budget)\b/i,
233
- /^\s*Caveat:/i,
234
- /Base directory for this skill:/i,
235
- /This session is being continued from a previous conversation/i,
236
- /^\s*#\s*claudeMd\b/im,
237
- /\[INTELLIGENCE\]/i,
238
- ];
237
+ export const HARNESS_TEMPLATES = HARNESS_GENERATED_PATTERNS;
239
238
 
240
239
  /**
241
240
  * Pasted content, not speech. Includes markdown structure — this repository's own ADRs and DDDs are
@@ -25,7 +25,7 @@
25
25
  import fs from 'node:fs';
26
26
  import path from 'node:path';
27
27
  import { pathToFileURL } from 'node:url';
28
- import { CLAUDE_TIERS, PRICING, estTokens, estimateCosts, receiptLine, receiptsPath, priceOf } from './route-cheap.mjs';
28
+ import { CLAUDE_MODEL_IDS, CLAUDE_TIERS, PRICING, estTokens, estimateCosts, receiptLine, receiptsPath, priceOf } from './route-cheap.mjs';
29
29
 
30
30
  // Default input share when only a MEASURED TOTAL is known. A subagent's tokens are dominated by input
31
31
  // (it re-reads files and tool output on every turn); its final report is small. 0.9 is an assumption,
@@ -33,7 +33,7 @@ import { CLAUDE_TIERS, PRICING, estTokens, estimateCosts, receiptLine, receiptsP
33
33
  const DEFAULT_INPUT_SHARE = 0.9;
34
34
 
35
35
  export function parseArgs(argv) {
36
- const args = { model: 'claude-haiku-4.5', inherited: 'claude-opus-4.8', class: 'mechanical' };
36
+ const args = { model: CLAUDE_MODEL_IDS.haiku, inherited: CLAUDE_MODEL_IDS.opus, class: 'mechanical' };
37
37
  for (let i = 0; i < argv.length; i++) {
38
38
  const k = argv[i];
39
39
  if (['--model', '--inherited', '--task', '--class', '--in-chars', '--out-chars', '--label', '--total-tokens', '--split'].includes(k)) {
@@ -5,13 +5,14 @@ import fs from 'node:fs';
5
5
  import { spawn } from 'node:child_process';
6
6
  import { fileURLToPath } from 'node:url';
7
7
  import { probeSubscriptionHosts, subscriptionOnlyEnv } from './subscription-hosts.mjs';
8
+ import { DUAL_HOST_MODEL_IDS } from './review-model-defaults.mjs';
8
9
 
9
10
  const HOSTS = Object.freeze(['claude-code', 'codex']);
10
11
  // Top subscription models verified on the native hosts on 2026-09-10.
11
12
  // Keep these explicit: an implicit host default silently weakens the dual review.
12
13
  export const TOP_SUBSCRIPTION_MODELS = Object.freeze({
13
- 'claude-code': 'claude-fable-5-1',
14
- codex: 'gpt-6-astra',
14
+ 'claude-code': DUAL_HOST_MODEL_IDS.claude,
15
+ codex: DUAL_HOST_MODEL_IDS.codex,
15
16
  });
16
17
  const HARD_PROBLEM = /\b(adr|architecture|architect|ddd|bounded context|aggregate|agentic[- ]?qe|holistic|security|production|migration|irreversible|threat model|experience)\b/i;
17
18
 
@@ -4,6 +4,7 @@
4
4
 
5
5
  const NATIVE_HOSTS = new Set(['claude', 'codex']);
6
6
  const API_EXECUTORS = new Set(['agent_execute', 'sdk', 'openrouter', 'api']);
7
+ const ACTIONS = new Set(['read', 'write', 'delegate', 'release', 'external']);
7
8
  const ARCHITECTURE_TERMS = /\b(adr|ddd|architecture|release|deploy|qa|security|schema|migration)\b/i;
8
9
  const CONSEQUENTIAL_ACTIONS = new Set(['delegate', 'write', 'release', 'external']);
9
10
  const FRESHNESS_MS = 30 * 60 * 1000;
@@ -33,8 +34,8 @@ export function validateEvidence(input = {}, now = Date.now()) {
33
34
  if (!memory || memory.status !== 'retrieved' || !fresh(memory.observedAt, now)) {
34
35
  failures.push('fresh exact AgentDB checkpoint retrieval is required');
35
36
  } else {
36
- if (!String(memory.path || '').endsWith('/.swarm/memory.db')) failures.push('memory receipt must use the project .swarm/memory.db');
37
- if (!/^project-state-current-\d+$/.test(String(memory.key || ''))) failures.push('memory receipt must retrieve an append-only project-state-current key');
37
+ if (!/[\\/]\.swarm[\\/]memory\.db$/.test(String(memory.path || ''))) failures.push('memory receipt must use the project .swarm/memory.db');
38
+ if (!/^project-state-current-\d+(?:-[A-Za-z0-9][A-Za-z0-9_-]*)?$/.test(String(memory.key || ''))) failures.push('memory receipt must retrieve an append-only project-state-current key');
38
39
  if (!hex64(memory.valueDigest)) failures.push('memory receipt must bind the retrieved value digest');
39
40
  }
40
41
  return { valid: failures.length === 0, failures };
@@ -46,6 +47,13 @@ export function classifyExecutionPolicy(input = {}) {
46
47
  const changedFiles = [...new Set((input.changedFiles || []).map(String).filter(Boolean))];
47
48
  const nativeHosts = [...new Set((input.nativeHosts || []).map(String).filter((h) => NATIVE_HOSTS.has(h)))];
48
49
  const requestedExecutor = String(input.requestedExecutor || '');
50
+ if (!ACTIONS.has(action)) {
51
+ return {
52
+ schema: 'ruvnet-brain.execution-policy.v1',
53
+ verdict: 'REFUSE', action, swarmRequired: false, swarmReason: 'invalid-action',
54
+ executor: 'unknown', reason: 'unsupported-action', evidence: { valid: false, failures: ['action must be one of read, write, delegate, release, external'] },
55
+ };
56
+ }
49
57
  const explicitSwarm = input.explicitSwarm === true;
50
58
  const multiFile = changedFiles.length >= 3;
51
59
  const architectureTask = ARCHITECTURE_TERMS.test(description) || ['release', 'external'].includes(action);
@@ -24,7 +24,6 @@ const STYLE = ' — Style: deep near-black background (#0a0c10), sophisticated p
24
24
 
25
25
  const IMAGES = [
26
26
  { slug: 'hero', size: '1536x1024', p: 'A warm amber intelligence gently understanding a computer: soft glowing amber and gold neural filaments and threads of light weaving and resolving out of a faint tangle on the left into an elegant, orderly, translucent crystalline lattice of floating glass panels and cards on the right — the feeling of messy machine settings being calmly brought into clear, beautiful order. Lots of soft dark negative space on the right for text.' },
27
- { slug: 'memory', size: '1024x1024', p: 'A single luminous softly-glowing sphere of warm amber and cyan light, made of countless fine interwoven filaments, holding its shape calmly in dark space — an abstract emblem of a mind that remembers; serene, alive, precise.' },
28
27
  ];
29
28
 
30
29
  async function gen(model, prompt, size) {
@@ -15,10 +15,6 @@ const STYLE = ' — Style: dark near-black background (#0b0d0f) with a faint blu
15
15
 
16
16
  const IMAGES = [
17
17
  { slug: 'hero', size: '1536x1024', p: 'A luminous intricate three-dimensional structure resembling a brain fused with a vast interconnected codebase: thousands of glowing amber and cyan filaments forming one elegant organized sphere of intelligence, floating in dark space, a sense of all knowledge made orderly and alive.' },
18
- { slug: 'problem-skim', size: '1536x1024', p: 'A vast deep canyon made of densely stacked layers of code and documents descending far into darkness; a single small fragile light hovers at the very top only grazing the surface, never reaching the immense depth below. The feeling of skimming and missing everything underneath.' },
19
- { slug: 'point-deeper', size: '1536x1024', p: 'One precise clean beam of warm amber light cutting straight down through many deep translucent strata of a vast structure to perfectly illuminate a single exact point far below; surgical precision locating the one true answer in the depths.' },
20
- { slug: 'architecture', size: '1536x1024', p: 'An elegant isometric exploded view of five translucent glass layers floating one above another in dark space, each a slightly different luminous tone, joined by thin vertical conduits of light; a refined premium product render of a clean layered system.' },
21
- { slug: 'proof', size: '1536x1024', p: 'Three distinct elegant luminous measuring instruments aim converging beams of light onto a single crystalline object at center that glows confident green, while one beam exposes a hidden flaw glowing warning red; independent rigorous verification against a single source of truth.' },
22
18
  ];
23
19
 
24
20
  async function gen(model, prompt, size) {