ruvnet-brain 4.3.27 → 4.3.29
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/bin/install.mjs +74 -7
- package/console/app.js +7 -2
- package/kb/corpus-release-identity.mjs +73 -0
- package/package.json +4 -3
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/mcp/managed-cli-interface.mjs +83 -9
- package/plugin/scripts/advocacy-route.mjs +4 -1
- package/plugin/scripts/capacity-aware-parallel-work.mjs +4 -0
- package/plugin/scripts/continuation-gate.mjs +71 -0
- package/plugin/scripts/decision-gate.mjs +29 -57
- package/plugin/scripts/ground-ruvnet.sh +77 -4
- package/plugin/scripts/grounding-stamp.sh +34 -5
- package/plugin/scripts/grounding-turn-gate.mjs +8 -2
- package/plugin/scripts/grounding-turn-mark.mjs +6 -1
- package/plugin/scripts/hook-input.mjs +54 -0
- package/plugin/scripts/hook-shim.mjs +10 -0
- package/plugin/scripts/memory-doctor.mjs +43 -0
- package/plugin/scripts/nightly-controller.mjs +6 -1
- package/plugin/scripts/project-progression-contract.mjs +17 -0
- package/plugin/scripts/project-progression-hook.mjs +18 -5
- package/plugin/scripts/project-progression-producer.mjs +16 -12
- package/plugin/scripts/ruvnet-gate1-pattern.mjs +17 -0
- package/plugin/scripts/session-start-core.mjs +11 -5
- package/plugin/scripts/session-start-update-plane.mjs +5 -1
- package/plugin/scripts/unprompted-runtime.mjs +29 -8
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +1 -1
- package/scripts/adr-072-completion.mjs +2 -1
- package/scripts/calibrate-router.mjs +6 -6
- package/scripts/candidate-host-evidence.mjs +44 -16
- package/scripts/corpus-reconcile.mjs +32 -3
- package/scripts/correction-detect.mjs +10 -11
- package/scripts/dispatch-receipt.mjs +2 -2
- package/scripts/dual-host-deliberation.mjs +3 -2
- package/scripts/execution-policy.mjs +10 -2
- package/scripts/gen-console-images.mjs +0 -1
- package/scripts/gen-images.mjs +0 -4
- package/scripts/host-install-matrix.mjs +60 -14
- package/scripts/independent-review-receipt.mjs +7 -8
- package/scripts/ingest-repo.mjs +45 -1
- package/scripts/learning-replay-cli.mjs +2 -1
- package/scripts/learning-replay-fixture.mjs +2 -1
- package/scripts/learnings.mjs +19 -5
- package/scripts/lesson-migrate-agentdb.mjs +635 -0
- package/scripts/loop-checkpoint.mjs +37 -1
- package/scripts/metaharness-receipts.mjs +3 -2
- package/scripts/nightly-two-run-proof.mjs +6 -6
- package/scripts/nightly-watchdog.mjs +10 -2
- package/scripts/onboarding-console.mjs +98 -49
- package/scripts/oracle/producer-hosts.mjs +2 -1
- package/scripts/prepublication-evidence.mjs +8 -0
- package/scripts/private-overlay.mjs +31 -9
- package/scripts/public-verification-aggregate.mjs +4 -3
- package/scripts/publication-receipt.mjs +80 -30
- package/scripts/qe/agentic-qe-4.3.mjs +0 -1
- package/scripts/qe/ux-suite.mjs +17 -5
- package/scripts/rebuild-gists-from-receipts.mjs +1 -1
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/release-qualification-contract.mjs +11 -1
- package/scripts/release.mjs +34 -82
- package/scripts/retrieval-canary.mjs +3 -2
- package/scripts/review-model-defaults.mjs +26 -0
- package/scripts/route-cheap.mjs +20 -7
- package/scripts/router-utilization.mjs +5 -5
- package/scripts/rvf-generation.mjs +68 -2
- package/scripts/single-source-check.mjs +270 -0
- package/scripts/staged-host-verifier.mjs +2 -1
- package/scripts/subscription-hosts.mjs +4 -0
- package/scripts/sync-version.mjs +22 -0
- package/scripts/trismart.mjs +3 -3
- package/scripts/wired-check.mjs +40 -4
- package/console/assets/memory.webp +0 -0
- package/plugin/scripts/version-bump-gate.sh +0 -124
- package/scripts/qe/aggregate-4.3.mjs +0 -42
- package/scripts/release-convergence-watchdog.mjs +0 -112
- package/scripts/stamp-existing-rvf-generations.mjs +0 -53
- package/scripts/verify-channels.mjs +0 -196
|
@@ -27,3 +27,20 @@ export const RUVNET_GATE1_PATTERN =
|
|
|
27
27
|
export function ruvnetGate1Matches(text) {
|
|
28
28
|
return new RegExp(RUVNET_GATE1_PATTERN, 'i').test(String(text ?? ''));
|
|
29
29
|
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* H1 / GitHub #316: the plain substring vocabulary mechanically derived from RUVNET_GATE1_PATTERN —
|
|
33
|
+
* one lowercased term per `|`-separated alternative, with the `\b` word-boundary anchors stripped
|
|
34
|
+
* (irrelevant to a substring scan) and the one optional-plural alternative ("swarms?") reduced to
|
|
35
|
+
* its shortest substring-safe form ("swarm", which is a substring of both "swarm" and "swarms").
|
|
36
|
+
*
|
|
37
|
+
* This is the ONE vocabulary grounding-stamp.sh's GATE1_ONLY_TERMS must mirror byte-for-byte
|
|
38
|
+
* (tests/unit/grounding-stamp-terms.test.mjs enforces it, same idiom as
|
|
39
|
+
* tests/unit/ruvnet-gate1-pattern.test.mjs's byte-identity check against ground-ruvnet.sh). Before
|
|
40
|
+
* that fix, grounding-stamp.sh hard-coded its own narrower 9-term list that omitted `ruvnet` itself
|
|
41
|
+
* — so a search literally about "ruvnet" minted no stamp and grounding-turn-gate.mjs's Stop-time
|
|
42
|
+
* check wrongly reported "no successful search_ruvnet call this turn".
|
|
43
|
+
*/
|
|
44
|
+
export const RUVNET_GATE1_TERMS = RUVNET_GATE1_PATTERN
|
|
45
|
+
.split('|')
|
|
46
|
+
.map((alt) => alt.replace(/\\b/g, '').replace(/s\?$/, '').toLowerCase());
|
|
@@ -283,8 +283,6 @@ export async function runSessionStart({
|
|
|
283
283
|
} else if (running) {
|
|
284
284
|
write(path.join(stateDir, '.dev-version'), `${running}\n`);
|
|
285
285
|
}
|
|
286
|
-
const source = json(path.join(home, '.cache', 'ruvnet-brain', 'kb', 'SOURCE.json'), {});
|
|
287
|
-
const kbVersion = typeof source?.releaseTag === 'string' ? source.releaseTag : '';
|
|
288
286
|
const readiness = mcpReadiness(env, home);
|
|
289
287
|
|
|
290
288
|
const grounding = json(path.join(stateDir, 'install-state.json'));
|
|
@@ -315,10 +313,18 @@ export async function runSessionStart({
|
|
|
315
313
|
// alarm mechanism as HEALTH ALARM above (never gated by maintainerIssueEntitlement).
|
|
316
314
|
emit(`[RuvNet Brain v${bannerVersion} — active this session${updated ? ` · updated ${updated}` : ''}]`);
|
|
317
315
|
bannerEmitted = true;
|
|
318
|
-
|
|
319
|
-
|
|
316
|
+
// S2 (ONE CURRENCY VERDICT): this no longer compares the KB's own SOURCE.json releaseTag to the
|
|
317
|
+
// plugin version — that heuristic fires a FALSE POSITIVE for the entire (normal, expected)
|
|
318
|
+
// window between the plugin auto-updating and the KB's own background --check catching up, and
|
|
319
|
+
// it duplicated, less accurately, a comparison kb/forge-update.mjs already makes properly
|
|
320
|
+
// (against the LIVE canonical release, not a locally-observed string). Instead this reads the
|
|
321
|
+
// recorded verdict from the SessionStart heartbeat's own `--check --result-file` run — the SAME
|
|
322
|
+
// structured verdict --apply and bin/install.mjs read — and only alarms on a genuine CODE-release
|
|
323
|
+
// mismatch (a corpus-only update does not mean the plugin's own code is out of sync).
|
|
324
|
+
const kbCheck = json(path.join(stateDir, '.last-kb-check-result.json'));
|
|
325
|
+
if (kbCheck?.currencyVerdict === 'UPDATE_AVAILABLE' && kbCheck?.candidateKind === 'code') {
|
|
320
326
|
emit('🚨 [RuvNet Brain — INSTALL ALARM: plugin and knowledge bundle are out of sync] 🚨');
|
|
321
|
-
emit(
|
|
327
|
+
emit(`${kbCheck.currencyReason || 'a newer code release is available'} — they are meant to ship together, so search results may not match this plugin's behavior yet. Fix: npx ruvnet-brain@latest --update (or reinstall: npx github:stuinfla/ruvnet-brain --force).`);
|
|
322
328
|
}
|
|
323
329
|
|
|
324
330
|
const hookContracts = readHookContracts(path.join(pluginRoot, 'hooks', 'hook-contracts.json'));
|
|
@@ -77,8 +77,12 @@ export const heartbeat = ({ env, hookDir, stateDir, home, running, seedDispatche
|
|
|
77
77
|
if (/\bBEHIND\b/.test(read(kbLog))) {
|
|
78
78
|
emit('[RuvNet Brain — a newer knowledge bundle is available. It is signed (Ed25519) and the updater verifies that signature before extracting anything. We do NOT auto-apply it: applying replaces executable tool files, which is your call. To update: cd ~/.cache/ruvnet-brain/kb && node forge-update.mjs --apply]');
|
|
79
79
|
}
|
|
80
|
+
// S2 (ONE CURRENCY VERDICT): --result-file records the SAME structured verdict --check/--apply
|
|
81
|
+
// and bin/install.mjs already read (forge-update.mjs's currencyVerdict()), at the well-known path
|
|
82
|
+
// session-start-core.mjs's banner stage reads it from — so the banner's own "is my code out of
|
|
83
|
+
// sync" alarm reads this recorded verdict instead of re-deriving its own comparison.
|
|
80
84
|
dispatchDetached(hookDir, 60, kbLog, process.execPath,
|
|
81
|
-
[path.join(kbDir, 'forge-update.mjs'), '--check'], env);
|
|
85
|
+
[path.join(kbDir, 'forge-update.mjs'), '--check', '--result-file', path.join(stateDir, '.last-kb-check-result.json')], env);
|
|
82
86
|
}
|
|
83
87
|
const versionLog = path.join(stateDir, '.last-version-check.log');
|
|
84
88
|
const remoteVersion = firstVersion(read(versionLog));
|
|
@@ -86,7 +86,7 @@ import fs from 'node:fs';
|
|
|
86
86
|
import path from 'node:path';
|
|
87
87
|
import { spawnSync } from 'node:child_process';
|
|
88
88
|
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
89
|
-
import { readStdinBounded } from './hook-input.mjs';
|
|
89
|
+
import { readStdinBounded, isHarnessGenerated } from './hook-input.mjs';
|
|
90
90
|
import { resolveBash } from './hook-shim-bash.mjs';
|
|
91
91
|
|
|
92
92
|
// WHERE THIS FILE'S SIBLINGS LIVE. Resolved from THIS file's own location so it is correct under the
|
|
@@ -112,9 +112,16 @@ const SCRIPTS_DIR = path.dirname(SELF); // the payload's scr
|
|
|
112
112
|
// The CC event name the shim forwarded. No event → nothing to run; stay silent.
|
|
113
113
|
const EVENT = process.argv[2] || '';
|
|
114
114
|
|
|
115
|
-
// Bound the whole runtime well under
|
|
116
|
-
//
|
|
117
|
-
|
|
115
|
+
// Bound the whole runtime well under this hook's DECLARED timeout (hooks.json / codex-hooks.json:
|
|
116
|
+
// 'unprompted-speech' is 3s, not the 5s this comment used to assume) — producers run sequentially and
|
|
117
|
+
// each has its own internal watchdog, but a backstop timeout here means a wedged producer can never
|
|
118
|
+
// hang the turn. Found live 2026-09-27: the old 4000ms default left NO real margin under a 3000ms
|
|
119
|
+
// declared timeout once Node startup + per-producer spawnSync overhead is counted, and slower
|
|
120
|
+
// per-process-spawn hosts (Windows CI) pushed measured wall-clock to 83% of budget — selfcheck.mjs's
|
|
121
|
+
// own 80%-margin rule exists exactly to catch a hook running this close to its declared contract.
|
|
122
|
+
// 2000ms leaves real headroom (Node/import startup + the 80% margin check at 2400ms) on every host,
|
|
123
|
+
// not just a Windows-specific patch.
|
|
124
|
+
const PRODUCER_TIMEOUT_MS = Number(process.env.RUVNET_UNPROMPTED_TIMEOUT_MS) || 2000;
|
|
118
125
|
const MAX_BUFFER = 1 << 20;
|
|
119
126
|
|
|
120
127
|
const VALID_CHANNELS = new Set(['advocacy', 'promotion', 'lesson', 'alarm']);
|
|
@@ -223,6 +230,17 @@ try {
|
|
|
223
230
|
} catch { /* not JSON → no occasion → silence, below */ }
|
|
224
231
|
if (!event) silent();
|
|
225
232
|
|
|
233
|
+
// H2: a background task notification, slash-command scaffold, or other harness-authored message
|
|
234
|
+
// arrives on UserPromptSubmit exactly like real user text (wrapped in tags such as
|
|
235
|
+
// <task-notification>, <local-command-caveat>, <system-reminder>, ...). None of that is something a
|
|
236
|
+
// user typed, so no producer here should react to it — an advocacy nudge, a lesson prompt, or a
|
|
237
|
+
// promotion offer fired at the harness's own bookkeeping is noise on every background-task turn.
|
|
238
|
+
// PreToolUse payloads carry no `prompt`/`user_prompt`/`input` field, so this is a no-op for them.
|
|
239
|
+
{
|
|
240
|
+
const promptText = event.prompt ?? event.user_prompt ?? event.input;
|
|
241
|
+
if (typeof promptText === 'string' && isHarnessGenerated(promptText)) silent();
|
|
242
|
+
}
|
|
243
|
+
|
|
226
244
|
const producers = resolveProducers(EVENT);
|
|
227
245
|
if (!producers.length) silent(); // unknown event, or nothing wired for it — never speak on a guess
|
|
228
246
|
|
|
@@ -396,10 +414,13 @@ for (const c of candidates) {
|
|
|
396
414
|
let offer = true;
|
|
397
415
|
try { offer = led.shouldStillOffer(findingId, { severity, stateHash }); } catch { offer = false; }
|
|
398
416
|
if (!offer) break; // dismissed / budget spent → drop
|
|
399
|
-
//
|
|
400
|
-
//
|
|
401
|
-
//
|
|
402
|
-
|
|
417
|
+
// Persist the OFFERED denominator before delivery. A recommendation whose delivery receipt was
|
|
418
|
+
// not durably written cannot participate in the later applied/dismissed lifecycle; emitting it
|
|
419
|
+
// anyway would create a card the next prompt cannot resolve and would make precision lie.
|
|
420
|
+
let receipt;
|
|
421
|
+
try { receipt = led.record({ id: findingId, action: led.ACTIONS.OFFERED, severity, stateHash }); }
|
|
422
|
+
catch { receipt = null; }
|
|
423
|
+
if (!receipt?.ok) break;
|
|
403
424
|
advisories.push({ copy, hookEventName });
|
|
404
425
|
break;
|
|
405
426
|
}
|
|
@@ -13,7 +13,7 @@ not all individually audited and remain diagnostics. A diagnostic pass is not pr
|
|
|
13
13
|
Promote only the clean exact candidate SHA and its sealed artifacts after qualification. Public
|
|
14
14
|
acceptance still requires all nine OS/host-mode leaves plus real native installed-update proof on
|
|
15
15
|
each platform, ending at `install-verified`. Imported upstream corpus freshness remains UNKNOWN
|
|
16
|
-
until separately proven. See `
|
|
16
|
+
until separately proven. See `CONTRIBUTING.md` and the audit records in `docs/reviews/`.
|
|
17
17
|
|
|
18
18
|
---
|
|
19
19
|
|
|
@@ -4,9 +4,10 @@ import path from 'node:path';
|
|
|
4
4
|
import { spawnSync } from 'node:child_process';
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { verifyCapabilityClaimAggregate } from '../plugin/scripts/capability-claim-evidence.mjs';
|
|
7
|
+
import { LEGACY_REVIEW_MODEL_IDS } from './review-model-defaults.mjs';
|
|
7
8
|
|
|
8
9
|
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
9
|
-
const REQUIRED_REVIEWERS =
|
|
10
|
+
const REQUIRED_REVIEWERS = LEGACY_REVIEW_MODEL_IDS;
|
|
10
11
|
|
|
11
12
|
const command = (cwd, bin, args) => {
|
|
12
13
|
const result = spawnSync(bin, args, { cwd, encoding: 'utf8', maxBuffer: 64 * 1024 * 1024 });
|
|
@@ -26,18 +26,18 @@ import fs from 'node:fs';
|
|
|
26
26
|
import os from 'node:os';
|
|
27
27
|
import path from 'node:path';
|
|
28
28
|
import { recordOutcome } from './metaharness-router.mjs';
|
|
29
|
-
import { estimateCosts, estTokens } from './route-cheap.mjs';
|
|
29
|
+
import { CLAUDE_MODEL_IDS, estimateCosts, estTokens } from './route-cheap.mjs';
|
|
30
30
|
|
|
31
31
|
const CLAUDE = path.join(os.homedir(), '.npm-global/bin/claude');
|
|
32
32
|
const RECEIPTS = process.env.METAHARNESS_RECEIPTS
|
|
33
33
|
|| path.join(os.homedir(), '.claude', 'metaharness', 'routing-receipts.jsonl');
|
|
34
34
|
|
|
35
35
|
const MODELS = [
|
|
36
|
-
{ alias: 'haiku', name:
|
|
37
|
-
{ alias: 'sonnet', name:
|
|
38
|
-
{ alias: 'opus', name:
|
|
36
|
+
{ alias: 'haiku', name: CLAUDE_MODEL_IDS.haiku },
|
|
37
|
+
{ alias: 'sonnet', name: CLAUDE_MODEL_IDS.sonnet },
|
|
38
|
+
{ alias: 'opus', name: CLAUDE_MODEL_IDS.opus }, // the baseline tier
|
|
39
39
|
];
|
|
40
|
-
const BASELINE =
|
|
40
|
+
const BASELINE = CLAUDE_MODEL_IDS.opus;
|
|
41
41
|
|
|
42
42
|
// Deterministic tasks: known answers, graded by regex — no LLM judge, no judgment calls.
|
|
43
43
|
const TASKS = [
|
|
@@ -90,7 +90,7 @@ for (const [ti, task] of TASKS.entries()) {
|
|
|
90
90
|
}
|
|
91
91
|
}
|
|
92
92
|
|
|
93
|
-
const cheap = results.flatMap((r) => [r.runs[
|
|
93
|
+
const cheap = results.flatMap((r) => [r.runs[CLAUDE_MODEL_IDS.haiku].ms]);
|
|
94
94
|
const base = results.map((r) => r.runs[BASELINE].ms);
|
|
95
95
|
const sum = (a) => a.reduce((s, x) => s + x, 0);
|
|
96
96
|
console.log(`\nhaiku total ${(sum(cheap) / 1000).toFixed(1)}s vs opus baseline ${(sum(base) / 1000).toFixed(1)}s on ${TASKS.length} tasks`);
|
|
@@ -5,6 +5,7 @@ import { fileURLToPath } from 'node:url';
|
|
|
5
5
|
import { stagedHostVerifier, readCandidateRetrieval, verifyCandidateRetrievalAssets } from './staged-host-verifier.mjs';
|
|
6
6
|
import { payloadIdFor } from './release-payload.mjs';
|
|
7
7
|
import { validateRetrievalCanaryReceipt } from './retrieval-canary.mjs';
|
|
8
|
+
import { HOST_WARMUP_TIMEOUT_MS, RELEASE_SEARCH_DEADLINE_MS } from './host-install-matrix.mjs';
|
|
8
9
|
|
|
9
10
|
const arg = (name) => {
|
|
10
11
|
const index = process.argv.indexOf(name);
|
|
@@ -12,13 +13,13 @@ const arg = (name) => {
|
|
|
12
13
|
};
|
|
13
14
|
|
|
14
15
|
export async function buildCandidateHostEvidence({ manifestFile, packagePath, bundlePath, planFile, coverageFile, failureFile },
|
|
15
|
-
{ createVerifier = stagedHostVerifier } = {}) {
|
|
16
|
+
{ createVerifier = stagedHostVerifier, sequentialSearches = false } = {}) {
|
|
16
17
|
const manifest = JSON.parse(fs.readFileSync(manifestFile, 'utf8'));
|
|
17
18
|
const payloadId = payloadIdFor(manifest);
|
|
18
19
|
const identity = { version: manifest.version, candidateSha: manifest.candidateSha, payloadId };
|
|
19
20
|
const retrieval = readCandidateRetrieval({ manifest, planFile, coverageFile });
|
|
20
21
|
verifyCandidateRetrievalAssets({ retrieval, assets: { packagePath, bundlePath } });
|
|
21
|
-
const result = await createVerifier({ assets: { packagePath, bundlePath }, identity, retrieval })
|
|
22
|
+
const result = await createVerifier({ assets: { packagePath, bundlePath }, identity, retrieval, sequentialSearches })
|
|
22
23
|
.verify({ source: 'candidate', assets: { packagePath, bundlePath } });
|
|
23
24
|
verifyCandidateRetrievalAssets({ retrieval, assets: { packagePath, bundlePath } });
|
|
24
25
|
if (result.verdict !== 'PASS') {
|
|
@@ -32,34 +33,60 @@ export async function buildCandidateHostEvidence({ manifestFile, packagePath, bu
|
|
|
32
33
|
}
|
|
33
34
|
|
|
34
35
|
const modeNames = { claude: 'claude-only', codex: 'codex-only', dual: 'dual-host' };
|
|
35
|
-
const leaves =
|
|
36
|
+
const leaves = [];
|
|
37
|
+
const failures = [];
|
|
38
|
+
for (const [mode, name] of Object.entries(modeNames)) {
|
|
36
39
|
const fixture = result.fixtures?.[mode];
|
|
37
40
|
const grounding = fixture?.grounding;
|
|
38
|
-
const
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
41
|
+
const warmupGrounding = fixture?.warmupGrounding;
|
|
42
|
+
const hasReceipt = (receipt) => receipt && ['repo', 'path', 'file', 'storedPath']
|
|
43
|
+
.every((field) => typeof receipt[field] === 'string' && receipt[field].trim());
|
|
44
|
+
if (!Number.isFinite(fixture?.warmupMs) || fixture.warmupMs < 0 || fixture.warmupMs > HOST_WARMUP_TIMEOUT_MS
|
|
45
|
+
|| !hasReceipt(warmupGrounding) || warmupGrounding.repo !== 'ruvnet-brain') {
|
|
46
|
+
failures.push(`${name} did not produce a grounded worker warmup within ${HOST_WARMUP_TIMEOUT_MS}ms (${fixture?.warmupMs ?? 'unmeasured'}ms)`);
|
|
42
47
|
}
|
|
43
|
-
|
|
44
|
-
|
|
48
|
+
if (!Number.isFinite(fixture?.searchMs) || fixture.searchMs < 0 || fixture.searchMs > RELEASE_SEARCH_DEADLINE_MS) {
|
|
49
|
+
failures.push(`${name} first measured cited search exceeded the ${RELEASE_SEARCH_DEADLINE_MS}ms candidate deadline (${fixture?.searchMs ?? 'unmeasured'}ms)`);
|
|
50
|
+
}
|
|
51
|
+
if (fixture?.status !== 'PASS' || fixture?.process?.status !== 0 || !hasReceipt(grounding)) {
|
|
52
|
+
failures.push(`${name} did not produce a clean installed-search grounding receipt`);
|
|
53
|
+
}
|
|
54
|
+
try { validateRetrievalCanaryReceipt(fixture?.retrieval, { plan: retrieval.plan }); }
|
|
55
|
+
catch (error) { failures.push(`${name} retrieval canary rejected: ${error.message}`); }
|
|
56
|
+
leaves.push({
|
|
45
57
|
name,
|
|
46
58
|
sha: manifest.candidateSha,
|
|
47
59
|
payloadId,
|
|
48
|
-
status: fixture
|
|
49
|
-
conclusion: 'success',
|
|
60
|
+
status: fixture?.status === 'PASS' && fixture?.process?.status === 0 && hasReceipt(grounding) ? 'completed' : 'failed',
|
|
61
|
+
conclusion: failures.length ? 'failure' : 'success',
|
|
50
62
|
verdict: 'PASS',
|
|
51
63
|
source: 'candidate-host-evidence',
|
|
52
64
|
mode,
|
|
53
|
-
functionalSearch:
|
|
54
|
-
searchExit: fixture
|
|
65
|
+
functionalSearch: fixture?.status === 'PASS',
|
|
66
|
+
searchExit: fixture?.process?.status ?? null,
|
|
67
|
+
warmupMs: fixture?.warmupMs,
|
|
68
|
+
warmupGrounding,
|
|
69
|
+
searchMs: fixture?.searchMs,
|
|
55
70
|
grounding,
|
|
56
|
-
retrieval: fixture
|
|
71
|
+
retrieval: fixture?.retrieval,
|
|
57
72
|
artifactSha256: retrieval.artifactSha256,
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
if (failures.length) {
|
|
76
|
+
const failure = {
|
|
77
|
+
schemaVersion: 1, kind: 'ruvnet-brain-candidate-host-failure', verdict: 'FAIL',
|
|
78
|
+
sha: manifest.candidateSha, payloadId, artifactSha256: retrieval.artifactSha256,
|
|
79
|
+
candidateArchiveSha256: retrieval.candidateArchiveSha256, planSha256: retrieval.plan.planSha256,
|
|
80
|
+
failures, result,
|
|
58
81
|
};
|
|
59
|
-
|
|
82
|
+
if (failureFile) fs.writeFileSync(path.resolve(failureFile), JSON.stringify(failure, null, 2) + '\n',
|
|
83
|
+
{ flag: 'wx', mode: 0o600 });
|
|
84
|
+
throw new Error(`candidate host matrix failed: ${failures.join('; ')}`);
|
|
85
|
+
}
|
|
60
86
|
return {
|
|
61
87
|
schemaVersion: 1,
|
|
62
88
|
sha: manifest.candidateSha,
|
|
89
|
+
hostPlatform: process.platform,
|
|
63
90
|
payloadId,
|
|
64
91
|
artifactSha256: retrieval.artifactSha256,
|
|
65
92
|
leaves,
|
|
@@ -68,7 +95,8 @@ export async function buildCandidateHostEvidence({ manifestFile, packagePath, bu
|
|
|
68
95
|
|
|
69
96
|
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
70
97
|
const evidence = await buildCandidateHostEvidence({ manifestFile: arg('--manifest'),
|
|
71
|
-
packagePath: arg('--package'), bundlePath: arg('--bundle'), planFile: arg('--plan'), coverageFile: arg('--coverage'), failureFile: arg('--out') ? `${arg('--out')}.failure.json` : null }
|
|
98
|
+
packagePath: arg('--package'), bundlePath: arg('--bundle'), planFile: arg('--plan'), coverageFile: arg('--coverage'), failureFile: arg('--out') ? `${arg('--out')}.failure.json` : null },
|
|
99
|
+
{ sequentialSearches: process.argv.includes('--sequential-searches') });
|
|
72
100
|
fs.writeFileSync(path.resolve(arg('--out')), `${JSON.stringify(evidence, null, 2)}\n`, { flag: 'wx', mode: 0o600 });
|
|
73
101
|
console.log(JSON.stringify({ verdict: 'PASS', payloadId: evidence.payloadId, leaves: evidence.leaves.map(({ name }) => name) }));
|
|
74
102
|
}
|
|
@@ -17,8 +17,10 @@ import { promoteArtifactSet } from '../kb/incremental-refresh.mjs';
|
|
|
17
17
|
import { rebuildCorpusAggregates } from './corpus-aggregates.mjs';
|
|
18
18
|
import { assertCapabilityOnlyStore, isCapabilityOnly, CAPABILITY_RETIRED_SUFFIXES } from '../kb/capability-only.mjs';
|
|
19
19
|
import { fileIdentity } from '../plugin/scripts/coverage-integrity.mjs';
|
|
20
|
+
import { readDiagnosticAccuracyReport } from './oracle/retrieval-accuracy.mjs';
|
|
20
21
|
import { storeRoot } from '../kb/store-root.mjs';
|
|
21
22
|
import { captureGistSources } from './gist-receipts.mjs';
|
|
23
|
+
import { projectSourceStore } from './rvf-generation.mjs';
|
|
22
24
|
|
|
23
25
|
export { rebuildCorpusAggregates };
|
|
24
26
|
|
|
@@ -540,7 +542,13 @@ export async function executeReconciliation({
|
|
|
540
542
|
promotedFiles.push(file.name);
|
|
541
543
|
}
|
|
542
544
|
mergedLedger.stores[result.store] = result.generation;
|
|
543
|
-
|
|
545
|
+
// S4 (ONE PROVENANCE RECORD): re-project the merged SOURCE.json entry FROM the merged ledger
|
|
546
|
+
// row rather than trusting the worker's own already-written SOURCE.json fragment verbatim —
|
|
547
|
+
// the merge boundary is where multiple workers' results combine, so it is the right place to
|
|
548
|
+
// assert "the ledger is the source of truth" rather than assume every worker upheld it.
|
|
549
|
+
// `result.source` (validateWorkerOutput's read of the worker's own output, already checked
|
|
550
|
+
// there to bind the exact upstream SHA) supplies the non-identity updater fields unchanged.
|
|
551
|
+
mergedSource.stores[result.store] = projectSourceStore(result.store, result.generation, result.source);
|
|
544
552
|
}
|
|
545
553
|
mergedLedger.stores = Object.fromEntries(Object.entries(mergedLedger.stores).sort(([a], [b]) => a.localeCompare(b)));
|
|
546
554
|
mergedSource.stores = Object.fromEntries(Object.entries(mergedSource.stores).sort(([a], [b]) => a.localeCompare(b)));
|
|
@@ -766,12 +774,33 @@ export function prepareCorpusCandidate({
|
|
|
766
774
|
// bounded run (--stores/--sample) still writes a report, but it marks itself incomplete and the
|
|
767
775
|
// seal refuses it, so a bounded measurement can never be presented as a corpus-wide pass.
|
|
768
776
|
const accuracyReportFile = `${bundleFile}.accuracy.json`;
|
|
769
|
-
|
|
777
|
+
// A stale leftover report from a prior run must never be mistaken for a fresh measurement of
|
|
778
|
+
// THIS bundle -- delete it before invoking the script so only a report the script just wrote
|
|
779
|
+
// (or none at all) can be found below.
|
|
780
|
+
fs.rmSync(accuracyReportFile, { force: true });
|
|
781
|
+
const accuracyResult = run(process.execPath, [accuracyScript, '--bundle', bundleFile,
|
|
770
782
|
'--oracle', accuracyOracle, '--out', accuracyReportFile,
|
|
771
783
|
...(accuracyStores != null ? ['--stores', String(accuracyStores)] : []),
|
|
772
784
|
...(accuracySample != null ? ['--sample', String(accuracySample)] : []),
|
|
773
785
|
...(accuracyTimeoutMs != null ? ['--timeout-ms', String(accuracyTimeoutMs)] : [])],
|
|
774
|
-
{ stdio: 'inherit' });
|
|
786
|
+
{ stdio: 'inherit' }) || {};
|
|
787
|
+
// C3 was demoted to a non-blocking diagnostic on 2026-09-15 (commit a20727b7, ADR-086
|
|
788
|
+
// amendment) -- every other caller (corpus-candidate.mjs, release.mjs, corpus-seed.yml) reads
|
|
789
|
+
// it through readDiagnosticAccuracyReport, which checks the report's integrity/archive binding,
|
|
790
|
+
// never its score. A nonzero exit here is therefore NOT immediately fatal: it may just mean the
|
|
791
|
+
// measured score fell below the (no-longer-enforced) threshold. What stays fatal is a CRASHED
|
|
792
|
+
// measurement -- no valid report bound to this exact archive was produced at all.
|
|
793
|
+
if (accuracyResult.error || accuracyResult.status !== 0) {
|
|
794
|
+
let diagnostic;
|
|
795
|
+
try {
|
|
796
|
+
diagnostic = readDiagnosticAccuracyReport({ reportFile: accuracyReportFile, archive: fileIdentity(bundleFile) });
|
|
797
|
+
} catch (error) {
|
|
798
|
+
fail(`retrieval-accuracy diagnostic (C3) crashed with no valid report to show for it: ${error.message}`);
|
|
799
|
+
}
|
|
800
|
+
console.log(`[corpus-reconcile] C3 retrieval-accuracy diagnostic scored below its (non-blocking) `
|
|
801
|
+
+ `threshold: state=${diagnostic.state} classification=${diagnostic.classification} `
|
|
802
|
+
+ `totals=${JSON.stringify(diagnostic.totals)} -- continuing, C3 is advisory only.`);
|
|
803
|
+
}
|
|
775
804
|
// THE BLOCKING RETRIEVAL GATE (ADR-086 amendment 2026-09-15). Same placement and same discipline
|
|
776
805
|
// as the C3 run above — the EXTRACTED final archive through the customer query path — but this is
|
|
777
806
|
// the measurement that can refuse a candidate. It asks the 194 frozen human questions, one per
|
|
@@ -188,6 +188,8 @@
|
|
|
188
188
|
// industrialised. `confidence` orders the ratification queue and does nothing else — a confidence
|
|
189
189
|
// threshold is just ratification with the human removed and the word "confidence" in front of it.
|
|
190
190
|
|
|
191
|
+
import { HARNESS_GENERATED_PATTERNS } from '../plugin/scripts/hook-input.mjs';
|
|
192
|
+
|
|
191
193
|
/** Corrections are short. The measured tightened detector used this bound; specs and briefs exceed it. */
|
|
192
194
|
export const MAX_UTTERANCE_CHARS = 800;
|
|
193
195
|
|
|
@@ -224,18 +226,15 @@ export const ACCEPTED_MISSES = Object.freeze([
|
|
|
224
226
|
* speech at all. The detector was right to ignore them; the pool handed them to a human to label
|
|
225
227
|
* anyway, burning ~29% of the scarcest resource in this whole problem (labelled examples) on rows
|
|
226
228
|
* whose answer is definitionally "no", and diluting the base rate with them.
|
|
229
|
+
*
|
|
230
|
+
* H2 (2026-09-26): re-exported from plugin/scripts/hook-input.mjs's HARNESS_GENERATED_PATTERNS
|
|
231
|
+
* rather than kept as this file's own literal array. Other UserPromptSubmit consumers
|
|
232
|
+
* (unprompted-runtime.mjs, capacity-aware-parallel-work.mjs, grounding-turn-mark.mjs) needed the
|
|
233
|
+
* exact same recognition and, before this fix, had no shared place to get it from — hook-input.mjs
|
|
234
|
+
* is that ONE owner now; this name stays so correction-detect-measure.mjs and this file's own test
|
|
235
|
+
* do not need to change.
|
|
227
236
|
*/
|
|
228
|
-
export const HARNESS_TEMPLATES =
|
|
229
|
-
/\[Your previous response/i,
|
|
230
|
-
/\[Request interrupted/i,
|
|
231
|
-
/<\/?system-reminder>/i,
|
|
232
|
-
/<\/?(?:command-name|command-message|command-args|local-command-stdout|local-command-stderr|local-command-caveat|task-notification|function_results|function_calls|budget)\b/i,
|
|
233
|
-
/^\s*Caveat:/i,
|
|
234
|
-
/Base directory for this skill:/i,
|
|
235
|
-
/This session is being continued from a previous conversation/i,
|
|
236
|
-
/^\s*#\s*claudeMd\b/im,
|
|
237
|
-
/\[INTELLIGENCE\]/i,
|
|
238
|
-
];
|
|
237
|
+
export const HARNESS_TEMPLATES = HARNESS_GENERATED_PATTERNS;
|
|
239
238
|
|
|
240
239
|
/**
|
|
241
240
|
* Pasted content, not speech. Includes markdown structure — this repository's own ADRs and DDDs are
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
import fs from 'node:fs';
|
|
26
26
|
import path from 'node:path';
|
|
27
27
|
import { pathToFileURL } from 'node:url';
|
|
28
|
-
import { CLAUDE_TIERS, PRICING, estTokens, estimateCosts, receiptLine, receiptsPath, priceOf } from './route-cheap.mjs';
|
|
28
|
+
import { CLAUDE_MODEL_IDS, CLAUDE_TIERS, PRICING, estTokens, estimateCosts, receiptLine, receiptsPath, priceOf } from './route-cheap.mjs';
|
|
29
29
|
|
|
30
30
|
// Default input share when only a MEASURED TOTAL is known. A subagent's tokens are dominated by input
|
|
31
31
|
// (it re-reads files and tool output on every turn); its final report is small. 0.9 is an assumption,
|
|
@@ -33,7 +33,7 @@ import { CLAUDE_TIERS, PRICING, estTokens, estimateCosts, receiptLine, receiptsP
|
|
|
33
33
|
const DEFAULT_INPUT_SHARE = 0.9;
|
|
34
34
|
|
|
35
35
|
export function parseArgs(argv) {
|
|
36
|
-
const args = { model:
|
|
36
|
+
const args = { model: CLAUDE_MODEL_IDS.haiku, inherited: CLAUDE_MODEL_IDS.opus, class: 'mechanical' };
|
|
37
37
|
for (let i = 0; i < argv.length; i++) {
|
|
38
38
|
const k = argv[i];
|
|
39
39
|
if (['--model', '--inherited', '--task', '--class', '--in-chars', '--out-chars', '--label', '--total-tokens', '--split'].includes(k)) {
|
|
@@ -5,13 +5,14 @@ import fs from 'node:fs';
|
|
|
5
5
|
import { spawn } from 'node:child_process';
|
|
6
6
|
import { fileURLToPath } from 'node:url';
|
|
7
7
|
import { probeSubscriptionHosts, subscriptionOnlyEnv } from './subscription-hosts.mjs';
|
|
8
|
+
import { DUAL_HOST_MODEL_IDS } from './review-model-defaults.mjs';
|
|
8
9
|
|
|
9
10
|
const HOSTS = Object.freeze(['claude-code', 'codex']);
|
|
10
11
|
// Top subscription models verified on the native hosts on 2026-09-10.
|
|
11
12
|
// Keep these explicit: an implicit host default silently weakens the dual review.
|
|
12
13
|
export const TOP_SUBSCRIPTION_MODELS = Object.freeze({
|
|
13
|
-
'claude-code':
|
|
14
|
-
codex:
|
|
14
|
+
'claude-code': DUAL_HOST_MODEL_IDS.claude,
|
|
15
|
+
codex: DUAL_HOST_MODEL_IDS.codex,
|
|
15
16
|
});
|
|
16
17
|
const HARD_PROBLEM = /\b(adr|architecture|architect|ddd|bounded context|aggregate|agentic[- ]?qe|holistic|security|production|migration|irreversible|threat model|experience)\b/i;
|
|
17
18
|
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
const NATIVE_HOSTS = new Set(['claude', 'codex']);
|
|
6
6
|
const API_EXECUTORS = new Set(['agent_execute', 'sdk', 'openrouter', 'api']);
|
|
7
|
+
const ACTIONS = new Set(['read', 'write', 'delegate', 'release', 'external']);
|
|
7
8
|
const ARCHITECTURE_TERMS = /\b(adr|ddd|architecture|release|deploy|qa|security|schema|migration)\b/i;
|
|
8
9
|
const CONSEQUENTIAL_ACTIONS = new Set(['delegate', 'write', 'release', 'external']);
|
|
9
10
|
const FRESHNESS_MS = 30 * 60 * 1000;
|
|
@@ -33,8 +34,8 @@ export function validateEvidence(input = {}, now = Date.now()) {
|
|
|
33
34
|
if (!memory || memory.status !== 'retrieved' || !fresh(memory.observedAt, now)) {
|
|
34
35
|
failures.push('fresh exact AgentDB checkpoint retrieval is required');
|
|
35
36
|
} else {
|
|
36
|
-
if (
|
|
37
|
-
if (!/^project-state-current-\d
|
|
37
|
+
if (!/[\\/]\.swarm[\\/]memory\.db$/.test(String(memory.path || ''))) failures.push('memory receipt must use the project .swarm/memory.db');
|
|
38
|
+
if (!/^project-state-current-\d+(?:-[A-Za-z0-9][A-Za-z0-9_-]*)?$/.test(String(memory.key || ''))) failures.push('memory receipt must retrieve an append-only project-state-current key');
|
|
38
39
|
if (!hex64(memory.valueDigest)) failures.push('memory receipt must bind the retrieved value digest');
|
|
39
40
|
}
|
|
40
41
|
return { valid: failures.length === 0, failures };
|
|
@@ -46,6 +47,13 @@ export function classifyExecutionPolicy(input = {}) {
|
|
|
46
47
|
const changedFiles = [...new Set((input.changedFiles || []).map(String).filter(Boolean))];
|
|
47
48
|
const nativeHosts = [...new Set((input.nativeHosts || []).map(String).filter((h) => NATIVE_HOSTS.has(h)))];
|
|
48
49
|
const requestedExecutor = String(input.requestedExecutor || '');
|
|
50
|
+
if (!ACTIONS.has(action)) {
|
|
51
|
+
return {
|
|
52
|
+
schema: 'ruvnet-brain.execution-policy.v1',
|
|
53
|
+
verdict: 'REFUSE', action, swarmRequired: false, swarmReason: 'invalid-action',
|
|
54
|
+
executor: 'unknown', reason: 'unsupported-action', evidence: { valid: false, failures: ['action must be one of read, write, delegate, release, external'] },
|
|
55
|
+
};
|
|
56
|
+
}
|
|
49
57
|
const explicitSwarm = input.explicitSwarm === true;
|
|
50
58
|
const multiFile = changedFiles.length >= 3;
|
|
51
59
|
const architectureTask = ARCHITECTURE_TERMS.test(description) || ['release', 'external'].includes(action);
|
|
@@ -24,7 +24,6 @@ const STYLE = ' — Style: deep near-black background (#0a0c10), sophisticated p
|
|
|
24
24
|
|
|
25
25
|
const IMAGES = [
|
|
26
26
|
{ slug: 'hero', size: '1536x1024', p: 'A warm amber intelligence gently understanding a computer: soft glowing amber and gold neural filaments and threads of light weaving and resolving out of a faint tangle on the left into an elegant, orderly, translucent crystalline lattice of floating glass panels and cards on the right — the feeling of messy machine settings being calmly brought into clear, beautiful order. Lots of soft dark negative space on the right for text.' },
|
|
27
|
-
{ slug: 'memory', size: '1024x1024', p: 'A single luminous softly-glowing sphere of warm amber and cyan light, made of countless fine interwoven filaments, holding its shape calmly in dark space — an abstract emblem of a mind that remembers; serene, alive, precise.' },
|
|
28
27
|
];
|
|
29
28
|
|
|
30
29
|
async function gen(model, prompt, size) {
|
package/scripts/gen-images.mjs
CHANGED
|
@@ -15,10 +15,6 @@ const STYLE = ' — Style: dark near-black background (#0b0d0f) with a faint blu
|
|
|
15
15
|
|
|
16
16
|
const IMAGES = [
|
|
17
17
|
{ slug: 'hero', size: '1536x1024', p: 'A luminous intricate three-dimensional structure resembling a brain fused with a vast interconnected codebase: thousands of glowing amber and cyan filaments forming one elegant organized sphere of intelligence, floating in dark space, a sense of all knowledge made orderly and alive.' },
|
|
18
|
-
{ slug: 'problem-skim', size: '1536x1024', p: 'A vast deep canyon made of densely stacked layers of code and documents descending far into darkness; a single small fragile light hovers at the very top only grazing the surface, never reaching the immense depth below. The feeling of skimming and missing everything underneath.' },
|
|
19
|
-
{ slug: 'point-deeper', size: '1536x1024', p: 'One precise clean beam of warm amber light cutting straight down through many deep translucent strata of a vast structure to perfectly illuminate a single exact point far below; surgical precision locating the one true answer in the depths.' },
|
|
20
|
-
{ slug: 'architecture', size: '1536x1024', p: 'An elegant isometric exploded view of five translucent glass layers floating one above another in dark space, each a slightly different luminous tone, joined by thin vertical conduits of light; a refined premium product render of a clean layered system.' },
|
|
21
|
-
{ slug: 'proof', size: '1536x1024', p: 'Three distinct elegant luminous measuring instruments aim converging beams of light onto a single crystalline object at center that glows confident green, while one beam exposes a hidden flaw glowing warning red; independent rigorous verification against a single source of truth.' },
|
|
22
18
|
];
|
|
23
19
|
|
|
24
20
|
async function gen(model, prompt, size) {
|