ruvnet-brain 4.0.1 → 4.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +100 -5
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +21 -0
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +7 -7
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +477 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +81 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +522 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +250 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +639 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +271 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +180 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2749 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +395 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +508 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +864 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
// private-fence.mjs — the fence that keeps private repos out of anything we ship (SEC-0010 #5).
|
|
2
|
+
//
|
|
3
|
+
// This logic used to live inline in build-concepts.mjs, executed at import time, and it called
|
|
4
|
+
// process.exit(1) on failure. That made it literally untestable: importing the module to test the
|
|
5
|
+
// fence would kill the test runner. So the highest-severity code in the repo — the thing standing
|
|
6
|
+
// between a private cognitum store and a public 512MB release — had zero tests.
|
|
7
|
+
//
|
|
8
|
+
// Here the decisions are pure: they RETURN {ok, reason} and never exit. The caller (a build script)
|
|
9
|
+
// owns the exit. Same contract as before, now assertable.
|
|
10
|
+
//
|
|
11
|
+
// FAIL-CLOSED is the whole point. Three ways to fail, all of which must refuse to build:
|
|
12
|
+
// 1. the fence file is missing (unless the caller explicitly allows a no-private fork)
|
|
13
|
+
// 2. the fence file is corrupt (a fence you can't read is not a fence)
|
|
14
|
+
// 3. a private repo's topics file is corrupt (same reasoning, one level down)
|
|
15
|
+
// Fail-OPEN already shipped once (QE-0011 security#1): an L2 article whose repo attribution was
|
|
16
|
+
// unknown defaulted to 'ruvnet' and went out the door. Hence the second, slug-based layer below.
|
|
17
|
+
|
|
18
|
+
import fs from 'node:fs';
|
|
19
|
+
import path from 'node:path';
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Read kb/PRIVATE-STORES.json into a lowercased Set of private repo names.
|
|
23
|
+
* `allowNoFence` is the documented ALLOW_NO_PRIVATE_FENCE=1 escape hatch for a fork with no private
|
|
24
|
+
* repos — the ONLY case where a missing fence is acceptable.
|
|
25
|
+
*/
|
|
26
|
+
export function loadPrivateFence(kbDir, { allowNoFence = false } = {}) {
|
|
27
|
+
const p = path.join(kbDir, 'PRIVATE-STORES.json');
|
|
28
|
+
if (!fs.existsSync(p)) {
|
|
29
|
+
if (allowNoFence) return { ok: true, privateSet: new Set(), reason: 'no-fence-allowed' };
|
|
30
|
+
return { ok: false, privateSet: null, reason: `private fence missing (${p})` };
|
|
31
|
+
}
|
|
32
|
+
try {
|
|
33
|
+
const j = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
34
|
+
if (!Array.isArray(j.privateStores)) throw new Error('no privateStores array');
|
|
35
|
+
return { ok: true, privateSet: new Set(j.privateStores.map((s) => String(s).toLowerCase())), reason: 'ok' };
|
|
36
|
+
} catch (e) {
|
|
37
|
+
return { ok: false, privateSet: null, reason: `PRIVATE-STORES.json unreadable/corrupt (${e.message})` };
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Case-insensitive membership — "Seed", "SEED" and "seed" must all fence. */
|
|
42
|
+
export const isPrivate = (privateSet, repo) => privateSet.has(String(repo).toLowerCase());
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Second layer: collect the SLUGS owned by private repos, by reading each one's l2-topics file.
|
|
46
|
+
* An absent topics file is normal (that repo has no L2 articles) — absence is not corruption.
|
|
47
|
+
* A present-but-corrupt one is corruption, and must abort the build.
|
|
48
|
+
*/
|
|
49
|
+
export function loadPrivateSlugs(kbDir, privateSet) {
|
|
50
|
+
const slugs = new Set();
|
|
51
|
+
for (const repo of privateSet) {
|
|
52
|
+
const tf = path.join(kbDir, `l2-topics.${repo}.json`);
|
|
53
|
+
if (!fs.existsSync(tf)) continue;
|
|
54
|
+
try {
|
|
55
|
+
for (const t of JSON.parse(fs.readFileSync(tf, 'utf8'))) if (t.slug) slugs.add(t.slug);
|
|
56
|
+
} catch (e) {
|
|
57
|
+
return { ok: false, slugs: null, reason: `private topics file ${tf} is corrupt (${e.message})` };
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
return { ok: true, slugs, reason: 'ok' };
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* THE decision. An L2 article is fenced when its repo is private OR its slug belongs to a private
|
|
65
|
+
* repo. The second clause is what catches the shipped bug: `slugRepo.get(slug) || 'ruvnet'` hands
|
|
66
|
+
* an unattributed article a PUBLIC repo name, so repo-based fencing alone would wave it through.
|
|
67
|
+
*/
|
|
68
|
+
export const shouldFenceL2 = ({ repo, slug }, privateSet, privateSlugs) =>
|
|
69
|
+
isPrivate(privateSet, repo) || privateSlugs.has(slug);
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// proactivity-metrics.mjs — ADR-041 detector-layer recall + false-alarm harness.
|
|
3
|
+
//
|
|
4
|
+
// Spawns the REAL scripts/capability-registry.mjs (a fresh child, so HOME=<scratch> redirects
|
|
5
|
+
// os.homedir() at module load — capability-registry.mjs:60) against a ground-truth scratch machine, and
|
|
6
|
+
// scores it against the manifest. It never imports the detector; it runs the real process and reads its
|
|
7
|
+
// real JSON, so a bug anywhere in the probe path is in scope — which is the whole point (ADR-028's shipped
|
|
8
|
+
// "26 hooks off" lies all lived in that probe path).
|
|
9
|
+
//
|
|
10
|
+
// detector-recall = cohort capabilities the detector reports 'off' on the DORMANT machine
|
|
11
|
+
// ------------------------------------------------------------------------
|
|
12
|
+
// cohort capabilities the manifest declares dormant (target >= 0.80)
|
|
13
|
+
// detector-false-alarm = cohort capabilities the detector reports 'off' on the HEALTHY machine
|
|
14
|
+
// (target = 0 — a verified-healthy machine must raise no dormancy flag)
|
|
15
|
+
//
|
|
16
|
+
// The registryPath argument exists so the mutation test can point this at a deliberately-broken copy and
|
|
17
|
+
// watch the numbers move — the falsifiability ADR-041 demands (a harness that cannot fall on a broken
|
|
18
|
+
// detector is not a measurement).
|
|
19
|
+
|
|
20
|
+
import { spawnSync } from 'node:child_process';
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import os from 'node:os';
|
|
23
|
+
import path from 'node:path';
|
|
24
|
+
import { fileURLToPath } from 'node:url';
|
|
25
|
+
import { buildState, readManifest } from '../tests/helpers/ground-truth-machine.mjs';
|
|
26
|
+
|
|
27
|
+
const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
28
|
+
export const REAL_REGISTRY = path.join(REPO, 'scripts', 'capability-registry.mjs');
|
|
29
|
+
|
|
30
|
+
/** Run the real detector against a scratch machine; return { key: state } for every row it emitted. */
|
|
31
|
+
export function runDetector(home, project, registryPath = REAL_REGISTRY) {
|
|
32
|
+
const res = spawnSync(process.execPath, [registryPath, '--json', '--project', project], {
|
|
33
|
+
env: { ...process.env, HOME: home },
|
|
34
|
+
encoding: 'utf8',
|
|
35
|
+
timeout: 60_000,
|
|
36
|
+
});
|
|
37
|
+
if (res.status !== 0 || !res.stdout) {
|
|
38
|
+
throw new Error(`detector exited ${res.status}; stderr: ${String(res.stderr).slice(0, 300)}`);
|
|
39
|
+
}
|
|
40
|
+
const rows = JSON.parse(res.stdout);
|
|
41
|
+
const out = {};
|
|
42
|
+
for (const r of rows) if (r && typeof r.key === 'string') out[r.key] = r.state;
|
|
43
|
+
return out;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Build both ground-truth machines, run the detector against each, and score. Returns the two headline
|
|
48
|
+
* numbers plus the per-capability detail (for a failing assertion to point at). Cleans up its scratch dirs.
|
|
49
|
+
*/
|
|
50
|
+
export function measure({ registryPath = REAL_REGISTRY } = {}) {
|
|
51
|
+
const manifest = readManifest();
|
|
52
|
+
const cohort = manifest.cohort;
|
|
53
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gtm-'));
|
|
54
|
+
try {
|
|
55
|
+
const dormant = buildState('dormant', path.join(root, 'dormant'));
|
|
56
|
+
const healthy = buildState('healthy', path.join(root, 'healthy'));
|
|
57
|
+
|
|
58
|
+
const dormantSeen = runDetector(dormant.home, dormant.project, registryPath);
|
|
59
|
+
const healthySeen = runDetector(healthy.home, healthy.project, registryPath);
|
|
60
|
+
|
|
61
|
+
// Recall: of the cohort caps the manifest says are dormant, how many did the detector get RIGHT?
|
|
62
|
+
//
|
|
63
|
+
// BOTH LINES USED TO TEST `=== 'off'`, AND THAT MADE THIS HARNESS BLIND TO STATE.IDLE — the very
|
|
64
|
+
// state added on 2026-07-24 because "off" was too coarse. Two failures came from the one literal:
|
|
65
|
+
//
|
|
66
|
+
// 1. SILENT EXCLUSION. A capability the manifest declares `idle` was filtered out of the
|
|
67
|
+
// denominator entirely, so adding one to the cohort moved no number and the harness reported
|
|
68
|
+
// an unchanged, healthy-looking 1.00 while measuring strictly less than before. A metric that
|
|
69
|
+
// quietly ignores what it cannot classify is worse than one that fails.
|
|
70
|
+
// 2. THE MUTANT SURVIVED. MEASURED: deleting learning-enable's staleness check — the exact bug
|
|
71
|
+
// fixed hours earlier, where a learner holding 457 trajectories and quiet for 30 days reports
|
|
72
|
+
// ON — left recall at 1.00 and the gate PASSING. The harness could not fall on the precise
|
|
73
|
+
// defect the pillar exists to catch.
|
|
74
|
+
//
|
|
75
|
+
// EXACT MATCH, not merely "some dormant state", is the bar. A detector answering `off` for an idle
|
|
76
|
+
// capability HAS noticed the dormancy, but it is still wrong in the way that costs the user: OFF
|
|
77
|
+
// points at `turnOn`, which is how "turn this on" got printed beside 457 already-learned patterns.
|
|
78
|
+
// Distinguishing them is the entire reason IDLE was introduced, so the metric must require it.
|
|
79
|
+
const DORMANT_STATES = new Set(['off', 'idle']);
|
|
80
|
+
const dormantKeys = cohort.filter((k) => DORMANT_STATES.has(manifest.states.dormant[k]));
|
|
81
|
+
const recalled = dormantKeys.filter((k) => dormantSeen[k] === manifest.states.dormant[k]);
|
|
82
|
+
const recall = dormantKeys.length ? recalled.length / dormantKeys.length : 1;
|
|
83
|
+
|
|
84
|
+
// False alarm: of the cohort, how many did the detector call DORMANT on the HEALTHY machine?
|
|
85
|
+
// `idle` counts here too — telling a user that a working capability has gone quiet is the same
|
|
86
|
+
// broken promise as calling it off, and ADR-028 is explicit that one false alarm costs more trust
|
|
87
|
+
// than ten true ones earn.
|
|
88
|
+
const falseAlarms = cohort.filter((k) => DORMANT_STATES.has(healthySeen[k]));
|
|
89
|
+
|
|
90
|
+
return {
|
|
91
|
+
recall,
|
|
92
|
+
falseAlarmCount: falseAlarms.length,
|
|
93
|
+
cohort,
|
|
94
|
+
dormantSeen: Object.fromEntries(cohort.map((k) => [k, dormantSeen[k]])),
|
|
95
|
+
healthySeen: Object.fromEntries(cohort.map((k) => [k, healthySeen[k]])),
|
|
96
|
+
// DERIVED from `recalled`, never restated. It used to carry its own `!== 'off'` predicate, and
|
|
97
|
+
// when the recall rule above was corrected for STATE.IDLE this line kept the old one — so the
|
|
98
|
+
// CLI printed "PASS — recall 1.00" while the test reading missedDormant on the same run saw a
|
|
99
|
+
// miss. Two independent definitions of one fact will always drift; the reported number and the
|
|
100
|
+
// list explaining it must come from the same comparison or one of them is lying.
|
|
101
|
+
missedDormant: dormantKeys.filter((k) => !recalled.includes(k)),
|
|
102
|
+
falseAlarms,
|
|
103
|
+
};
|
|
104
|
+
} finally {
|
|
105
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]).endsWith('proactivity-metrics.mjs');
|
|
110
|
+
if (invokedDirectly) {
|
|
111
|
+
const m = measure();
|
|
112
|
+
console.log(JSON.stringify({ recall: m.recall, falseAlarmCount: m.falseAlarmCount,
|
|
113
|
+
dormant: m.dormantSeen, healthy: m.healthySeen }, null, 2));
|
|
114
|
+
const ok = m.recall >= 0.80 && m.falseAlarmCount === 0;
|
|
115
|
+
console.log(ok ? `\nPASS — recall ${m.recall.toFixed(2)} (>=0.80), false-alarm ${m.falseAlarmCount} (=0)`
|
|
116
|
+
: `\nFAIL — recall ${m.recall.toFixed(2)}, false-alarm ${m.falseAlarmCount}; missed=${JSON.stringify(m.missedDormant)} falseAlarms=${JSON.stringify(m.falseAlarms)}`);
|
|
117
|
+
process.exit(ok ? 0 : 1);
|
|
118
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[
|
|
2
|
+
{ "set": "tuned", "query": "Can ruflo orchestrate agent swarms?", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
|
|
3
|
+
{ "set": "tuned", "query": "Does RuVector use HNSW for vector search?", "expectRepo": ["ruvector", "concepts"], "minRelevance": -3 },
|
|
4
|
+
{ "set": "tuned", "query": "Can AgentDB run Cypher graph queries over agent memory?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
|
|
5
|
+
{ "set": "tuned", "query": "Can agentic-flow switch between alternative low-cost AI models in Claude Code?", "expectRepo": ["agentic-flow", "concepts"], "minRelevance": -3 },
|
|
6
|
+
{ "set": "tuned", "query": "What phases does the SPARC methodology define?", "expectRepo": ["sparc", "concepts", "ruflo"], "minRelevance": -3 },
|
|
7
|
+
{ "set": "tuned", "query": "Does QuDAG provide quantum-resistant DAG-based anonymous communication?", "expectRepo": ["qudag", "concepts"], "minRelevance": -3 },
|
|
8
|
+
{ "set": "tuned", "query": "Can RuLake cache vector queries in front of a vector store?", "expectRepo": ["rulake", "concepts"], "minRelevance": -3 },
|
|
9
|
+
{ "set": "tuned", "query": "Does SAFLA implement a self-aware feedback loop?", "expectRepo": ["safla", "concepts"], "minRelevance": -3 },
|
|
10
|
+
{ "set": "tuned", "query": "What is agent-harness-generator (metaharness) and what does it scaffold?", "expectRepo": ["metaharness", "concepts"], "minRelevance": -3 },
|
|
11
|
+
|
|
12
|
+
{ "set": "held-out", "query": "How does RuVector store and search vectors on disk?", "expectRepo": ["ruvector", "concepts"], "minRelevance": -3 },
|
|
13
|
+
{ "set": "held-out", "query": "What graph-database features does AgentDB provide?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
|
|
14
|
+
{ "set": "held-out", "query": "Is ruv-FANN a fast neural network library written in Rust?", "expectRepo": ["ruv-fann", "concepts"], "minRelevance": -3 },
|
|
15
|
+
{ "set": "held-out", "query": "What does SynthLang compress or optimize?", "expectRepo": ["synthlang", "concepts"], "minRelevance": -3 },
|
|
16
|
+
{ "set": "held-out", "query": "Can rupixel retrieve documents using visual embeddings?", "expectRepo": ["rupixel", "concepts"], "minRelevance": -3 },
|
|
17
|
+
{ "set": "held-out", "query": "What is agenticow's copy-on-write approach to agent memory?", "expectRepo": ["agenticow", "concepts"], "minRelevance": -3 },
|
|
18
|
+
{ "set": "held-out", "query": "Does CVE-bench benchmark security fixes reproduced from real CVEs?", "expectRepo": ["cve-bench", "concepts"], "minRelevance": -3 },
|
|
19
|
+
{ "set": "held-out", "query": "What is dspy.ts and what does it do?", "expectRepo": ["dspy.ts", "concepts"], "minRelevance": -3 },
|
|
20
|
+
{ "set": "held-out", "query": "Does FACT provide fast-access cached tools for reliable AI integration?", "expectRepo": ["fact", "concepts"], "minRelevance": -3 },
|
|
21
|
+
{ "set": "held-out", "query": "What does the daa project enable for decentralized autonomous applications?", "expectRepo": ["daa", "concepts"], "minRelevance": -3 },
|
|
22
|
+
|
|
23
|
+
{ "set": "anti-drift", "query": "Can Ruflo agents actually create and edit files on disk, or are they limited to chat responses?", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
|
|
24
|
+
{ "set": "anti-drift", "query": "Should I reach for pgvector or Pinecone, or does the RuvNet stack have its own vector database?", "expectRepo": ["ruvector", "rulake", "concepts"], "minRelevance": -3 },
|
|
25
|
+
{ "set": "anti-drift", "query": "Is there a RuvNet tool for cheap model routing so I don't pay full price in Claude Code?", "expectRepo": ["agentic-flow", "concepts"], "minRelevance": -3 },
|
|
26
|
+
{ "set": "anti-drift", "query": "What is the RuvNet way to give an agent persistent structured memory across sessions?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
|
|
27
|
+
{ "set": "anti-drift", "query": "Does the RuvNet ecosystem include a methodology for planning a non-trivial build?", "expectRepo": ["sparc", "concepts", "ruflo"], "minRelevance": -3 },
|
|
28
|
+
{ "set": "anti-drift", "query": "Is there a RuvNet caching layer to make repeated vector queries sub-millisecond?", "expectRepo": ["rulake", "concepts"], "minRelevance": -3 },
|
|
29
|
+
{ "set": "anti-drift", "query": "Does RuvNet have anything that senses people, presence, or falls through walls using WiFi, with no camera?", "expectRepo": ["ruview", "concepts"], "minRelevance": -3 },
|
|
30
|
+
{ "set": "anti-drift", "query": "Does RuvNet have anything for quantum-resistant secure messaging between agents?", "expectRepo": ["qudag", "concepts"], "minRelevance": -3 },
|
|
31
|
+
{ "set": "anti-drift", "query": "Is there a RuvNet neural network library I can run in Rust or WASM?", "expectRepo": ["ruv-fann", "concepts"], "minRelevance": -3 },
|
|
32
|
+
{ "set": "anti-drift", "query": "Which RuvNet tool benchmarks an agent's ability to fix real security vulnerabilities?", "expectRepo": ["cve-bench", "concepts"], "minRelevance": -3 },
|
|
33
|
+
|
|
34
|
+
{ "set": "cross-repo", "query": "If I want both vector search and structured agent memory, which RuvNet tools combine?", "expectRepo": ["ruvector", "agentdb", "rulake", "concepts"], "minRelevance": -3 },
|
|
35
|
+
{ "set": "cross-repo", "query": "How do Ruflo and agentic-flow relate, orchestration versus model routing?", "expectRepo": ["ruflo", "agentic-flow", "concepts"], "minRelevance": -3 },
|
|
36
|
+
{ "set": "cross-repo", "query": "What is the difference between RuVector and RuLake?", "expectRepo": ["ruvector", "rulake", "concepts"], "minRelevance": -3 },
|
|
37
|
+
|
|
38
|
+
{ "set": "implementation", "query": "How does RuVector implement the HNSW navigable small-world graph on disk?", "expectRepo": ["ruvector", "concepts"], "minRelevance": -3 },
|
|
39
|
+
{ "set": "implementation", "query": "How does AgentDB store causal edges between memory nodes?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
|
|
40
|
+
{ "set": "implementation", "query": "How does SPARC structure its five phases with quality gates?", "expectRepo": ["sparc", "concepts", "ruflo"], "minRelevance": -3 },
|
|
41
|
+
{ "set": "implementation", "query": "How does SAFLA close its self-aware feedback learning loop?", "expectRepo": ["safla", "concepts"], "minRelevance": -3 },
|
|
42
|
+
{ "set": "implementation", "query": "How does SynthLang compress prompts to reduce token cost?", "expectRepo": ["synthlang", "concepts"], "minRelevance": -3 },
|
|
43
|
+
|
|
44
|
+
{ "set": "coverage", "query": "What is rupixel and what kind of embeddings does it use?", "expectRepo": ["rupixel", "concepts"], "minRelevance": -3 },
|
|
45
|
+
{ "set": "coverage", "query": "What is agenticow and how does copy-on-write help agent memory?", "expectRepo": ["agenticow", "concepts"], "minRelevance": -3 },
|
|
46
|
+
{ "set": "coverage", "query": "What is daa and what does it enable for decentralized autonomous agents?", "expectRepo": ["daa", "concepts"], "minRelevance": -3 },
|
|
47
|
+
{ "set": "coverage", "query": "What is dspy.ts and how does it bring DSPy-style programming to TypeScript?", "expectRepo": ["dspy.ts", "concepts"], "minRelevance": -3 },
|
|
48
|
+
{ "set": "coverage", "query": "What does FACT actually cache, and why is it about caching rather than retrieval?", "expectRepo": ["fact", "concepts"], "minRelevance": -3 },
|
|
49
|
+
{ "set": "coverage", "query": "What does agent-harness-generator / metaharness mint, and what is Darwin Mode?", "expectRepo": ["metaharness", "concepts"], "minRelevance": -3 },
|
|
50
|
+
|
|
51
|
+
{ "set": "adversarial", "query": "I heard RuvNet can't do real multi-agent swarms, is that true?", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
|
|
52
|
+
{ "set": "adversarial", "query": "Does ruv-FANN actually run, or is it just an idea?", "expectRepo": ["ruv-fann", "concepts"], "minRelevance": -3 },
|
|
53
|
+
{ "set": "adversarial", "query": "Give me the RuvNet building block for orchestrating parallel coding agents.", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
|
|
54
|
+
{ "set": "adversarial", "query": "What is the RuvNet equivalent of LangGraph for agent workflows?", "expectRepo": ["ruflo", "agentic-flow", "concepts"], "minRelevance": -3 },
|
|
55
|
+
{ "set": "adversarial", "query": "Which RuvNet repo would I use to A/B test and score different agent harnesses?", "expectRepo": ["metaharness", "concepts"], "minRelevance": -3 }
|
|
56
|
+
]
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// prove.mjs — run a question battery through the REAL retrieval engine and emit an auditable
|
|
3
|
+
// proof artifact. It calls searchAll() directly — the *exact* function the search_ruvnet MCP tool
|
|
4
|
+
// wraps — so the scores are the same ones a live Claude session sees, just without the JSON-RPC
|
|
5
|
+
// envelope. Models load once for the whole battery.
|
|
6
|
+
//
|
|
7
|
+
// KB_MODEL_CACHE=/path/to/models-cache node scripts/prove.mjs
|
|
8
|
+
// [--questions scripts/proof-questions.json] [--k 3]
|
|
9
|
+
//
|
|
10
|
+
// Writes: PROOF.md (human table + summary) and data/proof-results.json (machine-readable)
|
|
11
|
+
// Exit 0 if every question resolves to an expected repo at/above its threshold; 1 otherwise.
|
|
12
|
+
import path from 'node:path';
|
|
13
|
+
import fs from 'node:fs';
|
|
14
|
+
import { fileURLToPath } from 'node:url';
|
|
15
|
+
import { searchAll } from '../kb/forge-ask-all.mjs';
|
|
16
|
+
|
|
17
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
18
|
+
const KB = path.join(ROOT, 'kb');
|
|
19
|
+
const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
20
|
+
const QFILE = path.resolve(arg('--questions', path.join(ROOT, 'scripts', 'proof-questions.json')));
|
|
21
|
+
const K = parseInt(arg('--k', '3'), 10);
|
|
22
|
+
const OUTBASE = arg('--out', 'PROOF'); // PROOF.md + data/<lower>-results.json
|
|
23
|
+
|
|
24
|
+
const questions = JSON.parse(fs.readFileSync(QFILE, 'utf8'));
|
|
25
|
+
const started = new Date().toISOString();
|
|
26
|
+
console.log(`[prove] ${questions.length} questions · k=${K} · brain=${KB}\n`);
|
|
27
|
+
|
|
28
|
+
const results = [];
|
|
29
|
+
for (let i = 0; i < questions.length; i++) {
|
|
30
|
+
const q = questions[i];
|
|
31
|
+
let ranked = [];
|
|
32
|
+
try { const out = await searchAll({ dir: KB, query: q.query, k: K, pool: 8 }); ranked = out?.results || []; }
|
|
33
|
+
catch (e) { console.log(`ERR #${i + 1} ${e.message}`); }
|
|
34
|
+
const top = ranked[0] || null;
|
|
35
|
+
const got = top?.repo ?? null;
|
|
36
|
+
const eff = (got === 'concepts' && top?.path) ? (top.path.split('/')[0] || got) : got; // primer's home repo
|
|
37
|
+
const score = top?.ceScore ?? null;
|
|
38
|
+
const citedPath = top ? `${top.repo}/${top.path}` : null;
|
|
39
|
+
const exp = q.expectRepo;
|
|
40
|
+
const repoOk = !exp || (Array.isArray(exp) ? (exp.includes(got) || exp.includes(eff)) : (got === exp || eff === exp));
|
|
41
|
+
const relOk = score == null ? true : score >= (q.minRelevance ?? -3);
|
|
42
|
+
const pass = !!got && repoOk && relOk;
|
|
43
|
+
results.push({ n: i + 1, set: q.set || '', query: q.query, expect: exp, got, eff, score, citedPath, pass });
|
|
44
|
+
const s = score == null ? 'n/a' : score.toFixed(3);
|
|
45
|
+
console.log(`${pass ? 'PASS' : 'FAIL'} #${String(i + 1).padStart(2)} ${(eff || '(no hit)').padEnd(24)} @ ${String(s).padStart(7)} ${q.query.slice(0, 64)}`);
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const passed = results.filter((r) => r.pass).length;
|
|
49
|
+
const scored = results.filter((r) => r.score != null).map((r) => r.score).sort((a, b) => a - b);
|
|
50
|
+
const median = scored.length ? scored[Math.floor(scored.length / 2)] : null;
|
|
51
|
+
const bySet = {};
|
|
52
|
+
for (const r of results) { const k = r.set || 'other'; (bySet[k] ??= { pass: 0, total: 0 }); bySet[k].total++; if (r.pass) bySet[k].pass++; }
|
|
53
|
+
|
|
54
|
+
// machine artifact
|
|
55
|
+
fs.mkdirSync(path.join(ROOT, 'data'), { recursive: true });
|
|
56
|
+
fs.writeFileSync(path.join(ROOT, 'data', `${OUTBASE.toLowerCase()}-results.json`),
|
|
57
|
+
JSON.stringify({ started, finished: new Date().toISOString(), k: K, passed, total: results.length, bySet, results }, null, 2));
|
|
58
|
+
|
|
59
|
+
// human artifact
|
|
60
|
+
const esc = (s) => String(s).replace(/\|/g, '\\|');
|
|
61
|
+
const setLines = Object.entries(bySet).map(([k, v]) => `| ${k} | ${v.pass}/${v.total} |`).join('\n');
|
|
62
|
+
const rows = results.map((r) => `| ${r.n} | ${r.set} | ${esc(r.query)} | ${r.eff || '—'} | ${r.score == null ? 'n/a' : r.score.toFixed(3)} | ${esc(r.citedPath || '—')} | ${r.pass ? '✅' : '❌'} |`).join('\n');
|
|
63
|
+
const md = `# RuvNet Brain — Proof of Retrieval (live run)
|
|
64
|
+
|
|
65
|
+
> Generated by \`node scripts/prove.mjs\` on ${started}. Re-run it yourself; this file is the output, not a claim.
|
|
66
|
+
> Each row is a real query through \`searchAll()\` — the exact engine the \`search_ruvnet\` MCP tool wraps.
|
|
67
|
+
|
|
68
|
+
## Headline
|
|
69
|
+
|
|
70
|
+
**${passed}/${results.length} questions resolved to an expected repo** at/above threshold. Median top-hit relevance: **${median == null ? 'n/a' : median.toFixed(3)}**.
|
|
71
|
+
|
|
72
|
+
| Question set | Passed |
|
|
73
|
+
|---|---|
|
|
74
|
+
${setLines}
|
|
75
|
+
|
|
76
|
+
- **tuned / held-out** — the capability-confidence gate (held-out built *after* tuning ⇒ not overfit).
|
|
77
|
+
- **anti-drift** — questions phrased the way a skeptic would, where Claude normally reaches for pgvector/Pinecone or doubts a real tool.
|
|
78
|
+
- **adversarial** — negative-framed ("I heard it *can't*…") to prove the brain doesn't fold.
|
|
79
|
+
- **implementation / coverage / cross-repo** — depth and breadth across all 19 repos.
|
|
80
|
+
|
|
81
|
+
## Every question, every score
|
|
82
|
+
|
|
83
|
+
| # | set | question | resolved repo | relevance | cited source | ok |
|
|
84
|
+
|--:|---|---|---|--:|---|:--:|
|
|
85
|
+
${rows}
|
|
86
|
+
|
|
87
|
+
> "relevance" is the cross-encoder score after name-affinity boost; higher is sharper. Negative is normal for
|
|
88
|
+
> a cross-encoder and still ranks correctly — what matters for the *never-wrongly-doubt* guarantee is that the
|
|
89
|
+
> top hit lands on the right repo with a cited file you can open.
|
|
90
|
+
`;
|
|
91
|
+
fs.writeFileSync(path.join(ROOT, `${OUTBASE}.md`), md);
|
|
92
|
+
|
|
93
|
+
console.log(`\n[prove] ${passed}/${results.length} passed · median ${median == null ? 'n/a' : median.toFixed(3)}`);
|
|
94
|
+
console.log('[prove] wrote PROOF.md + data/proof-results.json');
|
|
95
|
+
process.exit(passed === results.length ? 0 : 1);
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# claude-proxied.sh — launch ONE Claude Code session routed through the proxy.
|
|
3
|
+
#
|
|
4
|
+
# THIS IS THE ISOLATION BOUNDARY OF THE WHOLE TRIAL.
|
|
5
|
+
#
|
|
6
|
+
# ANTHROPIC_BASE_URL is exported for this process only. Nothing is written to
|
|
7
|
+
# ~/.claude/settings.json, so every other Claude Code window on this machine —
|
|
8
|
+
# including the one you are reading this in — keeps talking directly to
|
|
9
|
+
# Anthropic, exactly as before. Close this session and the wiring is gone.
|
|
10
|
+
#
|
|
11
|
+
# What happens inside this session (ADR-313 addendum): setting
|
|
12
|
+
# ANTHROPIC_BASE_URL makes Claude Code stop managing its own Max/Pro OAuth
|
|
13
|
+
# session. That is fine here and ONLY because the proxy defaults to the
|
|
14
|
+
# Passthrough plane, which reads ~/.claude/.credentials.json (read-only) and
|
|
15
|
+
# forwards to the real api.anthropic.com with your actual subscription token.
|
|
16
|
+
# Your subscription is still what pays and still what answers.
|
|
17
|
+
#
|
|
18
|
+
# If the plane were ever anything other than passthrough, this script refuses
|
|
19
|
+
# to launch — a silent flip to a cheap tier while you believe you are on your
|
|
20
|
+
# subscription is the single worst failure mode here, so it is checked, not
|
|
21
|
+
# assumed.
|
|
22
|
+
set -euo pipefail
|
|
23
|
+
|
|
24
|
+
TOKEN_FILE="$HOME/.ruflo/proxy-token"
|
|
25
|
+
|
|
26
|
+
if ! ruflo proxy status --json 2>/dev/null | grep -q '"running":true'; then
|
|
27
|
+
echo "Proxy is not running. Start it first:"
|
|
28
|
+
echo " ./scripts/proxy/proxy-up.sh"
|
|
29
|
+
exit 1
|
|
30
|
+
fi
|
|
31
|
+
|
|
32
|
+
if [ ! -r "$TOKEN_FILE" ]; then
|
|
33
|
+
echo "Missing proxy token at $TOKEN_FILE — reinstall with ./scripts/proxy/proxy-up.sh"
|
|
34
|
+
exit 1
|
|
35
|
+
fi
|
|
36
|
+
|
|
37
|
+
PLANE=$(ruflo proxy config 2>/dev/null | head -1)
|
|
38
|
+
if ! echo "$PLANE" | grep -qi 'passthrough'; then
|
|
39
|
+
echo "REFUSING TO LAUNCH — data plane is not passthrough:"
|
|
40
|
+
echo " $PLANE"
|
|
41
|
+
echo
|
|
42
|
+
echo "This trial only sanctions the passthrough plane (your own subscription)."
|
|
43
|
+
echo "Revert to it with: ruflo proxy config --local-only (or re-read ADR-0026)"
|
|
44
|
+
exit 1
|
|
45
|
+
fi
|
|
46
|
+
|
|
47
|
+
export ANTHROPIC_BASE_URL="http://127.0.0.1:11435"
|
|
48
|
+
export ANTHROPIC_AUTH_TOKEN="$(cat "$TOKEN_FILE")"
|
|
49
|
+
|
|
50
|
+
echo "Launching a proxied Claude Code session."
|
|
51
|
+
echo " ANTHROPIC_BASE_URL = $ANTHROPIC_BASE_URL (this process only)"
|
|
52
|
+
echo " data plane = passthrough (your own Anthropic subscription)"
|
|
53
|
+
echo " other windows = unaffected"
|
|
54
|
+
echo " watch traffic with = ruflo proxy logs -f"
|
|
55
|
+
echo
|
|
56
|
+
|
|
57
|
+
exec claude "$@"
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# proxy-revert.sh — remove the Meta LLM Proxy trial completely.
|
|
3
|
+
#
|
|
4
|
+
# This is the safety net for the whole trial, so it is deliberately boring and
|
|
5
|
+
# total: stop the process, uninstall via rUv's own lifecycle command, then
|
|
6
|
+
# PROVE nothing is left rather than claiming it.
|
|
7
|
+
#
|
|
8
|
+
# It does not touch ~/.claude/settings.json — the trial never wrote there (see
|
|
9
|
+
# proxy-verify.mjs check 3), so there is nothing to undo.
|
|
10
|
+
set -uo pipefail
|
|
11
|
+
|
|
12
|
+
echo "Reverting the Meta LLM Proxy trial..."
|
|
13
|
+
echo
|
|
14
|
+
|
|
15
|
+
echo "--- stopping (if running) ---"
|
|
16
|
+
ruflo proxy stop 2>&1 | sed 's/^/ /' || true
|
|
17
|
+
echo
|
|
18
|
+
|
|
19
|
+
echo "--- uninstalling binary, token, consent receipt (ruflo proxy uninstall) ---"
|
|
20
|
+
ruflo proxy uninstall 2>&1 | sed 's/^/ /' || true
|
|
21
|
+
echo
|
|
22
|
+
|
|
23
|
+
echo "--- PROOF it is gone (derived, not asserted) ---"
|
|
24
|
+
FAIL=0
|
|
25
|
+
|
|
26
|
+
STATUS=$(ruflo proxy status --json 2>/dev/null || echo '{}')
|
|
27
|
+
echo " ruflo proxy status: $STATUS"
|
|
28
|
+
case "$STATUS" in
|
|
29
|
+
*'"installed":true'*) echo " ^ still installed" ; FAIL=1 ;;
|
|
30
|
+
esac
|
|
31
|
+
case "$STATUS" in
|
|
32
|
+
*'"running":true'*) echo " ^ still running" ; FAIL=1 ;;
|
|
33
|
+
esac
|
|
34
|
+
|
|
35
|
+
if lsof -iTCP:11435 -sTCP:LISTEN -n -P >/dev/null 2>&1; then
|
|
36
|
+
echo " port 11435: STILL LISTENING <-- revert incomplete"; FAIL=1
|
|
37
|
+
else
|
|
38
|
+
echo " port 11435: clear"
|
|
39
|
+
fi
|
|
40
|
+
|
|
41
|
+
if [ -f "$HOME/.ruflo/proxy-token" ]; then
|
|
42
|
+
echo " proxy-token: STILL PRESENT <-- revert incomplete"; FAIL=1
|
|
43
|
+
else
|
|
44
|
+
echo " proxy-token: removed"
|
|
45
|
+
fi
|
|
46
|
+
|
|
47
|
+
if grep -q 'ANTHROPIC_BASE_URL' "$HOME/.claude/settings.json" 2>/dev/null; then
|
|
48
|
+
echo " ~/.claude/settings.json: contains ANTHROPIC_BASE_URL <-- unexpected"; FAIL=1
|
|
49
|
+
else
|
|
50
|
+
echo " ~/.claude/settings.json: clean (never modified by this trial)"
|
|
51
|
+
fi
|
|
52
|
+
|
|
53
|
+
echo
|
|
54
|
+
if [ "$FAIL" = 0 ]; then
|
|
55
|
+
echo "Reverted. The machine is back to its pre-trial state."
|
|
56
|
+
exit 0
|
|
57
|
+
fi
|
|
58
|
+
echo "REVERT INCOMPLETE — see the lines marked above."
|
|
59
|
+
exit 1
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# proxy-up.sh — install + start the Meta LLM Proxy, then verify it.
|
|
3
|
+
#
|
|
4
|
+
# Thin orchestration over rUv's own lifecycle commands (ADR-307):
|
|
5
|
+
# ruflo proxy install / start / status / doctor
|
|
6
|
+
# It adds no logic of its own except the macOS workaround documented below.
|
|
7
|
+
set -uo pipefail
|
|
8
|
+
|
|
9
|
+
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
|
|
10
|
+
|
|
11
|
+
# --- macOS workaround: upstream bug in @claude-flow/security's PathValidator --
|
|
12
|
+
#
|
|
13
|
+
# `ruflo proxy install` extracts into a tmpdir and then validates that the
|
|
14
|
+
# extracted binary really lives inside that tmpdir (defense against a symlink
|
|
15
|
+
# swap). The validator canonicalizes the CANDIDATE path with fs.realpath but
|
|
16
|
+
# resolves the ALLOWED PREFIX with path.resolve only:
|
|
17
|
+
#
|
|
18
|
+
# path-validator.ts:177-178 prefixes -> path.resolve(p) (no realpath)
|
|
19
|
+
# path-validator.ts:229,234 candidate -> path.resolve + realpath
|
|
20
|
+
# path-validator.ts:262 resolvedPath.startsWith(prefix + sep)
|
|
21
|
+
#
|
|
22
|
+
# On macOS os.tmpdir() is /var/folders/... which is a symlink to
|
|
23
|
+
# /private/var/folders/... So the candidate becomes /private/var/... while the
|
|
24
|
+
# prefix stays /var/... and the startsWith check can NEVER pass. Install fails
|
|
25
|
+
# with "extracted binary path failed validation: Path is outside allowed
|
|
26
|
+
# directories" for every macOS user.
|
|
27
|
+
#
|
|
28
|
+
# Handing the installer an ALREADY-CANONICAL TMPDIR makes the validator's own
|
|
29
|
+
# assumption true. This changes nothing about what is downloaded or verified —
|
|
30
|
+
# the Ed25519 signature and sha256 checks run exactly as before.
|
|
31
|
+
#
|
|
32
|
+
# Upstream fix would be one line: realpath the prefixes too.
|
|
33
|
+
if [ "$(uname -s)" = "Darwin" ]; then
|
|
34
|
+
CANONICAL_TMP="$(node -e 'console.log(require("fs").realpathSync(require("os").tmpdir()))' 2>/dev/null || echo "")"
|
|
35
|
+
if [ -n "$CANONICAL_TMP" ]; then
|
|
36
|
+
export TMPDIR="$CANONICAL_TMP"
|
|
37
|
+
echo "macOS: TMPDIR canonicalized to $TMPDIR (upstream PathValidator symlink bug)"
|
|
38
|
+
echo
|
|
39
|
+
fi
|
|
40
|
+
fi
|
|
41
|
+
|
|
42
|
+
echo "--- install (idempotent; verifies Ed25519 signature + sha256) ---"
|
|
43
|
+
if ruflo proxy status --json 2>/dev/null | grep -q '"installed":true'; then
|
|
44
|
+
echo " already installed"
|
|
45
|
+
else
|
|
46
|
+
ruflo proxy install --yes 2>&1 | tail -4 | sed 's/^/ /'
|
|
47
|
+
fi
|
|
48
|
+
echo
|
|
49
|
+
|
|
50
|
+
echo "--- start (detached; survives terminal close, NOT a reboot) ---"
|
|
51
|
+
if ruflo proxy status --json 2>/dev/null | grep -q '"running":true'; then
|
|
52
|
+
echo " already running"
|
|
53
|
+
else
|
|
54
|
+
ruflo proxy start --service 2>&1 | tail -3 | sed 's/^/ /'
|
|
55
|
+
sleep 2
|
|
56
|
+
fi
|
|
57
|
+
echo
|
|
58
|
+
|
|
59
|
+
echo "--- verify ---"
|
|
60
|
+
node "$REPO_ROOT/scripts/proxy/proxy-verify.mjs"
|