ruvnet-brain 4.0.1 → 4.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/.claude-plugin/marketplace.json +1 -0
  2. package/README.md +4 -4
  3. package/bin/install.mjs +303 -24
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  25. package/package.json +8 -22
  26. package/plugin/.claude-plugin/marketplace.json +1 -0
  27. package/plugin/.claude-plugin/plugin.json +2 -3
  28. package/plugin/.codex-plugin/plugin.json +1 -1
  29. package/plugin/commands/brain-console.md +2 -2
  30. package/plugin/commands/configure.md +3 -2
  31. package/plugin/commands/rvbc.md +4 -3
  32. package/plugin/commands/rvcb.md +2 -2
  33. package/plugin/commands/whats-new.md +6 -6
  34. package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
  35. package/plugin/hooks/hooks.json +1 -2
  36. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  37. package/plugin/mcp/server.mjs +90 -32
  38. package/plugin/scripts/detach.mjs +14 -0
  39. package/plugin/scripts/first-session-worker.mjs +38 -0
  40. package/plugin/scripts/ground-ruvnet.sh +16 -6
  41. package/plugin/scripts/hook-shim.mjs +34 -29
  42. package/plugin/scripts/learn-capture.sh +22 -3
  43. package/plugin/scripts/learn-flush.mjs +21 -4
  44. package/plugin/scripts/runtime-preferences.mjs +269 -0
  45. package/plugin/scripts/session-start-core.mjs +503 -0
  46. package/plugin/scripts/session-start.sh +3 -858
  47. package/plugin/scripts/whats-new.mjs +42 -0
  48. package/plugin/skills/brain-console/SKILL.md +4 -2
  49. package/plugin/skills/release-proof/SKILL.md +98 -0
  50. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  51. package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
  52. package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
  53. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
  54. package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
  55. package/plugin/skills/rvbc/SKILL.md +9 -6
  56. package/plugin/skills/whats-new/SKILL.md +4 -4
  57. package/scripts/adr-backfill.mjs +107 -0
  58. package/scripts/advocacy-outcomes.mjs +808 -0
  59. package/scripts/agentdb-context.mjs +216 -0
  60. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  61. package/scripts/ascii-drift.mjs +236 -0
  62. package/scripts/behavioral-l1-l4.mjs +210 -0
  63. package/scripts/brain-capability-check.mjs +72 -0
  64. package/scripts/brain-grade-groundtruth.mjs +100 -0
  65. package/scripts/brain-latency-50.mjs +227 -0
  66. package/scripts/brain-novice-50.mjs +189 -0
  67. package/scripts/brain-stamp.mjs +94 -0
  68. package/scripts/brain-state.mjs +212 -0
  69. package/scripts/build-bundle.mjs +531 -0
  70. package/scripts/build-concepts.mjs +132 -0
  71. package/scripts/build-l2.mjs +71 -0
  72. package/scripts/build-primer.mjs +73 -0
  73. package/scripts/build-symbols.mjs +68 -0
  74. package/scripts/calibrate-router.mjs +97 -0
  75. package/scripts/capability-audit.mjs +321 -0
  76. package/scripts/capability-registry.mjs +876 -0
  77. package/scripts/check-indexation.mjs +108 -0
  78. package/scripts/check-legibility.mjs +189 -0
  79. package/scripts/ci/build-fixture-kb.mjs +67 -0
  80. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  81. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  82. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  83. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  84. package/scripts/ci/stranger-scenario.mjs +228 -0
  85. package/scripts/ci/stranger-timeout.mjs +25 -0
  86. package/scripts/ci-verdict.mjs +29 -0
  87. package/scripts/claims-verify.mjs +710 -0
  88. package/scripts/clear-claude-tmp.sh +31 -0
  89. package/scripts/console-engine.mjs +434 -0
  90. package/scripts/console-engine.test.mjs +125 -0
  91. package/scripts/corpus-qa.mjs +250 -0
  92. package/scripts/correction-detect-embed.mjs +346 -0
  93. package/scripts/correction-detect-measure.mjs +270 -0
  94. package/scripts/correction-detect.mjs +686 -0
  95. package/scripts/count-chunks.mjs +54 -0
  96. package/scripts/described-questions.json +30 -0
  97. package/scripts/design-grade.mjs +58 -0
  98. package/scripts/dev-plugin-link.sh +105 -0
  99. package/scripts/distill-project.mjs +200 -0
  100. package/scripts/doc-currency.mjs +801 -0
  101. package/scripts/eval-brain.mjs +244 -0
  102. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  103. package/scripts/fix-workstream.mjs +291 -0
  104. package/scripts/full-hints.mjs +87 -0
  105. package/scripts/gate.sh +39 -0
  106. package/scripts/gates.mjs +146 -0
  107. package/scripts/gen-console-images.mjs +54 -0
  108. package/scripts/gen-images.mjs +47 -0
  109. package/scripts/git-clone-refresh.mjs +52 -0
  110. package/scripts/git-hooks/pre-push +126 -0
  111. package/scripts/goal-match.mjs +398 -0
  112. package/scripts/goldie-research.mjs +223 -0
  113. package/scripts/goldie-weekly.sh +67 -0
  114. package/scripts/health-repair.mjs +237 -0
  115. package/scripts/helix-scenario-questions.json +10 -0
  116. package/scripts/ingest-gists.mjs +230 -0
  117. package/scripts/ingest-meeting.mjs +115 -0
  118. package/scripts/ingest-repo.mjs +79 -0
  119. package/scripts/install-npx-witness.sh +49 -0
  120. package/scripts/issue-fix.mjs +558 -0
  121. package/scripts/issue-watch.mjs +276 -0
  122. package/scripts/issue4-close-note.md +31 -0
  123. package/scripts/key-canary.mjs +91 -0
  124. package/scripts/latency-to-surface.mjs +233 -0
  125. package/scripts/learning-enable.mjs +380 -0
  126. package/scripts/learning-replay.mjs +1570 -0
  127. package/scripts/learnings.mjs +62 -0
  128. package/scripts/lesson-gate.mjs +680 -0
  129. package/scripts/lesson-lifecycle.mjs +449 -0
  130. package/scripts/lesson-promote.mjs +262 -0
  131. package/scripts/lesson-ratify.mjs +98 -0
  132. package/scripts/lesson-seed.mjs +252 -0
  133. package/scripts/lesson-store.mjs +447 -0
  134. package/scripts/loop-checkpoint.mjs +86 -0
  135. package/scripts/memdb-health.sh +14 -0
  136. package/scripts/memory-doctor.mjs +326 -0
  137. package/scripts/model-catalog.mjs +79 -0
  138. package/scripts/nightly-controller.mjs +66 -0
  139. package/scripts/nightly-gists.sh +72 -0
  140. package/scripts/nightly-wrapper.sh +172 -0
  141. package/scripts/notify.sh +12 -0
  142. package/scripts/npx-witness.sh +56 -0
  143. package/scripts/onboarding-console.mjs +2922 -0
  144. package/scripts/private-fence.mjs +69 -0
  145. package/scripts/proactivity-metrics.mjs +118 -0
  146. package/scripts/proof-questions.json +56 -0
  147. package/scripts/protected-release-invocation.mjs +76 -0
  148. package/scripts/prove.mjs +95 -0
  149. package/scripts/proxy/claude-proxied.sh +57 -0
  150. package/scripts/proxy/proxy-revert.sh +59 -0
  151. package/scripts/proxy/proxy-up.sh +60 -0
  152. package/scripts/proxy/proxy-verify.mjs +142 -0
  153. package/scripts/publication-receipt.mjs +307 -0
  154. package/scripts/published-surface-probe.mjs +241 -0
  155. package/scripts/qe/card-lane-gate.mjs +162 -0
  156. package/scripts/qe/session-start-gate.mjs +229 -0
  157. package/scripts/qe/ux-suite.mjs +323 -0
  158. package/scripts/reconcile-project.mjs +0 -0
  159. package/scripts/record-lesson.mjs +113 -0
  160. package/scripts/refresh-model-catalog.mjs +99 -0
  161. package/scripts/release-authority.mjs +93 -0
  162. package/scripts/release-proof.mjs +9 -0
  163. package/scripts/release-vector.mjs +281 -0
  164. package/scripts/release.mjs +439 -0
  165. package/scripts/remedy-registry.mjs +247 -0
  166. package/scripts/rerank-cap-eval.mjs +265 -0
  167. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  168. package/scripts/route-cheap.mjs +20 -15
  169. package/scripts/router-utilization.mjs +182 -0
  170. package/scripts/routing-flywheel.mjs +596 -0
  171. package/scripts/rvf-generation.mjs +104 -0
  172. package/scripts/rvf-index-audit.mjs +138 -0
  173. package/scripts/self-update.mjs +296 -0
  174. package/scripts/selfcheck.mjs +7 -1
  175. package/scripts/sign-bundle.mjs +69 -0
  176. package/scripts/signal-watch.mjs +171 -0
  177. package/scripts/stabilization-receipt.mjs +108 -0
  178. package/scripts/stack-sync.mjs +469 -0
  179. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  180. package/scripts/stamp-sweep.mjs +144 -0
  181. package/scripts/status-honesty.mjs +102 -0
  182. package/scripts/sync-version.mjs +217 -0
  183. package/scripts/token-report.mjs +102 -0
  184. package/scripts/top100-benchmark.mjs +479 -0
  185. package/scripts/top100-corpus.mjs +112 -0
  186. package/scripts/top100-semantic-assertions.mjs +449 -0
  187. package/scripts/update-apply.mjs +9 -0
  188. package/scripts/upgrade-notice.mjs +14 -0
  189. package/scripts/verify-bundle.mjs +51 -0
  190. package/scripts/verify-channels.mjs +184 -0
  191. package/scripts/verify-model-catalog.mjs +104 -0
  192. package/scripts/verify-nightly-close-issue4.sh +31 -0
  193. package/scripts/version.mjs +40 -0
  194. package/scripts/wired-check.mjs +867 -0
  195. package/plugin/scripts/finalize-token-meter.mjs +0 -25
@@ -0,0 +1,69 @@
1
+ // private-fence.mjs — the fence that keeps private repos out of anything we ship (SEC-0010 #5).
2
+ //
3
+ // This logic used to live inline in build-concepts.mjs, executed at import time, and it called
4
+ // process.exit(1) on failure. That made it literally untestable: importing the module to test the
5
+ // fence would kill the test runner. So the highest-severity code in the repo — the thing standing
6
+ // between a private cognitum store and a public 512MB release — had zero tests.
7
+ //
8
+ // Here the decisions are pure: they RETURN {ok, reason} and never exit. The caller (a build script)
9
+ // owns the exit. Same contract as before, now assertable.
10
+ //
11
+ // FAIL-CLOSED is the whole point. Three ways to fail, all of which must refuse to build:
12
+ // 1. the fence file is missing (unless the caller explicitly allows a no-private fork)
13
+ // 2. the fence file is corrupt (a fence you can't read is not a fence)
14
+ // 3. a private repo's topics file is corrupt (same reasoning, one level down)
15
+ // Fail-OPEN already shipped once (QE-0011 security#1): an L2 article whose repo attribution was
16
+ // unknown defaulted to 'ruvnet' and went out the door. Hence the second, slug-based layer below.
17
+
18
+ import fs from 'node:fs';
19
+ import path from 'node:path';
20
+
21
+ /**
22
+ * Read kb/PRIVATE-STORES.json into a lowercased Set of private repo names.
23
+ * `allowNoFence` is the documented ALLOW_NO_PRIVATE_FENCE=1 escape hatch for a fork with no private
24
+ * repos — the ONLY case where a missing fence is acceptable.
25
+ */
26
+ export function loadPrivateFence(kbDir, { allowNoFence = false } = {}) {
27
+ const p = path.join(kbDir, 'PRIVATE-STORES.json');
28
+ if (!fs.existsSync(p)) {
29
+ if (allowNoFence) return { ok: true, privateSet: new Set(), reason: 'no-fence-allowed' };
30
+ return { ok: false, privateSet: null, reason: `private fence missing (${p})` };
31
+ }
32
+ try {
33
+ const j = JSON.parse(fs.readFileSync(p, 'utf8'));
34
+ if (!Array.isArray(j.privateStores)) throw new Error('no privateStores array');
35
+ return { ok: true, privateSet: new Set(j.privateStores.map((s) => String(s).toLowerCase())), reason: 'ok' };
36
+ } catch (e) {
37
+ return { ok: false, privateSet: null, reason: `PRIVATE-STORES.json unreadable/corrupt (${e.message})` };
38
+ }
39
+ }
40
+
41
+ /** Case-insensitive membership — "Seed", "SEED" and "seed" must all fence. */
42
+ export const isPrivate = (privateSet, repo) => privateSet.has(String(repo).toLowerCase());
43
+
44
+ /**
45
+ * Second layer: collect the SLUGS owned by private repos, by reading each one's l2-topics file.
46
+ * An absent topics file is normal (that repo has no L2 articles) — absence is not corruption.
47
+ * A present-but-corrupt one is corruption, and must abort the build.
48
+ */
49
+ export function loadPrivateSlugs(kbDir, privateSet) {
50
+ const slugs = new Set();
51
+ for (const repo of privateSet) {
52
+ const tf = path.join(kbDir, `l2-topics.${repo}.json`);
53
+ if (!fs.existsSync(tf)) continue;
54
+ try {
55
+ for (const t of JSON.parse(fs.readFileSync(tf, 'utf8'))) if (t.slug) slugs.add(t.slug);
56
+ } catch (e) {
57
+ return { ok: false, slugs: null, reason: `private topics file ${tf} is corrupt (${e.message})` };
58
+ }
59
+ }
60
+ return { ok: true, slugs, reason: 'ok' };
61
+ }
62
+
63
+ /**
64
+ * THE decision. An L2 article is fenced when its repo is private OR its slug belongs to a private
65
+ * repo. The second clause is what catches the shipped bug: `slugRepo.get(slug) || 'ruvnet'` hands
66
+ * an unattributed article a PUBLIC repo name, so repo-based fencing alone would wave it through.
67
+ */
68
+ export const shouldFenceL2 = ({ repo, slug }, privateSet, privateSlugs) =>
69
+ isPrivate(privateSet, repo) || privateSlugs.has(slug);
@@ -0,0 +1,118 @@
1
+ #!/usr/bin/env node
2
+ // proactivity-metrics.mjs — ADR-041 detector-layer recall + false-alarm harness.
3
+ //
4
+ // Spawns the REAL scripts/capability-registry.mjs (a fresh child, so HOME=<scratch> redirects
5
+ // os.homedir() at module load — capability-registry.mjs:60) against a ground-truth scratch machine, and
6
+ // scores it against the manifest. It never imports the detector; it runs the real process and reads its
7
+ // real JSON, so a bug anywhere in the probe path is in scope — which is the whole point (ADR-028's shipped
8
+ // "26 hooks off" lies all lived in that probe path).
9
+ //
10
+ // detector-recall = cohort capabilities the detector reports 'off' on the DORMANT machine
11
+ // ------------------------------------------------------------------------
12
+ // cohort capabilities the manifest declares dormant (target >= 0.80)
13
+ // detector-false-alarm = cohort capabilities the detector reports 'off' on the HEALTHY machine
14
+ // (target = 0 — a verified-healthy machine must raise no dormancy flag)
15
+ //
16
+ // The registryPath argument exists so the mutation test can point this at a deliberately-broken copy and
17
+ // watch the numbers move — the falsifiability ADR-041 demands (a harness that cannot fall on a broken
18
+ // detector is not a measurement).
19
+
20
+ import { spawnSync } from 'node:child_process';
21
+ import fs from 'node:fs';
22
+ import os from 'node:os';
23
+ import path from 'node:path';
24
+ import { fileURLToPath } from 'node:url';
25
+ import { buildState, readManifest } from '../tests/helpers/ground-truth-machine.mjs';
26
+
27
+ const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
28
+ export const REAL_REGISTRY = path.join(REPO, 'scripts', 'capability-registry.mjs');
29
+
30
+ /** Run the real detector against a scratch machine; return { key: state } for every row it emitted. */
31
+ export function runDetector(home, project, registryPath = REAL_REGISTRY) {
32
+ const res = spawnSync(process.execPath, [registryPath, '--json', '--project', project], {
33
+ env: { ...process.env, HOME: home },
34
+ encoding: 'utf8',
35
+ timeout: 60_000,
36
+ });
37
+ if (res.status !== 0 || !res.stdout) {
38
+ throw new Error(`detector exited ${res.status}; stderr: ${String(res.stderr).slice(0, 300)}`);
39
+ }
40
+ const rows = JSON.parse(res.stdout);
41
+ const out = {};
42
+ for (const r of rows) if (r && typeof r.key === 'string') out[r.key] = r.state;
43
+ return out;
44
+ }
45
+
46
+ /**
47
+ * Build both ground-truth machines, run the detector against each, and score. Returns the two headline
48
+ * numbers plus the per-capability detail (for a failing assertion to point at). Cleans up its scratch dirs.
49
+ */
50
+ export function measure({ registryPath = REAL_REGISTRY } = {}) {
51
+ const manifest = readManifest();
52
+ const cohort = manifest.cohort;
53
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gtm-'));
54
+ try {
55
+ const dormant = buildState('dormant', path.join(root, 'dormant'));
56
+ const healthy = buildState('healthy', path.join(root, 'healthy'));
57
+
58
+ const dormantSeen = runDetector(dormant.home, dormant.project, registryPath);
59
+ const healthySeen = runDetector(healthy.home, healthy.project, registryPath);
60
+
61
+ // Recall: of the cohort caps the manifest says are dormant, how many did the detector get RIGHT?
62
+ //
63
+ // BOTH LINES USED TO TEST `=== 'off'`, AND THAT MADE THIS HARNESS BLIND TO STATE.IDLE — the very
64
+ // state added on 2026-07-24 because "off" was too coarse. Two failures came from the one literal:
65
+ //
66
+ // 1. SILENT EXCLUSION. A capability the manifest declares `idle` was filtered out of the
67
+ // denominator entirely, so adding one to the cohort moved no number and the harness reported
68
+ // an unchanged, healthy-looking 1.00 while measuring strictly less than before. A metric that
69
+ // quietly ignores what it cannot classify is worse than one that fails.
70
+ // 2. THE MUTANT SURVIVED. MEASURED: deleting learning-enable's staleness check — the exact bug
71
+ // fixed hours earlier, where a learner holding 457 trajectories and quiet for 30 days reports
72
+ // ON — left recall at 1.00 and the gate PASSING. The harness could not fall on the precise
73
+ // defect the pillar exists to catch.
74
+ //
75
+ // EXACT MATCH, not merely "some dormant state", is the bar. A detector answering `off` for an idle
76
+ // capability HAS noticed the dormancy, but it is still wrong in the way that costs the user: OFF
77
+ // points at `turnOn`, which is how "turn this on" got printed beside 457 already-learned patterns.
78
+ // Distinguishing them is the entire reason IDLE was introduced, so the metric must require it.
79
+ const DORMANT_STATES = new Set(['off', 'idle']);
80
+ const dormantKeys = cohort.filter((k) => DORMANT_STATES.has(manifest.states.dormant[k]));
81
+ const recalled = dormantKeys.filter((k) => dormantSeen[k] === manifest.states.dormant[k]);
82
+ const recall = dormantKeys.length ? recalled.length / dormantKeys.length : 1;
83
+
84
+ // False alarm: of the cohort, how many did the detector call DORMANT on the HEALTHY machine?
85
+ // `idle` counts here too — telling a user that a working capability has gone quiet is the same
86
+ // broken promise as calling it off, and ADR-028 is explicit that one false alarm costs more trust
87
+ // than ten true ones earn.
88
+ const falseAlarms = cohort.filter((k) => DORMANT_STATES.has(healthySeen[k]));
89
+
90
+ return {
91
+ recall,
92
+ falseAlarmCount: falseAlarms.length,
93
+ cohort,
94
+ dormantSeen: Object.fromEntries(cohort.map((k) => [k, dormantSeen[k]])),
95
+ healthySeen: Object.fromEntries(cohort.map((k) => [k, healthySeen[k]])),
96
+ // DERIVED from `recalled`, never restated. It used to carry its own `!== 'off'` predicate, and
97
+ // when the recall rule above was corrected for STATE.IDLE this line kept the old one — so the
98
+ // CLI printed "PASS — recall 1.00" while the test reading missedDormant on the same run saw a
99
+ // miss. Two independent definitions of one fact will always drift; the reported number and the
100
+ // list explaining it must come from the same comparison or one of them is lying.
101
+ missedDormant: dormantKeys.filter((k) => !recalled.includes(k)),
102
+ falseAlarms,
103
+ };
104
+ } finally {
105
+ fs.rmSync(root, { recursive: true, force: true });
106
+ }
107
+ }
108
+
109
+ const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]).endsWith('proactivity-metrics.mjs');
110
+ if (invokedDirectly) {
111
+ const m = measure();
112
+ console.log(JSON.stringify({ recall: m.recall, falseAlarmCount: m.falseAlarmCount,
113
+ dormant: m.dormantSeen, healthy: m.healthySeen }, null, 2));
114
+ const ok = m.recall >= 0.80 && m.falseAlarmCount === 0;
115
+ console.log(ok ? `\nPASS — recall ${m.recall.toFixed(2)} (>=0.80), false-alarm ${m.falseAlarmCount} (=0)`
116
+ : `\nFAIL — recall ${m.recall.toFixed(2)}, false-alarm ${m.falseAlarmCount}; missed=${JSON.stringify(m.missedDormant)} falseAlarms=${JSON.stringify(m.falseAlarms)}`);
117
+ process.exit(ok ? 0 : 1);
118
+ }
@@ -0,0 +1,56 @@
1
+ [
2
+ { "set": "tuned", "query": "Can ruflo orchestrate agent swarms?", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
3
+ { "set": "tuned", "query": "Does RuVector use HNSW for vector search?", "expectRepo": ["ruvector", "concepts"], "minRelevance": -3 },
4
+ { "set": "tuned", "query": "Can AgentDB run Cypher graph queries over agent memory?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
5
+ { "set": "tuned", "query": "Can agentic-flow switch between alternative low-cost AI models in Claude Code?", "expectRepo": ["agentic-flow", "concepts"], "minRelevance": -3 },
6
+ { "set": "tuned", "query": "What phases does the SPARC methodology define?", "expectRepo": ["sparc", "concepts", "ruflo"], "minRelevance": -3 },
7
+ { "set": "tuned", "query": "Does QuDAG provide quantum-resistant DAG-based anonymous communication?", "expectRepo": ["qudag", "concepts"], "minRelevance": -3 },
8
+ { "set": "tuned", "query": "Can RuLake cache vector queries in front of a vector store?", "expectRepo": ["rulake", "concepts"], "minRelevance": -3 },
9
+ { "set": "tuned", "query": "Does SAFLA implement a self-aware feedback loop?", "expectRepo": ["safla", "concepts"], "minRelevance": -3 },
10
+ { "set": "tuned", "query": "What is agent-harness-generator (metaharness) and what does it scaffold?", "expectRepo": ["metaharness", "concepts"], "minRelevance": -3 },
11
+
12
+ { "set": "held-out", "query": "How does RuVector store and search vectors on disk?", "expectRepo": ["ruvector", "concepts"], "minRelevance": -3 },
13
+ { "set": "held-out", "query": "What graph-database features does AgentDB provide?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
14
+ { "set": "held-out", "query": "Is ruv-FANN a fast neural network library written in Rust?", "expectRepo": ["ruv-fann", "concepts"], "minRelevance": -3 },
15
+ { "set": "held-out", "query": "What does SynthLang compress or optimize?", "expectRepo": ["synthlang", "concepts"], "minRelevance": -3 },
16
+ { "set": "held-out", "query": "Can rupixel retrieve documents using visual embeddings?", "expectRepo": ["rupixel", "concepts"], "minRelevance": -3 },
17
+ { "set": "held-out", "query": "What is agenticow's copy-on-write approach to agent memory?", "expectRepo": ["agenticow", "concepts"], "minRelevance": -3 },
18
+ { "set": "held-out", "query": "Does CVE-bench benchmark security fixes reproduced from real CVEs?", "expectRepo": ["cve-bench", "concepts"], "minRelevance": -3 },
19
+ { "set": "held-out", "query": "What is dspy.ts and what does it do?", "expectRepo": ["dspy.ts", "concepts"], "minRelevance": -3 },
20
+ { "set": "held-out", "query": "Does FACT provide fast-access cached tools for reliable AI integration?", "expectRepo": ["fact", "concepts"], "minRelevance": -3 },
21
+ { "set": "held-out", "query": "What does the daa project enable for decentralized autonomous applications?", "expectRepo": ["daa", "concepts"], "minRelevance": -3 },
22
+
23
+ { "set": "anti-drift", "query": "Can Ruflo agents actually create and edit files on disk, or are they limited to chat responses?", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
24
+ { "set": "anti-drift", "query": "Should I reach for pgvector or Pinecone, or does the RuvNet stack have its own vector database?", "expectRepo": ["ruvector", "rulake", "concepts"], "minRelevance": -3 },
25
+ { "set": "anti-drift", "query": "Is there a RuvNet tool for cheap model routing so I don't pay full price in Claude Code?", "expectRepo": ["agentic-flow", "concepts"], "minRelevance": -3 },
26
+ { "set": "anti-drift", "query": "What is the RuvNet way to give an agent persistent structured memory across sessions?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
27
+ { "set": "anti-drift", "query": "Does the RuvNet ecosystem include a methodology for planning a non-trivial build?", "expectRepo": ["sparc", "concepts", "ruflo"], "minRelevance": -3 },
28
+ { "set": "anti-drift", "query": "Is there a RuvNet caching layer to make repeated vector queries sub-millisecond?", "expectRepo": ["rulake", "concepts"], "minRelevance": -3 },
29
+ { "set": "anti-drift", "query": "Does RuvNet have anything that senses people, presence, or falls through walls using WiFi, with no camera?", "expectRepo": ["ruview", "concepts"], "minRelevance": -3 },
30
+ { "set": "anti-drift", "query": "Does RuvNet have anything for quantum-resistant secure messaging between agents?", "expectRepo": ["qudag", "concepts"], "minRelevance": -3 },
31
+ { "set": "anti-drift", "query": "Is there a RuvNet neural network library I can run in Rust or WASM?", "expectRepo": ["ruv-fann", "concepts"], "minRelevance": -3 },
32
+ { "set": "anti-drift", "query": "Which RuvNet tool benchmarks an agent's ability to fix real security vulnerabilities?", "expectRepo": ["cve-bench", "concepts"], "minRelevance": -3 },
33
+
34
+ { "set": "cross-repo", "query": "If I want both vector search and structured agent memory, which RuvNet tools combine?", "expectRepo": ["ruvector", "agentdb", "rulake", "concepts"], "minRelevance": -3 },
35
+ { "set": "cross-repo", "query": "How do Ruflo and agentic-flow relate, orchestration versus model routing?", "expectRepo": ["ruflo", "agentic-flow", "concepts"], "minRelevance": -3 },
36
+ { "set": "cross-repo", "query": "What is the difference between RuVector and RuLake?", "expectRepo": ["ruvector", "rulake", "concepts"], "minRelevance": -3 },
37
+
38
+ { "set": "implementation", "query": "How does RuVector implement the HNSW navigable small-world graph on disk?", "expectRepo": ["ruvector", "concepts"], "minRelevance": -3 },
39
+ { "set": "implementation", "query": "How does AgentDB store causal edges between memory nodes?", "expectRepo": ["agentdb", "concepts"], "minRelevance": -3 },
40
+ { "set": "implementation", "query": "How does SPARC structure its five phases with quality gates?", "expectRepo": ["sparc", "concepts", "ruflo"], "minRelevance": -3 },
41
+ { "set": "implementation", "query": "How does SAFLA close its self-aware feedback learning loop?", "expectRepo": ["safla", "concepts"], "minRelevance": -3 },
42
+ { "set": "implementation", "query": "How does SynthLang compress prompts to reduce token cost?", "expectRepo": ["synthlang", "concepts"], "minRelevance": -3 },
43
+
44
+ { "set": "coverage", "query": "What is rupixel and what kind of embeddings does it use?", "expectRepo": ["rupixel", "concepts"], "minRelevance": -3 },
45
+ { "set": "coverage", "query": "What is agenticow and how does copy-on-write help agent memory?", "expectRepo": ["agenticow", "concepts"], "minRelevance": -3 },
46
+ { "set": "coverage", "query": "What is daa and what does it enable for decentralized autonomous agents?", "expectRepo": ["daa", "concepts"], "minRelevance": -3 },
47
+ { "set": "coverage", "query": "What is dspy.ts and how does it bring DSPy-style programming to TypeScript?", "expectRepo": ["dspy.ts", "concepts"], "minRelevance": -3 },
48
+ { "set": "coverage", "query": "What does FACT actually cache, and why is it about caching rather than retrieval?", "expectRepo": ["fact", "concepts"], "minRelevance": -3 },
49
+ { "set": "coverage", "query": "What does agent-harness-generator / metaharness mint, and what is Darwin Mode?", "expectRepo": ["metaharness", "concepts"], "minRelevance": -3 },
50
+
51
+ { "set": "adversarial", "query": "I heard RuvNet can't do real multi-agent swarms, is that true?", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
52
+ { "set": "adversarial", "query": "Does ruv-FANN actually run, or is it just an idea?", "expectRepo": ["ruv-fann", "concepts"], "minRelevance": -3 },
53
+ { "set": "adversarial", "query": "Give me the RuvNet building block for orchestrating parallel coding agents.", "expectRepo": ["ruflo", "concepts"], "minRelevance": -3 },
54
+ { "set": "adversarial", "query": "What is the RuvNet equivalent of LangGraph for agent workflows?", "expectRepo": ["ruflo", "agentic-flow", "concepts"], "minRelevance": -3 },
55
+ { "set": "adversarial", "query": "Which RuvNet repo would I use to A/B test and score different agent harnesses?", "expectRepo": ["metaharness", "concepts"], "minRelevance": -3 }
56
+ ]
@@ -0,0 +1,76 @@
1
+ #!/usr/bin/env node
2
+ // Local publication is deliberately impossible. The one publisher may mutate public channels only
3
+ // when the reviewer-protected GitHub workflow carries a fully valid candidate seal into it.
4
+ import crypto from 'node:crypto';
5
+ import fs from 'node:fs';
6
+ import path from 'node:path';
7
+ import { evaluateCandidateReceipt } from './release-proof.mjs';
8
+ import { getVersion } from './version.mjs';
9
+
10
+ export const PROTECTED_WORKFLOW = 'protected-release';
11
+ export const PROTECTED_VERSION = getVersion();
12
+
13
+ const hex = (value, length) => new RegExp(`^[0-9a-f]{${length}}$`).test(String(value || ''));
14
+ const digestOf = (receipt) => String(receipt?.artifact?.sha256 || '').replace(/^sha256:/, '');
15
+
16
+ export function validateProtectedPublishInvocation({ root = process.cwd(), env = process.env } = {}) {
17
+ const failures = [];
18
+ let releaseMode = 'unknown';
19
+ if (env.GITHUB_ACTIONS !== 'true' || env.GITHUB_WORKFLOW !== PROTECTED_WORKFLOW) {
20
+ failures.push('publication is allowed only inside the protected-release GitHub workflow');
21
+ }
22
+
23
+ const expectedSha = String(env.RUVNET_EXPECTED_SHA || '');
24
+ const expectedDigest = String(env.RUVNET_EXPECTED_ARTIFACT_SHA256 || '').replace(/^sha256:/, '');
25
+ const expectedVersion = String(env.RUVNET_EXPECTED_VERSION || '');
26
+ if (!hex(expectedSha, 40)) failures.push('expected candidate SHA is missing or malformed');
27
+ if (!hex(expectedDigest, 64)) failures.push('expected artifact digest is missing or malformed');
28
+ if (expectedVersion !== PROTECTED_VERSION) failures.push(`release generation must be ${PROTECTED_VERSION}`);
29
+
30
+ const receiptRelative = String(env.RUVNET_CANDIDATE_RECEIPT || '');
31
+ const evidenceRoot = path.resolve(root, 'release-evidence');
32
+ const receiptPath = path.resolve(root, receiptRelative || '.missing-candidate-receipt');
33
+ if (!receiptRelative || !(receiptPath === evidenceRoot || receiptPath.startsWith(`${evidenceRoot}${path.sep}`))) {
34
+ failures.push('candidate receipt must be inside release-evidence');
35
+ }
36
+
37
+ let receipt = null;
38
+ try {
39
+ receipt = JSON.parse(fs.readFileSync(receiptPath, 'utf8'));
40
+ } catch (error) {
41
+ failures.push(`candidate receipt unavailable: ${error.message}`);
42
+ }
43
+
44
+ if (receipt) {
45
+ const seal = evaluateCandidateReceipt(receipt);
46
+ if (seal.verdict !== 'PASS') failures.push(`candidate seal failed: ${seal.failures.map(({ code }) => code).join(',')}`);
47
+ const receiptDigest = digestOf(receipt);
48
+ if (receipt.sha !== expectedSha) failures.push('candidate receipt SHA mismatch');
49
+ if (receiptDigest !== expectedDigest) failures.push('candidate artifact digest mismatch');
50
+ if (receipt.version !== expectedVersion) failures.push('candidate version mismatch');
51
+ const expectedMode = receipt.phase === 'stabilization-candidate' ? 'stabilization' : 'strict';
52
+ releaseMode = expectedMode;
53
+ if (env.RUVNET_RELEASE_MODE !== expectedMode) failures.push(`release mode must be ${expectedMode} for this receipt`);
54
+
55
+ const artifactPath = path.resolve(root, String(receipt.artifact?.path || '.missing-artifact'));
56
+ if (!(artifactPath === evidenceRoot || artifactPath.startsWith(`${evidenceRoot}${path.sep}`))) {
57
+ failures.push('candidate artifact must be inside release-evidence');
58
+ } else {
59
+ try {
60
+ const actual = crypto.createHash('sha256').update(fs.readFileSync(artifactPath)).digest('hex');
61
+ if (actual !== expectedDigest) failures.push('candidate artifact bytes do not match the seal');
62
+ } catch (error) {
63
+ failures.push(`candidate artifact unavailable: ${error.message}`);
64
+ }
65
+ }
66
+ }
67
+
68
+ return {
69
+ verdict: failures.length === 0 ? 'PASS' : 'FAIL',
70
+ sha: expectedSha,
71
+ artifactSha256: expectedDigest,
72
+ version: expectedVersion,
73
+ mode: releaseMode,
74
+ failures,
75
+ };
76
+ }
@@ -0,0 +1,95 @@
1
+ #!/usr/bin/env node
2
+ // prove.mjs — run a question battery through the REAL retrieval engine and emit an auditable
3
+ // proof artifact. It calls searchAll() directly — the *exact* function the search_ruvnet MCP tool
4
+ // wraps — so the scores are the same ones a live Claude session sees, just without the JSON-RPC
5
+ // envelope. Models load once for the whole battery.
6
+ //
7
+ // KB_MODEL_CACHE=/path/to/models-cache node scripts/prove.mjs
8
+ // [--questions scripts/proof-questions.json] [--k 3]
9
+ //
10
+ // Writes: PROOF.md (human table + summary) and data/proof-results.json (machine-readable)
11
+ // Exit 0 if every question resolves to an expected repo at/above its threshold; 1 otherwise.
12
+ import path from 'node:path';
13
+ import fs from 'node:fs';
14
+ import { fileURLToPath } from 'node:url';
15
+ import { searchAll } from '../kb/forge-ask-all.mjs';
16
+
17
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
18
+ const KB = path.join(ROOT, 'kb');
19
+ const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
20
+ const QFILE = path.resolve(arg('--questions', path.join(ROOT, 'scripts', 'proof-questions.json')));
21
+ const K = parseInt(arg('--k', '3'), 10);
22
+ const OUTBASE = arg('--out', 'PROOF'); // PROOF.md + data/<lower>-results.json
23
+
24
+ const questions = JSON.parse(fs.readFileSync(QFILE, 'utf8'));
25
+ const started = new Date().toISOString();
26
+ console.log(`[prove] ${questions.length} questions · k=${K} · brain=${KB}\n`);
27
+
28
+ const results = [];
29
+ for (let i = 0; i < questions.length; i++) {
30
+ const q = questions[i];
31
+ let ranked = [];
32
+ try { const out = await searchAll({ dir: KB, query: q.query, k: K, pool: 8 }); ranked = out?.results || []; }
33
+ catch (e) { console.log(`ERR #${i + 1} ${e.message}`); }
34
+ const top = ranked[0] || null;
35
+ const got = top?.repo ?? null;
36
+ const eff = (got === 'concepts' && top?.path) ? (top.path.split('/')[0] || got) : got; // primer's home repo
37
+ const score = top?.ceScore ?? null;
38
+ const citedPath = top ? `${top.repo}/${top.path}` : null;
39
+ const exp = q.expectRepo;
40
+ const repoOk = !exp || (Array.isArray(exp) ? (exp.includes(got) || exp.includes(eff)) : (got === exp || eff === exp));
41
+ const relOk = score == null ? true : score >= (q.minRelevance ?? -3);
42
+ const pass = !!got && repoOk && relOk;
43
+ results.push({ n: i + 1, set: q.set || '', query: q.query, expect: exp, got, eff, score, citedPath, pass });
44
+ const s = score == null ? 'n/a' : score.toFixed(3);
45
+ console.log(`${pass ? 'PASS' : 'FAIL'} #${String(i + 1).padStart(2)} ${(eff || '(no hit)').padEnd(24)} @ ${String(s).padStart(7)} ${q.query.slice(0, 64)}`);
46
+ }
47
+
48
+ const passed = results.filter((r) => r.pass).length;
49
+ const scored = results.filter((r) => r.score != null).map((r) => r.score).sort((a, b) => a - b);
50
+ const median = scored.length ? scored[Math.floor(scored.length / 2)] : null;
51
+ const bySet = {};
52
+ for (const r of results) { const k = r.set || 'other'; (bySet[k] ??= { pass: 0, total: 0 }); bySet[k].total++; if (r.pass) bySet[k].pass++; }
53
+
54
+ // machine artifact
55
+ fs.mkdirSync(path.join(ROOT, 'data'), { recursive: true });
56
+ fs.writeFileSync(path.join(ROOT, 'data', `${OUTBASE.toLowerCase()}-results.json`),
57
+ JSON.stringify({ started, finished: new Date().toISOString(), k: K, passed, total: results.length, bySet, results }, null, 2));
58
+
59
+ // human artifact
60
+ const esc = (s) => String(s).replace(/\|/g, '\\|');
61
+ const setLines = Object.entries(bySet).map(([k, v]) => `| ${k} | ${v.pass}/${v.total} |`).join('\n');
62
+ const rows = results.map((r) => `| ${r.n} | ${r.set} | ${esc(r.query)} | ${r.eff || '—'} | ${r.score == null ? 'n/a' : r.score.toFixed(3)} | ${esc(r.citedPath || '—')} | ${r.pass ? '✅' : '❌'} |`).join('\n');
63
+ const md = `# RuvNet Brain — Proof of Retrieval (live run)
64
+
65
+ > Generated by \`node scripts/prove.mjs\` on ${started}. Re-run it yourself; this file is the output, not a claim.
66
+ > Each row is a real query through \`searchAll()\` — the exact engine the \`search_ruvnet\` MCP tool wraps.
67
+
68
+ ## Headline
69
+
70
+ **${passed}/${results.length} questions resolved to an expected repo** at/above threshold. Median top-hit relevance: **${median == null ? 'n/a' : median.toFixed(3)}**.
71
+
72
+ | Question set | Passed |
73
+ |---|---|
74
+ ${setLines}
75
+
76
+ - **tuned / held-out** — the capability-confidence gate (held-out built *after* tuning ⇒ not overfit).
77
+ - **anti-drift** — questions phrased the way a skeptic would, where Claude normally reaches for pgvector/Pinecone or doubts a real tool.
78
+ - **adversarial** — negative-framed ("I heard it *can't*…") to prove the brain doesn't fold.
79
+ - **implementation / coverage / cross-repo** — depth and breadth across all 19 repos.
80
+
81
+ ## Every question, every score
82
+
83
+ | # | set | question | resolved repo | relevance | cited source | ok |
84
+ |--:|---|---|---|--:|---|:--:|
85
+ ${rows}
86
+
87
+ > "relevance" is the cross-encoder score after name-affinity boost; higher is sharper. Negative is normal for
88
+ > a cross-encoder and still ranks correctly — what matters for the *never-wrongly-doubt* guarantee is that the
89
+ > top hit lands on the right repo with a cited file you can open.
90
+ `;
91
+ fs.writeFileSync(path.join(ROOT, `${OUTBASE}.md`), md);
92
+
93
+ console.log(`\n[prove] ${passed}/${results.length} passed · median ${median == null ? 'n/a' : median.toFixed(3)}`);
94
+ console.log('[prove] wrote PROOF.md + data/proof-results.json');
95
+ process.exit(passed === results.length ? 0 : 1);
@@ -0,0 +1,57 @@
1
+ #!/usr/bin/env bash
2
+ # claude-proxied.sh — launch ONE Claude Code session routed through the proxy.
3
+ #
4
+ # THIS IS THE ISOLATION BOUNDARY OF THE WHOLE TRIAL.
5
+ #
6
+ # ANTHROPIC_BASE_URL is exported for this process only. Nothing is written to
7
+ # ~/.claude/settings.json, so every other Claude Code window on this machine —
8
+ # including the one you are reading this in — keeps talking directly to
9
+ # Anthropic, exactly as before. Close this session and the wiring is gone.
10
+ #
11
+ # What happens inside this session (ADR-313 addendum): setting
12
+ # ANTHROPIC_BASE_URL makes Claude Code stop managing its own Max/Pro OAuth
13
+ # session. That is fine here and ONLY because the proxy defaults to the
14
+ # Passthrough plane, which reads ~/.claude/.credentials.json (read-only) and
15
+ # forwards to the real api.anthropic.com with your actual subscription token.
16
+ # Your subscription is still what pays and still what answers.
17
+ #
18
+ # If the plane were ever anything other than passthrough, this script refuses
19
+ # to launch — a silent flip to a cheap tier while you believe you are on your
20
+ # subscription is the single worst failure mode here, so it is checked, not
21
+ # assumed.
22
+ set -euo pipefail
23
+
24
+ TOKEN_FILE="$HOME/.ruflo/proxy-token"
25
+
26
+ if ! ruflo proxy status --json 2>/dev/null | grep -q '"running":true'; then
27
+ echo "Proxy is not running. Start it first:"
28
+ echo " ./scripts/proxy/proxy-up.sh"
29
+ exit 1
30
+ fi
31
+
32
+ if [ ! -r "$TOKEN_FILE" ]; then
33
+ echo "Missing proxy token at $TOKEN_FILE — reinstall with ./scripts/proxy/proxy-up.sh"
34
+ exit 1
35
+ fi
36
+
37
+ PLANE=$(ruflo proxy config 2>/dev/null | head -1)
38
+ if ! echo "$PLANE" | grep -qi 'passthrough'; then
39
+ echo "REFUSING TO LAUNCH — data plane is not passthrough:"
40
+ echo " $PLANE"
41
+ echo
42
+ echo "This trial only sanctions the passthrough plane (your own subscription)."
43
+ echo "Revert to it with: ruflo proxy config --local-only (or re-read ADR-0026)"
44
+ exit 1
45
+ fi
46
+
47
+ export ANTHROPIC_BASE_URL="http://127.0.0.1:11435"
48
+ export ANTHROPIC_AUTH_TOKEN="$(cat "$TOKEN_FILE")"
49
+
50
+ echo "Launching a proxied Claude Code session."
51
+ echo " ANTHROPIC_BASE_URL = $ANTHROPIC_BASE_URL (this process only)"
52
+ echo " data plane = passthrough (your own Anthropic subscription)"
53
+ echo " other windows = unaffected"
54
+ echo " watch traffic with = ruflo proxy logs -f"
55
+ echo
56
+
57
+ exec claude "$@"
@@ -0,0 +1,59 @@
1
+ #!/usr/bin/env bash
2
+ # proxy-revert.sh — remove the Meta LLM Proxy trial completely.
3
+ #
4
+ # This is the safety net for the whole trial, so it is deliberately boring and
5
+ # total: stop the process, uninstall via rUv's own lifecycle command, then
6
+ # PROVE nothing is left rather than claiming it.
7
+ #
8
+ # It does not touch ~/.claude/settings.json — the trial never wrote there (see
9
+ # proxy-verify.mjs check 3), so there is nothing to undo.
10
+ set -uo pipefail
11
+
12
+ echo "Reverting the Meta LLM Proxy trial..."
13
+ echo
14
+
15
+ echo "--- stopping (if running) ---"
16
+ ruflo proxy stop 2>&1 | sed 's/^/ /' || true
17
+ echo
18
+
19
+ echo "--- uninstalling binary, token, consent receipt (ruflo proxy uninstall) ---"
20
+ ruflo proxy uninstall 2>&1 | sed 's/^/ /' || true
21
+ echo
22
+
23
+ echo "--- PROOF it is gone (derived, not asserted) ---"
24
+ FAIL=0
25
+
26
+ STATUS=$(ruflo proxy status --json 2>/dev/null || echo '{}')
27
+ echo " ruflo proxy status: $STATUS"
28
+ case "$STATUS" in
29
+ *'"installed":true'*) echo " ^ still installed" ; FAIL=1 ;;
30
+ esac
31
+ case "$STATUS" in
32
+ *'"running":true'*) echo " ^ still running" ; FAIL=1 ;;
33
+ esac
34
+
35
+ if lsof -iTCP:11435 -sTCP:LISTEN -n -P >/dev/null 2>&1; then
36
+ echo " port 11435: STILL LISTENING <-- revert incomplete"; FAIL=1
37
+ else
38
+ echo " port 11435: clear"
39
+ fi
40
+
41
+ if [ -f "$HOME/.ruflo/proxy-token" ]; then
42
+ echo " proxy-token: STILL PRESENT <-- revert incomplete"; FAIL=1
43
+ else
44
+ echo " proxy-token: removed"
45
+ fi
46
+
47
+ if grep -q 'ANTHROPIC_BASE_URL' "$HOME/.claude/settings.json" 2>/dev/null; then
48
+ echo " ~/.claude/settings.json: contains ANTHROPIC_BASE_URL <-- unexpected"; FAIL=1
49
+ else
50
+ echo " ~/.claude/settings.json: clean (never modified by this trial)"
51
+ fi
52
+
53
+ echo
54
+ if [ "$FAIL" = 0 ]; then
55
+ echo "Reverted. The machine is back to its pre-trial state."
56
+ exit 0
57
+ fi
58
+ echo "REVERT INCOMPLETE — see the lines marked above."
59
+ exit 1
@@ -0,0 +1,60 @@
1
+ #!/usr/bin/env bash
2
+ # proxy-up.sh — install + start the Meta LLM Proxy, then verify it.
3
+ #
4
+ # Thin orchestration over rUv's own lifecycle commands (ADR-307):
5
+ # ruflo proxy install / start / status / doctor
6
+ # It adds no logic of its own except the macOS workaround documented below.
7
+ set -uo pipefail
8
+
9
+ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
10
+
11
+ # --- macOS workaround: upstream bug in @claude-flow/security's PathValidator --
12
+ #
13
+ # `ruflo proxy install` extracts into a tmpdir and then validates that the
14
+ # extracted binary really lives inside that tmpdir (defense against a symlink
15
+ # swap). The validator canonicalizes the CANDIDATE path with fs.realpath but
16
+ # resolves the ALLOWED PREFIX with path.resolve only:
17
+ #
18
+ # path-validator.ts:177-178 prefixes -> path.resolve(p) (no realpath)
19
+ # path-validator.ts:229,234 candidate -> path.resolve + realpath
20
+ # path-validator.ts:262 resolvedPath.startsWith(prefix + sep)
21
+ #
22
+ # On macOS os.tmpdir() is /var/folders/... which is a symlink to
23
+ # /private/var/folders/... So the candidate becomes /private/var/... while the
24
+ # prefix stays /var/... and the startsWith check can NEVER pass. Install fails
25
+ # with "extracted binary path failed validation: Path is outside allowed
26
+ # directories" for every macOS user.
27
+ #
28
+ # Handing the installer an ALREADY-CANONICAL TMPDIR makes the validator's own
29
+ # assumption true. This changes nothing about what is downloaded or verified —
30
+ # the Ed25519 signature and sha256 checks run exactly as before.
31
+ #
32
+ # Upstream fix would be one line: realpath the prefixes too.
33
+ if [ "$(uname -s)" = "Darwin" ]; then
34
+ CANONICAL_TMP="$(node -e 'console.log(require("fs").realpathSync(require("os").tmpdir()))' 2>/dev/null || echo "")"
35
+ if [ -n "$CANONICAL_TMP" ]; then
36
+ export TMPDIR="$CANONICAL_TMP"
37
+ echo "macOS: TMPDIR canonicalized to $TMPDIR (upstream PathValidator symlink bug)"
38
+ echo
39
+ fi
40
+ fi
41
+
42
+ echo "--- install (idempotent; verifies Ed25519 signature + sha256) ---"
43
+ if ruflo proxy status --json 2>/dev/null | grep -q '"installed":true'; then
44
+ echo " already installed"
45
+ else
46
+ ruflo proxy install --yes 2>&1 | tail -4 | sed 's/^/ /'
47
+ fi
48
+ echo
49
+
50
+ echo "--- start (detached; survives terminal close, NOT a reboot) ---"
51
+ if ruflo proxy status --json 2>/dev/null | grep -q '"running":true'; then
52
+ echo " already running"
53
+ else
54
+ ruflo proxy start --service 2>&1 | tail -3 | sed 's/^/ /'
55
+ sleep 2
56
+ fi
57
+ echo
58
+
59
+ echo "--- verify ---"
60
+ node "$REPO_ROOT/scripts/proxy/proxy-verify.mjs"