ruvnet-brain 4.0.1 → 4.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +303 -24
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/commands/whats-new.md +6 -6
- package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +90 -32
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +34 -29
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +503 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/scripts/whats-new.mjs +42 -0
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +98 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/plugin/skills/whats-new/SKILL.md +4 -4
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +531 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/fix-workstream.mjs +291 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +237 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +558 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +326 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +172 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2922 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/protected-release-invocation.mjs +76 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/publication-receipt.mjs +307 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-authority.mjs +93 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +439 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +296 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stabilization-receipt.mjs +108 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +867 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
// ux-suite.mjs — the UX-experience QE suite runner (owner request 2026-07-24).
|
|
2
|
+
//
|
|
3
|
+
// Runs the deterministic UX probes, prints a table of MEASURED numbers, writes a machine-readable
|
|
4
|
+
// receipt when UX_QE_EVIDENCE is set, and exits non-zero on any HARD failure.
|
|
5
|
+
// 1. Environment-sensitive timings (server-ready, console/tips paint, command→explanation,
|
|
6
|
+
// dead-air) are HARD user-experience budgets. Platform calibration gives slower hosted runners
|
|
7
|
+
// honest headroom without turning "slow enough for a person to notice" into advisory green.
|
|
8
|
+
// 2. kb/card-lane.mjs's decision lane is MODEL-FREE, ML-FREE keyword overlap with a measured warm
|
|
9
|
+
// baseline of 0.1158ms. Its budget (kb/card-lane-budget.json, p95 <= 250ms / absolute fail
|
|
10
|
+
// >1000ms — ~2,159x / ~8,600x the baseline) has so much headroom that a breach cannot be
|
|
11
|
+
// scheduler jitter — it can only be a correctness regression. THIS is a genuine hard gate: a
|
|
12
|
+
// breach here fails the suite, not warns it. See scripts/qe/card-lane-gate.mjs for the full
|
|
13
|
+
// reasoning and the in-process (no subprocess per firing) measurement method.
|
|
14
|
+
// 3. SESSION-START WALL TIME (added 2026-07-28) is the SAME tier as 2, and is here because tier 2
|
|
15
|
+
// alone was not enough. An independent grader's words: the card-lane gate "measures a
|
|
16
|
+
// 0.03–0.22ms in-process function against a 250ms budget (~1000x headroom — it can only catch
|
|
17
|
+
// catastrophic regression classes)", while "everything the user actually FEELS — heavy-lane
|
|
18
|
+
// query seconds, session-start WALL TIME, install minutes, dead air, refusal clarity — is
|
|
19
|
+
// advisory or unmeasured". Session-start wall time is the first of those promoted out of tier 1:
|
|
20
|
+
// it is the hook a stranger's Claude Code fires before their first prompt is answered, it is
|
|
21
|
+
// already measured by scripts/selfcheck.mjs's external process-group watchdog (no second timer
|
|
22
|
+
// was written), and its budget is set from a measured distribution — p95 1000ms is ~3.1x the
|
|
23
|
+
// worst measured p95 (323ms over n=110), NOT 1000x. See scripts/qe/session-start-gate.mjs.
|
|
24
|
+
//
|
|
25
|
+
// HONESTY (same rules as the product):
|
|
26
|
+
// • Every number is measured on THIS run. Nothing is asserted from memory.
|
|
27
|
+
// • A probe that could not execute is reported "not run" and HARD-fails — silence is not success.
|
|
28
|
+
// • The probes are MODEL-FREE (render + PTY-style timing, plus the in-process card-lane firings).
|
|
29
|
+
// They call no LLM, use no API key, touch no account — the cleanest satisfaction of the owner's
|
|
30
|
+
// "no API keys, run on our account" rule.
|
|
31
|
+
// • aqe orchestration: we OPTIONALLY register this run as an `aqe task` for visibility in
|
|
32
|
+
// `aqe status`, but the MEASUREMENT is a plain deterministic probe, NOT aqe-internal. Verified live
|
|
33
|
+
// 2026-07-24: `aqe domain` supports only list/health (not create), so inventing an "onboarding-ux"
|
|
34
|
+
// domain would be fiction. We do not. If aqe isn't present, the suite runs identically and says so.
|
|
35
|
+
import { spawn, spawnSync } from 'node:child_process';
|
|
36
|
+
import fs from 'node:fs';
|
|
37
|
+
import path from 'node:path';
|
|
38
|
+
import os from 'node:os';
|
|
39
|
+
import { fileURLToPath } from 'node:url';
|
|
40
|
+
import { runCommandProbe } from '../../tests/ux/command-probe.mjs';
|
|
41
|
+
import { runCardLaneGate } from './card-lane-gate.mjs';
|
|
42
|
+
import { runSessionStartGate } from './session-start-gate.mjs';
|
|
43
|
+
|
|
44
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
45
|
+
const RENDER_PROBE = path.resolve(HERE, '../../tests/ux/render-probe.mjs');
|
|
46
|
+
// The child now performs seven acceptance assertions, two real settings writes + reload, one real
|
|
47
|
+
// batch remedy and one real undo in addition to paint timings. Its total wall clock is test-runtime,
|
|
48
|
+
// not user-visible latency; each user action has its own hard 4s assertion inside the probe.
|
|
49
|
+
const RENDER_PROBE_TIMEOUT_MS = 60_000;
|
|
50
|
+
|
|
51
|
+
function stopProcessTree(child) {
|
|
52
|
+
if (!child?.pid) return;
|
|
53
|
+
if (process.platform === 'win32') {
|
|
54
|
+
try { spawnSync('taskkill', ['/pid', String(child.pid), '/T', '/F'], { stdio: 'ignore' }); } catch {}
|
|
55
|
+
return;
|
|
56
|
+
}
|
|
57
|
+
try { process.kill(-child.pid, 'SIGTERM'); } catch {}
|
|
58
|
+
try { process.kill(-child.pid, 'SIGKILL'); } catch {}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Browser drivers can wedge below JavaScript, so an in-process Promise timeout is not a bound.
|
|
63
|
+
* Run the render probe in its own process group and kill the whole group at the deadline.
|
|
64
|
+
*/
|
|
65
|
+
export function runRenderProbeIsolated({
|
|
66
|
+
probeFile = RENDER_PROBE,
|
|
67
|
+
timeoutMs = RENDER_PROBE_TIMEOUT_MS,
|
|
68
|
+
} = {}) {
|
|
69
|
+
return new Promise((resolve) => {
|
|
70
|
+
const child = spawn(process.execPath, [probeFile], {
|
|
71
|
+
cwd: path.resolve(HERE, '../..'),
|
|
72
|
+
env: process.env,
|
|
73
|
+
detached: process.platform !== 'win32',
|
|
74
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
75
|
+
windowsHide: true,
|
|
76
|
+
});
|
|
77
|
+
let stdout = '';
|
|
78
|
+
let stderr = '';
|
|
79
|
+
let settled = false;
|
|
80
|
+
child.stdout.on('data', (chunk) => { stdout += String(chunk); });
|
|
81
|
+
child.stderr.on('data', (chunk) => { stderr += String(chunk); });
|
|
82
|
+
|
|
83
|
+
const finish = (result) => {
|
|
84
|
+
if (settled) return;
|
|
85
|
+
settled = true;
|
|
86
|
+
clearTimeout(timer);
|
|
87
|
+
resolve(result);
|
|
88
|
+
};
|
|
89
|
+
const timer = setTimeout(() => {
|
|
90
|
+
stopProcessTree(child);
|
|
91
|
+
const trace = stderr.trim().split('\n').filter(Boolean).slice(-4).join(' | ');
|
|
92
|
+
finish({
|
|
93
|
+
results: [],
|
|
94
|
+
notes: [`render probe exceeded ${timeoutMs}ms process deadline${trace ? `; last stages: ${trace}` : ''}`],
|
|
95
|
+
});
|
|
96
|
+
}, timeoutMs);
|
|
97
|
+
|
|
98
|
+
child.on('error', (error) => finish({ results: [], notes: [`render probe spawn failed: ${error.message}`] }));
|
|
99
|
+
child.on('close', () => {
|
|
100
|
+
try {
|
|
101
|
+
const parsed = JSON.parse(stdout);
|
|
102
|
+
finish(parsed);
|
|
103
|
+
} catch {
|
|
104
|
+
const trace = stderr.trim().split('\n').filter(Boolean).slice(-4).join(' | ');
|
|
105
|
+
finish({ results: [], notes: [`render probe returned no readable JSON${trace ? `; last stages: ${trace}` : ''}`] });
|
|
106
|
+
}
|
|
107
|
+
});
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// Darwin values are frozen from the measured 2026-07-24 baseline in docs/qe/ux-first-run.md.
|
|
112
|
+
// Linux and Windows receive bounded hosted-runner startup headroom; the visible-paint and dead-air
|
|
113
|
+
// product promises stay tight. These are release budgets, not performance claims about GitHub's
|
|
114
|
+
// hardware. CI receipts make future recalibration evidence-based rather than guessed.
|
|
115
|
+
export const PLATFORM_BUDGETS = Object.freeze({
|
|
116
|
+
darwin: Object.freeze({
|
|
117
|
+
'server-ready': 2500,
|
|
118
|
+
'console time-to-visible': 2500,
|
|
119
|
+
'tips time-to-visible (hero)': 2000,
|
|
120
|
+
'tips first-section': 2000,
|
|
121
|
+
commandToExplanationMs: 1500,
|
|
122
|
+
maxDeadAirMs: 3000,
|
|
123
|
+
}),
|
|
124
|
+
linux: Object.freeze({
|
|
125
|
+
'server-ready': 4000,
|
|
126
|
+
'console time-to-visible': 3000,
|
|
127
|
+
'tips time-to-visible (hero)': 2500,
|
|
128
|
+
'tips first-section': 2500,
|
|
129
|
+
commandToExplanationMs: 2500,
|
|
130
|
+
maxDeadAirMs: 3000,
|
|
131
|
+
}),
|
|
132
|
+
win32: Object.freeze({
|
|
133
|
+
'server-ready': 6000,
|
|
134
|
+
'console time-to-visible': 4000,
|
|
135
|
+
'tips time-to-visible (hero)': 3500,
|
|
136
|
+
'tips first-section': 3500,
|
|
137
|
+
commandToExplanationMs: 3000,
|
|
138
|
+
maxDeadAirMs: 3000,
|
|
139
|
+
}),
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
export function budgetsForPlatform(platform = process.platform) {
|
|
143
|
+
const budgets = PLATFORM_BUDGETS[platform];
|
|
144
|
+
if (!budgets) throw new Error(`unsupported UX-QE platform: ${platform}`);
|
|
145
|
+
return budgets;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export function timingFailure(label, measured, budget) {
|
|
149
|
+
if (measured == null) return `${label}: could not measure`;
|
|
150
|
+
if (measured > budget) return `${label}: ${measured}ms exceeds HARD ${budget}ms budget`;
|
|
151
|
+
return null;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function line(label, measured, unit, hardAt) {
|
|
155
|
+
const val = measured == null ? 'NOT RUN' : `${measured}${unit}`;
|
|
156
|
+
let flag = '';
|
|
157
|
+
if (measured == null) flag = ' ✗ could not measure';
|
|
158
|
+
else if (hardAt != null && measured > hardAt) flag = ` ✗ HARD FAIL (>${hardAt}${unit})`;
|
|
159
|
+
else if (hardAt != null) flag = ` ✓ HARD budget ${hardAt}${unit}`;
|
|
160
|
+
return ` ${label.padEnd(30)} ${String(val).padStart(10)}${flag}`;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
function writeEvidence(receipt) {
|
|
164
|
+
const jsonOutIndex = process.argv.indexOf('--json-out');
|
|
165
|
+
if (jsonOutIndex >= 0 && !process.argv[jsonOutIndex + 1]) {
|
|
166
|
+
throw new Error('--json-out requires a file path');
|
|
167
|
+
}
|
|
168
|
+
const target = jsonOutIndex >= 0
|
|
169
|
+
? process.argv[jsonOutIndex + 1]
|
|
170
|
+
: process.env.UX_QE_EVIDENCE;
|
|
171
|
+
if (!target) return;
|
|
172
|
+
const resolved = path.resolve(target);
|
|
173
|
+
fs.mkdirSync(path.dirname(resolved), { recursive: true });
|
|
174
|
+
fs.writeFileSync(resolved, `${JSON.stringify(receipt, null, 2)}\n`);
|
|
175
|
+
};
|
|
176
|
+
|
|
177
|
+
function tryRegisterAqeTask() {
|
|
178
|
+
// Best-effort visibility only. Never fails the suite; never bills a model. `submit` enqueues
|
|
179
|
+
// metadata to the Queen Coordinator; `--no-progress` and no `--wait` keep it fire-and-forget, so no
|
|
180
|
+
// model is invoked. Flags grounded live 2026-07-24 against `aqe task submit --help` (type positional,
|
|
181
|
+
// -p/-d/-t/--payload — there is NO --description).
|
|
182
|
+
const payload = JSON.stringify({ probe: 'ruvnet-brain-ux-time-to-visible', model_free: true });
|
|
183
|
+
const r = spawnSync('aqe', ['task', 'submit', 'quality-assessment', '-p', 'p3', '--payload', payload, '--no-progress'], { encoding: 'utf8', timeout: 15000 });
|
|
184
|
+
if (r.error || r.status !== 0) return { registered: false, why: (r.error && r.error.message) || (r.stderr || '').trim().split('\n').filter(Boolean).pop() || `exit ${r.status}` };
|
|
185
|
+
const id = ((r.stdout || '').match(/task[- ]?id[:\s]+(\S+)/i) || [])[1] || 'submitted';
|
|
186
|
+
return { registered: true, id };
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
export async function runUxSuite() {
|
|
190
|
+
console.log('\n RuvNet Brain — UX-experience QE suite (deterministic · model-free · runs on your account)\n');
|
|
191
|
+
const platform = process.platform;
|
|
192
|
+
const budgets = budgetsForPlatform(platform);
|
|
193
|
+
const startedAt = new Date().toISOString();
|
|
194
|
+
|
|
195
|
+
const aqe = tryRegisterAqeTask();
|
|
196
|
+
console.log(aqe.registered
|
|
197
|
+
? ` aqe: registered task ${aqe.id} for orchestration visibility (measurement is a plain probe)\n`
|
|
198
|
+
: ` aqe: not registered (${aqe.why}) — probes run identically; orchestration visibility only\n`);
|
|
199
|
+
|
|
200
|
+
const hardFailures = [];
|
|
201
|
+
|
|
202
|
+
// ── Probe 1: render time-to-visible ──────────────────────────────────────────────────────────
|
|
203
|
+
console.log(' ── time-to-visible (console + tips) ──');
|
|
204
|
+
const render = await runRenderProbeIsolated();
|
|
205
|
+
for (const r of render.results) {
|
|
206
|
+
console.log(line(r.label, r.ms, 'ms', budgets[r.label]));
|
|
207
|
+
const failure = timingFailure(r.label, r.ms, budgets[r.label]);
|
|
208
|
+
if (failure) hardFailures.push(failure);
|
|
209
|
+
}
|
|
210
|
+
for (const n of render.notes) { console.log(` ! ${n}`); hardFailures.push(`render: ${n}`); }
|
|
211
|
+
console.log('\n ── console control acceptance ──');
|
|
212
|
+
for (const row of render.acceptance || []) {
|
|
213
|
+
console.log(` ${row.pass ? '✓' : '✗'} ${row.label}: ${row.detail}`);
|
|
214
|
+
if (!row.pass) hardFailures.push(`console control acceptance: ${row.label} — ${row.detail}`);
|
|
215
|
+
}
|
|
216
|
+
if (!(render.acceptance || []).length) hardFailures.push('console control acceptance: NOT RUN');
|
|
217
|
+
// Any expected render row missing entirely = not run = hard fail.
|
|
218
|
+
const gotConsole = render.results.some((r) => r.label === 'console time-to-visible' && r.ms != null);
|
|
219
|
+
if (!gotConsole) hardFailures.push('console time-to-visible: NOT RUN');
|
|
220
|
+
|
|
221
|
+
// ── Probe 2/3: command → explanation → "it's live" ──────────────────────────────────────────
|
|
222
|
+
console.log('\n ── command → explanation → completion signal ──');
|
|
223
|
+
const cmd = await runCommandProbe();
|
|
224
|
+
console.log(line('command→explanation', cmd.commandToExplanationMs, 'ms', budgets.commandToExplanationMs));
|
|
225
|
+
console.log(line('command→"it\'s live"', cmd.commandToLiveMs, 'ms', null) + ' (reported, not gated)');
|
|
226
|
+
console.log(line('max dead-air gap', cmd.maxDeadAirMs, 'ms', budgets.maxDeadAirMs));
|
|
227
|
+
console.log(` completion signal present ${cmd.completionSignalPresent ? ' YES ✓' : ' NO ✗ (GAP)'}`);
|
|
228
|
+
if (cmd.liveSignalText) console.log(` signal: "${cmd.liveSignalText}"`);
|
|
229
|
+
|
|
230
|
+
const explanationFailure = timingFailure('command→explanation', cmd.commandToExplanationMs, budgets.commandToExplanationMs);
|
|
231
|
+
if (explanationFailure) hardFailures.push(explanationFailure);
|
|
232
|
+
const deadAirFailure = timingFailure('max dead-air gap', cmd.maxDeadAirMs, budgets.maxDeadAirMs);
|
|
233
|
+
if (deadAirFailure) hardFailures.push(deadAirFailure);
|
|
234
|
+
if (!cmd.completionSignalPresent) hardFailures.push('completion signal MISSING — the "it\'s live, take a look at your page" line never printed');
|
|
235
|
+
|
|
236
|
+
// ── Probe 4: decision-lane latency — HARD GATE, not advisory (ADR-058 D6) ───────────────────
|
|
237
|
+
// Deliberately NOT reusing line()'s warnAt/"(proposed)" formatting above: that phrasing is correct
|
|
238
|
+
// for the advisory timings but would misreport a HARD budget breach as merely "proposed".
|
|
239
|
+
console.log('\n ── decision-lane latency (kb/card-lane.mjs) — HARD GATE, deterministic, model-free ──');
|
|
240
|
+
try {
|
|
241
|
+
const laneResult = await runCardLaneGate();
|
|
242
|
+
const b = laneResult.budget;
|
|
243
|
+
const tag = (ok) => (ok ? '✓' : '✗ HARD FAIL');
|
|
244
|
+
console.log(` ${'card-lane p50'.padEnd(30)} ${laneResult.p50.toFixed(4).padStart(10)}ms (reported, not gated)`);
|
|
245
|
+
console.log(` ${'card-lane p95'.padEnd(30)} ${laneResult.p95.toFixed(4).padStart(10)}ms budget ${b.p95BudgetMs}ms ${tag(laneResult.p95 <= b.p95BudgetMs)}`);
|
|
246
|
+
console.log(` ${'card-lane max'.padEnd(30)} ${laneResult.max.toFixed(4).padStart(10)}ms absolute-fail ${b.absoluteFailMs}ms ${tag(laneResult.max <= b.absoluteFailMs)}`);
|
|
247
|
+
console.log(` firings: ${laneResult.n} in-process (no subprocess per firing — see card-lane-gate.mjs)`);
|
|
248
|
+
if (!laneResult.pass) for (const r of laneResult.reasons) hardFailures.push(`card-lane latency: ${r}`);
|
|
249
|
+
} catch (e) {
|
|
250
|
+
console.log(` ! could not run the card-lane latency gate: ${e.message}`);
|
|
251
|
+
hardFailures.push(`card-lane latency gate: could not run — ${e.message}`);
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// ── Probe 5: session-start wall time — HARD GATE, the first USER-FELT number (ADR-058 D6) ───
|
|
255
|
+
// Wired exactly like probe 4 above and for the same reason: same tier, same "could not measure is
|
|
256
|
+
// never success" handling, same refusal to reuse line()'s "(proposed)" phrasing, which is correct
|
|
257
|
+
// for an advisory row and would misreport a HARD breach.
|
|
258
|
+
console.log('\n ── session-start wall time (plugin/hooks/hooks.json SessionStart) — HARD GATE, user-felt ──');
|
|
259
|
+
try {
|
|
260
|
+
const ss = await runSessionStartGate();
|
|
261
|
+
const b = ss.budget;
|
|
262
|
+
const tag = (ok) => (ok ? '✓' : '✗ HARD FAIL');
|
|
263
|
+
console.log(` ${'session-start cold first fire'.padEnd(30)} ${ss.warmupMs.toFixed(0).padStart(10)}ms ${ss.warmupTimedOut ? '✗ HARD FAIL (declared timeout exceeded)' : '✓ inside declared timeout'}`);
|
|
264
|
+
console.log(` ${'session-start p50'.padEnd(30)} ${ss.p50.toFixed(0).padStart(10)}ms (reported, not gated)`);
|
|
265
|
+
console.log(` ${'session-start p95'.padEnd(30)} ${ss.p95.toFixed(0).padStart(10)}ms budget ${b.p95BudgetMs}ms ${tag(ss.p95 <= b.p95BudgetMs)}`);
|
|
266
|
+
console.log(` ${'session-start max'.padEnd(30)} ${ss.max.toFixed(0).padStart(10)}ms absolute-fail ${b.absoluteFailMs}ms ${tag(ss.max <= b.absoluteFailMs)}`);
|
|
267
|
+
console.log(` firings: ${ss.n} sequential fires of the REAL registered command via selfcheck.mjs's watchdog, from ${ss.surface.source}`);
|
|
268
|
+
if (ss.warmupStderr) {
|
|
269
|
+
console.log(` cold trace: ${ss.warmupStderr.trim().split('\n').join(' | ')}`);
|
|
270
|
+
}
|
|
271
|
+
if (!ss.pass) for (const r of ss.reasons) hardFailures.push(`session-start wall time: ${r}`);
|
|
272
|
+
} catch (e) {
|
|
273
|
+
console.log(` ! could not run the session-start wall-time gate: ${e.message}`);
|
|
274
|
+
hardFailures.push(`session-start wall-time gate: could not run — ${e.message}`);
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
// ── Not run on this host (stated, never faked) ──────────────────────────────────────────────
|
|
278
|
+
console.log('\n ── execution scope ──');
|
|
279
|
+
console.log(` Platform — ${platform} ${os.arch()} (this process; other OSes execute as separate CI jobs)`);
|
|
280
|
+
console.log(' Codex host — NOT RUN: GitHub-hosted runners do not provide a configured Codex host; this probes the shipped console process directly');
|
|
281
|
+
|
|
282
|
+
// ── Verdict ─────────────────────────────────────────────────────────────────────────────────
|
|
283
|
+
const receipt = {
|
|
284
|
+
schemaVersion: 1,
|
|
285
|
+
suite: 'ruvnet-brain-ux-qe',
|
|
286
|
+
startedAt,
|
|
287
|
+
finishedAt: new Date().toISOString(),
|
|
288
|
+
gitSha: process.env.GITHUB_SHA || null,
|
|
289
|
+
platform,
|
|
290
|
+
arch: os.arch(),
|
|
291
|
+
node: process.version,
|
|
292
|
+
budgetsMs: budgets,
|
|
293
|
+
render,
|
|
294
|
+
command: cmd,
|
|
295
|
+
hardFailures,
|
|
296
|
+
pass: hardFailures.length === 0,
|
|
297
|
+
scope: {
|
|
298
|
+
browser: 'Playwright Chromium, real local console HTTP server',
|
|
299
|
+
command: 'direct shipped console process',
|
|
300
|
+
codexHost: 'not-run',
|
|
301
|
+
},
|
|
302
|
+
};
|
|
303
|
+
writeEvidence(receipt);
|
|
304
|
+
|
|
305
|
+
console.log('\n ── verdict ──');
|
|
306
|
+
if (hardFailures.length === 0) {
|
|
307
|
+
console.log(' PASS — every probe ran and every render, explanation, dead-air, decision-lane, and session-start HARD budget passed.\n');
|
|
308
|
+
return receipt;
|
|
309
|
+
}
|
|
310
|
+
console.log(' FAIL (hard):');
|
|
311
|
+
for (const f of hardFailures) console.log(` ✗ ${f}`);
|
|
312
|
+
console.log('');
|
|
313
|
+
const error = new Error(`UX QE failed with ${hardFailures.length} hard failure(s)`);
|
|
314
|
+
error.receipt = receipt;
|
|
315
|
+
throw error;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
319
|
+
runUxSuite().catch((e) => {
|
|
320
|
+
if (!e.receipt) console.error(' ux-suite crashed:', e.message);
|
|
321
|
+
process.exit(e.receipt ? 1 : 2);
|
|
322
|
+
});
|
|
323
|
+
}
|
|
Binary file
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* record-lesson.mjs — the durable "capture a lesson the RIGHT way" habit.
|
|
4
|
+
*
|
|
5
|
+
* WHY: AgentDB auto-capture records session transcripts (logging), not lessons
|
|
6
|
+
* (learning), and that telemetry drowns real lessons in recall. This records a
|
|
7
|
+
* lesson *structured* (task / tried / worked / critique / outcome) into a dedicated
|
|
8
|
+
* `lessons` signal namespace, refines it via native distill, and proves recall.
|
|
9
|
+
*
|
|
10
|
+
* NATIVE ONLY — shells to `ruflo memory` (store + distill + search). It does NOT
|
|
11
|
+
* reimplement any rUv capability; it enforces the structured-capture discipline
|
|
12
|
+
* that rUv's own `/remember` command recommends (agentdb-memory/commands/remember.md).
|
|
13
|
+
*
|
|
14
|
+
* Usage:
|
|
15
|
+
* node scripts/record-lesson.mjs \
|
|
16
|
+
* --task "..." --tried "..." --worked "..." --critique "..." --outcome success \
|
|
17
|
+
* [--slug short-name] [--dir <projectDir>] [--namespace lessons]
|
|
18
|
+
*/
|
|
19
|
+
import { execFileSync } from 'node:child_process';
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
|
|
23
|
+
const arg = (name, def = '') => {
|
|
24
|
+
const i = process.argv.indexOf(`--${name}`);
|
|
25
|
+
return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : def;
|
|
26
|
+
};
|
|
27
|
+
|
|
28
|
+
const task = arg('task');
|
|
29
|
+
if (!task) {
|
|
30
|
+
console.error('ERROR: --task is required (what were you trying to do?)');
|
|
31
|
+
process.exit(2);
|
|
32
|
+
}
|
|
33
|
+
const tried = arg('tried');
|
|
34
|
+
const worked = arg('worked');
|
|
35
|
+
const critique = arg('critique');
|
|
36
|
+
const outcome = arg('outcome', 'success');
|
|
37
|
+
const dir = path.resolve(arg('dir', process.cwd()));
|
|
38
|
+
const ns = arg('namespace', 'lessons');
|
|
39
|
+
const slug =
|
|
40
|
+
arg('slug') ||
|
|
41
|
+
task.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').slice(0, 40);
|
|
42
|
+
|
|
43
|
+
const db = path.join(dir, '.swarm', 'memory.db');
|
|
44
|
+
if (!fs.existsSync(db)) {
|
|
45
|
+
console.error(`ERROR: no AgentDB at ${db}\n -> run \`ruflo memory init\` in that project first.`);
|
|
46
|
+
process.exit(2);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const key = `lesson-${slug}`;
|
|
50
|
+
const value = [
|
|
51
|
+
`TASK: ${task}`,
|
|
52
|
+
tried ? `TRIED(failed): ${tried}` : null,
|
|
53
|
+
worked ? `WORKED: ${worked}` : null,
|
|
54
|
+
critique ? `CRITIQUE: ${critique}` : null,
|
|
55
|
+
`OUTCOME: ${outcome}`,
|
|
56
|
+
].filter(Boolean).join(' ');
|
|
57
|
+
|
|
58
|
+
const ruflo = (args) =>
|
|
59
|
+
execFileSync('ruflo', args, { cwd: dir, encoding: 'utf8', timeout: 60000 });
|
|
60
|
+
|
|
61
|
+
console.log(`\nRecording lesson into ${path.basename(dir)}/.swarm/memory.db (namespace: ${ns})`);
|
|
62
|
+
console.log(` key: ${key}`);
|
|
63
|
+
|
|
64
|
+
// 1. STORE (native, signal namespace) — L1 content + L2 embedding
|
|
65
|
+
let stored = false;
|
|
66
|
+
try {
|
|
67
|
+
const out = ruflo(['memory', 'store', '-k', key, '-n', ns, '--value', value]);
|
|
68
|
+
stored = /OK|stored/i.test(out);
|
|
69
|
+
} catch (e) {
|
|
70
|
+
console.error(' store FAILED:', String(e.stdout || e.message).split('\n')[0]);
|
|
71
|
+
process.exit(1);
|
|
72
|
+
}
|
|
73
|
+
console.log(` 1. store -> ${stored ? 'OK' : '?'}`);
|
|
74
|
+
|
|
75
|
+
// 2. REFINE (native) — L3 patterns + L4 episodes
|
|
76
|
+
let batchEpisodes = '?';
|
|
77
|
+
let distillOk = false;
|
|
78
|
+
try {
|
|
79
|
+
const dist = ruflo(['memory', 'distill', 'run']);
|
|
80
|
+
const m = dist.match(/Episodes\s*\|\s*(\d+)/i);
|
|
81
|
+
if (m) batchEpisodes = m[1];
|
|
82
|
+
distillOk = true;
|
|
83
|
+
} catch (e) {
|
|
84
|
+
/* distill is best-effort; the store already succeeded */
|
|
85
|
+
}
|
|
86
|
+
// DERIVED, not asserted (F15): say what actually happened — the old line printed "refined into
|
|
87
|
+
// episodes+patterns" even when distill threw.
|
|
88
|
+
console.log(distillOk
|
|
89
|
+
? ` 2. distill -> refined into episodes+patterns (batch: ${batchEpisodes})`
|
|
90
|
+
: ' 2. distill -> FAILED (best-effort; the raw lesson is stored, refinement will catch up on a later distill)');
|
|
91
|
+
|
|
92
|
+
// 3. VERIFY recall by the task text (paraphrase-ish), filtered to the namespace
|
|
93
|
+
let recalled = false;
|
|
94
|
+
try {
|
|
95
|
+
const search = ruflo(['memory', 'search', '-q', task, '-n', ns]);
|
|
96
|
+
recalled = search.includes(key.slice(0, 16));
|
|
97
|
+
} catch (e) {
|
|
98
|
+
/* search failure shouldn't fail the record */
|
|
99
|
+
}
|
|
100
|
+
console.log(
|
|
101
|
+
` 3. recall -> ${
|
|
102
|
+
recalled
|
|
103
|
+
? `✅ "${task.slice(0, 44)}…" returns ${key}`
|
|
104
|
+
: '⚠️ not the top in-namespace hit (stored fine; ranking improves as signal grows)'
|
|
105
|
+
}`,
|
|
106
|
+
);
|
|
107
|
+
|
|
108
|
+
// DERIVED, not asserted (F15): the closing line reports exactly what was verified, never more. The
|
|
109
|
+
// old line claimed "captured, refined, and recall-verified" even when distill failed and recall
|
|
110
|
+
// didn't return the key — asserted prose over an honest exit code.
|
|
111
|
+
const parts = ['captured', distillOk ? 'refined' : 'NOT refined (distill failed)', recalled ? 'recall-verified' : 'recall NOT verified'];
|
|
112
|
+
console.log(`\nDone. Lesson is ${parts.join(', ')}.\n`);
|
|
113
|
+
process.exit(stored ? 0 : 1);
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// scripts/refresh-model-catalog.mjs — pulls the LIVE OpenRouter model catalog and writes the committed
|
|
3
|
+
// snapshot data/openrouter-catalog-snapshot.json (with a pulledAt stamp). verify-model-catalog.mjs then
|
|
4
|
+
// enforces, offline in CI, that every model in data/model-catalog.json exists and is priced correctly
|
|
5
|
+
// against this snapshot — and that the snapshot is fresh. Run nightly (the anti-rot mechanism) + on demand.
|
|
6
|
+
// ADR-0016.
|
|
7
|
+
//
|
|
8
|
+
// It ALSO flags drift so a new flagship (e.g. a GPT-5.6-class release) surfaces instead of rotting:
|
|
9
|
+
// any catalog model that has VANISHED from the live catalog. OpenRouter /models is a free metadata
|
|
10
|
+
// endpoint (no generation, no spend).
|
|
11
|
+
//
|
|
12
|
+
// Usage: node scripts/refresh-model-catalog.mjs [--check] (--check: fail if the pull would change the snapshot)
|
|
13
|
+
|
|
14
|
+
import fs from 'node:fs';
|
|
15
|
+
import path from 'node:path';
|
|
16
|
+
import { fileURLToPath } from 'node:url';
|
|
17
|
+
|
|
18
|
+
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
19
|
+
const ROOT = path.resolve(__dirname, '..');
|
|
20
|
+
const SNAPSHOT = path.join(ROOT, 'data/openrouter-catalog-snapshot.json');
|
|
21
|
+
const CATALOG = path.join(ROOT, 'data/model-catalog.json');
|
|
22
|
+
|
|
23
|
+
/** Turn OpenRouter's /models `data` array into a compact {id: {in,out}} price map ($/Mtok). */
|
|
24
|
+
export function priceMap(orData) {
|
|
25
|
+
const models = {};
|
|
26
|
+
for (const m of orData || []) {
|
|
27
|
+
const p = m.pricing || {};
|
|
28
|
+
models[m.id] = { in: +(+p.prompt * 1e6).toFixed(4), out: +(+p.completion * 1e6).toFixed(4) };
|
|
29
|
+
}
|
|
30
|
+
return models;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** Which catalog models are no longer present in the live map (tolerating bare Claude ids)? */
|
|
34
|
+
export function detectDrift(catalog, models) {
|
|
35
|
+
const missing = [];
|
|
36
|
+
for (const [pid, p] of Object.entries(catalog.providers || {})) {
|
|
37
|
+
if (p.aliasOf) continue;
|
|
38
|
+
for (const tier of ['frontier', 'mid', 'cheap']) {
|
|
39
|
+
const id = p[tier]?.model;
|
|
40
|
+
if (!id) continue;
|
|
41
|
+
const found = models[id] || (!id.includes('/') && models['anthropic/' + id]);
|
|
42
|
+
if (!found) missing.push(`${pid}.${tier} "${id}"`);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
return missing;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function shapeSnapshot(models, pulledAt) {
|
|
49
|
+
return { _meta: { source: 'OpenRouter /api/v1/models (live metadata)', pulledAt, modelCount: Object.keys(models).length }, models };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function loadKey() {
|
|
53
|
+
if (process.env.OPENROUTER_API_KEY) return process.env.OPENROUTER_API_KEY;
|
|
54
|
+
try {
|
|
55
|
+
const env = fs.readFileSync(path.join(ROOT, '.env'), 'utf8');
|
|
56
|
+
const m = env.match(/^OPENROUTER_API_KEY=(.+)$/m);
|
|
57
|
+
if (m) return m[1].trim().replace(/^["']|["']$/g, '');
|
|
58
|
+
} catch { /* no .env */ }
|
|
59
|
+
return null;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
async function pullLive() {
|
|
63
|
+
const key = loadKey();
|
|
64
|
+
const res = await fetch('https://openrouter.ai/api/v1/models', { headers: key ? { Authorization: `Bearer ${key}` } : {} });
|
|
65
|
+
if (!res.ok) throw new Error(`OpenRouter /models HTTP ${res.status}`);
|
|
66
|
+
return priceMap((await res.json()).data || []);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
async function main() {
|
|
70
|
+
const check = process.argv.includes('--check');
|
|
71
|
+
const models = await pullLive();
|
|
72
|
+
if (Object.keys(models).length < 50) throw new Error(`live catalog returned only ${Object.keys(models).length} models — refusing to overwrite the snapshot with a suspiciously thin pull`);
|
|
73
|
+
|
|
74
|
+
const catalog = JSON.parse(fs.readFileSync(CATALOG, 'utf8'));
|
|
75
|
+
const missing = detectDrift(catalog, models);
|
|
76
|
+
const snapshot = shapeSnapshot(models, new Date().toISOString());
|
|
77
|
+
const next = JSON.stringify(snapshot, null, 2) + '\n';
|
|
78
|
+
|
|
79
|
+
if (check) {
|
|
80
|
+
const cur = fs.existsSync(SNAPSHOT) ? JSON.parse(fs.readFileSync(SNAPSHOT, 'utf8')).models : null;
|
|
81
|
+
if (JSON.stringify(cur) !== JSON.stringify(models)) {
|
|
82
|
+
console.error('✗ snapshot is out of date vs the live OpenRouter catalog — run: node scripts/refresh-model-catalog.mjs');
|
|
83
|
+
process.exit(1);
|
|
84
|
+
}
|
|
85
|
+
console.log(`✓ snapshot matches live (${Object.keys(models).length} models).`);
|
|
86
|
+
return;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
fs.writeFileSync(SNAPSHOT, next);
|
|
90
|
+
console.log(`✓ wrote ${SNAPSHOT.replace(ROOT + '/', '')} — ${Object.keys(models).length} live models, pulledAt ${snapshot._meta.pulledAt}`);
|
|
91
|
+
if (missing.length) {
|
|
92
|
+
console.log(`\n⚠ DRIFT: ${missing.length} catalog model(s) no longer in the live catalog — update data/model-catalog.json:`);
|
|
93
|
+
for (const m of missing) console.log(' - ' + m);
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
98
|
+
main().catch((e) => { console.error('refresh-model-catalog FAILED:', e.message); process.exit(1); });
|
|
99
|
+
}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// One-publisher source gate for issue #77. Rebuild and maintenance jobs may prepare bytes, but
|
|
3
|
+
// only scripts/release.mjs may contain operations that create a GitHub Release, publish npm, or
|
|
4
|
+
// move an npm dist-tag. CI and the canonical release path both execute this check.
|
|
5
|
+
import fs from 'node:fs';
|
|
6
|
+
import path from 'node:path';
|
|
7
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
8
|
+
|
|
9
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
10
|
+
const CANONICAL_PUBLISHER = 'scripts/release.mjs';
|
|
11
|
+
const SOURCE_EXTENSIONS = new Set(['.mjs', '.js', '.cjs', '.sh']);
|
|
12
|
+
|
|
13
|
+
function executableSource(source) {
|
|
14
|
+
const withoutBlocks = source.replace(/\/\*[\s\S]*?\*\//g, '');
|
|
15
|
+
return withoutBlocks
|
|
16
|
+
.split('\n')
|
|
17
|
+
.filter((line) => !line.trimStart().startsWith('#'))
|
|
18
|
+
.map((line) => line.replace(/\/\/.*$/, ''))
|
|
19
|
+
.join('\n');
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
const ACTIONS = [
|
|
23
|
+
{
|
|
24
|
+
action: 'github-release-create',
|
|
25
|
+
jsPatterns: [
|
|
26
|
+
/\[\s*['"]release['"]\s*,\s*['"]create['"]/,
|
|
27
|
+
],
|
|
28
|
+
shellPatterns: [/^\s*gh\s+release\s+create\b/m],
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
action: 'npm-publish',
|
|
32
|
+
jsPatterns: [
|
|
33
|
+
/(?:execFileSync|spawnSync)\(\s*['"]npm['"]\s*,\s*\[\s*['"]publish['"]/,
|
|
34
|
+
/runOrDie\(\s*['"]npm publish['"]/,
|
|
35
|
+
],
|
|
36
|
+
shellPatterns: [/^\s*npm\s+publish\b/m],
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
action: 'npm-dist-tag',
|
|
40
|
+
jsPatterns: [
|
|
41
|
+
/(?:execFileSync|spawnSync)\(\s*['"]npm['"]\s*,\s*\[\s*['"]dist-tag['"]\s*,\s*['"]add['"]/,
|
|
42
|
+
/runOrDie\(\s*['"]npm dist-tag[^'"]*['"]/,
|
|
43
|
+
],
|
|
44
|
+
shellPatterns: [/^\s*npm\s+dist-tag\s+add\b/m],
|
|
45
|
+
},
|
|
46
|
+
];
|
|
47
|
+
|
|
48
|
+
export function detectPublisherActions(file, source) {
|
|
49
|
+
const relative = file.split(path.sep).join('/');
|
|
50
|
+
if (relative === CANONICAL_PUBLISHER) return [];
|
|
51
|
+
const executable = executableSource(source);
|
|
52
|
+
const kind = path.extname(relative) === '.sh' ? 'shellPatterns' : 'jsPatterns';
|
|
53
|
+
return ACTIONS
|
|
54
|
+
.filter((action) => action[kind].some((pattern) => pattern.test(executable)))
|
|
55
|
+
.map(({ action }) => ({ file: relative, action }));
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function sourceFiles(root) {
|
|
59
|
+
const files = [];
|
|
60
|
+
const visit = (absolute) => {
|
|
61
|
+
if (!fs.existsSync(absolute)) return;
|
|
62
|
+
for (const entry of fs.readdirSync(absolute, { withFileTypes: true })) {
|
|
63
|
+
const child = path.join(absolute, entry.name);
|
|
64
|
+
if (entry.isDirectory()) visit(child);
|
|
65
|
+
else if (entry.isFile() && SOURCE_EXTENSIONS.has(path.extname(entry.name))) files.push(child);
|
|
66
|
+
}
|
|
67
|
+
};
|
|
68
|
+
visit(path.join(root, 'scripts'));
|
|
69
|
+
visit(path.join(root, 'deploy'));
|
|
70
|
+
return files;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export function findUnauthorizedPublishers(root = ROOT) {
|
|
74
|
+
return sourceFiles(root).flatMap((absolute) => {
|
|
75
|
+
const relative = path.relative(root, absolute).split(path.sep).join('/');
|
|
76
|
+
return detectPublisherActions(relative, fs.readFileSync(absolute, 'utf8'));
|
|
77
|
+
});
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
export function main(root = ROOT) {
|
|
81
|
+
const findings = findUnauthorizedPublishers(root);
|
|
82
|
+
if (findings.length === 0) {
|
|
83
|
+
console.log('[release-authority] PASS: scripts/release.mjs is the only publisher');
|
|
84
|
+
return 0;
|
|
85
|
+
}
|
|
86
|
+
console.error('[release-authority] FAIL: publication action found outside scripts/release.mjs');
|
|
87
|
+
for (const finding of findings) console.error(` ${finding.file}: ${finding.action}`);
|
|
88
|
+
return 1;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
if (process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url) {
|
|
92
|
+
process.exitCode = main();
|
|
93
|
+
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
export * from '../plugin/skills/release-proof/scripts/release-proof.mjs';
|
|
3
|
+
import { main } from '../plugin/skills/release-proof/scripts/release-proof.mjs';
|
|
4
|
+
import path from 'node:path';
|
|
5
|
+
import { pathToFileURL } from 'node:url';
|
|
6
|
+
|
|
7
|
+
if (process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url) {
|
|
8
|
+
process.exitCode = main(process.argv.slice(2));
|
|
9
|
+
}
|