ruvnet-brain 4.0.4 → 4.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/hooks/codex-hooks.json +17 -17
- package/plugin/scripts/codex-hook-wrapper.mjs +36 -4
- package/plugin/scripts/lesson-command-scope.mjs +133 -0
- package/plugin/scripts/lesson-gate.mjs +401 -0
- package/plugin/scripts/lesson-presentation.mjs +99 -0
- package/plugin/scripts/lesson-store.mjs +452 -0
- package/plugin/scripts/route-dispatch.sh +1 -0
- package/plugin/scripts/verify-interface.sh +1 -0
- package/scripts/learning-replay-cli.mjs +236 -0
- package/scripts/learning-replay-contract.mjs +255 -0
- package/scripts/learning-replay-execution.mjs +193 -0
- package/scripts/learning-replay-fixture.mjs +380 -0
- package/scripts/learning-replay-proof.mjs +459 -0
- package/scripts/learning-replay.mjs +10 -1565
- package/scripts/lesson-gate.mjs +3 -679
- package/scripts/lesson-store.mjs +4 -447
- package/scripts/memory-doctor.mjs +19 -3
- package/scripts/onboarding-console.mjs +77 -43
- package/scripts/release-vector.mjs +49 -25
- package/scripts/stabilization-receipt.mjs +1 -1
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
import fs from 'node:fs';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
import { spawnSync } from 'node:child_process';
|
|
4
|
+
import { loadLessons, saveLessons } from './lesson-store.mjs';
|
|
5
|
+
import {
|
|
6
|
+
EXIT,
|
|
7
|
+
INVARIANT,
|
|
8
|
+
MUTANTS,
|
|
9
|
+
MUTANT_RESULT_FILES,
|
|
10
|
+
PORTFOLIO_RESULT_FILES,
|
|
11
|
+
TRAP,
|
|
12
|
+
VERDICT,
|
|
13
|
+
WRONG_SUBCOMMAND_COMMAND,
|
|
14
|
+
aggregate,
|
|
15
|
+
trapSpec,
|
|
16
|
+
verdictForRun,
|
|
17
|
+
} from './learning-replay-contract.mjs';
|
|
18
|
+
import { verifyPostTaskContract, verifyRufloFlag } from './learning-replay-execution.mjs';
|
|
19
|
+
import {
|
|
20
|
+
allocateRunBase,
|
|
21
|
+
buildFixtures,
|
|
22
|
+
nightlyRefresh,
|
|
23
|
+
recordInProjectA,
|
|
24
|
+
removeFixture,
|
|
25
|
+
runArm,
|
|
26
|
+
seedProjectBMemory,
|
|
27
|
+
} from './learning-replay-fixture.mjs';
|
|
28
|
+
import {
|
|
29
|
+
checkArtifact,
|
|
30
|
+
checkMutantArtifacts,
|
|
31
|
+
checkPortfolio,
|
|
32
|
+
checkSourceIdentity,
|
|
33
|
+
cleanupFixtureDaemons,
|
|
34
|
+
pruneArchive,
|
|
35
|
+
writeArtifact,
|
|
36
|
+
} from './learning-replay-proof.mjs';
|
|
37
|
+
import { ROOT } from './learning-replay-contract.mjs';
|
|
38
|
+
|
|
39
|
+
const usage = () => `Usage:
|
|
40
|
+
node scripts/learning-replay.mjs [--trap ${TRAP.MEMORY_SEARCH}|${TRAP.POST_TASK}] [--n N] [--host codex|claude-code] [--model MODEL]
|
|
41
|
+
node scripts/learning-replay.mjs --check
|
|
42
|
+
node scripts/learning-replay.mjs --check-portfolio
|
|
43
|
+
node scripts/learning-replay.mjs --check-mutants
|
|
44
|
+
node scripts/learning-replay.mjs --dry-run
|
|
45
|
+
node scripts/learning-replay.mjs --mutant <${Object.keys(MUTANTS).join('|')}>
|
|
46
|
+
|
|
47
|
+
Exit: 0=PASS, 1=FAIL, 3=INCONCLUSIVE, 4=UNKNOWN.`;
|
|
48
|
+
|
|
49
|
+
function parse(argv) {
|
|
50
|
+
const has = (flag) => argv.includes(flag);
|
|
51
|
+
const arg = (flag, fallback) => {
|
|
52
|
+
const index = argv.indexOf(flag);
|
|
53
|
+
return index >= 0 && argv[index + 1] ? argv[index + 1] : fallback;
|
|
54
|
+
};
|
|
55
|
+
return { has, arg };
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function wireProbe(dirs, spec, stateDir) {
|
|
59
|
+
const result = spawnSync(process.execPath, [
|
|
60
|
+
path.join(ROOT, 'plugin', 'scripts', 'hook-shim.mjs'),
|
|
61
|
+
'unprompted-speech', 'UserPromptSubmit',
|
|
62
|
+
], {
|
|
63
|
+
input: JSON.stringify({
|
|
64
|
+
prompt: spec.prompt,
|
|
65
|
+
session_id: `dry-${Date.now()}`,
|
|
66
|
+
cwd: dirs.projectB,
|
|
67
|
+
}),
|
|
68
|
+
encoding: 'utf8',
|
|
69
|
+
cwd: dirs.projectB,
|
|
70
|
+
env: {
|
|
71
|
+
...process.env,
|
|
72
|
+
RUVNET_BRAIN_HOME: dirs.brainHome,
|
|
73
|
+
RUVNET_BRAIN_STATE_DIR: stateDir,
|
|
74
|
+
RUVNET_CONFIG_ROOT: path.join(dirs.base, 'config'),
|
|
75
|
+
RUVNET_LESSON_STORE: dirs.lessons,
|
|
76
|
+
RUVNET_LESSON_GATE_STATE: dirs.gateState,
|
|
77
|
+
CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin'),
|
|
78
|
+
},
|
|
79
|
+
});
|
|
80
|
+
return (result.stdout || '').length;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
function printed(label, result) {
|
|
84
|
+
console.log(`\n ${label}: ${result.status}\n ${result.why}\n`);
|
|
85
|
+
return EXIT[result.status] ?? EXIT.UNKNOWN;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export async function main(argv = process.argv.slice(2)) {
|
|
89
|
+
const { has, arg } = parse(argv);
|
|
90
|
+
if (has('--help') || has('-h')) {
|
|
91
|
+
console.log(usage());
|
|
92
|
+
return EXIT.PASS;
|
|
93
|
+
}
|
|
94
|
+
const mutant = arg('--mutant', null);
|
|
95
|
+
const trap = arg('--trap', TRAP.MEMORY_SEARCH);
|
|
96
|
+
const host = arg('--host', 'codex');
|
|
97
|
+
const model = arg('--model', host === 'codex' ? 'gpt-5.6-sol' : 'haiku');
|
|
98
|
+
const n = Math.max(1, parseInt(arg('--n', mutant ? '1' : '3'), 10) || 1);
|
|
99
|
+
const outFile = arg('--out', mutant
|
|
100
|
+
? MUTANT_RESULT_FILES[trap]?.[mutant]
|
|
101
|
+
: PORTFOLIO_RESULT_FILES[trap]);
|
|
102
|
+
if (mutant && !MUTANTS[mutant]) {
|
|
103
|
+
console.error(`unknown mutant "${mutant}"`);
|
|
104
|
+
return EXIT.UNKNOWN;
|
|
105
|
+
}
|
|
106
|
+
if (![TRAP.MEMORY_SEARCH, TRAP.POST_TASK].includes(trap) || !outFile) {
|
|
107
|
+
console.error(`unsupported trap/mutant: ${trap}/${mutant || 'normal'}`);
|
|
108
|
+
return EXIT.UNKNOWN;
|
|
109
|
+
}
|
|
110
|
+
if (has('--check')) {
|
|
111
|
+
return printed(INVARIANT, checkArtifact({ file: PORTFOLIO_RESULT_FILES[trap] }));
|
|
112
|
+
}
|
|
113
|
+
if (has('--check-portfolio')) return printed(`${INVARIANT}-PORTFOLIO`, checkPortfolio());
|
|
114
|
+
if (has('--check-mutants')) return printed(`${INVARIANT}-MUTANTS`, checkMutantArtifacts());
|
|
115
|
+
|
|
116
|
+
const source = checkSourceIdentity();
|
|
117
|
+
if (!source.clean) {
|
|
118
|
+
console.error(` ${INVARIANT}: UNKNOWN — ${source.why}`);
|
|
119
|
+
return EXIT.UNKNOWN;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
const spec = trapSpec(trap);
|
|
123
|
+
const premise = trap === TRAP.POST_TASK ? verifyPostTaskContract() : verifyRufloFlag();
|
|
124
|
+
console.log(`\n=== ${INVARIANT} — ${trap} ===`);
|
|
125
|
+
console.log(` premise: ${premise.ok ? `VERIFIED: ${premise.evidence}` : `NOT VERIFIED: ${premise.why}`}`);
|
|
126
|
+
if (!premise.ok) {
|
|
127
|
+
writeArtifact(outFile, aggregate([]), { host, model, mutant, trap, task: spec.prompt });
|
|
128
|
+
return EXIT.UNKNOWN;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
const base = allocateRunBase();
|
|
132
|
+
const dirs = buildFixtures(base);
|
|
133
|
+
process.once('exit', () => cleanupFixtureDaemons(dirs));
|
|
134
|
+
const record = recordInProjectA(dirs, { trap });
|
|
135
|
+
const seed = trap === TRAP.MEMORY_SEARCH
|
|
136
|
+
? seedProjectBMemory(dirs)
|
|
137
|
+
: { key: null, storeExit: null, ok: true, skipped: 'not required' };
|
|
138
|
+
const refresh = nightlyRefresh(dirs);
|
|
139
|
+
console.log(` record: ${record.projectCount} sources; promoted=${record.promoted}; ok=${record.ok}`);
|
|
140
|
+
console.log(` seed: ${seed.skipped || `ok=${seed.ok}`}`);
|
|
141
|
+
console.log(` refresh: ${refresh.generation}; lessonSurvived=${refresh.lessonSurvived}`);
|
|
142
|
+
|
|
143
|
+
if (mutant === 'delete-lesson') {
|
|
144
|
+
saveLessons([], dirs.lessons);
|
|
145
|
+
removeFixture(path.join(dirs.projectA, '.swarm', 'memory.db'));
|
|
146
|
+
console.log(` mutant delete-lesson: ${loadLessons(dirs.lessons).length} lessons remain`);
|
|
147
|
+
}
|
|
148
|
+
if (mutant === 'empty-store') {
|
|
149
|
+
removeFixture(path.join(dirs.projectB, '.swarm', 'memory.db'));
|
|
150
|
+
console.log(' mutant empty-store: target memory removed');
|
|
151
|
+
}
|
|
152
|
+
if (has('--dry-run')) {
|
|
153
|
+
const treatedBytes = wireProbe(dirs, spec, dirs.stateOn);
|
|
154
|
+
const controlBytes = wireProbe(dirs, spec, dirs.stateOff);
|
|
155
|
+
const dry = aggregate([]);
|
|
156
|
+
dry.why = `--dry-run: no model called; treated ${treatedBytes}B, control ${controlBytes}B`;
|
|
157
|
+
writeArtifact(outFile, dry, {
|
|
158
|
+
host, model, mutant, trap, task: spec.prompt, record, seed, refresh, flag: premise,
|
|
159
|
+
});
|
|
160
|
+
if (!has('--keep-fixtures')) {
|
|
161
|
+
cleanupFixtureDaemons(dirs);
|
|
162
|
+
removeFixture(base);
|
|
163
|
+
}
|
|
164
|
+
return EXIT.UNKNOWN;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const runs = [];
|
|
168
|
+
for (let i = 1; i <= n; i++) {
|
|
169
|
+
const treated = runArm({
|
|
170
|
+
dirs,
|
|
171
|
+
arm: 'treated',
|
|
172
|
+
stateDir: mutant === 'brain-off-treated' ? dirs.stateOff : dirs.stateOn,
|
|
173
|
+
model,
|
|
174
|
+
host,
|
|
175
|
+
tag: `run${i}-treated`,
|
|
176
|
+
trap,
|
|
177
|
+
forceCommand: mutant === 'wrong-subcommand' ? WRONG_SUBCOMMAND_COMMAND : null,
|
|
178
|
+
});
|
|
179
|
+
const control = runArm({
|
|
180
|
+
dirs,
|
|
181
|
+
arm: 'control',
|
|
182
|
+
stateDir: dirs.stateOff,
|
|
183
|
+
model,
|
|
184
|
+
host,
|
|
185
|
+
tag: `run${i}-control`,
|
|
186
|
+
trap,
|
|
187
|
+
appendSystemPrompt: mutant === 'seed-control' ? spec.lesson : null,
|
|
188
|
+
});
|
|
189
|
+
const run = {
|
|
190
|
+
i,
|
|
191
|
+
treatedClass: treated.class,
|
|
192
|
+
controlClass: control.class,
|
|
193
|
+
lessonBeforeFirstToolCall: treated.lessonBeforeFirstToolCall,
|
|
194
|
+
controlLessonDelivered: control.lessonDelivered,
|
|
195
|
+
treatedSubcommandCorrect: treated.subcommandCorrect,
|
|
196
|
+
treatedExecOk: treated.exec?.exitOk === true,
|
|
197
|
+
treatedRetrieved: treated.exec?.retrieved === true,
|
|
198
|
+
treatedExecWhy: treated.exec?.why || null,
|
|
199
|
+
controlWorked: control.exec?.exitOk === true && control.exec?.retrieved === true,
|
|
200
|
+
treated,
|
|
201
|
+
control,
|
|
202
|
+
error: treated.spawnError || control.spawnError || null,
|
|
203
|
+
};
|
|
204
|
+
const verdict = verdictForRun(run);
|
|
205
|
+
console.log(` run ${i}: ${verdict.verdict} — ${verdict.why}`);
|
|
206
|
+
runs.push(run);
|
|
207
|
+
}
|
|
208
|
+
const result = aggregate(runs);
|
|
209
|
+
const costUsd = runs.reduce((sum, run) =>
|
|
210
|
+
sum + (run.treated.costUsd || 0) + (run.control.costUsd || 0), 0);
|
|
211
|
+
const wallMs = runs.reduce((sum, run) =>
|
|
212
|
+
sum + (run.treated.wallMs || 0) + (run.control.wallMs || 0), 0);
|
|
213
|
+
const artifact = writeArtifact(outFile, result, {
|
|
214
|
+
host,
|
|
215
|
+
model,
|
|
216
|
+
mutant,
|
|
217
|
+
trap,
|
|
218
|
+
task: spec.prompt,
|
|
219
|
+
record,
|
|
220
|
+
seed,
|
|
221
|
+
refresh,
|
|
222
|
+
flag: premise,
|
|
223
|
+
costUsd,
|
|
224
|
+
wallMs,
|
|
225
|
+
});
|
|
226
|
+
if (!has('--keep-fixtures')) {
|
|
227
|
+
const preservePaths = artifact.runs.flatMap((run) => [
|
|
228
|
+
path.resolve(ROOT, run.treated.transcript.path),
|
|
229
|
+
path.resolve(ROOT, run.control.transcript.path),
|
|
230
|
+
]);
|
|
231
|
+
pruneArchive(dirs, { preservePaths });
|
|
232
|
+
}
|
|
233
|
+
console.log(` ${INVARIANT}: ${result.verdict} — ${result.why}`);
|
|
234
|
+
console.log(` artifact: ${path.relative(ROOT, outFile)}`);
|
|
235
|
+
return EXIT[result.verdict] ?? EXIT.UNKNOWN;
|
|
236
|
+
}
|
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
import path from 'node:path';
|
|
2
|
+
import { fileURLToPath } from 'node:url';
|
|
3
|
+
import { findInvocations } from '../plugin/scripts/hook-input.mjs';
|
|
4
|
+
|
|
5
|
+
export const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
6
|
+
export const INVARIANT = 'LEARNING-REPLAY';
|
|
7
|
+
export const VERDICT = Object.freeze({
|
|
8
|
+
PASS: 'PASS',
|
|
9
|
+
FAIL: 'FAIL',
|
|
10
|
+
UNKNOWN: 'UNKNOWN',
|
|
11
|
+
INCONCLUSIVE: 'INCONCLUSIVE',
|
|
12
|
+
});
|
|
13
|
+
export const EXIT = Object.freeze({ PASS: 0, FAIL: 1, INCONCLUSIVE: 3, UNKNOWN: 4 });
|
|
14
|
+
export const RESULT_FILE = path.join(ROOT, 'data', 'learning-replay-result.json');
|
|
15
|
+
export const POST_TASK_RESULT_FILE = path.join(ROOT, 'data', 'learning-replay-post-task-result.json');
|
|
16
|
+
export const TRAP = Object.freeze({
|
|
17
|
+
MEMORY_SEARCH: 'memory-search-query',
|
|
18
|
+
POST_TASK: 'hooks-post-task-persistence',
|
|
19
|
+
});
|
|
20
|
+
|
|
21
|
+
export const REPLAY_PROMPT =
|
|
22
|
+
'Earlier in this project someone recorded a note about the caching strategy. '
|
|
23
|
+
+ "Recall it from this project's agent memory with the ruflo CLI. "
|
|
24
|
+
+ 'Run the recall command now, then tell me what you ran.';
|
|
25
|
+
export const LESSON_STATEMENT =
|
|
26
|
+
'When you look something up in agent memory with the ruflo CLI, run exactly '
|
|
27
|
+
+ '`ruflo memory search -q "<query>" --path .swarm/memory.db` as the first and only Ruflo invocation; '
|
|
28
|
+
+ 'a bare quoted phrase placed after the subcommand is rejected.';
|
|
29
|
+
export const POST_TASK_PROMPT =
|
|
30
|
+
'Record successful completion of the release retry-budget investigation in ruflo hooks learning, '
|
|
31
|
+
+ 'assign it to the tester agent, and persist the routing decision for later sessions. '
|
|
32
|
+
+ 'Use the terminal CLI, not an MCP tool. Do not inspect help or run any setup command. '
|
|
33
|
+
+ 'Run exactly one post-task command now, then tell me what you ran.';
|
|
34
|
+
export const POST_TASK_LESSON =
|
|
35
|
+
'A ruflo hooks post-task call persists a reusable routing outcome only when it includes all three: '
|
|
36
|
+
+ '--task with the task description, --agent with the executor, and --store-results; a success/task-id alone '
|
|
37
|
+
+ 'may print success but does not create the routing decision that later learning reads.';
|
|
38
|
+
|
|
39
|
+
export function trapSpec(id = TRAP.MEMORY_SEARCH) {
|
|
40
|
+
if (id === TRAP.POST_TASK) {
|
|
41
|
+
return {
|
|
42
|
+
id,
|
|
43
|
+
lessonId: 'FX-D4-ruflo-hooks-post-task-persistence',
|
|
44
|
+
prompt: POST_TASK_PROMPT,
|
|
45
|
+
lesson: POST_TASK_LESSON,
|
|
46
|
+
memoryKey: 'lesson-ruflo-hooks-post-task-persistence',
|
|
47
|
+
recordQuery: 'ruflo hooks post task routing persistence',
|
|
48
|
+
check: 'the produced ruflo hooks post-task command includes --task, --agent, and --store-results',
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
return {
|
|
52
|
+
id: TRAP.MEMORY_SEARCH,
|
|
53
|
+
lessonId: 'FX-D4-ruflo-memory-search-flag',
|
|
54
|
+
prompt: REPLAY_PROMPT,
|
|
55
|
+
lesson: LESSON_STATEMENT,
|
|
56
|
+
memoryKey: 'lesson-ruflo-memory-search-flag',
|
|
57
|
+
recordQuery: 'ruflo CLI memory query flag',
|
|
58
|
+
check: 'the produced ruflo memory search command delivers its query through -q/--query',
|
|
59
|
+
};
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export const LOAD_BEARING = Object.freeze([
|
|
63
|
+
'scripts/learning-replay.mjs',
|
|
64
|
+
'scripts/learning-replay-contract.mjs',
|
|
65
|
+
'scripts/learning-replay-execution.mjs',
|
|
66
|
+
'scripts/learning-replay-fixture.mjs',
|
|
67
|
+
'scripts/learning-replay-proof.mjs',
|
|
68
|
+
'scripts/learning-replay-cli.mjs',
|
|
69
|
+
'scripts/ci/learning-replay-recorder.mjs',
|
|
70
|
+
'scripts/ci/learning-replay-codex-adapter.mjs',
|
|
71
|
+
'scripts/lesson-store.mjs',
|
|
72
|
+
'plugin/scripts/lesson-store.mjs',
|
|
73
|
+
'plugin/scripts/lesson-gate.mjs',
|
|
74
|
+
'plugin/scripts/lesson-command-scope.mjs',
|
|
75
|
+
'plugin/scripts/lesson-presentation.mjs',
|
|
76
|
+
'plugin/scripts/lesson-hooks.sh',
|
|
77
|
+
'plugin/scripts/unprompted-runtime.mjs',
|
|
78
|
+
'plugin/scripts/hook-shim.mjs',
|
|
79
|
+
'plugin/scripts/hook-input.mjs',
|
|
80
|
+
]);
|
|
81
|
+
|
|
82
|
+
export function classifyCommand(cmd) {
|
|
83
|
+
const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
|
|
84
|
+
if (!invocations.length) return 'none';
|
|
85
|
+
let sawPositional = false;
|
|
86
|
+
for (const inv of invocations) {
|
|
87
|
+
const args = inv.args.filter((a) => a !== '');
|
|
88
|
+
if (args.some((a) => a === '-q' || a === '--query' || a.startsWith('--query='))) return 'flagged';
|
|
89
|
+
const bare = [];
|
|
90
|
+
for (let i = 0; i < args.length; i++) {
|
|
91
|
+
const value = args[i];
|
|
92
|
+
if (value.startsWith('-')) {
|
|
93
|
+
if (!value.includes('=') && args[i + 1] && !args[i + 1].startsWith('-')) i++;
|
|
94
|
+
continue;
|
|
95
|
+
}
|
|
96
|
+
bare.push(value);
|
|
97
|
+
}
|
|
98
|
+
const queryish = (value) => /\s/.test(value) || value.length > 24;
|
|
99
|
+
if (bare.length >= 3 || bare.slice(1).some(queryish)) sawPositional = true;
|
|
100
|
+
}
|
|
101
|
+
return sawPositional ? 'positional' : 'other';
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function subcommandCorrect(cmd) {
|
|
105
|
+
for (const inv of findInvocations(String(cmd || ''), ['ruflo', 'claude-flow'])) {
|
|
106
|
+
const words = inv.args.filter((a) => a !== '' && !a.startsWith('-'));
|
|
107
|
+
const memory = words.indexOf('memory');
|
|
108
|
+
if (memory !== -1 && words[memory + 1] === 'search') return true;
|
|
109
|
+
}
|
|
110
|
+
return false;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
export const carriesToken = (classification) => classification === 'flagged';
|
|
114
|
+
|
|
115
|
+
export function optionValue(args, short, long) {
|
|
116
|
+
for (let i = 0; i < args.length; i++) {
|
|
117
|
+
const value = args[i];
|
|
118
|
+
if (value === short || value === long) {
|
|
119
|
+
return args[i + 1] && !args[i + 1].startsWith('-') ? args[i + 1] : null;
|
|
120
|
+
}
|
|
121
|
+
if (value.startsWith(`${long}=`)) return value.slice(long.length + 1);
|
|
122
|
+
}
|
|
123
|
+
return null;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
export function classifyPostTaskCommand(cmd) {
|
|
127
|
+
const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
|
|
128
|
+
if (!invocations.length) return 'none';
|
|
129
|
+
let sawPostTask = false;
|
|
130
|
+
for (const inv of invocations) {
|
|
131
|
+
const args = inv.args.filter(Boolean);
|
|
132
|
+
const hooks = args.indexOf('hooks');
|
|
133
|
+
if (hooks === -1 || args[hooks + 1] !== 'post-task') continue;
|
|
134
|
+
sawPostTask = true;
|
|
135
|
+
const task = optionValue(args, '-t', '--task');
|
|
136
|
+
const agent = optionValue(args, '-a', '--agent');
|
|
137
|
+
const store = args.includes('--store-results')
|
|
138
|
+
|| args.some((a) => a.startsWith('--store-results=') && !/=false$/i.test(a));
|
|
139
|
+
if (task && agent && store) return 'flagged';
|
|
140
|
+
}
|
|
141
|
+
return sawPostTask ? 'partial' : 'other';
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
export function postTaskSubcommandCorrect(cmd) {
|
|
145
|
+
return findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']).some((inv) => {
|
|
146
|
+
const words = inv.args.filter((a) => a !== '' && !a.startsWith('-'));
|
|
147
|
+
const hooks = words.indexOf('hooks');
|
|
148
|
+
return hooks !== -1 && words[hooks + 1] === 'post-task';
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
export function verdictForRun(run) {
|
|
153
|
+
const {
|
|
154
|
+
treatedClass, controlClass, lessonBeforeFirstToolCall, error,
|
|
155
|
+
treatedSubcommandCorrect, treatedExecOk, treatedRetrieved, treatedExecWhy, controlWorked,
|
|
156
|
+
} = run;
|
|
157
|
+
if (error) return { verdict: VERDICT.UNKNOWN, why: `harness could not measure this run: ${error}` };
|
|
158
|
+
if (controlClass === 'none') {
|
|
159
|
+
return { verdict: VERDICT.UNKNOWN, why: 'the control arm produced no ruflo invocation at all — there is no comparable artifact to difference against' };
|
|
160
|
+
}
|
|
161
|
+
if (treatedClass === 'none') return { verdict: VERDICT.FAIL, why: 'the treated arm produced no ruflo invocation at all' };
|
|
162
|
+
if (carriesToken(controlClass) || controlWorked === true) {
|
|
163
|
+
return {
|
|
164
|
+
verdict: VERDICT.INCONCLUSIVE,
|
|
165
|
+
why: carriesToken(controlClass)
|
|
166
|
+
? `the CONTROL arm produced the token (${controlClass}) — the model would have got it right without the lesson, so this trap measured nothing`
|
|
167
|
+
: `the CONTROL arm's command EXECUTED AND RETRIEVED (class "${controlClass}") — the model would have got it right without the lesson, so this trap measured nothing`,
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
if (!carriesToken(treatedClass)) return { verdict: VERDICT.FAIL, why: `treated arm produced "${treatedClass}", not the token` };
|
|
171
|
+
if (treatedSubcommandCorrect !== true) {
|
|
172
|
+
return { verdict: VERDICT.FAIL, why: 'treated arm carried the token on the WRONG SUBCOMMAND — the corrective action is unusable' };
|
|
173
|
+
}
|
|
174
|
+
if (treatedExecOk !== true) {
|
|
175
|
+
return { verdict: VERDICT.FAIL, why: `treated arm's produced command did not execute successfully: ${treatedExecWhy || 'not executed'}` };
|
|
176
|
+
}
|
|
177
|
+
if (treatedRetrieved !== true) {
|
|
178
|
+
return { verdict: VERDICT.FAIL, why: `treated arm's command exited 0 but RETRIEVED NOTHING: ${treatedExecWhy || 'no retrieval evidence'}` };
|
|
179
|
+
}
|
|
180
|
+
if (lessonBeforeFirstToolCall !== true) {
|
|
181
|
+
return { verdict: VERDICT.FAIL, why: 'treated arm carried the token but the lesson was NOT observed before the first tool call' };
|
|
182
|
+
}
|
|
183
|
+
return { verdict: VERDICT.PASS, why: `treated "${treatedClass}" vs control "${controlClass}"; the produced command executed and returned the required outcome` };
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
export function aggregate(runs, { threshold = 2 / 3 } = {}) {
|
|
187
|
+
const perRun = runs.map((run) => ({ ...run, ...verdictForRun(run) }));
|
|
188
|
+
const n = perRun.length;
|
|
189
|
+
const passes = perRun.filter((run) => run.verdict === VERDICT.PASS).length;
|
|
190
|
+
const fails = perRun.filter((run) => run.verdict === VERDICT.FAIL).length;
|
|
191
|
+
const unknowns = perRun.filter((run) => run.verdict === VERDICT.UNKNOWN).length;
|
|
192
|
+
const controlTokenRuns = perRun.filter((run) => carriesToken(run.controlClass)).length;
|
|
193
|
+
const controlWorkedRuns = perRun.filter((run) => run.controlWorked === true).length;
|
|
194
|
+
const treatedTokenRuns = perRun.filter((run) => carriesToken(run.treatedClass)).length;
|
|
195
|
+
const treatedSubcommandRuns = perRun.filter((run) => run.treatedSubcommandCorrect === true).length;
|
|
196
|
+
const treatedExecutedRuns = perRun.filter((run) => run.treatedExecOk === true).length;
|
|
197
|
+
const treatedRetrievedRuns = perRun.filter((run) => run.treatedRetrieved === true).length;
|
|
198
|
+
let verdict;
|
|
199
|
+
let why;
|
|
200
|
+
if (n === 0) {
|
|
201
|
+
verdict = VERDICT.UNKNOWN;
|
|
202
|
+
why = 'zero runs executed — an empty run is not a pass';
|
|
203
|
+
} else if (controlTokenRuns > 0 || controlWorkedRuns > 0) {
|
|
204
|
+
verdict = VERDICT.INCONCLUSIVE;
|
|
205
|
+
why = `${controlTokenRuns}/${n} CONTROL run(s) produced the token and ${controlWorkedRuns}/${n} executed+retrieved — the trap is invalid`;
|
|
206
|
+
} else if (passes / n >= threshold) {
|
|
207
|
+
verdict = VERDICT.PASS;
|
|
208
|
+
why = `${passes}/${n} runs passed (bar ${Math.ceil(threshold * n)}/${n})`;
|
|
209
|
+
} else if (unknowns > 0) {
|
|
210
|
+
verdict = VERDICT.UNKNOWN;
|
|
211
|
+
const error = perRun.map((run) => run.error).find(Boolean);
|
|
212
|
+
why = error
|
|
213
|
+
? `${unknowns}/${n} run(s) could not be measured; executor error: ${error}`
|
|
214
|
+
: `${unknowns}/${n} run(s) could not be measured; ${passes}/${n} passed`;
|
|
215
|
+
} else {
|
|
216
|
+
verdict = VERDICT.FAIL;
|
|
217
|
+
why = `${passes}/${n} runs passed — below the ${Math.ceil(threshold * n)}/${n} bar`;
|
|
218
|
+
}
|
|
219
|
+
if (verdict === VERDICT.PASS && (controlTokenRuns > 0 || controlWorkedRuns > 0)) {
|
|
220
|
+
throw new Error('LEARNING-REPLAY: refusing PASS while a control succeeded');
|
|
221
|
+
}
|
|
222
|
+
const brokenPass = perRun.find((run) => run.verdict === VERDICT.PASS
|
|
223
|
+
&& (run.treatedSubcommandCorrect !== true || run.treatedExecOk !== true || run.treatedRetrieved !== true));
|
|
224
|
+
if (brokenPass) throw new Error(`LEARNING-REPLAY: unusable PASS run ${brokenPass.i}`);
|
|
225
|
+
return {
|
|
226
|
+
verdict, why, n, passes, fails, unknowns,
|
|
227
|
+
controlTokenRuns, controlWorkedRuns, treatedTokenRuns,
|
|
228
|
+
treatedSubcommandRuns, treatedExecutedRuns, treatedRetrievedRuns,
|
|
229
|
+
rate: n ? +(passes / n).toFixed(4) : 0,
|
|
230
|
+
runs: perRun,
|
|
231
|
+
};
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
export const WRONG_SUBCOMMAND_COMMAND = 'ruflo recall -q "caching strategy"';
|
|
235
|
+
export const MUTANTS = Object.freeze({
|
|
236
|
+
'delete-lesson': 'delete the recorded lesson after refresh',
|
|
237
|
+
'brain-off-treated': 'run the treated arm with the brain disabled',
|
|
238
|
+
'seed-control': 'pre-seed the control arm with the lesson',
|
|
239
|
+
'wrong-subcommand': 'substitute the treated artifact with a right flag on a wrong verb',
|
|
240
|
+
'empty-store': 'remove the seeded target memory before the gate runs',
|
|
241
|
+
});
|
|
242
|
+
export const MUTANT_RESULT_FILES = Object.freeze({
|
|
243
|
+
[TRAP.MEMORY_SEARCH]: Object.freeze({
|
|
244
|
+
'delete-lesson': path.join(ROOT, 'data', 'learning-replay-delete-lesson-result.json'),
|
|
245
|
+
'brain-off-treated': path.join(ROOT, 'data', 'learning-replay-brain-off-result.json'),
|
|
246
|
+
}),
|
|
247
|
+
[TRAP.POST_TASK]: Object.freeze({
|
|
248
|
+
'delete-lesson': path.join(ROOT, 'data', 'learning-replay-post-task-delete-lesson-result.json'),
|
|
249
|
+
'brain-off-treated': path.join(ROOT, 'data', 'learning-replay-post-task-brain-off-result.json'),
|
|
250
|
+
}),
|
|
251
|
+
});
|
|
252
|
+
export const PORTFOLIO_RESULT_FILES = Object.freeze({
|
|
253
|
+
[TRAP.MEMORY_SEARCH]: RESULT_FILE,
|
|
254
|
+
[TRAP.POST_TASK]: POST_TASK_RESULT_FILE,
|
|
255
|
+
});
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
import fs from 'node:fs';
|
|
2
|
+
import os from 'node:os';
|
|
3
|
+
import path from 'node:path';
|
|
4
|
+
import { spawnSync } from 'node:child_process';
|
|
5
|
+
import { commandNodes, findInvocations } from '../plugin/scripts/hook-input.mjs';
|
|
6
|
+
import { optionValue, TRAP } from './learning-replay-contract.mjs';
|
|
7
|
+
|
|
8
|
+
export const PROJECT_B_MEMORY_KEY = 'note-caching-strategy';
|
|
9
|
+
export const PROJECT_B_MEMORY_VALUE =
|
|
10
|
+
'The caching strategy for this project: responses are memoized in a two-tier LRU, '
|
|
11
|
+
+ 'warm tier in memory and cold tier on disk, invalidated by content hash.';
|
|
12
|
+
export const RETRIEVAL_EVIDENCE = Object.freeze({
|
|
13
|
+
positive: Object.freeze([PROJECT_B_MEMORY_KEY.slice(0, 12), PROJECT_B_MEMORY_VALUE.slice(0, 20)]),
|
|
14
|
+
negative: Object.freeze([
|
|
15
|
+
/No results found/i,
|
|
16
|
+
/Unknown command/i,
|
|
17
|
+
/Required option missing/i,
|
|
18
|
+
/Usage:\s*claude-flow memory/i,
|
|
19
|
+
/\[ERROR\]/,
|
|
20
|
+
]),
|
|
21
|
+
});
|
|
22
|
+
|
|
23
|
+
export const RUFLO_BIN = process.env.RUVNET_RUFLO_BIN
|
|
24
|
+
|| path.join(os.homedir(), '.npm-global', 'bin', 'ruflo');
|
|
25
|
+
const MUTATING_SUBCOMMANDS = new Set([
|
|
26
|
+
'store', 'delete', 'rm', 'purge', 'cleanup', 'compress', 'import', 'export',
|
|
27
|
+
'backup', 'init', 'configure',
|
|
28
|
+
]);
|
|
29
|
+
|
|
30
|
+
function spawnRuflo(bin, args, options) {
|
|
31
|
+
return /\.[cm]?js$/i.test(bin)
|
|
32
|
+
? spawnSync(process.execPath, [bin, ...args], options)
|
|
33
|
+
: spawnSync(bin, args, options);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export function assertRetrieved(out) {
|
|
37
|
+
const text = String(out || '');
|
|
38
|
+
for (const pattern of RETRIEVAL_EVIDENCE.negative) {
|
|
39
|
+
if (pattern.test(text)) return { retrieved: false, why: `output matched failure shape ${pattern}` };
|
|
40
|
+
}
|
|
41
|
+
const hit = RETRIEVAL_EVIDENCE.positive.find((value) => text.includes(value));
|
|
42
|
+
if (!hit) return { retrieved: false, why: 'output names neither the seeded key nor its text' };
|
|
43
|
+
return { retrieved: true, why: `output carries the seeded memory (matched "${hit}")` };
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export function assertPostTaskPersisted({ args, output, cwd }) {
|
|
47
|
+
const task = optionValue(args, '-t', '--task');
|
|
48
|
+
const agent = optionValue(args, '-a', '--agent');
|
|
49
|
+
const taskId = optionValue(args, '-i', '--task-id')
|
|
50
|
+
|| String(output || '').match(/Recording outcome for task:\s*([a-zA-Z0-9_-]+)/)?.[1]
|
|
51
|
+
|| null;
|
|
52
|
+
if (!task || !agent || !taskId || !args.includes('--store-results')) {
|
|
53
|
+
return { retrieved: false, why: 'command lacked --task, --agent, --store-results, or a task id' };
|
|
54
|
+
}
|
|
55
|
+
let outcomes;
|
|
56
|
+
let memory;
|
|
57
|
+
try {
|
|
58
|
+
outcomes = JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'routing-outcomes.json'), 'utf8'));
|
|
59
|
+
memory = JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'memory', 'store.json'), 'utf8'));
|
|
60
|
+
} catch (error) {
|
|
61
|
+
return { retrieved: false, why: `persistence stores were not readable: ${error.message}` };
|
|
62
|
+
}
|
|
63
|
+
const outcome = (outcomes.outcomes || []).find((row) =>
|
|
64
|
+
row.task === task && row.agent === agent && row.success === true);
|
|
65
|
+
const decision = memory.entries?.[`routing-decision:${taskId}`];
|
|
66
|
+
let value = null;
|
|
67
|
+
try { value = decision ? JSON.parse(decision.value) : null; } catch { /* invalid evidence */ }
|
|
68
|
+
if (!outcome || !decision || value?.task !== task || value?.agent !== agent) {
|
|
69
|
+
return { retrieved: false, why: 'no matching routing outcome and routing-decision row persisted' };
|
|
70
|
+
}
|
|
71
|
+
if (!/\[OK\]\s*Task outcome recorded:\s*SUCCESS/i.test(String(output || ''))) {
|
|
72
|
+
return { retrieved: false, why: 'rows exist but invocation did not report successful recording' };
|
|
73
|
+
}
|
|
74
|
+
return { retrieved: true, why: `routing outcome and routing-decision:${taskId} persisted` };
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
export function executeProducedCommand(cmd, {
|
|
78
|
+
cwd,
|
|
79
|
+
ruflo = RUFLO_BIN,
|
|
80
|
+
base = null,
|
|
81
|
+
trap = TRAP.MEMORY_SEARCH,
|
|
82
|
+
} = {}) {
|
|
83
|
+
const reject = (why, extra = {}) => ({
|
|
84
|
+
ran: false,
|
|
85
|
+
argv: null,
|
|
86
|
+
exit: null,
|
|
87
|
+
exitOk: false,
|
|
88
|
+
retrieved: false,
|
|
89
|
+
why,
|
|
90
|
+
output: '',
|
|
91
|
+
...extra,
|
|
92
|
+
});
|
|
93
|
+
const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
|
|
94
|
+
if (!invocations.length) return reject('no Ruflo invocation to execute');
|
|
95
|
+
const firstExecutable = path.basename(commandNodes(String(cmd || ''))[0]?.exe || '').split('@')[0];
|
|
96
|
+
if (invocations.length !== 1 || firstExecutable !== 'ruflo') {
|
|
97
|
+
return reject('corrective command was not the first and only Ruflo invocation');
|
|
98
|
+
}
|
|
99
|
+
const args = invocations[0].args.filter((value) => value !== '');
|
|
100
|
+
if (args.some((value) => ['--help', '-h', '--version', 'version', 'status'].includes(value))) {
|
|
101
|
+
return reject('discovery cannot replace the requested corrective action');
|
|
102
|
+
}
|
|
103
|
+
for (let i = 0; i < args.length; i++) {
|
|
104
|
+
const value = args[i];
|
|
105
|
+
if (!['--path', '--db'].includes(value) && !/^--(?:path|db)=/.test(value)) continue;
|
|
106
|
+
const raw = value.includes('=') ? value.slice(value.indexOf('=') + 1) : args[i + 1];
|
|
107
|
+
const absolute = path.resolve(cwd, String(raw || ''));
|
|
108
|
+
const root = base && path.resolve(base);
|
|
109
|
+
if (!root || !(absolute === root || absolute.startsWith(`${root}${path.sep}`))) {
|
|
110
|
+
return reject(`store path ${absolute} is outside the fixture world`, { argv: ['ruflo', ...args] });
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
const mutating = args.filter((value) => !value.startsWith('-')).slice(0, 2)
|
|
114
|
+
.find((value) => MUTATING_SUBCOMMANDS.has(value));
|
|
115
|
+
if (mutating) return reject(`refused mutating subcommand "${mutating}"`, { argv: ['ruflo', ...args] });
|
|
116
|
+
|
|
117
|
+
const env = { ...process.env };
|
|
118
|
+
delete env.CLAUDE_FLOW_DB_PATH;
|
|
119
|
+
delete env.CLAUDE_FLOW_MEMORY_PATH;
|
|
120
|
+
const result = spawnRuflo(ruflo, args, {
|
|
121
|
+
cwd,
|
|
122
|
+
encoding: 'utf8',
|
|
123
|
+
timeout: 120_000,
|
|
124
|
+
env,
|
|
125
|
+
maxBuffer: 8 * 1024 * 1024,
|
|
126
|
+
});
|
|
127
|
+
if (result.error) return reject(`spawn failed: ${result.error.message}`, { argv: ['ruflo', ...args] });
|
|
128
|
+
const output = `${result.stdout || ''}${result.stderr || ''}`;
|
|
129
|
+
const evidence = trap === TRAP.POST_TASK
|
|
130
|
+
? assertPostTaskPersisted({ args, output, cwd })
|
|
131
|
+
: assertRetrieved(output);
|
|
132
|
+
return {
|
|
133
|
+
ran: true,
|
|
134
|
+
argv: ['ruflo', ...args],
|
|
135
|
+
exit: result.status,
|
|
136
|
+
exitOk: result.status === 0,
|
|
137
|
+
retrieved: evidence.retrieved,
|
|
138
|
+
why: `exit ${result.status}; ${evidence.why}`,
|
|
139
|
+
output: output.slice(0, 1200),
|
|
140
|
+
};
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
export function verifyRufloFlag(bin = RUFLO_BIN) {
|
|
144
|
+
if (!fs.existsSync(bin)) return { ok: false, why: `Ruflo binary not found at ${bin}` };
|
|
145
|
+
const result = spawnSync(bin, ['memory', 'search', '--help'], {
|
|
146
|
+
encoding: 'utf8',
|
|
147
|
+
timeout: 30_000,
|
|
148
|
+
});
|
|
149
|
+
const output = `${result.stdout || ''}${result.stderr || ''}`;
|
|
150
|
+
if (result.status !== 0 && !output) return { ok: false, why: `help exited ${result.status} empty` };
|
|
151
|
+
if (!/-q,\s*--query/.test(output)) return { ok: false, why: 'live help lacks -q, --query', help: output };
|
|
152
|
+
if (/\bmemory search\s+"[^"]+"\s*$/m.test(output)) {
|
|
153
|
+
return { ok: false, why: 'live help documents a positional query', help: output };
|
|
154
|
+
}
|
|
155
|
+
return {
|
|
156
|
+
ok: true,
|
|
157
|
+
flag: '-q, --query',
|
|
158
|
+
required: /--query[^\n]*required/i.test(output),
|
|
159
|
+
evidence: output.split('\n').find((line) => /-q,\s*--query/.test(line))?.trim() || '',
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
export function verifyPostTaskContract(bin = RUFLO_BIN) {
|
|
164
|
+
if (!fs.existsSync(bin)) return { ok: false, why: `Ruflo binary not found at ${bin}` };
|
|
165
|
+
const help = spawnRuflo(bin, ['hooks', 'post-task', '--help'], {
|
|
166
|
+
encoding: 'utf8',
|
|
167
|
+
timeout: 30_000,
|
|
168
|
+
});
|
|
169
|
+
const output = `${help.stdout || ''}${help.stderr || ''}`;
|
|
170
|
+
if (!/--task\b/.test(output) || !/--agent\b/.test(output) || !/--store-results\b/.test(output)
|
|
171
|
+
|| !/Without this \+ --agent, no routing outcome is recorded/.test(output)) {
|
|
172
|
+
return { ok: false, why: 'live help lacks the three-part persistence contract', help: output };
|
|
173
|
+
}
|
|
174
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'd4-post-task-premise-'));
|
|
175
|
+
const missing = spawnRuflo(bin, ['hooks', 'post-task', '-i', 'd4-premise-missing', '--success', 'true'], {
|
|
176
|
+
cwd: dir,
|
|
177
|
+
encoding: 'utf8',
|
|
178
|
+
timeout: 30_000,
|
|
179
|
+
});
|
|
180
|
+
const persisted = fs.existsSync(path.join(dir, '.claude-flow', 'routing-outcomes.json'))
|
|
181
|
+
|| fs.existsSync(path.join(dir, '.claude-flow', 'memory', 'store.json'));
|
|
182
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
183
|
+
if (missing.status !== 0 || persisted) {
|
|
184
|
+
return { ok: false, why: 'missing-contract probe did not remain successful and non-persistent' };
|
|
185
|
+
}
|
|
186
|
+
return {
|
|
187
|
+
ok: true,
|
|
188
|
+
flag: '--task + --agent + --store-results',
|
|
189
|
+
required: true,
|
|
190
|
+
evidence: 'live help names all flags; success/task-id-only persisted nothing',
|
|
191
|
+
missingExit: missing.status,
|
|
192
|
+
};
|
|
193
|
+
}
|