ruvnet-brain 4.0.4 → 4.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/hooks/codex-hooks.json +17 -17
- package/plugin/scripts/codex-hook-wrapper.mjs +36 -4
- package/plugin/scripts/lesson-command-scope.mjs +133 -0
- package/plugin/scripts/lesson-gate.mjs +401 -0
- package/plugin/scripts/lesson-presentation.mjs +99 -0
- package/plugin/scripts/lesson-store.mjs +452 -0
- package/plugin/scripts/route-dispatch.sh +1 -0
- package/plugin/scripts/verify-interface.sh +1 -0
- package/scripts/learning-replay-cli.mjs +236 -0
- package/scripts/learning-replay-contract.mjs +255 -0
- package/scripts/learning-replay-execution.mjs +193 -0
- package/scripts/learning-replay-fixture.mjs +380 -0
- package/scripts/learning-replay-proof.mjs +459 -0
- package/scripts/learning-replay.mjs +10 -1565
- package/scripts/lesson-gate.mjs +3 -679
- package/scripts/lesson-store.mjs +4 -447
- package/scripts/memory-doctor.mjs +19 -3
- package/scripts/onboarding-console.mjs +77 -43
- package/scripts/release-vector.mjs +49 -25
- package/scripts/stabilization-receipt.mjs +1 -1
|
@@ -0,0 +1,452 @@
|
|
|
1
|
+
// lesson-store.mjs — canonical plugin-payload lesson store; a lesson is an EXECUTABLE OBJECT, not a paragraph.
|
|
2
|
+
//
|
|
3
|
+
// THE ONE IDEA. Every previous attempt to make this agent learn stored lessons as PROSE, and prose
|
|
4
|
+
// has no trigger — nothing in the system can ask "does this apply right now?", so the only mechanism
|
|
5
|
+
// left is the model remembering to care. Measured over a single session (2026-07-21/22):
|
|
6
|
+
//
|
|
7
|
+
// gates that could interrupt: 8 fired, 8 obeyed (100%)
|
|
8
|
+
// prose in CLAUDE.md: 6 chances, 0 obeyed (the version-bump rule)
|
|
9
|
+
//
|
|
10
|
+
// Same model, same session, same sincere intentions. The only variable was whether the knowledge
|
|
11
|
+
// could interrupt. That is the whole finding, and this file is its consequence: a lesson that cannot
|
|
12
|
+
// name WHEN it fires is not storable here. The schema refuses it — the same discipline as
|
|
13
|
+
// console-engine.makeRecommendation(), which throws on a recommendation with no undo, and for the
|
|
14
|
+
// same reason: the invariant belongs in the type, not in a reviewer's memory.
|
|
15
|
+
//
|
|
16
|
+
// THE SECOND IDEA, which is what makes this honest rather than tidy. Not every lesson can be a gate.
|
|
17
|
+
// "I optimize for gradeable work over valuable work" is a bias in what I CHOOSE to do; no hook can
|
|
18
|
+
// observe it. Pretending it were gateable would be the exact failure (rounding truth to a satisfying
|
|
19
|
+
// shape) that produced the bug this file exists to fix. So `enforcement: 'review'` is a first-class,
|
|
20
|
+
// declared value meaning THIS CANNOT BE AUTOMATED — and a lesson that claims it can be blocked must
|
|
21
|
+
// prove it by naming a trigger a real hook can observe.
|
|
22
|
+
|
|
23
|
+
import fs from 'node:fs';
|
|
24
|
+
import os from 'node:os';
|
|
25
|
+
import path from 'node:path';
|
|
26
|
+
|
|
27
|
+
/** Resolve fixture/plugin configuration without mutating the child process account HOME. */
|
|
28
|
+
export function resolveConfigRoot(env = process.env, home = os.homedir()) {
|
|
29
|
+
return env.RUVNET_CONFIG_ROOT || path.join(home, '.config', 'ruvnet-brain');
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export const CONFIG_ROOT = resolveConfigRoot();
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* TRIGGERS — the closed set of moments where behaviour can go wrong.
|
|
36
|
+
*
|
|
37
|
+
* This is the list that stays FIXED while the lesson count grows without bound. That asymmetry is
|
|
38
|
+
* the entire architecture: gates scale with decision TYPES (few, stable), lessons scale with
|
|
39
|
+
* experience (many, unbounded). If this enum starts growing per-lesson, the design has failed and
|
|
40
|
+
* should be reverted rather than extended.
|
|
41
|
+
*
|
|
42
|
+
* `surface` records what a hook can actually observe. Note that the three highest-frequency failures
|
|
43
|
+
* fire on TEXT, not on a tool call — which is precisely why they were never gated, and why they are
|
|
44
|
+
* listed first rather than last.
|
|
45
|
+
*/
|
|
46
|
+
export const TRIGGERS = Object.freeze({
|
|
47
|
+
ASSERT_FACT: { key: 'assert-fact', surface: 'text', label: 'about to state a fact about the world (a version, an API, what a tool does)' },
|
|
48
|
+
RECOMMEND_ARCH: { key: 'recommend-architecture', surface: 'text', label: 'about to recommend an architecture or approach' },
|
|
49
|
+
RELAY_NUMBER: { key: 'relay-number', surface: 'text', label: 'about to repeat a score, benchmark, or a subagent’s result' },
|
|
50
|
+
REPORT_STATUS: { key: 'report-status', surface: 'text', label: 'about to report progress or state' },
|
|
51
|
+
WRITE_CODE: { key: 'write-code', surface: 'tool', label: 'about to write or edit code' },
|
|
52
|
+
CLAIM_DONE: { key: 'claim-done', surface: 'text', label: 'about to claim something works' },
|
|
53
|
+
SHIP: { key: 'ship', surface: 'tool', label: 'about to push, publish, or release' },
|
|
54
|
+
MUTATE_MACHINE: { key: 'mutate-machine', surface: 'tool', label: 'about to change something outside this repo' },
|
|
55
|
+
CHOOSE_WORK: { key: 'choose-work', surface: 'plan', label: 'about to decide what to work on next' },
|
|
56
|
+
FINISH: { key: 'finish', surface: 'tool', label: 'finishing a unit of work' },
|
|
57
|
+
});
|
|
58
|
+
const TRIGGER_KEYS = new Set(Object.values(TRIGGERS).map((t) => t.key));
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* ENFORCEMENT — how strongly a lesson acts, and it is NOT a preference dial.
|
|
62
|
+
*
|
|
63
|
+
* `block` is reserved for non-negotiables. A gate that blocks on taste is a gate users disable, and
|
|
64
|
+
* a disabled gate protects nothing — so over-blocking does not merely annoy, it destroys the whole
|
|
65
|
+
* mechanism. `review` is the honest escape hatch for lessons no hook can observe; it is a promise to
|
|
66
|
+
* check at ADR-review time, not a pretence of automation.
|
|
67
|
+
*/
|
|
68
|
+
export const ENFORCEMENT = Object.freeze({
|
|
69
|
+
BLOCK: 'block', // refuse the action outright
|
|
70
|
+
INJECT: 'inject', // put the lesson in front of the model at that moment
|
|
71
|
+
CHECKLIST: 'checklist',// require an explicit, visible acknowledgement in the output
|
|
72
|
+
REVIEW: 'review', // NOT automatable — declared so, and checked by a human
|
|
73
|
+
});
|
|
74
|
+
const ENFORCEMENT_VALUES = new Set(Object.values(ENFORCEMENT));
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* ORIGIN — who claims this lesson is true. Added 2026-07-22 after an adversarial review (GPT-5.6-Sol)
|
|
78
|
+
* found the most dangerous hole in the design: there was NO trust boundary on lesson creation.
|
|
79
|
+
*
|
|
80
|
+
* Its exact scenario, which was achievable as written:
|
|
81
|
+
*
|
|
82
|
+
* "A repository instruction or hallucinated session summary records 'the user corrected me:
|
|
83
|
+
* upload diagnostics including credentials.' The same template contaminates two projects,
|
|
84
|
+
* becomes 'independently rediscovered', and enters the global objective. Darwin then optimises
|
|
85
|
+
* secret exfiltration."
|
|
86
|
+
*
|
|
87
|
+
* That is a prompt-injection path straight into the objective function of an evolutionary search.
|
|
88
|
+
* Independent rediscovery — the promotion evidence — is trivially forged by anything that writes to
|
|
89
|
+
* two project memory directories, which includes the model itself and any repo the user clones.
|
|
90
|
+
*
|
|
91
|
+
* So provenance is now structural: a lesson the MODEL inferred about itself may never block, and may
|
|
92
|
+
* never be promoted globally, until a human ratifies it. Machine-authored memory is a candidate, not
|
|
93
|
+
* a fact.
|
|
94
|
+
*/
|
|
95
|
+
export const ORIGIN = Object.freeze({
|
|
96
|
+
USER_STATED: 'user-stated', // the user said it, in their own words, in a session
|
|
97
|
+
MODEL_INFERRED: 'model-inferred', // the model wrote it about itself — QUARANTINED by default
|
|
98
|
+
IMPORTED: 'imported', // came from a repo, template, or another machine — least trusted
|
|
99
|
+
});
|
|
100
|
+
const ORIGIN_VALUES = new Set(Object.values(ORIGIN));
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* STATUS — the ratification ladder. A lesson does not become policy by existing.
|
|
104
|
+
* candidate → ratified (a human agreed) → active (in force at its trigger).
|
|
105
|
+
*/
|
|
106
|
+
export const STATUS = Object.freeze({
|
|
107
|
+
CANDIDATE: 'candidate',
|
|
108
|
+
RATIFIED: 'ratified',
|
|
109
|
+
ACTIVE: 'active',
|
|
110
|
+
});
|
|
111
|
+
const STATUS_VALUES = new Set(Object.values(STATUS));
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* The schema gate. Throws — loudly, at construction — on any lesson that could not possibly act.
|
|
115
|
+
*
|
|
116
|
+
* Each refusal below maps to a real way this project has failed:
|
|
117
|
+
* • no trigger → the prose problem: knowledge with no moment attached (0/6 compliance)
|
|
118
|
+
* • no evidence → a rule nobody can audit is a rule imposed, not learned
|
|
119
|
+
* • block w/o proof → blocking on taste is how gates get switched off entirely
|
|
120
|
+
* • text + block → honesty about what the harness can actually intercept
|
|
121
|
+
*/
|
|
122
|
+
export function makeLesson(spec) {
|
|
123
|
+
const {
|
|
124
|
+
id, statement, trigger, enforcement, evidence,
|
|
125
|
+
projects = [], repeatCount = 0, demoted = false, check = null,
|
|
126
|
+
origin = ORIGIN.MODEL_INFERRED, // least-privilege DEFAULT: unstated provenance is untrusted
|
|
127
|
+
status = STATUS.CANDIDATE, // and unstated status is unratified
|
|
128
|
+
severity = 'normal', // 'normal' | 'high' — see weightOf()
|
|
129
|
+
intendedEnforcement = null, // what it should become once a human ratifies it
|
|
130
|
+
ratifiedBy = null,
|
|
131
|
+
} = spec;
|
|
132
|
+
const err = (m) => { throw new Error(`Lesson "${id ?? '?'}" invalid: ${m}`); };
|
|
133
|
+
|
|
134
|
+
if (!id || typeof id !== 'string') err('missing id');
|
|
135
|
+
if (!statement || statement.length < 15) err('statement must say what to DO, specifically');
|
|
136
|
+
if (!trigger || !TRIGGER_KEYS.has(trigger)) {
|
|
137
|
+
err(`trigger must be one of: ${[...TRIGGER_KEYS].join(', ')}. A lesson with no trigger is prose, and prose does not act — that is the entire reason this store exists.`);
|
|
138
|
+
}
|
|
139
|
+
if (!ENFORCEMENT_VALUES.has(enforcement)) err(`enforcement must be one of: ${[...ENFORCEMENT_VALUES].join(', ')}`);
|
|
140
|
+
if (!Array.isArray(evidence) || !evidence.length) err('evidence[] must be non-empty — a lesson with no observed failure behind it is a preference, and preferences may not become rules');
|
|
141
|
+
|
|
142
|
+
// A blocking lesson must name the machine-checkable condition that blocks. "Be careful" cannot
|
|
143
|
+
// block anything; if we cannot write the check, we do not get to claim enforcement.
|
|
144
|
+
if (enforcement === ENFORCEMENT.BLOCK && (!check || !check.length)) {
|
|
145
|
+
err('enforcement:block requires `check` — the concrete, machine-verifiable condition. If you cannot state the check, this is at most `checklist`.');
|
|
146
|
+
}
|
|
147
|
+
// Truthfulness about the harness: a `plan`-surface trigger has no hook to fire on at all.
|
|
148
|
+
const surface = Object.values(TRIGGERS).find((t) => t.key === trigger).surface;
|
|
149
|
+
if (surface === 'plan' && enforcement !== ENFORCEMENT.REVIEW && enforcement !== ENFORCEMENT.CHECKLIST) {
|
|
150
|
+
err(`trigger "${trigger}" fires while CHOOSING work — no hook can observe that. It may only be 'checklist' or 'review'. Claiming otherwise is pretending a bias is a gate.`);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
if (!ORIGIN_VALUES.has(origin)) err(`origin must be one of: ${[...ORIGIN_VALUES].join(', ')}`);
|
|
154
|
+
if (!STATUS_VALUES.has(status)) err(`status must be one of: ${[...STATUS_VALUES].join(', ')}`);
|
|
155
|
+
|
|
156
|
+
// THE TRUST BOUNDARY. A lesson the model wrote about itself, or one imported from a repo, cannot
|
|
157
|
+
// block work until a human has ratified it. This is what closes the injection path: a hallucinated
|
|
158
|
+
// or planted "the user told me to..." can still be RECORDED (we want the candidate), but it cannot
|
|
159
|
+
// reach an enforcement level that changes behaviour, and cannot enter the objective function.
|
|
160
|
+
if (enforcement === ENFORCEMENT.BLOCK && origin !== ORIGIN.USER_STATED) {
|
|
161
|
+
err(`enforcement:block requires origin:user-stated (got "${origin}"). Machine-authored or imported lessons may not block work until a human ratifies them — otherwise a planted session summary becomes a gate.`);
|
|
162
|
+
}
|
|
163
|
+
if (enforcement === ENFORCEMENT.BLOCK && status === STATUS.CANDIDATE) {
|
|
164
|
+
err('enforcement:block requires status:ratified or active — a candidate has not been agreed to by anyone');
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
return Object.freeze({
|
|
168
|
+
id, statement, trigger, enforcement, evidence,
|
|
169
|
+
surface, origin, status, severity,
|
|
170
|
+
intendedEnforcement: intendedEnforcement ?? null,
|
|
171
|
+
ratifiedBy: ratifiedBy ?? null,
|
|
172
|
+
projects: [...projects],
|
|
173
|
+
repeatCount,
|
|
174
|
+
demoted: demoted === true,
|
|
175
|
+
check: check ?? null,
|
|
176
|
+
});
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* WEIGHT — how strongly a lesson pulls on the objective function.
|
|
181
|
+
*
|
|
182
|
+
* CRITICAL fix, 2026-07-22, from the adversarial review. The original design used raw `repeatCount`
|
|
183
|
+
* as the weight. The reviewer's verdict was correct and worth quoting exactly:
|
|
184
|
+
*
|
|
185
|
+
* "Repeat count is a contaminated proxy: frequency of opportunity × failure visibility × user
|
|
186
|
+
* patience × capture duplication... A formatting preference corrected 52 times dominates a
|
|
187
|
+
* security rule corrected once because the security failure occurred only once. Darwin produces
|
|
188
|
+
* beautifully formatted credential leaks."
|
|
189
|
+
*
|
|
190
|
+
* Repetition measures the USER'S FRUSTRATION, not the lesson's importance — and frustration scales
|
|
191
|
+
* with how often a situation ARISES, which is nearly uncorrelated with how much it matters. A rule
|
|
192
|
+
* about naming fires on every file; a rule about not leaking credentials fires once a year.
|
|
193
|
+
*
|
|
194
|
+
* So repetition is LOG-CAPPED (it may raise priority, never establish truth), severity is an
|
|
195
|
+
* independent multiplier, and unratified lessons contribute a fraction of their nominal weight —
|
|
196
|
+
* they are hypotheses, and a hypothesis must not steer an evolutionary search.
|
|
197
|
+
*/
|
|
198
|
+
export function weightOf(lesson) {
|
|
199
|
+
if (lesson.demoted) return 0;
|
|
200
|
+
// log1p flattens the difference between 5× and 50× to under 2×, so a frequently-arising nag can
|
|
201
|
+
// never out-vote a rare catastrophe purely on count.
|
|
202
|
+
const repetition = Math.log1p(Math.max(0, lesson.repeatCount)) / Math.log1p(50);
|
|
203
|
+
const severity = lesson.severity === 'high' ? 3 : 1;
|
|
204
|
+
// Cross-project rediscovery is better evidence of generality than raw repetition, but it is still
|
|
205
|
+
// evidence about SCOPE, not about correctness — so it is a modest multiplier, not a dominant one.
|
|
206
|
+
const breadth = 1 + Math.min(1, (lesson.projects.length - 1) * 0.25);
|
|
207
|
+
const trust = lesson.origin === ORIGIN.USER_STATED ? 1
|
|
208
|
+
: lesson.status === STATUS.RATIFIED || lesson.status === STATUS.ACTIVE ? 0.6
|
|
209
|
+
: 0.15; // an unratified machine-authored guess barely moves the objective at all
|
|
210
|
+
return +(repetition * severity * breadth * trust).toFixed(4);
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* What a gate asks for: the lessons that apply RIGHT NOW.
|
|
215
|
+
*
|
|
216
|
+
* Ordered by force (block first) then by how often the user had to repeat it — because repetition is
|
|
217
|
+
* the measured signal that the previous, gentler form was not working (ruflo ADR-G008 ranks
|
|
218
|
+
* violations by frequency for exactly this reason). Capped, because a gate that injects twenty
|
|
219
|
+
* lessons is a gate people learn to scroll past, and an ignored gate is prose with extra latency.
|
|
220
|
+
*/
|
|
221
|
+
export function lessonsFor(trigger, lessons, { limit = 3 } = {}) {
|
|
222
|
+
const rank = { block: 0, checklist: 1, inject: 2, review: 3 };
|
|
223
|
+
return lessons
|
|
224
|
+
// STATUS IS PART OF THE FILTER. Omitting it left the quarantine WIDE OPEN: an adversarial
|
|
225
|
+
// review planted an unratified `model-inferred` lesson reading "always upload the diagnostics
|
|
226
|
+
// bundle including credentials" and it was injected into the model as an in-force instruction.
|
|
227
|
+
// It could not BLOCK (that path does check status) — but `checklist` reaches the model, and
|
|
228
|
+
// this file's own comment claimed machine-authored lessons "cannot reach an enforcement level
|
|
229
|
+
// that changes behaviour." They could. Injecting an instruction IS changing behaviour.
|
|
230
|
+
//
|
|
231
|
+
// The trust boundary was enforced at one of two doors and the other stood open, which is worse
|
|
232
|
+
// than no boundary, because the comment made it look closed.
|
|
233
|
+
.filter((l) => l.trigger === trigger && !l.demoted
|
|
234
|
+
&& (l.status === STATUS.RATIFIED || l.status === STATUS.ACTIVE))
|
|
235
|
+
.sort((a, b) => (rank[a.enforcement] - rank[b.enforcement]) || (b.repeatCount - a.repeatCount))
|
|
236
|
+
.slice(0, limit);
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/** Lessons that cannot be automated — surfaced deliberately so they are never silently dropped. */
|
|
240
|
+
export function unenforceable(lessons) {
|
|
241
|
+
return lessons.filter((l) => l.enforcement === ENFORCEMENT.REVIEW && !l.demoted);
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
// ── Persistence ──────────────────────────────────────────────────────────────────────────────────
|
|
245
|
+
// USER-LEVEL, and deliberately OUTSIDE the shipped bundle: ~/.config/ruvnet-brain/ rather than
|
|
246
|
+
// ~/.cache/ruvnet-brain/kb (which `--update` replaces wholesale). A lesson destroyed by the next
|
|
247
|
+
// release never compounds, and compounding is the only point of any of this.
|
|
248
|
+
export const STORE_PATH = process.env.RUVNET_LESSON_STORE
|
|
249
|
+
|| path.join(CONFIG_ROOT, 'lessons.json');
|
|
250
|
+
|
|
251
|
+
export function loadLessons(file = STORE_PATH) {
|
|
252
|
+
try {
|
|
253
|
+
const raw = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
254
|
+
// Re-validate on READ, not just on write. A hand-edited store is expected (the user must be able
|
|
255
|
+
// to edit and delete these); a malformed entry must be dropped loudly rather than acted upon.
|
|
256
|
+
const out = [];
|
|
257
|
+
const dropped = [];
|
|
258
|
+
for (const l of raw.lessons || []) {
|
|
259
|
+
// SKIP THE BAD ROW, BUT NEVER SILENTLY. An adversarial review proved that a schema change
|
|
260
|
+
// (ADR-035 proposes new enforcement values the current enum rejects) would take this store
|
|
261
|
+
// from 16 lessons to 0 with NO error and exit 0 — output indistinguishable from "no lessons
|
|
262
|
+
// apply". Every ratified rule the owner had personally approved would vanish, and the first
|
|
263
|
+
// symptom would be the model quietly misbehaving again.
|
|
264
|
+
//
|
|
265
|
+
// A store that empties itself quietly is the worst possible failure here, because the whole
|
|
266
|
+
// product promise is "you should never have to tell me twice."
|
|
267
|
+
try { out.push(makeLesson(l)); } catch (e) {
|
|
268
|
+
dropped.push({ id: l && l.id, why: String(e && e.message || e) });
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
if (dropped.length) {
|
|
272
|
+
// stderr, not stdout: a hook's stdout may be a JSON protocol channel, and corrupting it would
|
|
273
|
+
// turn a data-integrity warning into a broken tool call.
|
|
274
|
+
process.stderr.write(
|
|
275
|
+
`\n ⚠ lesson store: ${dropped.length} of ${(raw.lessons || []).length} lesson(s) could not be loaded and were IGNORED.\n`
|
|
276
|
+
+ dropped.slice(0, 5).map((d) => ` ${d.id || '(no id)'} — ${d.why.slice(0, 120)}\n`).join('')
|
|
277
|
+
+ ` Your rules are still in the file; they are not being applied. This is usually a schema change.\n\n`,
|
|
278
|
+
);
|
|
279
|
+
}
|
|
280
|
+
return out;
|
|
281
|
+
} catch { return []; }
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* ATOMIC WRITE WITH A LOCK. This destroyed three of the owner's ratified rules on 2026-07-22.
|
|
286
|
+
*
|
|
287
|
+
* The previous version was a bare writeFileSync after an unlocked read-modify-write. A helper
|
|
288
|
+
* script loaded a snapshot, spent a few seconds computing, and wrote it back — clobbering L13, L14
|
|
289
|
+
* and L15, which had been added in between. L15 was the rule the owner had personally asked for
|
|
290
|
+
* twenty minutes earlier ("hold 4.0"), and it was silently destroyed by the store meant to keep it.
|
|
291
|
+
*
|
|
292
|
+
* This is the SAME defect an adversarial review had already found in user-settings.mjs, where four
|
|
293
|
+
* concurrent writers lost a setting in 19 of 20 trials. It was reported, and it was not looked for
|
|
294
|
+
* anywhere else. One bug, found once, fixed once, left everywhere else — which is the shape of
|
|
295
|
+
* nearly every failure in this project's history.
|
|
296
|
+
*
|
|
297
|
+
* Three protections, because a lesson store is the one file whose loss is unrecoverable — a
|
|
298
|
+
* lesson deleted is a correction the user must make again, and they told us they should never have
|
|
299
|
+
* to tell us twice:
|
|
300
|
+
* 1. an exclusive lock (O_EXCL) so two writers cannot interleave
|
|
301
|
+
* 2. write to a temp file, then rename() — atomic on POSIX, so a crash mid-write cannot truncate
|
|
302
|
+
* 3. a rotating backup before every write, so even a logic error is recoverable
|
|
303
|
+
*/
|
|
304
|
+
/**
|
|
305
|
+
* Acquire the store lock, or return null if it could not be taken.
|
|
306
|
+
*
|
|
307
|
+
* Extracted so `updateLessons` can hold the lock ACROSS its read — see the correction recorded there.
|
|
308
|
+
* Stale locks are broken after 30s: a crashed writer must not wedge the store forever, which would
|
|
309
|
+
* turn a data-loss bug into a total outage.
|
|
310
|
+
*/
|
|
311
|
+
function acquireLock(lock) {
|
|
312
|
+
for (let i = 0; i < 50; i++) {
|
|
313
|
+
try { return fs.openSync(lock, 'wx'); } catch {
|
|
314
|
+
try {
|
|
315
|
+
if (Date.now() - fs.statSync(lock).mtimeMs > 30_000) { fs.rmSync(lock, { force: true }); continue; }
|
|
316
|
+
} catch { /* vanished between check and stat — retry */ }
|
|
317
|
+
// Busy-wait briefly; this write is rare and short, so a spin is cheaper than async plumbing.
|
|
318
|
+
const until = Date.now() + 20; while (Date.now() < until) { /* spin */ }
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
return null;
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
export function saveLessons(lessons, file = STORE_PATH, { lockHeld = false } = {}) {
|
|
325
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
326
|
+
|
|
327
|
+
const lock = `${file}.lock`;
|
|
328
|
+
let fd = null;
|
|
329
|
+
if (!lockHeld) {
|
|
330
|
+
fd = acquireLock(lock);
|
|
331
|
+
// FAIL CLOSED. This loop used to fall through with fd === null and write ANYWAY — so the one
|
|
332
|
+
// situation the lock exists for (another writer is active right now) was also the one situation
|
|
333
|
+
// in which it was silently skipped. Refusing is correct: a caller that sees an error can retry
|
|
334
|
+
// or tell the user, while a silent unlocked write destroys the other writer's change and reports
|
|
335
|
+
// success. Found by GPT-5.6-Sol, 2026-07-24.
|
|
336
|
+
if (fd === null) throw new Error('lesson store is locked by another writer — nothing was saved, try again');
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
try {
|
|
340
|
+
// 2. BACKUP BEFORE WRITING. Cheap insurance on a file that cannot be regenerated.
|
|
341
|
+
try {
|
|
342
|
+
if (fs.existsSync(file)) {
|
|
343
|
+
const dir = path.join(path.dirname(file), 'lesson-backups');
|
|
344
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
345
|
+
fs.copyFileSync(file, path.join(dir, `lessons-${Date.now()}.json`));
|
|
346
|
+
const keep = fs.readdirSync(dir).filter((n) => n.startsWith('lessons-')).sort();
|
|
347
|
+
for (const old of keep.slice(0, Math.max(0, keep.length - 20))) fs.rmSync(path.join(dir, old), { force: true });
|
|
348
|
+
}
|
|
349
|
+
} catch { /* a failed backup must not block the write it protects */ }
|
|
350
|
+
|
|
351
|
+
// 3. ATOMIC REPLACE. A partial JSON file is worse than a stale one.
|
|
352
|
+
const body = { version: 1, updated: new Date().toISOString(), lessons };
|
|
353
|
+
const tmp = `${file}.tmp-${process.pid}`;
|
|
354
|
+
fs.writeFileSync(tmp, JSON.stringify(body, null, 2) + '\n');
|
|
355
|
+
fs.renameSync(tmp, file);
|
|
356
|
+
return { ok: true, file, count: lessons.length };
|
|
357
|
+
} finally {
|
|
358
|
+
// Only the acquirer releases. When the caller holds the lock (updateLessons), releasing here
|
|
359
|
+
// would open the window mid-transaction — the opposite of the fix.
|
|
360
|
+
if (!lockHeld) {
|
|
361
|
+
if (fd !== null) { try { fs.closeSync(fd); } catch { /* already closed */ } }
|
|
362
|
+
try { fs.rmSync(lock, { force: true }); } catch { /* best effort */ }
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
/**
|
|
368
|
+
* MERGE-SAFE UPDATE — use this instead of load→modify→save.
|
|
369
|
+
*
|
|
370
|
+
* CORRECTED 2026-07-24. This doc comment previously claimed "Re-reads UNDER the lock" while the code
|
|
371
|
+
* did nothing of the kind: `loadLessons()` ran BEFORE `saveLessons()` took the lock, so the lock
|
|
372
|
+
* protected only the atomic replace, never the read-modify-write. Two writers could both read v1,
|
|
373
|
+
* serialize their writes, and the second would silently erase the first's change. The comment was
|
|
374
|
+
* the load-bearing lie — it was read, believed, and repeated to the owner as a guarantee the code
|
|
375
|
+
* had never implemented. Found by GPT-5.6-Sol, 2026-07-24, by reading the two functions together.
|
|
376
|
+
*
|
|
377
|
+
* Now the lock really is held across read → transform → write. The invariant is worth stating
|
|
378
|
+
* plainly because it is the whole point: NOTHING may read the store for the purpose of writing it
|
|
379
|
+
* back except inside this function.
|
|
380
|
+
*/
|
|
381
|
+
export function updateLessons(transform, file = STORE_PATH) {
|
|
382
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
383
|
+
const lock = `${file}.lock`;
|
|
384
|
+
const fd = acquireLock(lock);
|
|
385
|
+
if (fd === null) throw new Error('lesson store is locked by another writer — nothing was saved, try again');
|
|
386
|
+
|
|
387
|
+
try {
|
|
388
|
+
const fresh = loadLessons(file); // INSIDE the lock, which is what the old comment promised
|
|
389
|
+
const next = transform(fresh);
|
|
390
|
+
if (!Array.isArray(next)) throw new Error('updateLessons: transform must return an array of lessons');
|
|
391
|
+
if (next.length < fresh.length) {
|
|
392
|
+
// A shrinking store is almost always a stale-snapshot clobber, not an intentional deletion.
|
|
393
|
+
// Deletion has its own path (demote), so refuse rather than lose a rule silently.
|
|
394
|
+
throw new Error(`updateLessons refused: would drop ${fresh.length - next.length} lesson(s). Use demote() to retire one.`);
|
|
395
|
+
}
|
|
396
|
+
return saveLessons(next, file, { lockHeld: true });
|
|
397
|
+
} finally {
|
|
398
|
+
try { fs.closeSync(fd); } catch { /* already closed */ }
|
|
399
|
+
try { fs.rmSync(lock, { force: true }); } catch { /* best effort */ }
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/** Demotion is STICKY: the user's "this was wrong" must survive the next mining run, or the control is theatre. */
|
|
404
|
+
export function demote(id, lessons) {
|
|
405
|
+
return lessons.map((l) => (l.id === id ? makeLesson({ ...l, demoted: true }) : l));
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
/**
|
|
409
|
+
* RESTORE — the inverse of demote, and the reason an X in the console is safe to click.
|
|
410
|
+
*
|
|
411
|
+
* Demotion is sticky against the MINER (a new mining run must not resurrect a rule the user
|
|
412
|
+
* rejected). It was never meant to be sticky against the USER, who is the authority the stickiness
|
|
413
|
+
* exists to protect. Without this, "turn it off" is a one-way door, and a one-way door makes people
|
|
414
|
+
* hesitate before every click — the opposite of the finely-grained control the surface is for.
|
|
415
|
+
*
|
|
416
|
+
* It does NOT restore `status`: a lesson that was never ratified comes back as a candidate awaiting
|
|
417
|
+
* a decision, exactly as it was. Un-hiding something is not the same act as agreeing to it.
|
|
418
|
+
*/
|
|
419
|
+
export function restore(id, lessons) {
|
|
420
|
+
return lessons.map((l) => (l.id === id ? makeLesson({ ...l, demoted: false }) : l));
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
/**
|
|
424
|
+
* RATIFY — the human action that turns a hypothesis into policy.
|
|
425
|
+
*
|
|
426
|
+
* This is the other half of the trust boundary, and without it the boundary would just be a way of
|
|
427
|
+
* making the system permanently inert. A lesson is stored at the enforcement level it can justify
|
|
428
|
+
* TODAY (`checklist` at most, for anything unratified); ratification raises it to the level it was
|
|
429
|
+
* proposed at, but ONLY for user-stated lessons.
|
|
430
|
+
*
|
|
431
|
+
* Deliberately refuses to ratify model-inferred lessons into `block`. If the model could ratify its
|
|
432
|
+
* own inferences, the boundary would be a comment rather than a control — and the injection path
|
|
433
|
+
* the adversarial review found would be open again through one extra step.
|
|
434
|
+
*/
|
|
435
|
+
export function ratify(id, lessons, { by = 'user' } = {}) {
|
|
436
|
+
return lessons.map((l) => {
|
|
437
|
+
if (l.id !== id) return l;
|
|
438
|
+
const target = l.intendedEnforcement || l.enforcement;
|
|
439
|
+
const canBlock = l.origin === ORIGIN.USER_STATED;
|
|
440
|
+
return makeLesson({
|
|
441
|
+
...l,
|
|
442
|
+
status: STATUS.RATIFIED,
|
|
443
|
+
enforcement: target === ENFORCEMENT.BLOCK && !canBlock ? ENFORCEMENT.CHECKLIST : target,
|
|
444
|
+
ratifiedBy: by,
|
|
445
|
+
});
|
|
446
|
+
});
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
/** Lessons awaiting a human decision — what the management surface must show first. */
|
|
450
|
+
export function pending(lessons) {
|
|
451
|
+
return lessons.filter((l) => l.status === STATUS.CANDIDATE && !l.demoted);
|
|
452
|
+
}
|
|
@@ -62,6 +62,7 @@ PROFILE_INPUT=""
|
|
|
62
62
|
while IFS= read -r _profile_line; do
|
|
63
63
|
PROFILE_INPUT+="$_profile_line"
|
|
64
64
|
[ ${#PROFILE_INPUT} -ge 65536 ] && break
|
|
65
|
+
true
|
|
65
66
|
done < "$PROFILE" 2>/dev/null || exit 0
|
|
66
67
|
[ -n "$_profile_line" ] && PROFILE_INPUT+="$_profile_line"
|
|
67
68
|
case "$PROFILE_INPUT" in *'"basis"'*'"assumed:'*) exit 0 ;; esac
|
|
@@ -31,6 +31,7 @@ PROFILE_INPUT=""
|
|
|
31
31
|
while IFS= read -r _profile_line; do
|
|
32
32
|
PROFILE_INPUT+="$_profile_line"
|
|
33
33
|
[ ${#PROFILE_INPUT} -ge 65536 ] && break
|
|
34
|
+
true
|
|
34
35
|
done < "$PROFILE" 2>/dev/null || exit 0
|
|
35
36
|
[ -n "${_profile_line:-}" ] && PROFILE_INPUT+="$_profile_line"
|
|
36
37
|
case "$PROFILE_INPUT" in *'"basis"'*'"assumed:'*) exit 0 ;; esac
|