ruvnet-brain 4.0.12 → 4.0.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +3 -3
  2. package/package.json +1 -1
  3. package/plugin/.claude-plugin/plugin.json +2 -2
  4. package/plugin/.codex-plugin/plugin.json +1 -1
  5. package/plugin/scripts/advocacy-outcomes.mjs +808 -0
  6. package/plugin/scripts/anticipate.sh +80 -14
  7. package/plugin/scripts/capability-registry.mjs +994 -0
  8. package/plugin/scripts/codex-hook-wrapper.mjs +1 -0
  9. package/plugin/scripts/continuation-gate.mjs +129 -1
  10. package/plugin/scripts/gates.mjs +146 -0
  11. package/plugin/scripts/goal-match.mjs +398 -0
  12. package/plugin/scripts/hijack-ruvnet.sh +69 -1
  13. package/plugin/scripts/hook-registry.mjs +616 -0
  14. package/plugin/scripts/hook-shim.mjs +13 -2
  15. package/plugin/scripts/learning-enable.mjs +382 -0
  16. package/plugin/scripts/lesson-promote.mjs +262 -0
  17. package/plugin/scripts/lesson-provenance.mjs +43 -0
  18. package/plugin/scripts/lesson-store.mjs +67 -56
  19. package/plugin/scripts/memory-doctor.mjs +345 -0
  20. package/plugin/scripts/nightly-controller.mjs +98 -0
  21. package/plugin/scripts/runtime-preferences.mjs +18 -0
  22. package/plugin/scripts/unprompted-runtime.mjs +22 -7
  23. package/plugin/scripts/user-settings.mjs +672 -0
  24. package/plugin/skills/ruvnet-brain/SKILL.md +2 -2
  25. package/scripts/advocacy-outcomes.mjs +4 -808
  26. package/scripts/capability-registry.mjs +4 -876
  27. package/scripts/corpus-qa.mjs +44 -6
  28. package/scripts/doc-currency.mjs +30 -2
  29. package/scripts/gates.mjs +4 -146
  30. package/scripts/goal-match.mjs +4 -398
  31. package/scripts/hook-registry.mjs +4 -567
  32. package/scripts/issue-watch.mjs +108 -0
  33. package/scripts/learning-enable.mjs +4 -380
  34. package/scripts/lesson-promote.mjs +4 -262
  35. package/scripts/memory-doctor.mjs +4 -345
  36. package/scripts/nightly-controller.mjs +4 -66
  37. package/scripts/nightly-wrapper.sh +23 -1
  38. package/scripts/proactivity-metrics.mjs +8 -1
  39. package/scripts/qe/ux-suite.mjs +72 -1
  40. package/scripts/release-abort-stale.mjs +111 -0
  41. package/scripts/release-convergence-watchdog.mjs +119 -0
  42. package/scripts/release-transaction-provider.mjs +46 -6
  43. package/scripts/release-transaction.mjs +55 -17
  44. package/scripts/self-update.mjs +63 -10
  45. package/scripts/user-settings.mjs +4 -640
@@ -0,0 +1,808 @@
1
+ // advocacy-outcomes.mjs — the ledger that tells us whether our own advocacy was RIGHT.
2
+ //
3
+ // THE MISSING HALF. ADR-027 gave the brain a voice: detect a dormant capability, recommend it,
4
+ // execute it, reverse it. ADR-028 then defined the honest measure of that voice — precision,
5
+ // "recommendations acted on ÷ recommendations fired, target ≥ 0.60. Below this we are nagging, and a
6
+ // nag trains users to ignore the real alarm." That number has never been computable, because nothing
7
+ // in this system records what happened AFTER a recommendation was shown. Every offer vanished the
8
+ // moment the page closed. A system that cannot see its own outcomes cannot improve, and one that
9
+ // reports a metric it cannot source is doing the fabrication this repo has a CI gate against.
10
+ //
11
+ // So: an append-only outcome ledger. Every offer resolves into exactly one record — applied,
12
+ // dismissed, or ignored — and those three records are the only evidence any claim about proactivity
13
+ // is allowed to rest on.
14
+ //
15
+ // THE ONE WAY TO FABRICATE THIS METRIC, named here so a reviewer can check for it: record only the
16
+ // applies. Precision is applied ÷ (applied + dismissed + ignored), so a caller that forgets to
17
+ // record the misses reports a beautiful 1.0. The invariant is therefore not "record outcomes" but
18
+ // "every offer produces exactly one record" — and `ignored` is what an unresolved offer becomes when
19
+ // the session ends. If you are adding a caller, the ignored-path is the one to write first.
20
+ //
21
+ // THE ASYMMETRY THIS FILE EXISTS TO ENCODE. The adversarial review of ADR-031 (GPT-5.6-Sol,
22
+ // 2026-07-22) killed the previous learning signal with one sentence: "repeat count measures the
23
+ // USER'S FRUSTRATION, not the lesson's correctness... a formatting preference corrected 52 times
24
+ // dominates a security rule corrected once." A dismissal ledger repeats that mistake exactly if it is
25
+ // read as a popularity contest, so it is not read as one here:
26
+ //
27
+ // A dismissal is evidence about FIT, not about IMPORTANCE.
28
+ //
29
+ // "Not for me" and "not worth an interruption" are the same click. So the click cannot be allowed to
30
+ // mean the same thing for a cosmetic suggestion and for a corrupt-database warning — a nag dismissed
31
+ // once should vanish, and a high-severity finding dismissed once must not. That asymmetry is
32
+ // DISMISSAL_BUDGET below, and it is the whole design; the rest is bookkeeping.
33
+ //
34
+ // STORAGE. `~/.config/ruvnet-brain/` — user-level, and deliberately OUTSIDE `~/.cache/ruvnet-brain/`
35
+ // which `--update` replaces wholesale. Same reasoning as lesson-store.mjs: an outcome destroyed by
36
+ // the next release never compounds, and compounding (ADR-028 L5) is the only point of any of this.
37
+ //
38
+ // PURITY: node builtins only, no spawn, no network. It is read by surfaces; it does not render.
39
+ //
40
+ // WIRED (2026-07-23). Until this build `shouldStillOffer()` had ZERO production callers —
41
+ // `anticipate.sh` kept its OWN binary dismissed-Set (one dismissal muted forever, no severity, no
42
+ // budget) as a second, disconnected suppression policy, and this file's asymmetric budget sat
43
+ // uncalled. `anticipate.sh` is now the single caller for every mode (suggest AND
44
+ // dismiss/undismiss/status): it asks `shouldStillOffer()` and nothing else decides. `reconcileIgnored()`
45
+ // likewise had zero callers; `onboarding-console.mjs`'s `/api/capabilities` handler now supplies its
46
+ // pending-and-stale ids via this file's `pendingOffers()` (see `findStaleOffers()` there for the
47
+ // staleness rule, which is deliberately this file's caller's decision, not this file's).
48
+
49
+ import crypto from 'node:crypto';
50
+ import fs from 'node:fs';
51
+ import os from 'node:os';
52
+ import path from 'node:path';
53
+
54
+ const HOME = os.homedir();
55
+
56
+ /**
57
+ * ACTIONS — what became of an offer.
58
+ *
59
+ * The first three are the closed set ADR-028's precision metric is defined over: an offer that was
60
+ * shown ends as exactly one of them. `ignored` is a real, declared value rather than an absence,
61
+ * for the same reason UNDO_KINDS.NONE is one in remedy-registry.mjs: "the user did nothing" and
62
+ * "nobody wrote the code to record it" must never look identical, and the second is what silently
63
+ * inflates precision.
64
+ *
65
+ * RESET is the fourth, and it is here because of a house rule, not because the metric needs it.
66
+ * Dismissal is a control — it makes the brain stop speaking — and this repo does not ship a control
67
+ * without a real inverse (remedy-registry.mjs exists because a recommendation once promised an undo
68
+ * that had no branch behind it, and reported "nothing to undo" instead of failing). Suppression with
69
+ * no way back would be that same dead button pointed at silence. A reset is a CHECKPOINT, never a
70
+ * deletion: the ledger stays append-only and complete, and only the suppression arithmetic starts
71
+ * counting again after it.
72
+ */
73
+ export const ACTIONS = Object.freeze({
74
+ OFFERED: 'offered',
75
+ APPLIED: 'applied',
76
+ DISMISSED: 'dismissed',
77
+ IGNORED: 'ignored',
78
+ RESET: 'reset',
79
+ });
80
+ const ACTION_VALUES = new Set(Object.values(ACTIONS));
81
+
82
+ /**
83
+ * `OFFERED` is the PENDING marker: the card was shown, and nothing has become of it yet. It is NOT a
84
+ * resolution and never enters the precision denominator (that would let merely showing a card move
85
+ * the metric). It exists for one reason — so `reconcileApplied()` can tell "we suggested this and the
86
+ * user then turned it on" (an APPLIED) apart from "it was already on". Formalising it here also ends a
87
+ * real schema drift: `anticipate.sh` was already writing `action:'offered'` through its own inline
88
+ * recorder, while the canonical `record()` below rejected that action — so every offered row it wrote
89
+ * was inert, counted by nothing. Now both writers speak one vocabulary and `record()` accepts it.
90
+ */
91
+ /** The three RESOLUTIONS. `offered` is pending (not a resolution); `reset` is a ledger checkpoint. */
92
+ const OFFER_ACTIONS = new Set([ACTIONS.APPLIED, ACTIONS.DISMISSED, ACTIONS.IGNORED]);
93
+
94
+ /**
95
+ * THE ASYMMETRY, as numbers.
96
+ *
97
+ * A `normal` item spends its whole budget on ONE dismissal: the user said no, and for a suggestion
98
+ * that is the end of the conversation. Cheap to honour, and the cost of being wrong is that they
99
+ * miss a nicety.
100
+ *
101
+ * A `high` item costs three, because the cost of being wrong runs the other way. The finding this
102
+ * mechanism will most often suppress is the 2026-07-21 case: a corrupt AgentDB store, detected,
103
+ * scored 49/100, rendered — and the owner had to notice it himself. If one distracted click could
104
+ * bury that class of finding permanently, this file would have shipped a regression dressed as a
105
+ * feature. Three refusals is a considered no; one is a busy hand.
106
+ *
107
+ * IGNORE_WEIGHT prices silence at a fifth of a refusal. Silence is the weakest signal we have — it is
108
+ * consistent with "no", with "later", and with "I never saw the card" — so it may accumulate into
109
+ * suppression (a card ignored fifteen times IS a nag) but it may never be mistaken for an answer.
110
+ *
111
+ * APPLIED_CREDIT lets acting on a recommendation buy back a stretch of ignores, because clicking it
112
+ * is the single strongest evidence of fit we can observe, and a wanted card that fires again when
113
+ * the state recurs is not a nag.
114
+ *
115
+ * HARD_DISMISSAL_CAP is the ceiling above severity: after five explicit refusals nothing re-fires,
116
+ * ever, at any severity, whatever the evidence says. At that point we are wrong about the user, not
117
+ * about the machine — and ADR-028's own anti-goal list puts "interruption without an off switch"
118
+ * beside nagging.
119
+ */
120
+ export const DISMISSAL_BUDGET = Object.freeze({ normal: 1, high: 3 });
121
+ export const IGNORE_WEIGHT = 0.2;
122
+ export const APPLIED_CREDIT = 1;
123
+ export const HARD_DISMISSAL_CAP = 5;
124
+
125
+ /** ADR-028's stated target and the sample floor below which reporting against it would be noise. */
126
+ export const PRECISION_TARGET = 0.60;
127
+ export const MIN_PRECISION_SAMPLES = 5;
128
+
129
+ /**
130
+ * THE GOODHART GUARD FOR PRECISION — the one the 4.0 briefing named and left open ("the fixture's
131
+ * separation-of-authorities design guards recall; the equivalent guard for precision needs the real
132
+ * ledger to exist first").
133
+ *
134
+ * It turns out not to need the ledger at all, because the hole is arithmetic. Once a metric gates a
135
+ * release it becomes a target, and precision = applied/offered has an obvious exploit: OFFER LESS.
136
+ * Suggest only the sure thing, and precision approaches 1.00 while the product helps nobody — which
137
+ * is the exact behaviour ADR-028 exists to prevent, certified by the metric meant to detect it.
138
+ *
139
+ * The old `meetsTarget: value >= PRECISION_TARGET` compared the POINT ESTIMATE, so 4 applied out of
140
+ * 6 offers read 0.667 >= 0.60 and passed, while the true rate consistent with that evidence goes far
141
+ * below 0.30. Worse, MIN_PRECISION_SAMPLES = 5 cannot support the target under ANY outcome: a
142
+ * perfect 5/5 bounds at 0.05^(1/5) = 54.9%, still short of 0.60. The floor was unreachable and the
143
+ * comparison was to the wrong number.
144
+ *
145
+ * Comparing the 95% LOWER BOUND to the target closes both. Fewer offers widen the interval, so
146
+ * withholding offers can no longer manufacture a passing score — it makes the metric report "not
147
+ * yet judgeable" instead. The incentive now points the right way: the only route to a certified
148
+ * precision is to offer MORE and be right.
149
+ *
150
+ * Exact Clopper-Pearson: the lower bound solves P(X >= k | n, p) = alpha. Computed by bisection on
151
+ * the binomial tail with exact terms (n is tiny here). Self-checkable: at k = n it must reduce to
152
+ * alpha^(1/n) — asserted in the tests rather than trusted, because a hand-rolled incomplete-beta in
153
+ * this same session returned 98.3% for n=3, which is absurd on its face.
154
+ */
155
+ export const PRECISION_ALPHA = 0.05;
156
+
157
+ export function precisionLowerBound(k, n, alpha = PRECISION_ALPHA) {
158
+ if (!Number.isFinite(k) || !Number.isFinite(n) || n <= 0 || k < 0 || k > n) return null;
159
+ if (k === 0) return 0;
160
+ const tailAtLeastK = (p) => {
161
+ // sum_{i=k}^{n} C(n,i) p^i (1-p)^(n-i), computed with a running coefficient to avoid factorials.
162
+ let sum = 0;
163
+ let coeff = 1; // C(n,0)
164
+ for (let i = 0; i <= n; i++) {
165
+ if (i >= k) sum += coeff * Math.pow(p, i) * Math.pow(1 - p, n - i);
166
+ coeff = coeff * (n - i) / (i + 1); // C(n,i) -> C(n,i+1)
167
+ }
168
+ return sum;
169
+ };
170
+ let lo = 0;
171
+ let hi = 1;
172
+ for (let i = 0; i < 200; i++) {
173
+ const mid = (lo + hi) / 2;
174
+ if (tailAtLeastK(mid) < alpha) lo = mid; else hi = mid;
175
+ }
176
+ return (lo + hi) / 2;
177
+ }
178
+
179
+ // Field caps. These are a CORRECTNESS property, not tidiness — see appendLine() below: the atomicity
180
+ // of a concurrent append depends on each record being one small write. EXPORTED so a test can assert
181
+ // the arithmetic that makes the guard in appendLine() unreachable: every field is bounded, and the
182
+ // bounds sum to well under MAX_RECORD_BYTES. The guard stays anyway, as the tripwire that fires the
183
+ // day somebody raises one of these caps without redoing that sum.
184
+ export const MAX_ID = 200;
185
+ export const MAX_PROJECT = 120;
186
+ export const MAX_HASH = 64;
187
+ export const MAX_SEVERITY = 32;
188
+ export const MAX_RECORD_BYTES = 1024;
189
+
190
+ export const OUTCOMES_PATH = process.env.RUVNET_ADVOCACY_OUTCOMES
191
+ || path.join(HOME, '.config', 'ruvnet-brain', 'advocacy-outcomes.jsonl');
192
+
193
+ /**
194
+ * Severity → the two classes the budget is defined over.
195
+ *
196
+ * Accepts console-engine's vocabulary (`INFO` | `SUGGESTED` | `IMPORTANT`) and lesson-store's
197
+ * (`normal` | `high`), because both produce things that get offered and neither is going to change
198
+ * to suit this file.
199
+ *
200
+ * UNKNOWN SEVERITY RESOLVES TO `normal`, i.e. to the quieter class, and that direction is deliberate.
201
+ * It means a caller that forgets to pass severity gets an item silenced after one dismissal rather
202
+ * than one that is nearly unsilenceable. ADR-028: "One false alarm costs more trust than ten true
203
+ * ones earn. Non-negotiable." When we do not know, we err toward respecting the refusal — and
204
+ * because record() stores the severity it was told, the history stays self-describing rather than
205
+ * quietly re-classified later.
206
+ */
207
+ export function weightClass(severity) {
208
+ const s = String(severity ?? '').trim().toLowerCase();
209
+ return (s === 'important' || s === 'high' || s === 'critical') ? 'high' : 'normal';
210
+ }
211
+
212
+ /**
213
+ * A stable fingerprint of the evidence a recommendation was built from.
214
+ *
215
+ * ADR-027's rule is "offered once per state change, dismissible, never re-fires while dismissed" —
216
+ * which is only implementable if "the state" is a value something can compare. This is that value:
217
+ * hash what we OBSERVED, not what we said about it, so rewording a card does not read as new
218
+ * evidence and re-open a settled question.
219
+ *
220
+ * Returns null for no evidence. Null is honest ("we cannot tell whether the state changed") and it
221
+ * is inert by construction: the state-change reprieve in shouldStillOffer() requires a real hash on
222
+ * both sides, so an unknown state can never argue its way past a dismissal.
223
+ */
224
+ export function stateHashOf(evidence) {
225
+ const items = (Array.isArray(evidence) ? evidence : [evidence])
226
+ .map((e) => {
227
+ if (e === null || e === undefined) return '';
228
+ if (typeof e === 'object') return String(e.observed ?? JSON.stringify(e));
229
+ return String(e);
230
+ })
231
+ .map((s) => s.trim())
232
+ .filter(Boolean)
233
+ .sort(); // order of evidence is presentation, not state
234
+ if (!items.length) return null;
235
+ return crypto.createHash('sha256').update(items.join('')).digest('hex').slice(0, 16);
236
+ }
237
+
238
+ function toIso(at) {
239
+ if (at instanceof Date) return Number.isNaN(at.getTime()) ? new Date().toISOString() : at.toISOString();
240
+ const d = new Date(at);
241
+ return Number.isNaN(d.getTime()) ? new Date().toISOString() : d.toISOString();
242
+ }
243
+
244
+ /**
245
+ * THE WRITE. One line, one open-with-O_APPEND, one write() — and no read step at all.
246
+ *
247
+ * This repo has already paid for the alternative. saveSettings() did read-modify-write on a JSON
248
+ * object, and MEASURED across 20 trials of four simultaneous writers, at least one setting was lost
249
+ * in 19 of them — every writer returning ok:true, no error, no warning. The fix there was a lock,
250
+ * because a settings file genuinely is a single mutable object.
251
+ *
252
+ * A ledger is not. Append-only removes the read, and with the read goes the entire class of bug:
253
+ * there is no prior value to clobber. That is why this file is JSONL and not a JSON array, and the
254
+ * shape is load-bearing rather than stylistic — an array would reintroduce read-modify-write and
255
+ * with it the 19-in-20 silent loss, on the surface whose only job is to remember what the user chose.
256
+ *
257
+ * The remaining hazard is a partial write interleaving with another process's. POSIX makes the
258
+ * offset-advance-and-write atomic for a single write() on an O_APPEND fd; Node issues one write()
259
+ * for a single small buffer. So the size cap is the guarantee: every field is truncated, and a
260
+ * record that still exceeds MAX_LINE_BYTES is refused rather than written and hoped for. And because
261
+ * a torn line is still conceivable on an exotic filesystem, loadOutcomes() drops unparseable lines
262
+ * instead of failing — one damaged record costs one record, never the ledger.
263
+ */
264
+ function appendLine(file, row) {
265
+ const line = JSON.stringify(row) + '\n';
266
+ if (Buffer.byteLength(line) > MAX_RECORD_BYTES) {
267
+ throw new Error(`Outcome for "${row.id}" invalid: record is ${Buffer.byteLength(line)} bytes, over the ${MAX_RECORD_BYTES}-byte cap that keeps a concurrent append atomic`);
268
+ }
269
+ fs.mkdirSync(path.dirname(file), { recursive: true });
270
+ fs.appendFileSync(file, line);
271
+ return line;
272
+ }
273
+
274
+ /**
275
+ * Record what became of one offer. Append-only; nothing here ever rewrites history.
276
+ *
277
+ * THROWS on a malformed record — same discipline as makeRecommendation() and makeLesson(): an
278
+ * invariant belongs in the constructor, not in a reviewer's memory. An unknown `action` written
279
+ * quietly would corrupt the precision denominator forever, and the wrongness would show up as a
280
+ * plausible number rather than as an error.
281
+ *
282
+ * DOES NOT THROW on an I/O failure — it returns `{ ok: false, reason }`, because callers are
283
+ * surfaces and a read-only home directory must not take down the console. But a caller MUST surface
284
+ * a failed `dismissed`: if the write fails silently, the user's "stop showing me this" does not
285
+ * stick, they see the same card tomorrow, and the off switch has become theatre. That is the exact
286
+ * failure shape as the undo that reported "nothing to undo" — a control that reports success and
287
+ * does nothing.
288
+ *
289
+ * @param {{id:string, action:string, at?:Date|string, project?:string, severity?:string|null,
290
+ * stateHash?:string|null, scope?:'forever'|null}} spec
291
+ */
292
+ /**
293
+ * Under a test runner, writing to the DEFAULT (real, user-level) ledger is a bug, not a choice.
294
+ *
295
+ * Found by an independent grader on 2026-07-24: the live ledger at ~/.config/ruvnet-brain/ held
296
+ * exactly one row, `{"id":"f-adv-1","stateHash":"hash-1",...}` — fixture-shaped data in the user's
297
+ * real outcome record, describing an event that never happened. Every precision number this product
298
+ * reports is computed over that file, and it had junk in it from the first day it existed.
299
+ *
300
+ * It got there because a test called record() without passing `{file}`, so the default path won. The
301
+ * tempting fix is to blocklist ids that look like fixtures (`f-*`, `hash-*`), but that is a guess
302
+ * about naming, and the next fixture that does not match the pattern lands in the ledger exactly the
303
+ * same way. The defect is not the id — it is that a test can address the real file at all.
304
+ *
305
+ * So the write refuses instead. Under vitest, an explicit `file` (or RUVNET_ADVOCACY_OUTCOMES) is
306
+ * mandatory; the default is unreachable. That makes the pollution impossible by construction rather
307
+ * than unlikely by convention, and it fails LOUD at the moment the test is written rather than
308
+ * silently into a file nobody reads until a grader opens it thirteen months of commits later.
309
+ */
310
+ const UNDER_TEST = !!(process.env.VITEST || process.env.VITEST_WORKER_ID);
311
+
312
+ export function record(spec, { file = OUTCOMES_PATH } = {}) {
313
+ if (UNDER_TEST && file === OUTCOMES_PATH && !process.env.RUVNET_ADVOCACY_OUTCOMES) {
314
+ throw new Error(
315
+ 'advocacy-outcomes.record() refused: a test tried to write to the REAL user ledger at '
316
+ + `${OUTCOMES_PATH}. Pass {file: <tmp path>} or set RUVNET_ADVOCACY_OUTCOMES. `
317
+ + '(A fixture row reached the live ledger this way once and was found only by an outside grader.)',
318
+ );
319
+ }
320
+ const {
321
+ id, action, at = new Date(), project = null,
322
+ severity = null, stateHash = null, scope = null,
323
+ } = spec || {};
324
+ const err = (m) => { throw new Error(`Outcome for "${id ?? '?'}" invalid: ${m}`); };
325
+
326
+ if (!id || typeof id !== 'string') err('missing id — an outcome that cannot name the recommendation it belongs to measures nothing');
327
+ if (!ACTION_VALUES.has(action)) err(`action must be one of: ${[...ACTION_VALUES].join(', ')}`);
328
+ // `scope:'forever'` is the one-action permanent silence ADR-028 requires ("anything that speaks
329
+ // in-session must be silenceable in one action, permanently, without penalty"). It is meaningless
330
+ // on anything but a dismissal, and accepting it elsewhere would let a stray field mute a card
331
+ // nobody asked to mute.
332
+ if (scope !== null && scope !== 'forever') err(`scope must be null or "forever" (got ${JSON.stringify(scope)})`);
333
+ if (scope === 'forever' && action !== ACTIONS.DISMISSED) err('scope:"forever" is only meaningful on a dismissal');
334
+
335
+ const row = {
336
+ v: 1,
337
+ id: id.slice(0, MAX_ID),
338
+ action,
339
+ at: toIso(at),
340
+ // The project is recorded but is NOT a scope — see shouldStillOffer(). It is here so the ledger
341
+ // can answer "where did this happen", which is what ADR-028's L5 test is phrased in terms of.
342
+ project: String(project ?? path.basename(process.cwd())).slice(0, MAX_PROJECT),
343
+ severity: severity === null ? null : String(severity).slice(0, MAX_SEVERITY),
344
+ stateHash: stateHash === null ? null : String(stateHash).slice(0, MAX_HASH),
345
+ scope: scope ?? null,
346
+ };
347
+
348
+ try {
349
+ appendLine(file, row);
350
+ return { ok: true, file, row };
351
+ } catch (e) {
352
+ // Over-cap is a programming error and was already thrown by appendLine before any write; an
353
+ // ENOSPC/EACCES/EROFS is the environment. Both arrive here as a receipt so the caller can decide
354
+ // how loud to be, and the reason is preserved rather than flattened to a boolean.
355
+ return { ok: false, reason: e.code || e.message, row };
356
+ }
357
+ }
358
+
359
+ /**
360
+ * Read the ledger. NEVER THROWS — a missing file, a corrupt file, a half-written last line, a file
361
+ * full of someone else's JSON: all of them degrade to "no outcomes yet".
362
+ *
363
+ * This is the same contract lesson-gate.mjs holds itself to and for the same reason: a mechanism
364
+ * that suppresses recommendations must fail toward SPEAKING. If an unreadable ledger threw, or worse
365
+ * returned a partial count that happened to look like a spent budget, a corrupt file would silence
366
+ * the brain — and it would be silent in exactly the way it is silent when everything is healthy, so
367
+ * nobody would ever find out.
368
+ */
369
+ export function loadOutcomes(file = OUTCOMES_PATH) {
370
+ let raw;
371
+ try { raw = fs.readFileSync(file, 'utf8'); } catch { return []; }
372
+ const out = [];
373
+ for (const line of raw.split('\n')) {
374
+ const s = line.trim();
375
+ if (!s) continue;
376
+ let r;
377
+ try { r = JSON.parse(s); } catch { continue; } // torn or hand-mangled line: drop it, keep the rest
378
+ if (!r || typeof r !== 'object') continue;
379
+ if (typeof r.id !== 'string' || !r.id) continue;
380
+ if (!ACTION_VALUES.has(r.action)) continue; // an action we do not understand is not counted as one we do
381
+ out.push({
382
+ v: Number(r.v) || 1,
383
+ id: r.id,
384
+ action: r.action,
385
+ at: typeof r.at === 'string' ? r.at : null,
386
+ project: typeof r.project === 'string' ? r.project : null,
387
+ severity: typeof r.severity === 'string' ? r.severity : null,
388
+ stateHash: typeof r.stateHash === 'string' ? r.stateHash : null,
389
+ scope: r.scope === 'forever' ? 'forever' : null,
390
+ });
391
+ }
392
+ return out;
393
+ }
394
+
395
+ /**
396
+ * The records for one id that the suppression arithmetic is allowed to see: everything appended
397
+ * after the most recent `reset`.
398
+ *
399
+ * ORDERED BY FILE POSITION, NOT BY `at`. The timestamp comes from whichever process wrote it, and a
400
+ * machine with a skewed clock (or a caller passing its own `at`, which record() permits) could
401
+ * otherwise re-order a reset behind the dismissals it was meant to clear — resurrecting a
402
+ * suppression the user explicitly lifted. Append order is the one ordering we actually control.
403
+ */
404
+ function liveRecords(id, all) {
405
+ const mine = all.filter((r) => r.id === id);
406
+ let start = 0;
407
+ for (let i = mine.length - 1; i >= 0; i--) {
408
+ if (mine[i].action === ACTIONS.RESET) { start = i + 1; break; }
409
+ }
410
+ return mine.slice(start);
411
+ }
412
+
413
+ /**
414
+ * What we know about one recommendation.
415
+ *
416
+ * `precision` is null — not 0 — when nothing has been offered yet. This is the repo's oldest live
417
+ * rule: a detector once read a CLI's table and reported "26 hooks off" while the learner held 457
418
+ * trajectories, because unknown rendered as off. A recommendation nobody has seen has an UNKNOWN
419
+ * precision; rendering that as 0.00 would say "this advice is always rejected" about advice that has
420
+ * never been given.
421
+ */
422
+ export function outcomesFor(id, { file = OUTCOMES_PATH, all = null, project = null } = {}) {
423
+ let recs = liveRecords(id, all ?? loadOutcomes(file));
424
+ if (project) recs = recs.filter((r) => r.project === project);
425
+
426
+ const count = (a) => recs.filter((r) => r.action === a).length;
427
+ const applied = count(ACTIONS.APPLIED);
428
+ const dismissed = count(ACTIONS.DISMISSED);
429
+ const ignored = count(ACTIONS.IGNORED);
430
+ const offered = applied + dismissed + ignored;
431
+
432
+ const dismissals = recs.filter((r) => r.action === ACTIONS.DISMISSED);
433
+ const offers = recs.filter((r) => OFFER_ACTIONS.has(r.action));
434
+ const last = offers.length ? offers[offers.length - 1] : null;
435
+
436
+ return {
437
+ id,
438
+ applied,
439
+ dismissed,
440
+ ignored,
441
+ offered,
442
+ precision: offered ? +(applied / offered).toFixed(4) : null,
443
+ projects: [...new Set(recs.map((r) => r.project).filter(Boolean))],
444
+ silencedForever: dismissals.some((r) => r.scope === 'forever'),
445
+ lastAction: last?.action ?? null,
446
+ lastAt: last?.at ?? null,
447
+ lastSeverity: [...offers].reverse().find((r) => r.severity)?.severity ?? null,
448
+ lastDismissal: dismissals.length ? dismissals[dismissals.length - 1] : null,
449
+ };
450
+ }
451
+
452
+ /**
453
+ * Should this recommendation be offered again? The question ADR-027 phrases as "dismissible, never
454
+ * re-fires while dismissed".
455
+ *
456
+ * NOT SCOPED BY PROJECT, AND THAT IS THE POINT. A dismissal recorded while working in project A
457
+ * suppresses the same recommendation in project B. This is the falsifiable L5 claim in ADR-028 —
458
+ * "a lesson validated in project A demonstrably changes behaviour in project B" — expressed on the
459
+ * signal we can actually observe today, and it is also just true of the subject matter: these
460
+ * recommendations are about the user's MACHINE (a dormant learner, a corrupt store, a stale install),
461
+ * so per-repo suppression would ask the same person the same question once per checkout.
462
+ *
463
+ * The order of the checks is the safety argument:
464
+ * 1. Never offered → offer. Silence has to be earned.
465
+ * 2. Silenced forever → never. One action, permanent, no penalty, no severity override. A finding
466
+ * important enough to argue past an explicit permanent mute does not exist; that argument is
467
+ * what turns a notification system into spam.
468
+ * 3. Budget by severity class → the asymmetry. A nag dies on one dismissal; a high-severity
469
+ * finding needs three, so a distracted click cannot bury a corrupt database.
470
+ * 4. State-change reprieve, HIGH SEVERITY ONLY. New evidence re-opens a high-severity question,
471
+ * because the underlying risk genuinely changed. It does NOT re-open a suggestion: for a nag, a
472
+ * changed number is not new information worth interrupting a person for, and granting it a
473
+ * reprieve would let a flapping metric nag forever through a budget it had already spent.
474
+ * 5. HARD_DISMISSAL_CAP overrides even that.
475
+ */
476
+ export function shouldStillOffer(id, {
477
+ severity = null, stateHash = null, file = OUTCOMES_PATH, all = null,
478
+ } = {}) {
479
+ const o = outcomesFor(id, { file, all });
480
+
481
+ if (o.silencedForever) return false;
482
+ if (!o.offered) return true;
483
+
484
+ // Severity is DERIVED per offer from evidence measured on this machine (ADR-028: "Severity is
485
+ // derived from measured evidence on this machine. Nothing is IMPORTANT because it would be good
486
+ // for adoption."), so the CURRENT call's severity wins over what history recorded. A capability
487
+ // whose dormancy has become serious must not stay suppressed because it was cosmetic last month.
488
+ const cls = weightClass(severity ?? o.lastSeverity);
489
+ const budget = DISMISSAL_BUDGET[cls];
490
+
491
+ // Dismissals count in full; silence counts at a fifth; having actually used it buys credit back.
492
+ // Floor at zero so a long history of applies cannot bank immunity against a later refusal.
493
+ const spend = Math.max(0, o.dismissed + (IGNORE_WEIGHT * o.ignored) - (APPLIED_CREDIT * o.applied));
494
+ if (spend < budget) return true;
495
+
496
+ if (o.dismissed >= HARD_DISMISSAL_CAP) return false;
497
+ if (cls === 'high' && stateHash && o.lastDismissal?.stateHash && stateHash !== o.lastDismissal.stateHash) {
498
+ return true;
499
+ }
500
+ return false;
501
+ }
502
+
503
+ /**
504
+ * claimOffer — the ATOMIC step shouldStillOffer() cannot provide, and the reason it needs one.
505
+ *
506
+ * GPT-5.6-Sol found this in the ADR-047 duel and it is a genuine race, not a theoretical one: he
507
+ * drove shouldStillOffer() to `true` while TWENTY offers for the same finding sat pending. The cause
508
+ * is structural — shouldStillOffer() is a pure READ over the ledger. Two Claude Code sessions in two
509
+ * terminals (the normal way this product is used) both read "not yet offered", both conclude yes,
510
+ * and the user is told the same thing twice. Nothing between the read and the write said "mine".
511
+ *
512
+ * A lock around the whole decision would be the obvious fix and the wrong one: the decision reads
513
+ * the ledger, and this repo has already been burned by holding a lock across a read (updateLessons
514
+ * read outside its own lock and the "safe" version raced anyway). So the claim is narrow — it does
515
+ * not protect the decision, it protects the RIGHT TO SPEAK.
516
+ *
517
+ * The primitive is `open(..., 'wx')`: exclusive create, which the OS guarantees is atomic. Exactly
518
+ * one caller can create a given claim file; everyone else gets EEXIST and stays quiet.
519
+ *
520
+ * TTL, because a crashed session must not silence a capability forever. A claim older than ttlMs is
521
+ * abandoned and may be taken over — the same reasoning as any lease. The default is deliberately
522
+ * short: the cost of a stale claim is a MISSED offer (the product's whole reason to exist), while
523
+ * the cost of taking one over early is a duplicate — annoying, not silencing. Between those two
524
+ * failure modes, this system must always fail toward speaking.
525
+ *
526
+ * Returns true if THIS caller owns the right to offer. The caller then record()s the `offered` row.
527
+ */
528
+ export function claimOffer(id, { dir = null, ttlMs = 60_000, now = Date.now() } = {}) {
529
+ if (!id || typeof id !== 'string') return false;
530
+ const base = dir || path.join(path.dirname(OUTCOMES_PATH), 'offer-claims');
531
+ const key = crypto.createHash('sha256').update(id).digest('hex').slice(0, 24);
532
+ const file = path.join(base, `${key}.claim`);
533
+
534
+ try { fs.mkdirSync(base, { recursive: true }); } catch { return true; } // cannot claim ⇒ fail toward speaking
535
+
536
+ // WRITE-THEN-LINK, and the reason is a bug this file's own concurrency test caught.
537
+ //
538
+ // The obvious implementation is open(file,'wx') followed by write(). It is wrong, and it fails
539
+ // exactly where it matters: `wx` publishes the filename BEFORE the content is written, so there is
540
+ // a window in which the claim exists and is EMPTY. Competing processes read it, fail to parse it,
541
+ // conclude "unknown age ⇒ stale ⇒ take it over", and speak. MEASURED with 12 real OS processes:
542
+ // FIVE of twelve won. A single-process test would have shown one winner and hidden it completely.
543
+ //
544
+ // link() closes the window. The content is written to a private temp file first, so the moment the
545
+ // claim name becomes visible it is already complete and parseable. link() itself fails with EEXIST
546
+ // when the target exists, giving the same atomic exactly-one-winner guarantee — with no torn state
547
+ // for the losers to misread.
548
+ const take = () => {
549
+ const tmp = `${file}.${process.pid}.${Math.abs(now % 1e9)}.tmp`;
550
+ try {
551
+ fs.writeFileSync(tmp, JSON.stringify({ id, at: new Date(now).toISOString(), pid: process.pid }));
552
+ try {
553
+ fs.linkSync(tmp, file); // ATOMIC create-if-absent, content already durable
554
+ return true;
555
+ } catch (e) {
556
+ if (e.code !== 'EEXIST') return true; // an unexpected FS error must not silence us
557
+ return null; // genuinely held — staleness decided below
558
+ } finally {
559
+ try { fs.unlinkSync(tmp); } catch { /* best effort */ }
560
+ }
561
+ } catch { return true; } // cannot even stage a claim ⇒ fail toward speaking
562
+ };
563
+
564
+ const first = take();
565
+ if (first !== null) return first;
566
+
567
+ // Someone holds it. Stale?
568
+ let heldAt = 0;
569
+ try { heldAt = Date.parse(JSON.parse(fs.readFileSync(file, 'utf8')).at) || 0; } catch { heldAt = 0; }
570
+ if (now - heldAt < ttlMs) return false; // live claim — stay quiet, this is the duplicate we came to prevent
571
+
572
+ // Abandoned. Take it over by REPLACING atomically, so two reapers cannot both win.
573
+ const tmp = `${file}.${process.pid}.${key.slice(0, 6)}`;
574
+ try {
575
+ fs.writeFileSync(tmp, JSON.stringify({ id, at: new Date(now).toISOString(), pid: process.pid, tookOver: true }));
576
+ fs.renameSync(tmp, file); // atomic replace
577
+ return true;
578
+ } catch {
579
+ try { fs.unlinkSync(tmp); } catch { /* best effort */ }
580
+ return true; // could not arbitrate ⇒ fail toward speaking
581
+ }
582
+ }
583
+
584
+ /** Release a claim once the offer is resolved (applied/dismissed), so a later dormancy can re-offer. */
585
+ export function releaseClaim(id, { dir = null } = {}) {
586
+ if (!id || typeof id !== 'string') return false;
587
+ const base = dir || path.join(path.dirname(OUTCOMES_PATH), 'offer-claims');
588
+ const key = crypto.createHash('sha256').update(id).digest('hex').slice(0, 24);
589
+ try { fs.unlinkSync(path.join(base, `${key}.claim`)); return true; } catch { return false; }
590
+ }
591
+
592
+ /**
593
+ * The pending offer for one id, or null. Pending = the most recent `offered` (since the last reset)
594
+ * has no resolution after it. Ordered by file position, not `at`, for the same clock-skew reason as
595
+ * liveRecords().
596
+ */
597
+ function pendingOffer(id, all) {
598
+ const mine = liveRecords(id, all);
599
+ let idx = -1;
600
+ for (let i = mine.length - 1; i >= 0; i--) {
601
+ if (mine[i].action === ACTIONS.OFFERED) { idx = i; break; }
602
+ }
603
+ if (idx === -1) return null; // never offered since the last reset
604
+ for (let i = idx + 1; i < mine.length; i++) {
605
+ if (OFFER_ACTIONS.has(mine[i].action)) return null; // already resolved
606
+ }
607
+ return mine[idx];
608
+ }
609
+
610
+ /**
611
+ * Every id with a currently-pending offer — the bulk, read-only form of the per-id check
612
+ * pendingOffer() already makes inside reconcileApplied()/reconcileIgnored(). Exists so a CALLER can
613
+ * decide its OWN staleness rule (wall-clock age, a newer offer superseding it, a session count) over
614
+ * a real `at` timestamp, without re-implementing the reset-aware, position-ordered definition of
615
+ * "pending" that lives here. Read-only: it records nothing and never throws.
616
+ *
617
+ * @returns {Array<{id:string, at:string|null, severity:string|null, project:string|null, stateHash:string|null}>}
618
+ */
619
+ export function pendingOffers({ file = OUTCOMES_PATH, all = null } = {}) {
620
+ let recs;
621
+ try { recs = all ?? loadOutcomes(file); } catch { return []; }
622
+ const ids = [...new Set(recs.map((r) => r.id))];
623
+ const out = [];
624
+ for (const id of ids) {
625
+ const offer = pendingOffer(id, recs);
626
+ if (offer) out.push({ id, at: offer.at, severity: offer.severity, project: offer.project, stateHash: offer.stateHash });
627
+ }
628
+ return out;
629
+ }
630
+
631
+ /**
632
+ * THE NUMERATOR, DERIVED — not asserted. precision = applied ÷ (applied+dismissed+ignored), and until
633
+ * now `applied` was recorded by nothing, so the number could only ever be 0 (once a dismissal landed)
634
+ * or null. That is the inverse of the fabrication this file warns about in its header: not a beautiful
635
+ * 1.0 from recording only the applies, but a permanent 0.0 from recording none of them — advocacy that
636
+ * looks like pure nagging no matter how well it lands.
637
+ *
638
+ * The honest signal for "the user acted on our suggestion" is a state transition we can OBSERVE: a
639
+ * capability we OFFERED, still pending, is now measured `on`. That is an APPLIED. We do not guess and
640
+ * we do not credit an offer the user resolved some other way — only a pending offer whose capability
641
+ * the audit now reports on. A capability that was already on when we offered it cannot go on again, so
642
+ * it cannot be double-counted; and a dismissed or ignored offer is no longer pending, so turning it on
643
+ * later (for reasons of their own) is not miscredited to us.
644
+ *
645
+ * NEVER THROWS — surfaces call it. Its writes go through record(), which returns a receipt on I/O
646
+ * failure rather than throwing; a lost applied costs one row, never the caller.
647
+ *
648
+ * @param {Array<{key?:string,id?:string,state?:string}>} auditRows the capability audit (auditAll()'s output)
649
+ * @returns {string[]} the ids reconciled to `applied` this call
650
+ */
651
+ export function reconcileApplied(auditRows, { file = OUTCOMES_PATH } = {}) {
652
+ if (!Array.isArray(auditRows)) return [];
653
+ let all;
654
+ try { all = loadOutcomes(file); } catch { return []; }
655
+ const done = [];
656
+ for (const row of auditRows) {
657
+ if (!row || typeof row !== 'object') continue;
658
+ const id = typeof row.key === 'string' ? row.key : (typeof row.id === 'string' ? row.id : '');
659
+ if (!id) continue;
660
+ if (row.state !== 'on') continue; // only a real, now-observed on-state
661
+ const offer = pendingOffer(id, all);
662
+ if (!offer) continue; // nothing pending to credit
663
+ const res = record({ id, action: ACTIONS.APPLIED, severity: offer.severity ?? null, project: offer.project ?? null }, { file });
664
+ if (res.ok) {
665
+ done.push(id);
666
+ // keep the in-call view consistent so a duplicate id in auditRows can't be applied twice
667
+ all.push({ id, action: ACTIONS.APPLIED, at: res.row.at, project: res.row.project, severity: res.row.severity, stateHash: null, scope: null });
668
+ }
669
+ }
670
+ return done;
671
+ }
672
+
673
+ /**
674
+ * THE DENOMINATOR'S MISSING THIRD — `ignored`, DERIVED, not guessed.
675
+ *
676
+ * ADR-028: precision = applied ÷ (applied + dismissed + ignored). `applied` was wired above by
677
+ * reconcileApplied(); `dismissed` was always recorded, because a dismissal is a click and a click has
678
+ * an event to hang a record on. `ignored` has no click — it is the ABSENCE of one — and this module
679
+ * already refuses to record an absence on a guess (see stateHashOf returning null for no evidence,
680
+ * outcomesFor's precision:null for no offers). An offer shown and never acted on nor dismissed is
681
+ * invisible today, and invisible is optimistic: it silently shrinks the denominator, the mirror image
682
+ * of the fabrication this file's header already names ("record only the applies").
683
+ *
684
+ * THE TRIGGER THIS BUILD CHOSE, AND WHY IT IS NOT A GUESS. "Ignored" is fundamentally a claim about
685
+ * TIME — the offer sat there, unresolved, long enough that the silence means something rather than
686
+ * "the user hasn't looked yet". This module has no clock of its own worth trusting for that: record()
687
+ * lets a caller pass an arbitrary `at`, and liveRecords()/pendingOffer() deliberately order by file
688
+ * position rather than timestamp for exactly the clock-skew reason documented on both of them. Picking
689
+ * a threshold HERE (say, "offered more than N days ago") would be inventing evidence this file does
690
+ * not have. So the staleness decision is left where the evidence actually lives — with the caller, who
691
+ * can say "this offer is N sessions old" or "a newer offer for the same capability just superseded
692
+ * it" — and reconcileIgnored() takes that decision as a plain list of ids rather than a clock. It stays
693
+ * pure: no Date.now(), no session counter, nothing but the ledger already on disk.
694
+ *
695
+ * WHAT MAKES IT SAFE TO CALL WITH A WRONG OR STALE LIST. Passing an id is a PROPOSAL, not a command —
696
+ * the ledger is the sole arbiter. For each id, the only question this function answers on its own
697
+ * evidence is: is there a `pendingOffer` for this id right now (an `offered` since the last reset with
698
+ * NO resolution after it)? That single check is what buys the three guarantees this build requires:
699
+ * - CANNOT double-count: the moment an id is recorded ignored, it IS a resolution — so any later
700
+ * call with the same id (a cron re-run, a duplicate in the same list) finds nothing pending.
701
+ * - CANNOT convert a real resolution: a caller that (wrongly) still lists an id the user applied or
702
+ * dismissed five minutes ago is a no-op, never an overwrite — pendingOffer() already sees the
703
+ * resolution and returns null, same as it does for reconcileApplied().
704
+ * - CANNOT invent an offer: an id that was never offered has no pendingOffer either, so a stray or
705
+ * misspelled id records nothing.
706
+ *
707
+ * NEVER THROWS — same contract as reconcileApplied(): a caller here is a surface or a scheduled job,
708
+ * and a lost `ignored` costs one row, never the caller.
709
+ *
710
+ * @param {Array<string>} pendingIds ids the CALLER has already judged pending AND stale — staleness
711
+ * (session age, wall-clock age, supersession by a newer offer) is entirely the caller's evidence;
712
+ * this function neither computes nor infers it, only verifies each id still has a real pendingOffer.
713
+ * @returns {string[]} the ids actually reconciled to `ignored` this call — a subset of pendingIds,
714
+ * only those that still had an unresolved offer to resolve.
715
+ */
716
+ export function reconcileIgnored(pendingIds, { file = OUTCOMES_PATH } = {}) {
717
+ if (!Array.isArray(pendingIds)) return [];
718
+ let all;
719
+ try { all = loadOutcomes(file); } catch { return []; }
720
+ const done = [];
721
+ for (const raw of pendingIds) {
722
+ const id = typeof raw === 'string' ? raw : '';
723
+ if (!id) continue;
724
+ const offer = pendingOffer(id, all);
725
+ if (!offer) continue; // already resolved, reset since, or never offered — nothing pending to mark
726
+ const res = record({ id, action: ACTIONS.IGNORED, severity: offer.severity ?? null, project: offer.project ?? null }, { file });
727
+ if (res.ok) {
728
+ done.push(id);
729
+ // keep the in-call view consistent so a duplicate id in pendingIds can't be recorded twice
730
+ all.push({ id, action: ACTIONS.IGNORED, at: res.row.at, project: res.row.project, severity: res.row.severity, stateHash: null, scope: null });
731
+ }
732
+ }
733
+ return done;
734
+ }
735
+
736
+ /**
737
+ * ADR-028's precision metric: recommendations acted on ÷ recommendations fired. Target ≥ 0.60.
738
+ *
739
+ * A DISMISSAL IS NOT AN ACTION. It is in the denominator and never the numerator, even though the
740
+ * user did click something. Counting it as "acted on" would let us hit target by annoying people
741
+ * into clicking X, which is the precise behaviour the metric exists to catch — the number would rise
742
+ * as the product got worse, and a metric that inverts under pressure is worse than no metric.
743
+ *
744
+ * `precision: null` when nothing has been offered, and `meetsTarget: null` below the sample floor.
745
+ * One rejected offer is not a 0.00 precision rate, and reporting it as one would be the same
746
+ * unknown-rendered-as-a-number failure this repo has a gate against. A grade we have not earned the
747
+ * right to state is stated as "not yet measurable", loudly, in the return value.
748
+ *
749
+ * COUNTS EVERY RECORDED OFFER, INCLUDING BEFORE A RESET. shouldStillOffer() honours the reset
750
+ * checkpoint because that is a user preference about the future; this is a measurement of how the
751
+ * product has actually behaved, and letting a reset launder a bad precision score would make the one
752
+ * number that judges us the one number we can clear.
753
+ */
754
+ export function precision({ file = OUTCOMES_PATH, all = null, since = null, id = null } = {}) {
755
+ let recs = (all ?? loadOutcomes(file)).filter((r) => OFFER_ACTIONS.has(r.action));
756
+ if (id) recs = recs.filter((r) => r.id === id);
757
+ if (since) {
758
+ const cut = toIso(since);
759
+ recs = recs.filter((r) => typeof r.at === 'string' && r.at >= cut);
760
+ }
761
+
762
+ const count = (a) => recs.filter((r) => r.action === a).length;
763
+ const applied = count(ACTIONS.APPLIED);
764
+ const dismissed = count(ACTIONS.DISMISSED);
765
+ const ignored = count(ACTIONS.IGNORED);
766
+ const offered = applied + dismissed + ignored;
767
+
768
+ if (!offered) {
769
+ return {
770
+ precision: null, offered: 0, applied: 0, dismissed: 0, ignored: 0,
771
+ target: PRECISION_TARGET, sufficient: false, meetsTarget: null,
772
+ reason: 'no offers recorded yet — precision is unknown, not zero',
773
+ };
774
+ }
775
+
776
+ const value = +(applied / offered).toFixed(4);
777
+ const sufficient = offered >= MIN_PRECISION_SAMPLES;
778
+ return {
779
+ precision: value,
780
+ offered, applied, dismissed, ignored,
781
+ target: PRECISION_TARGET,
782
+ sufficient,
783
+ // The lower bound, not the point estimate — see PRECISION_ALPHA above. `offered` is the sample
784
+ // and `applied` the successes, so withholding offers WIDENS this and can never buy a pass.
785
+ lowerBound: +(precisionLowerBound(applied, offered) ?? 0).toFixed(4),
786
+ meetsTarget: sufficient ? precisionLowerBound(applied, offered) >= PRECISION_TARGET : null,
787
+ reason: sufficient
788
+ ? (precisionLowerBound(applied, offered) >= PRECISION_TARGET
789
+ ? null
790
+ : `${applied}/${offered} applied — the point estimate is ${value}, but the 95% lower bound is `
791
+ + `${(precisionLowerBound(applied, offered) ?? 0).toFixed(3)}, below the ${PRECISION_TARGET} target. `
792
+ + 'More offers, not fewer, is the only way this clears.')
793
+ : `only ${offered} offer(s) recorded — below the ${MIN_PRECISION_SAMPLES}-sample floor, so this is not yet judgeable against the target`,
794
+ };
795
+ }
796
+
797
+ /**
798
+ * Every id the ledger knows about, with its derived state. What a management surface renders — and
799
+ * every field is computed from records on disk, never asserted.
800
+ */
801
+ export function summarize({ file = OUTCOMES_PATH, all = null } = {}) {
802
+ const recs = all ?? loadOutcomes(file);
803
+ const ids = [...new Set(recs.map((r) => r.id))];
804
+ return ids.map((id) => {
805
+ const o = outcomesFor(id, { all: recs });
806
+ return { ...o, suppressed: !shouldStillOffer(id, { all: recs, severity: o.lastSeverity }) };
807
+ }).sort((a, b) => b.offered - a.offered);
808
+ }