ruvnet-brain 4.0.1 → 4.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +303 -24
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/commands/whats-new.md +6 -6
- package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +90 -32
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +34 -29
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +503 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/scripts/whats-new.mjs +42 -0
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +98 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/plugin/skills/whats-new/SKILL.md +4 -4
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +531 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/fix-workstream.mjs +291 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +237 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +558 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +326 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +172 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2922 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/protected-release-invocation.mjs +76 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/publication-receipt.mjs +307 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-authority.mjs +93 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +439 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +296 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stabilization-receipt.mjs +108 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +867 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,808 @@
|
|
|
1
|
+
// advocacy-outcomes.mjs — the ledger that tells us whether our own advocacy was RIGHT.
|
|
2
|
+
//
|
|
3
|
+
// THE MISSING HALF. ADR-027 gave the brain a voice: detect a dormant capability, recommend it,
|
|
4
|
+
// execute it, reverse it. ADR-028 then defined the honest measure of that voice — precision,
|
|
5
|
+
// "recommendations acted on ÷ recommendations fired, target ≥ 0.60. Below this we are nagging, and a
|
|
6
|
+
// nag trains users to ignore the real alarm." That number has never been computable, because nothing
|
|
7
|
+
// in this system records what happened AFTER a recommendation was shown. Every offer vanished the
|
|
8
|
+
// moment the page closed. A system that cannot see its own outcomes cannot improve, and one that
|
|
9
|
+
// reports a metric it cannot source is doing the fabrication this repo has a CI gate against.
|
|
10
|
+
//
|
|
11
|
+
// So: an append-only outcome ledger. Every offer resolves into exactly one record — applied,
|
|
12
|
+
// dismissed, or ignored — and those three records are the only evidence any claim about proactivity
|
|
13
|
+
// is allowed to rest on.
|
|
14
|
+
//
|
|
15
|
+
// THE ONE WAY TO FABRICATE THIS METRIC, named here so a reviewer can check for it: record only the
|
|
16
|
+
// applies. Precision is applied ÷ (applied + dismissed + ignored), so a caller that forgets to
|
|
17
|
+
// record the misses reports a beautiful 1.0. The invariant is therefore not "record outcomes" but
|
|
18
|
+
// "every offer produces exactly one record" — and `ignored` is what an unresolved offer becomes when
|
|
19
|
+
// the session ends. If you are adding a caller, the ignored-path is the one to write first.
|
|
20
|
+
//
|
|
21
|
+
// THE ASYMMETRY THIS FILE EXISTS TO ENCODE. The adversarial review of ADR-031 (GPT-5.6-Sol,
|
|
22
|
+
// 2026-07-22) killed the previous learning signal with one sentence: "repeat count measures the
|
|
23
|
+
// USER'S FRUSTRATION, not the lesson's correctness... a formatting preference corrected 52 times
|
|
24
|
+
// dominates a security rule corrected once." A dismissal ledger repeats that mistake exactly if it is
|
|
25
|
+
// read as a popularity contest, so it is not read as one here:
|
|
26
|
+
//
|
|
27
|
+
// A dismissal is evidence about FIT, not about IMPORTANCE.
|
|
28
|
+
//
|
|
29
|
+
// "Not for me" and "not worth an interruption" are the same click. So the click cannot be allowed to
|
|
30
|
+
// mean the same thing for a cosmetic suggestion and for a corrupt-database warning — a nag dismissed
|
|
31
|
+
// once should vanish, and a high-severity finding dismissed once must not. That asymmetry is
|
|
32
|
+
// DISMISSAL_BUDGET below, and it is the whole design; the rest is bookkeeping.
|
|
33
|
+
//
|
|
34
|
+
// STORAGE. `~/.config/ruvnet-brain/` — user-level, and deliberately OUTSIDE `~/.cache/ruvnet-brain/`
|
|
35
|
+
// which `--update` replaces wholesale. Same reasoning as lesson-store.mjs: an outcome destroyed by
|
|
36
|
+
// the next release never compounds, and compounding (ADR-028 L5) is the only point of any of this.
|
|
37
|
+
//
|
|
38
|
+
// PURITY: node builtins only, no spawn, no network. It is read by surfaces; it does not render.
|
|
39
|
+
//
|
|
40
|
+
// WIRED (2026-07-23). Until this build `shouldStillOffer()` had ZERO production callers —
|
|
41
|
+
// `anticipate.sh` kept its OWN binary dismissed-Set (one dismissal muted forever, no severity, no
|
|
42
|
+
// budget) as a second, disconnected suppression policy, and this file's asymmetric budget sat
|
|
43
|
+
// uncalled. `anticipate.sh` is now the single caller for every mode (suggest AND
|
|
44
|
+
// dismiss/undismiss/status): it asks `shouldStillOffer()` and nothing else decides. `reconcileIgnored()`
|
|
45
|
+
// likewise had zero callers; `onboarding-console.mjs`'s `/api/capabilities` handler now supplies its
|
|
46
|
+
// pending-and-stale ids via this file's `pendingOffers()` (see `findStaleOffers()` there for the
|
|
47
|
+
// staleness rule, which is deliberately this file's caller's decision, not this file's).
|
|
48
|
+
|
|
49
|
+
import crypto from 'node:crypto';
|
|
50
|
+
import fs from 'node:fs';
|
|
51
|
+
import os from 'node:os';
|
|
52
|
+
import path from 'node:path';
|
|
53
|
+
|
|
54
|
+
const HOME = os.homedir();
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* ACTIONS — what became of an offer.
|
|
58
|
+
*
|
|
59
|
+
* The first three are the closed set ADR-028's precision metric is defined over: an offer that was
|
|
60
|
+
* shown ends as exactly one of them. `ignored` is a real, declared value rather than an absence,
|
|
61
|
+
* for the same reason UNDO_KINDS.NONE is one in remedy-registry.mjs: "the user did nothing" and
|
|
62
|
+
* "nobody wrote the code to record it" must never look identical, and the second is what silently
|
|
63
|
+
* inflates precision.
|
|
64
|
+
*
|
|
65
|
+
* RESET is the fourth, and it is here because of a house rule, not because the metric needs it.
|
|
66
|
+
* Dismissal is a control — it makes the brain stop speaking — and this repo does not ship a control
|
|
67
|
+
* without a real inverse (remedy-registry.mjs exists because a recommendation once promised an undo
|
|
68
|
+
* that had no branch behind it, and reported "nothing to undo" instead of failing). Suppression with
|
|
69
|
+
* no way back would be that same dead button pointed at silence. A reset is a CHECKPOINT, never a
|
|
70
|
+
* deletion: the ledger stays append-only and complete, and only the suppression arithmetic starts
|
|
71
|
+
* counting again after it.
|
|
72
|
+
*/
|
|
73
|
+
export const ACTIONS = Object.freeze({
|
|
74
|
+
OFFERED: 'offered',
|
|
75
|
+
APPLIED: 'applied',
|
|
76
|
+
DISMISSED: 'dismissed',
|
|
77
|
+
IGNORED: 'ignored',
|
|
78
|
+
RESET: 'reset',
|
|
79
|
+
});
|
|
80
|
+
const ACTION_VALUES = new Set(Object.values(ACTIONS));
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* `OFFERED` is the PENDING marker: the card was shown, and nothing has become of it yet. It is NOT a
|
|
84
|
+
* resolution and never enters the precision denominator (that would let merely showing a card move
|
|
85
|
+
* the metric). It exists for one reason — so `reconcileApplied()` can tell "we suggested this and the
|
|
86
|
+
* user then turned it on" (an APPLIED) apart from "it was already on". Formalising it here also ends a
|
|
87
|
+
* real schema drift: `anticipate.sh` was already writing `action:'offered'` through its own inline
|
|
88
|
+
* recorder, while the canonical `record()` below rejected that action — so every offered row it wrote
|
|
89
|
+
* was inert, counted by nothing. Now both writers speak one vocabulary and `record()` accepts it.
|
|
90
|
+
*/
|
|
91
|
+
/** The three RESOLUTIONS. `offered` is pending (not a resolution); `reset` is a ledger checkpoint. */
|
|
92
|
+
const OFFER_ACTIONS = new Set([ACTIONS.APPLIED, ACTIONS.DISMISSED, ACTIONS.IGNORED]);
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* THE ASYMMETRY, as numbers.
|
|
96
|
+
*
|
|
97
|
+
* A `normal` item spends its whole budget on ONE dismissal: the user said no, and for a suggestion
|
|
98
|
+
* that is the end of the conversation. Cheap to honour, and the cost of being wrong is that they
|
|
99
|
+
* miss a nicety.
|
|
100
|
+
*
|
|
101
|
+
* A `high` item costs three, because the cost of being wrong runs the other way. The finding this
|
|
102
|
+
* mechanism will most often suppress is the 2026-07-21 case: a corrupt AgentDB store, detected,
|
|
103
|
+
* scored 49/100, rendered — and the owner had to notice it himself. If one distracted click could
|
|
104
|
+
* bury that class of finding permanently, this file would have shipped a regression dressed as a
|
|
105
|
+
* feature. Three refusals is a considered no; one is a busy hand.
|
|
106
|
+
*
|
|
107
|
+
* IGNORE_WEIGHT prices silence at a fifth of a refusal. Silence is the weakest signal we have — it is
|
|
108
|
+
* consistent with "no", with "later", and with "I never saw the card" — so it may accumulate into
|
|
109
|
+
* suppression (a card ignored fifteen times IS a nag) but it may never be mistaken for an answer.
|
|
110
|
+
*
|
|
111
|
+
* APPLIED_CREDIT lets acting on a recommendation buy back a stretch of ignores, because clicking it
|
|
112
|
+
* is the single strongest evidence of fit we can observe, and a wanted card that fires again when
|
|
113
|
+
* the state recurs is not a nag.
|
|
114
|
+
*
|
|
115
|
+
* HARD_DISMISSAL_CAP is the ceiling above severity: after five explicit refusals nothing re-fires,
|
|
116
|
+
* ever, at any severity, whatever the evidence says. At that point we are wrong about the user, not
|
|
117
|
+
* about the machine — and ADR-028's own anti-goal list puts "interruption without an off switch"
|
|
118
|
+
* beside nagging.
|
|
119
|
+
*/
|
|
120
|
+
export const DISMISSAL_BUDGET = Object.freeze({ normal: 1, high: 3 });
|
|
121
|
+
export const IGNORE_WEIGHT = 0.2;
|
|
122
|
+
export const APPLIED_CREDIT = 1;
|
|
123
|
+
export const HARD_DISMISSAL_CAP = 5;
|
|
124
|
+
|
|
125
|
+
/** ADR-028's stated target and the sample floor below which reporting against it would be noise. */
|
|
126
|
+
export const PRECISION_TARGET = 0.60;
|
|
127
|
+
export const MIN_PRECISION_SAMPLES = 5;
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* THE GOODHART GUARD FOR PRECISION — the one the 4.0 briefing named and left open ("the fixture's
|
|
131
|
+
* separation-of-authorities design guards recall; the equivalent guard for precision needs the real
|
|
132
|
+
* ledger to exist first").
|
|
133
|
+
*
|
|
134
|
+
* It turns out not to need the ledger at all, because the hole is arithmetic. Once a metric gates a
|
|
135
|
+
* release it becomes a target, and precision = applied/offered has an obvious exploit: OFFER LESS.
|
|
136
|
+
* Suggest only the sure thing, and precision approaches 1.00 while the product helps nobody — which
|
|
137
|
+
* is the exact behaviour ADR-028 exists to prevent, certified by the metric meant to detect it.
|
|
138
|
+
*
|
|
139
|
+
* The old `meetsTarget: value >= PRECISION_TARGET` compared the POINT ESTIMATE, so 4 applied out of
|
|
140
|
+
* 6 offers read 0.667 >= 0.60 and passed, while the true rate consistent with that evidence goes far
|
|
141
|
+
* below 0.30. Worse, MIN_PRECISION_SAMPLES = 5 cannot support the target under ANY outcome: a
|
|
142
|
+
* perfect 5/5 bounds at 0.05^(1/5) = 54.9%, still short of 0.60. The floor was unreachable and the
|
|
143
|
+
* comparison was to the wrong number.
|
|
144
|
+
*
|
|
145
|
+
* Comparing the 95% LOWER BOUND to the target closes both. Fewer offers widen the interval, so
|
|
146
|
+
* withholding offers can no longer manufacture a passing score — it makes the metric report "not
|
|
147
|
+
* yet judgeable" instead. The incentive now points the right way: the only route to a certified
|
|
148
|
+
* precision is to offer MORE and be right.
|
|
149
|
+
*
|
|
150
|
+
* Exact Clopper-Pearson: the lower bound solves P(X >= k | n, p) = alpha. Computed by bisection on
|
|
151
|
+
* the binomial tail with exact terms (n is tiny here). Self-checkable: at k = n it must reduce to
|
|
152
|
+
* alpha^(1/n) — asserted in the tests rather than trusted, because a hand-rolled incomplete-beta in
|
|
153
|
+
* this same session returned 98.3% for n=3, which is absurd on its face.
|
|
154
|
+
*/
|
|
155
|
+
export const PRECISION_ALPHA = 0.05;
|
|
156
|
+
|
|
157
|
+
export function precisionLowerBound(k, n, alpha = PRECISION_ALPHA) {
|
|
158
|
+
if (!Number.isFinite(k) || !Number.isFinite(n) || n <= 0 || k < 0 || k > n) return null;
|
|
159
|
+
if (k === 0) return 0;
|
|
160
|
+
const tailAtLeastK = (p) => {
|
|
161
|
+
// sum_{i=k}^{n} C(n,i) p^i (1-p)^(n-i), computed with a running coefficient to avoid factorials.
|
|
162
|
+
let sum = 0;
|
|
163
|
+
let coeff = 1; // C(n,0)
|
|
164
|
+
for (let i = 0; i <= n; i++) {
|
|
165
|
+
if (i >= k) sum += coeff * Math.pow(p, i) * Math.pow(1 - p, n - i);
|
|
166
|
+
coeff = coeff * (n - i) / (i + 1); // C(n,i) -> C(n,i+1)
|
|
167
|
+
}
|
|
168
|
+
return sum;
|
|
169
|
+
};
|
|
170
|
+
let lo = 0;
|
|
171
|
+
let hi = 1;
|
|
172
|
+
for (let i = 0; i < 200; i++) {
|
|
173
|
+
const mid = (lo + hi) / 2;
|
|
174
|
+
if (tailAtLeastK(mid) < alpha) lo = mid; else hi = mid;
|
|
175
|
+
}
|
|
176
|
+
return (lo + hi) / 2;
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Field caps. These are a CORRECTNESS property, not tidiness — see appendLine() below: the atomicity
|
|
180
|
+
// of a concurrent append depends on each record being one small write. EXPORTED so a test can assert
|
|
181
|
+
// the arithmetic that makes the guard in appendLine() unreachable: every field is bounded, and the
|
|
182
|
+
// bounds sum to well under MAX_RECORD_BYTES. The guard stays anyway, as the tripwire that fires the
|
|
183
|
+
// day somebody raises one of these caps without redoing that sum.
|
|
184
|
+
export const MAX_ID = 200;
|
|
185
|
+
export const MAX_PROJECT = 120;
|
|
186
|
+
export const MAX_HASH = 64;
|
|
187
|
+
export const MAX_SEVERITY = 32;
|
|
188
|
+
export const MAX_RECORD_BYTES = 1024;
|
|
189
|
+
|
|
190
|
+
export const OUTCOMES_PATH = process.env.RUVNET_ADVOCACY_OUTCOMES
|
|
191
|
+
|| path.join(HOME, '.config', 'ruvnet-brain', 'advocacy-outcomes.jsonl');
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Severity → the two classes the budget is defined over.
|
|
195
|
+
*
|
|
196
|
+
* Accepts console-engine's vocabulary (`INFO` | `SUGGESTED` | `IMPORTANT`) and lesson-store's
|
|
197
|
+
* (`normal` | `high`), because both produce things that get offered and neither is going to change
|
|
198
|
+
* to suit this file.
|
|
199
|
+
*
|
|
200
|
+
* UNKNOWN SEVERITY RESOLVES TO `normal`, i.e. to the quieter class, and that direction is deliberate.
|
|
201
|
+
* It means a caller that forgets to pass severity gets an item silenced after one dismissal rather
|
|
202
|
+
* than one that is nearly unsilenceable. ADR-028: "One false alarm costs more trust than ten true
|
|
203
|
+
* ones earn. Non-negotiable." When we do not know, we err toward respecting the refusal — and
|
|
204
|
+
* because record() stores the severity it was told, the history stays self-describing rather than
|
|
205
|
+
* quietly re-classified later.
|
|
206
|
+
*/
|
|
207
|
+
export function weightClass(severity) {
|
|
208
|
+
const s = String(severity ?? '').trim().toLowerCase();
|
|
209
|
+
return (s === 'important' || s === 'high' || s === 'critical') ? 'high' : 'normal';
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* A stable fingerprint of the evidence a recommendation was built from.
|
|
214
|
+
*
|
|
215
|
+
* ADR-027's rule is "offered once per state change, dismissible, never re-fires while dismissed" —
|
|
216
|
+
* which is only implementable if "the state" is a value something can compare. This is that value:
|
|
217
|
+
* hash what we OBSERVED, not what we said about it, so rewording a card does not read as new
|
|
218
|
+
* evidence and re-open a settled question.
|
|
219
|
+
*
|
|
220
|
+
* Returns null for no evidence. Null is honest ("we cannot tell whether the state changed") and it
|
|
221
|
+
* is inert by construction: the state-change reprieve in shouldStillOffer() requires a real hash on
|
|
222
|
+
* both sides, so an unknown state can never argue its way past a dismissal.
|
|
223
|
+
*/
|
|
224
|
+
export function stateHashOf(evidence) {
|
|
225
|
+
const items = (Array.isArray(evidence) ? evidence : [evidence])
|
|
226
|
+
.map((e) => {
|
|
227
|
+
if (e === null || e === undefined) return '';
|
|
228
|
+
if (typeof e === 'object') return String(e.observed ?? JSON.stringify(e));
|
|
229
|
+
return String(e);
|
|
230
|
+
})
|
|
231
|
+
.map((s) => s.trim())
|
|
232
|
+
.filter(Boolean)
|
|
233
|
+
.sort(); // order of evidence is presentation, not state
|
|
234
|
+
if (!items.length) return null;
|
|
235
|
+
return crypto.createHash('sha256').update(items.join('')).digest('hex').slice(0, 16);
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
function toIso(at) {
|
|
239
|
+
if (at instanceof Date) return Number.isNaN(at.getTime()) ? new Date().toISOString() : at.toISOString();
|
|
240
|
+
const d = new Date(at);
|
|
241
|
+
return Number.isNaN(d.getTime()) ? new Date().toISOString() : d.toISOString();
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* THE WRITE. One line, one open-with-O_APPEND, one write() — and no read step at all.
|
|
246
|
+
*
|
|
247
|
+
* This repo has already paid for the alternative. saveSettings() did read-modify-write on a JSON
|
|
248
|
+
* object, and MEASURED across 20 trials of four simultaneous writers, at least one setting was lost
|
|
249
|
+
* in 19 of them — every writer returning ok:true, no error, no warning. The fix there was a lock,
|
|
250
|
+
* because a settings file genuinely is a single mutable object.
|
|
251
|
+
*
|
|
252
|
+
* A ledger is not. Append-only removes the read, and with the read goes the entire class of bug:
|
|
253
|
+
* there is no prior value to clobber. That is why this file is JSONL and not a JSON array, and the
|
|
254
|
+
* shape is load-bearing rather than stylistic — an array would reintroduce read-modify-write and
|
|
255
|
+
* with it the 19-in-20 silent loss, on the surface whose only job is to remember what the user chose.
|
|
256
|
+
*
|
|
257
|
+
* The remaining hazard is a partial write interleaving with another process's. POSIX makes the
|
|
258
|
+
* offset-advance-and-write atomic for a single write() on an O_APPEND fd; Node issues one write()
|
|
259
|
+
* for a single small buffer. So the size cap is the guarantee: every field is truncated, and a
|
|
260
|
+
* record that still exceeds MAX_LINE_BYTES is refused rather than written and hoped for. And because
|
|
261
|
+
* a torn line is still conceivable on an exotic filesystem, loadOutcomes() drops unparseable lines
|
|
262
|
+
* instead of failing — one damaged record costs one record, never the ledger.
|
|
263
|
+
*/
|
|
264
|
+
function appendLine(file, row) {
|
|
265
|
+
const line = JSON.stringify(row) + '\n';
|
|
266
|
+
if (Buffer.byteLength(line) > MAX_RECORD_BYTES) {
|
|
267
|
+
throw new Error(`Outcome for "${row.id}" invalid: record is ${Buffer.byteLength(line)} bytes, over the ${MAX_RECORD_BYTES}-byte cap that keeps a concurrent append atomic`);
|
|
268
|
+
}
|
|
269
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
270
|
+
fs.appendFileSync(file, line);
|
|
271
|
+
return line;
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* Record what became of one offer. Append-only; nothing here ever rewrites history.
|
|
276
|
+
*
|
|
277
|
+
* THROWS on a malformed record — same discipline as makeRecommendation() and makeLesson(): an
|
|
278
|
+
* invariant belongs in the constructor, not in a reviewer's memory. An unknown `action` written
|
|
279
|
+
* quietly would corrupt the precision denominator forever, and the wrongness would show up as a
|
|
280
|
+
* plausible number rather than as an error.
|
|
281
|
+
*
|
|
282
|
+
* DOES NOT THROW on an I/O failure — it returns `{ ok: false, reason }`, because callers are
|
|
283
|
+
* surfaces and a read-only home directory must not take down the console. But a caller MUST surface
|
|
284
|
+
* a failed `dismissed`: if the write fails silently, the user's "stop showing me this" does not
|
|
285
|
+
* stick, they see the same card tomorrow, and the off switch has become theatre. That is the exact
|
|
286
|
+
* failure shape as the undo that reported "nothing to undo" — a control that reports success and
|
|
287
|
+
* does nothing.
|
|
288
|
+
*
|
|
289
|
+
* @param {{id:string, action:string, at?:Date|string, project?:string, severity?:string|null,
|
|
290
|
+
* stateHash?:string|null, scope?:'forever'|null}} spec
|
|
291
|
+
*/
|
|
292
|
+
/**
|
|
293
|
+
* Under a test runner, writing to the DEFAULT (real, user-level) ledger is a bug, not a choice.
|
|
294
|
+
*
|
|
295
|
+
* Found by an independent grader on 2026-07-24: the live ledger at ~/.config/ruvnet-brain/ held
|
|
296
|
+
* exactly one row, `{"id":"f-adv-1","stateHash":"hash-1",...}` — fixture-shaped data in the user's
|
|
297
|
+
* real outcome record, describing an event that never happened. Every precision number this product
|
|
298
|
+
* reports is computed over that file, and it had junk in it from the first day it existed.
|
|
299
|
+
*
|
|
300
|
+
* It got there because a test called record() without passing `{file}`, so the default path won. The
|
|
301
|
+
* tempting fix is to blocklist ids that look like fixtures (`f-*`, `hash-*`), but that is a guess
|
|
302
|
+
* about naming, and the next fixture that does not match the pattern lands in the ledger exactly the
|
|
303
|
+
* same way. The defect is not the id — it is that a test can address the real file at all.
|
|
304
|
+
*
|
|
305
|
+
* So the write refuses instead. Under vitest, an explicit `file` (or RUVNET_ADVOCACY_OUTCOMES) is
|
|
306
|
+
* mandatory; the default is unreachable. That makes the pollution impossible by construction rather
|
|
307
|
+
* than unlikely by convention, and it fails LOUD at the moment the test is written rather than
|
|
308
|
+
* silently into a file nobody reads until a grader opens it thirteen months of commits later.
|
|
309
|
+
*/
|
|
310
|
+
const UNDER_TEST = !!(process.env.VITEST || process.env.VITEST_WORKER_ID);
|
|
311
|
+
|
|
312
|
+
export function record(spec, { file = OUTCOMES_PATH } = {}) {
|
|
313
|
+
if (UNDER_TEST && file === OUTCOMES_PATH && !process.env.RUVNET_ADVOCACY_OUTCOMES) {
|
|
314
|
+
throw new Error(
|
|
315
|
+
'advocacy-outcomes.record() refused: a test tried to write to the REAL user ledger at '
|
|
316
|
+
+ `${OUTCOMES_PATH}. Pass {file: <tmp path>} or set RUVNET_ADVOCACY_OUTCOMES. `
|
|
317
|
+
+ '(A fixture row reached the live ledger this way once and was found only by an outside grader.)',
|
|
318
|
+
);
|
|
319
|
+
}
|
|
320
|
+
const {
|
|
321
|
+
id, action, at = new Date(), project = null,
|
|
322
|
+
severity = null, stateHash = null, scope = null,
|
|
323
|
+
} = spec || {};
|
|
324
|
+
const err = (m) => { throw new Error(`Outcome for "${id ?? '?'}" invalid: ${m}`); };
|
|
325
|
+
|
|
326
|
+
if (!id || typeof id !== 'string') err('missing id — an outcome that cannot name the recommendation it belongs to measures nothing');
|
|
327
|
+
if (!ACTION_VALUES.has(action)) err(`action must be one of: ${[...ACTION_VALUES].join(', ')}`);
|
|
328
|
+
// `scope:'forever'` is the one-action permanent silence ADR-028 requires ("anything that speaks
|
|
329
|
+
// in-session must be silenceable in one action, permanently, without penalty"). It is meaningless
|
|
330
|
+
// on anything but a dismissal, and accepting it elsewhere would let a stray field mute a card
|
|
331
|
+
// nobody asked to mute.
|
|
332
|
+
if (scope !== null && scope !== 'forever') err(`scope must be null or "forever" (got ${JSON.stringify(scope)})`);
|
|
333
|
+
if (scope === 'forever' && action !== ACTIONS.DISMISSED) err('scope:"forever" is only meaningful on a dismissal');
|
|
334
|
+
|
|
335
|
+
const row = {
|
|
336
|
+
v: 1,
|
|
337
|
+
id: id.slice(0, MAX_ID),
|
|
338
|
+
action,
|
|
339
|
+
at: toIso(at),
|
|
340
|
+
// The project is recorded but is NOT a scope — see shouldStillOffer(). It is here so the ledger
|
|
341
|
+
// can answer "where did this happen", which is what ADR-028's L5 test is phrased in terms of.
|
|
342
|
+
project: String(project ?? path.basename(process.cwd())).slice(0, MAX_PROJECT),
|
|
343
|
+
severity: severity === null ? null : String(severity).slice(0, MAX_SEVERITY),
|
|
344
|
+
stateHash: stateHash === null ? null : String(stateHash).slice(0, MAX_HASH),
|
|
345
|
+
scope: scope ?? null,
|
|
346
|
+
};
|
|
347
|
+
|
|
348
|
+
try {
|
|
349
|
+
appendLine(file, row);
|
|
350
|
+
return { ok: true, file, row };
|
|
351
|
+
} catch (e) {
|
|
352
|
+
// Over-cap is a programming error and was already thrown by appendLine before any write; an
|
|
353
|
+
// ENOSPC/EACCES/EROFS is the environment. Both arrive here as a receipt so the caller can decide
|
|
354
|
+
// how loud to be, and the reason is preserved rather than flattened to a boolean.
|
|
355
|
+
return { ok: false, reason: e.code || e.message, row };
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
/**
|
|
360
|
+
* Read the ledger. NEVER THROWS — a missing file, a corrupt file, a half-written last line, a file
|
|
361
|
+
* full of someone else's JSON: all of them degrade to "no outcomes yet".
|
|
362
|
+
*
|
|
363
|
+
* This is the same contract lesson-gate.mjs holds itself to and for the same reason: a mechanism
|
|
364
|
+
* that suppresses recommendations must fail toward SPEAKING. If an unreadable ledger threw, or worse
|
|
365
|
+
* returned a partial count that happened to look like a spent budget, a corrupt file would silence
|
|
366
|
+
* the brain — and it would be silent in exactly the way it is silent when everything is healthy, so
|
|
367
|
+
* nobody would ever find out.
|
|
368
|
+
*/
|
|
369
|
+
export function loadOutcomes(file = OUTCOMES_PATH) {
|
|
370
|
+
let raw;
|
|
371
|
+
try { raw = fs.readFileSync(file, 'utf8'); } catch { return []; }
|
|
372
|
+
const out = [];
|
|
373
|
+
for (const line of raw.split('\n')) {
|
|
374
|
+
const s = line.trim();
|
|
375
|
+
if (!s) continue;
|
|
376
|
+
let r;
|
|
377
|
+
try { r = JSON.parse(s); } catch { continue; } // torn or hand-mangled line: drop it, keep the rest
|
|
378
|
+
if (!r || typeof r !== 'object') continue;
|
|
379
|
+
if (typeof r.id !== 'string' || !r.id) continue;
|
|
380
|
+
if (!ACTION_VALUES.has(r.action)) continue; // an action we do not understand is not counted as one we do
|
|
381
|
+
out.push({
|
|
382
|
+
v: Number(r.v) || 1,
|
|
383
|
+
id: r.id,
|
|
384
|
+
action: r.action,
|
|
385
|
+
at: typeof r.at === 'string' ? r.at : null,
|
|
386
|
+
project: typeof r.project === 'string' ? r.project : null,
|
|
387
|
+
severity: typeof r.severity === 'string' ? r.severity : null,
|
|
388
|
+
stateHash: typeof r.stateHash === 'string' ? r.stateHash : null,
|
|
389
|
+
scope: r.scope === 'forever' ? 'forever' : null,
|
|
390
|
+
});
|
|
391
|
+
}
|
|
392
|
+
return out;
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
/**
|
|
396
|
+
* The records for one id that the suppression arithmetic is allowed to see: everything appended
|
|
397
|
+
* after the most recent `reset`.
|
|
398
|
+
*
|
|
399
|
+
* ORDERED BY FILE POSITION, NOT BY `at`. The timestamp comes from whichever process wrote it, and a
|
|
400
|
+
* machine with a skewed clock (or a caller passing its own `at`, which record() permits) could
|
|
401
|
+
* otherwise re-order a reset behind the dismissals it was meant to clear — resurrecting a
|
|
402
|
+
* suppression the user explicitly lifted. Append order is the one ordering we actually control.
|
|
403
|
+
*/
|
|
404
|
+
function liveRecords(id, all) {
|
|
405
|
+
const mine = all.filter((r) => r.id === id);
|
|
406
|
+
let start = 0;
|
|
407
|
+
for (let i = mine.length - 1; i >= 0; i--) {
|
|
408
|
+
if (mine[i].action === ACTIONS.RESET) { start = i + 1; break; }
|
|
409
|
+
}
|
|
410
|
+
return mine.slice(start);
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
/**
|
|
414
|
+
* What we know about one recommendation.
|
|
415
|
+
*
|
|
416
|
+
* `precision` is null — not 0 — when nothing has been offered yet. This is the repo's oldest live
|
|
417
|
+
* rule: a detector once read a CLI's table and reported "26 hooks off" while the learner held 457
|
|
418
|
+
* trajectories, because unknown rendered as off. A recommendation nobody has seen has an UNKNOWN
|
|
419
|
+
* precision; rendering that as 0.00 would say "this advice is always rejected" about advice that has
|
|
420
|
+
* never been given.
|
|
421
|
+
*/
|
|
422
|
+
export function outcomesFor(id, { file = OUTCOMES_PATH, all = null, project = null } = {}) {
|
|
423
|
+
let recs = liveRecords(id, all ?? loadOutcomes(file));
|
|
424
|
+
if (project) recs = recs.filter((r) => r.project === project);
|
|
425
|
+
|
|
426
|
+
const count = (a) => recs.filter((r) => r.action === a).length;
|
|
427
|
+
const applied = count(ACTIONS.APPLIED);
|
|
428
|
+
const dismissed = count(ACTIONS.DISMISSED);
|
|
429
|
+
const ignored = count(ACTIONS.IGNORED);
|
|
430
|
+
const offered = applied + dismissed + ignored;
|
|
431
|
+
|
|
432
|
+
const dismissals = recs.filter((r) => r.action === ACTIONS.DISMISSED);
|
|
433
|
+
const offers = recs.filter((r) => OFFER_ACTIONS.has(r.action));
|
|
434
|
+
const last = offers.length ? offers[offers.length - 1] : null;
|
|
435
|
+
|
|
436
|
+
return {
|
|
437
|
+
id,
|
|
438
|
+
applied,
|
|
439
|
+
dismissed,
|
|
440
|
+
ignored,
|
|
441
|
+
offered,
|
|
442
|
+
precision: offered ? +(applied / offered).toFixed(4) : null,
|
|
443
|
+
projects: [...new Set(recs.map((r) => r.project).filter(Boolean))],
|
|
444
|
+
silencedForever: dismissals.some((r) => r.scope === 'forever'),
|
|
445
|
+
lastAction: last?.action ?? null,
|
|
446
|
+
lastAt: last?.at ?? null,
|
|
447
|
+
lastSeverity: [...offers].reverse().find((r) => r.severity)?.severity ?? null,
|
|
448
|
+
lastDismissal: dismissals.length ? dismissals[dismissals.length - 1] : null,
|
|
449
|
+
};
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
/**
|
|
453
|
+
* Should this recommendation be offered again? The question ADR-027 phrases as "dismissible, never
|
|
454
|
+
* re-fires while dismissed".
|
|
455
|
+
*
|
|
456
|
+
* NOT SCOPED BY PROJECT, AND THAT IS THE POINT. A dismissal recorded while working in project A
|
|
457
|
+
* suppresses the same recommendation in project B. This is the falsifiable L5 claim in ADR-028 —
|
|
458
|
+
* "a lesson validated in project A demonstrably changes behaviour in project B" — expressed on the
|
|
459
|
+
* signal we can actually observe today, and it is also just true of the subject matter: these
|
|
460
|
+
* recommendations are about the user's MACHINE (a dormant learner, a corrupt store, a stale install),
|
|
461
|
+
* so per-repo suppression would ask the same person the same question once per checkout.
|
|
462
|
+
*
|
|
463
|
+
* The order of the checks is the safety argument:
|
|
464
|
+
* 1. Never offered → offer. Silence has to be earned.
|
|
465
|
+
* 2. Silenced forever → never. One action, permanent, no penalty, no severity override. A finding
|
|
466
|
+
* important enough to argue past an explicit permanent mute does not exist; that argument is
|
|
467
|
+
* what turns a notification system into spam.
|
|
468
|
+
* 3. Budget by severity class → the asymmetry. A nag dies on one dismissal; a high-severity
|
|
469
|
+
* finding needs three, so a distracted click cannot bury a corrupt database.
|
|
470
|
+
* 4. State-change reprieve, HIGH SEVERITY ONLY. New evidence re-opens a high-severity question,
|
|
471
|
+
* because the underlying risk genuinely changed. It does NOT re-open a suggestion: for a nag, a
|
|
472
|
+
* changed number is not new information worth interrupting a person for, and granting it a
|
|
473
|
+
* reprieve would let a flapping metric nag forever through a budget it had already spent.
|
|
474
|
+
* 5. HARD_DISMISSAL_CAP overrides even that.
|
|
475
|
+
*/
|
|
476
|
+
export function shouldStillOffer(id, {
|
|
477
|
+
severity = null, stateHash = null, file = OUTCOMES_PATH, all = null,
|
|
478
|
+
} = {}) {
|
|
479
|
+
const o = outcomesFor(id, { file, all });
|
|
480
|
+
|
|
481
|
+
if (o.silencedForever) return false;
|
|
482
|
+
if (!o.offered) return true;
|
|
483
|
+
|
|
484
|
+
// Severity is DERIVED per offer from evidence measured on this machine (ADR-028: "Severity is
|
|
485
|
+
// derived from measured evidence on this machine. Nothing is IMPORTANT because it would be good
|
|
486
|
+
// for adoption."), so the CURRENT call's severity wins over what history recorded. A capability
|
|
487
|
+
// whose dormancy has become serious must not stay suppressed because it was cosmetic last month.
|
|
488
|
+
const cls = weightClass(severity ?? o.lastSeverity);
|
|
489
|
+
const budget = DISMISSAL_BUDGET[cls];
|
|
490
|
+
|
|
491
|
+
// Dismissals count in full; silence counts at a fifth; having actually used it buys credit back.
|
|
492
|
+
// Floor at zero so a long history of applies cannot bank immunity against a later refusal.
|
|
493
|
+
const spend = Math.max(0, o.dismissed + (IGNORE_WEIGHT * o.ignored) - (APPLIED_CREDIT * o.applied));
|
|
494
|
+
if (spend < budget) return true;
|
|
495
|
+
|
|
496
|
+
if (o.dismissed >= HARD_DISMISSAL_CAP) return false;
|
|
497
|
+
if (cls === 'high' && stateHash && o.lastDismissal?.stateHash && stateHash !== o.lastDismissal.stateHash) {
|
|
498
|
+
return true;
|
|
499
|
+
}
|
|
500
|
+
return false;
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
/**
|
|
504
|
+
* claimOffer — the ATOMIC step shouldStillOffer() cannot provide, and the reason it needs one.
|
|
505
|
+
*
|
|
506
|
+
* GPT-5.6-Sol found this in the ADR-047 duel and it is a genuine race, not a theoretical one: he
|
|
507
|
+
* drove shouldStillOffer() to `true` while TWENTY offers for the same finding sat pending. The cause
|
|
508
|
+
* is structural — shouldStillOffer() is a pure READ over the ledger. Two Claude Code sessions in two
|
|
509
|
+
* terminals (the normal way this product is used) both read "not yet offered", both conclude yes,
|
|
510
|
+
* and the user is told the same thing twice. Nothing between the read and the write said "mine".
|
|
511
|
+
*
|
|
512
|
+
* A lock around the whole decision would be the obvious fix and the wrong one: the decision reads
|
|
513
|
+
* the ledger, and this repo has already been burned by holding a lock across a read (updateLessons
|
|
514
|
+
* read outside its own lock and the "safe" version raced anyway). So the claim is narrow — it does
|
|
515
|
+
* not protect the decision, it protects the RIGHT TO SPEAK.
|
|
516
|
+
*
|
|
517
|
+
* The primitive is `open(..., 'wx')`: exclusive create, which the OS guarantees is atomic. Exactly
|
|
518
|
+
* one caller can create a given claim file; everyone else gets EEXIST and stays quiet.
|
|
519
|
+
*
|
|
520
|
+
* TTL, because a crashed session must not silence a capability forever. A claim older than ttlMs is
|
|
521
|
+
* abandoned and may be taken over — the same reasoning as any lease. The default is deliberately
|
|
522
|
+
* short: the cost of a stale claim is a MISSED offer (the product's whole reason to exist), while
|
|
523
|
+
* the cost of taking one over early is a duplicate — annoying, not silencing. Between those two
|
|
524
|
+
* failure modes, this system must always fail toward speaking.
|
|
525
|
+
*
|
|
526
|
+
* Returns true if THIS caller owns the right to offer. The caller then record()s the `offered` row.
|
|
527
|
+
*/
|
|
528
|
+
export function claimOffer(id, { dir = null, ttlMs = 60_000, now = Date.now() } = {}) {
|
|
529
|
+
if (!id || typeof id !== 'string') return false;
|
|
530
|
+
const base = dir || path.join(path.dirname(OUTCOMES_PATH), 'offer-claims');
|
|
531
|
+
const key = crypto.createHash('sha256').update(id).digest('hex').slice(0, 24);
|
|
532
|
+
const file = path.join(base, `${key}.claim`);
|
|
533
|
+
|
|
534
|
+
try { fs.mkdirSync(base, { recursive: true }); } catch { return true; } // cannot claim ⇒ fail toward speaking
|
|
535
|
+
|
|
536
|
+
// WRITE-THEN-LINK, and the reason is a bug this file's own concurrency test caught.
|
|
537
|
+
//
|
|
538
|
+
// The obvious implementation is open(file,'wx') followed by write(). It is wrong, and it fails
|
|
539
|
+
// exactly where it matters: `wx` publishes the filename BEFORE the content is written, so there is
|
|
540
|
+
// a window in which the claim exists and is EMPTY. Competing processes read it, fail to parse it,
|
|
541
|
+
// conclude "unknown age ⇒ stale ⇒ take it over", and speak. MEASURED with 12 real OS processes:
|
|
542
|
+
// FIVE of twelve won. A single-process test would have shown one winner and hidden it completely.
|
|
543
|
+
//
|
|
544
|
+
// link() closes the window. The content is written to a private temp file first, so the moment the
|
|
545
|
+
// claim name becomes visible it is already complete and parseable. link() itself fails with EEXIST
|
|
546
|
+
// when the target exists, giving the same atomic exactly-one-winner guarantee — with no torn state
|
|
547
|
+
// for the losers to misread.
|
|
548
|
+
const take = () => {
|
|
549
|
+
const tmp = `${file}.${process.pid}.${Math.abs(now % 1e9)}.tmp`;
|
|
550
|
+
try {
|
|
551
|
+
fs.writeFileSync(tmp, JSON.stringify({ id, at: new Date(now).toISOString(), pid: process.pid }));
|
|
552
|
+
try {
|
|
553
|
+
fs.linkSync(tmp, file); // ATOMIC create-if-absent, content already durable
|
|
554
|
+
return true;
|
|
555
|
+
} catch (e) {
|
|
556
|
+
if (e.code !== 'EEXIST') return true; // an unexpected FS error must not silence us
|
|
557
|
+
return null; // genuinely held — staleness decided below
|
|
558
|
+
} finally {
|
|
559
|
+
try { fs.unlinkSync(tmp); } catch { /* best effort */ }
|
|
560
|
+
}
|
|
561
|
+
} catch { return true; } // cannot even stage a claim ⇒ fail toward speaking
|
|
562
|
+
};
|
|
563
|
+
|
|
564
|
+
const first = take();
|
|
565
|
+
if (first !== null) return first;
|
|
566
|
+
|
|
567
|
+
// Someone holds it. Stale?
|
|
568
|
+
let heldAt = 0;
|
|
569
|
+
try { heldAt = Date.parse(JSON.parse(fs.readFileSync(file, 'utf8')).at) || 0; } catch { heldAt = 0; }
|
|
570
|
+
if (now - heldAt < ttlMs) return false; // live claim — stay quiet, this is the duplicate we came to prevent
|
|
571
|
+
|
|
572
|
+
// Abandoned. Take it over by REPLACING atomically, so two reapers cannot both win.
|
|
573
|
+
const tmp = `${file}.${process.pid}.${key.slice(0, 6)}`;
|
|
574
|
+
try {
|
|
575
|
+
fs.writeFileSync(tmp, JSON.stringify({ id, at: new Date(now).toISOString(), pid: process.pid, tookOver: true }));
|
|
576
|
+
fs.renameSync(tmp, file); // atomic replace
|
|
577
|
+
return true;
|
|
578
|
+
} catch {
|
|
579
|
+
try { fs.unlinkSync(tmp); } catch { /* best effort */ }
|
|
580
|
+
return true; // could not arbitrate ⇒ fail toward speaking
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
|
|
584
|
+
/** Release a claim once the offer is resolved (applied/dismissed), so a later dormancy can re-offer. */
|
|
585
|
+
export function releaseClaim(id, { dir = null } = {}) {
|
|
586
|
+
if (!id || typeof id !== 'string') return false;
|
|
587
|
+
const base = dir || path.join(path.dirname(OUTCOMES_PATH), 'offer-claims');
|
|
588
|
+
const key = crypto.createHash('sha256').update(id).digest('hex').slice(0, 24);
|
|
589
|
+
try { fs.unlinkSync(path.join(base, `${key}.claim`)); return true; } catch { return false; }
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
/**
|
|
593
|
+
* The pending offer for one id, or null. Pending = the most recent `offered` (since the last reset)
|
|
594
|
+
* has no resolution after it. Ordered by file position, not `at`, for the same clock-skew reason as
|
|
595
|
+
* liveRecords().
|
|
596
|
+
*/
|
|
597
|
+
function pendingOffer(id, all) {
|
|
598
|
+
const mine = liveRecords(id, all);
|
|
599
|
+
let idx = -1;
|
|
600
|
+
for (let i = mine.length - 1; i >= 0; i--) {
|
|
601
|
+
if (mine[i].action === ACTIONS.OFFERED) { idx = i; break; }
|
|
602
|
+
}
|
|
603
|
+
if (idx === -1) return null; // never offered since the last reset
|
|
604
|
+
for (let i = idx + 1; i < mine.length; i++) {
|
|
605
|
+
if (OFFER_ACTIONS.has(mine[i].action)) return null; // already resolved
|
|
606
|
+
}
|
|
607
|
+
return mine[idx];
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
/**
|
|
611
|
+
* Every id with a currently-pending offer — the bulk, read-only form of the per-id check
|
|
612
|
+
* pendingOffer() already makes inside reconcileApplied()/reconcileIgnored(). Exists so a CALLER can
|
|
613
|
+
* decide its OWN staleness rule (wall-clock age, a newer offer superseding it, a session count) over
|
|
614
|
+
* a real `at` timestamp, without re-implementing the reset-aware, position-ordered definition of
|
|
615
|
+
* "pending" that lives here. Read-only: it records nothing and never throws.
|
|
616
|
+
*
|
|
617
|
+
* @returns {Array<{id:string, at:string|null, severity:string|null, project:string|null, stateHash:string|null}>}
|
|
618
|
+
*/
|
|
619
|
+
export function pendingOffers({ file = OUTCOMES_PATH, all = null } = {}) {
|
|
620
|
+
let recs;
|
|
621
|
+
try { recs = all ?? loadOutcomes(file); } catch { return []; }
|
|
622
|
+
const ids = [...new Set(recs.map((r) => r.id))];
|
|
623
|
+
const out = [];
|
|
624
|
+
for (const id of ids) {
|
|
625
|
+
const offer = pendingOffer(id, recs);
|
|
626
|
+
if (offer) out.push({ id, at: offer.at, severity: offer.severity, project: offer.project, stateHash: offer.stateHash });
|
|
627
|
+
}
|
|
628
|
+
return out;
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
/**
|
|
632
|
+
* THE NUMERATOR, DERIVED — not asserted. precision = applied ÷ (applied+dismissed+ignored), and until
|
|
633
|
+
* now `applied` was recorded by nothing, so the number could only ever be 0 (once a dismissal landed)
|
|
634
|
+
* or null. That is the inverse of the fabrication this file warns about in its header: not a beautiful
|
|
635
|
+
* 1.0 from recording only the applies, but a permanent 0.0 from recording none of them — advocacy that
|
|
636
|
+
* looks like pure nagging no matter how well it lands.
|
|
637
|
+
*
|
|
638
|
+
* The honest signal for "the user acted on our suggestion" is a state transition we can OBSERVE: a
|
|
639
|
+
* capability we OFFERED, still pending, is now measured `on`. That is an APPLIED. We do not guess and
|
|
640
|
+
* we do not credit an offer the user resolved some other way — only a pending offer whose capability
|
|
641
|
+
* the audit now reports on. A capability that was already on when we offered it cannot go on again, so
|
|
642
|
+
* it cannot be double-counted; and a dismissed or ignored offer is no longer pending, so turning it on
|
|
643
|
+
* later (for reasons of their own) is not miscredited to us.
|
|
644
|
+
*
|
|
645
|
+
* NEVER THROWS — surfaces call it. Its writes go through record(), which returns a receipt on I/O
|
|
646
|
+
* failure rather than throwing; a lost applied costs one row, never the caller.
|
|
647
|
+
*
|
|
648
|
+
* @param {Array<{key?:string,id?:string,state?:string}>} auditRows the capability audit (auditAll()'s output)
|
|
649
|
+
* @returns {string[]} the ids reconciled to `applied` this call
|
|
650
|
+
*/
|
|
651
|
+
export function reconcileApplied(auditRows, { file = OUTCOMES_PATH } = {}) {
|
|
652
|
+
if (!Array.isArray(auditRows)) return [];
|
|
653
|
+
let all;
|
|
654
|
+
try { all = loadOutcomes(file); } catch { return []; }
|
|
655
|
+
const done = [];
|
|
656
|
+
for (const row of auditRows) {
|
|
657
|
+
if (!row || typeof row !== 'object') continue;
|
|
658
|
+
const id = typeof row.key === 'string' ? row.key : (typeof row.id === 'string' ? row.id : '');
|
|
659
|
+
if (!id) continue;
|
|
660
|
+
if (row.state !== 'on') continue; // only a real, now-observed on-state
|
|
661
|
+
const offer = pendingOffer(id, all);
|
|
662
|
+
if (!offer) continue; // nothing pending to credit
|
|
663
|
+
const res = record({ id, action: ACTIONS.APPLIED, severity: offer.severity ?? null, project: offer.project ?? null }, { file });
|
|
664
|
+
if (res.ok) {
|
|
665
|
+
done.push(id);
|
|
666
|
+
// keep the in-call view consistent so a duplicate id in auditRows can't be applied twice
|
|
667
|
+
all.push({ id, action: ACTIONS.APPLIED, at: res.row.at, project: res.row.project, severity: res.row.severity, stateHash: null, scope: null });
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
return done;
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
/**
|
|
674
|
+
* THE DENOMINATOR'S MISSING THIRD — `ignored`, DERIVED, not guessed.
|
|
675
|
+
*
|
|
676
|
+
* ADR-028: precision = applied ÷ (applied + dismissed + ignored). `applied` was wired above by
|
|
677
|
+
* reconcileApplied(); `dismissed` was always recorded, because a dismissal is a click and a click has
|
|
678
|
+
* an event to hang a record on. `ignored` has no click — it is the ABSENCE of one — and this module
|
|
679
|
+
* already refuses to record an absence on a guess (see stateHashOf returning null for no evidence,
|
|
680
|
+
* outcomesFor's precision:null for no offers). An offer shown and never acted on nor dismissed is
|
|
681
|
+
* invisible today, and invisible is optimistic: it silently shrinks the denominator, the mirror image
|
|
682
|
+
* of the fabrication this file's header already names ("record only the applies").
|
|
683
|
+
*
|
|
684
|
+
* THE TRIGGER THIS BUILD CHOSE, AND WHY IT IS NOT A GUESS. "Ignored" is fundamentally a claim about
|
|
685
|
+
* TIME — the offer sat there, unresolved, long enough that the silence means something rather than
|
|
686
|
+
* "the user hasn't looked yet". This module has no clock of its own worth trusting for that: record()
|
|
687
|
+
* lets a caller pass an arbitrary `at`, and liveRecords()/pendingOffer() deliberately order by file
|
|
688
|
+
* position rather than timestamp for exactly the clock-skew reason documented on both of them. Picking
|
|
689
|
+
* a threshold HERE (say, "offered more than N days ago") would be inventing evidence this file does
|
|
690
|
+
* not have. So the staleness decision is left where the evidence actually lives — with the caller, who
|
|
691
|
+
* can say "this offer is N sessions old" or "a newer offer for the same capability just superseded
|
|
692
|
+
* it" — and reconcileIgnored() takes that decision as a plain list of ids rather than a clock. It stays
|
|
693
|
+
* pure: no Date.now(), no session counter, nothing but the ledger already on disk.
|
|
694
|
+
*
|
|
695
|
+
* WHAT MAKES IT SAFE TO CALL WITH A WRONG OR STALE LIST. Passing an id is a PROPOSAL, not a command —
|
|
696
|
+
* the ledger is the sole arbiter. For each id, the only question this function answers on its own
|
|
697
|
+
* evidence is: is there a `pendingOffer` for this id right now (an `offered` since the last reset with
|
|
698
|
+
* NO resolution after it)? That single check is what buys the three guarantees this build requires:
|
|
699
|
+
* - CANNOT double-count: the moment an id is recorded ignored, it IS a resolution — so any later
|
|
700
|
+
* call with the same id (a cron re-run, a duplicate in the same list) finds nothing pending.
|
|
701
|
+
* - CANNOT convert a real resolution: a caller that (wrongly) still lists an id the user applied or
|
|
702
|
+
* dismissed five minutes ago is a no-op, never an overwrite — pendingOffer() already sees the
|
|
703
|
+
* resolution and returns null, same as it does for reconcileApplied().
|
|
704
|
+
* - CANNOT invent an offer: an id that was never offered has no pendingOffer either, so a stray or
|
|
705
|
+
* misspelled id records nothing.
|
|
706
|
+
*
|
|
707
|
+
* NEVER THROWS — same contract as reconcileApplied(): a caller here is a surface or a scheduled job,
|
|
708
|
+
* and a lost `ignored` costs one row, never the caller.
|
|
709
|
+
*
|
|
710
|
+
* @param {Array<string>} pendingIds ids the CALLER has already judged pending AND stale — staleness
|
|
711
|
+
* (session age, wall-clock age, supersession by a newer offer) is entirely the caller's evidence;
|
|
712
|
+
* this function neither computes nor infers it, only verifies each id still has a real pendingOffer.
|
|
713
|
+
* @returns {string[]} the ids actually reconciled to `ignored` this call — a subset of pendingIds,
|
|
714
|
+
* only those that still had an unresolved offer to resolve.
|
|
715
|
+
*/
|
|
716
|
+
export function reconcileIgnored(pendingIds, { file = OUTCOMES_PATH } = {}) {
|
|
717
|
+
if (!Array.isArray(pendingIds)) return [];
|
|
718
|
+
let all;
|
|
719
|
+
try { all = loadOutcomes(file); } catch { return []; }
|
|
720
|
+
const done = [];
|
|
721
|
+
for (const raw of pendingIds) {
|
|
722
|
+
const id = typeof raw === 'string' ? raw : '';
|
|
723
|
+
if (!id) continue;
|
|
724
|
+
const offer = pendingOffer(id, all);
|
|
725
|
+
if (!offer) continue; // already resolved, reset since, or never offered — nothing pending to mark
|
|
726
|
+
const res = record({ id, action: ACTIONS.IGNORED, severity: offer.severity ?? null, project: offer.project ?? null }, { file });
|
|
727
|
+
if (res.ok) {
|
|
728
|
+
done.push(id);
|
|
729
|
+
// keep the in-call view consistent so a duplicate id in pendingIds can't be recorded twice
|
|
730
|
+
all.push({ id, action: ACTIONS.IGNORED, at: res.row.at, project: res.row.project, severity: res.row.severity, stateHash: null, scope: null });
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
return done;
|
|
734
|
+
}
|
|
735
|
+
|
|
736
|
+
/**
|
|
737
|
+
* ADR-028's precision metric: recommendations acted on ÷ recommendations fired. Target ≥ 0.60.
|
|
738
|
+
*
|
|
739
|
+
* A DISMISSAL IS NOT AN ACTION. It is in the denominator and never the numerator, even though the
|
|
740
|
+
* user did click something. Counting it as "acted on" would let us hit target by annoying people
|
|
741
|
+
* into clicking X, which is the precise behaviour the metric exists to catch — the number would rise
|
|
742
|
+
* as the product got worse, and a metric that inverts under pressure is worse than no metric.
|
|
743
|
+
*
|
|
744
|
+
* `precision: null` when nothing has been offered, and `meetsTarget: null` below the sample floor.
|
|
745
|
+
* One rejected offer is not a 0.00 precision rate, and reporting it as one would be the same
|
|
746
|
+
* unknown-rendered-as-a-number failure this repo has a gate against. A grade we have not earned the
|
|
747
|
+
* right to state is stated as "not yet measurable", loudly, in the return value.
|
|
748
|
+
*
|
|
749
|
+
* COUNTS EVERY RECORDED OFFER, INCLUDING BEFORE A RESET. shouldStillOffer() honours the reset
|
|
750
|
+
* checkpoint because that is a user preference about the future; this is a measurement of how the
|
|
751
|
+
* product has actually behaved, and letting a reset launder a bad precision score would make the one
|
|
752
|
+
* number that judges us the one number we can clear.
|
|
753
|
+
*/
|
|
754
|
+
export function precision({ file = OUTCOMES_PATH, all = null, since = null, id = null } = {}) {
|
|
755
|
+
let recs = (all ?? loadOutcomes(file)).filter((r) => OFFER_ACTIONS.has(r.action));
|
|
756
|
+
if (id) recs = recs.filter((r) => r.id === id);
|
|
757
|
+
if (since) {
|
|
758
|
+
const cut = toIso(since);
|
|
759
|
+
recs = recs.filter((r) => typeof r.at === 'string' && r.at >= cut);
|
|
760
|
+
}
|
|
761
|
+
|
|
762
|
+
const count = (a) => recs.filter((r) => r.action === a).length;
|
|
763
|
+
const applied = count(ACTIONS.APPLIED);
|
|
764
|
+
const dismissed = count(ACTIONS.DISMISSED);
|
|
765
|
+
const ignored = count(ACTIONS.IGNORED);
|
|
766
|
+
const offered = applied + dismissed + ignored;
|
|
767
|
+
|
|
768
|
+
if (!offered) {
|
|
769
|
+
return {
|
|
770
|
+
precision: null, offered: 0, applied: 0, dismissed: 0, ignored: 0,
|
|
771
|
+
target: PRECISION_TARGET, sufficient: false, meetsTarget: null,
|
|
772
|
+
reason: 'no offers recorded yet — precision is unknown, not zero',
|
|
773
|
+
};
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
const value = +(applied / offered).toFixed(4);
|
|
777
|
+
const sufficient = offered >= MIN_PRECISION_SAMPLES;
|
|
778
|
+
return {
|
|
779
|
+
precision: value,
|
|
780
|
+
offered, applied, dismissed, ignored,
|
|
781
|
+
target: PRECISION_TARGET,
|
|
782
|
+
sufficient,
|
|
783
|
+
// The lower bound, not the point estimate — see PRECISION_ALPHA above. `offered` is the sample
|
|
784
|
+
// and `applied` the successes, so withholding offers WIDENS this and can never buy a pass.
|
|
785
|
+
lowerBound: +(precisionLowerBound(applied, offered) ?? 0).toFixed(4),
|
|
786
|
+
meetsTarget: sufficient ? precisionLowerBound(applied, offered) >= PRECISION_TARGET : null,
|
|
787
|
+
reason: sufficient
|
|
788
|
+
? (precisionLowerBound(applied, offered) >= PRECISION_TARGET
|
|
789
|
+
? null
|
|
790
|
+
: `${applied}/${offered} applied — the point estimate is ${value}, but the 95% lower bound is `
|
|
791
|
+
+ `${(precisionLowerBound(applied, offered) ?? 0).toFixed(3)}, below the ${PRECISION_TARGET} target. `
|
|
792
|
+
+ 'More offers, not fewer, is the only way this clears.')
|
|
793
|
+
: `only ${offered} offer(s) recorded — below the ${MIN_PRECISION_SAMPLES}-sample floor, so this is not yet judgeable against the target`,
|
|
794
|
+
};
|
|
795
|
+
}
|
|
796
|
+
|
|
797
|
+
/**
|
|
798
|
+
* Every id the ledger knows about, with its derived state. What a management surface renders — and
|
|
799
|
+
* every field is computed from records on disk, never asserted.
|
|
800
|
+
*/
|
|
801
|
+
export function summarize({ file = OUTCOMES_PATH, all = null } = {}) {
|
|
802
|
+
const recs = all ?? loadOutcomes(file);
|
|
803
|
+
const ids = [...new Set(recs.map((r) => r.id))];
|
|
804
|
+
return ids.map((id) => {
|
|
805
|
+
const o = outcomesFor(id, { all: recs });
|
|
806
|
+
return { ...o, suppressed: !shouldStillOffer(id, { all: recs, severity: o.lastSeverity }) };
|
|
807
|
+
}).sort((a, b) => b.offered - a.offered);
|
|
808
|
+
}
|