switchroom 0.21.17 → 0.21.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,153 @@
1
+ /**
2
+ * Unit suite for the flip-gate verdict logic + report rendering. Runs under
3
+ * `bun test` (this tree is vitest-excluded) via the `uat/flip/` entry in
4
+ * telegram-plugin/scripts/bun-test-ci.sh. Pure fixtures — no IO.
5
+ */
6
+
7
+ import { describe, it, expect } from "vitest";
8
+ import { evaluateGate, runGate, type GateInput, type Tier2ProbeResults } from "./gate.js";
9
+ import { renderFlipReport, renderVerdictLine } from "./report.js";
10
+ import type { EquivalenceReport } from "./tier1-equivalence.js";
11
+ import type { DirectiveInjectionDelta } from "./recall-log.js";
12
+
13
+ function passingTier1(): EquivalenceReport {
14
+ return {
15
+ pass: true,
16
+ missing_from_rules: [],
17
+ truncated_or_drifted: [],
18
+ unsourced_rules: [],
19
+ renderedBytes: 1000,
20
+ budgetBytes: 6144,
21
+ withinBudget: true,
22
+ sentinelCount: 3,
23
+ ruleCount: 3,
24
+ sentinelMatchesCount: true,
25
+ residueDirectiveCount: 3,
26
+ };
27
+ }
28
+
29
+ function suppressedDelta(): DirectiveInjectionDelta {
30
+ return {
31
+ baseline: { rowCount: 2, maxDirectiveCount: 6, lastDirectiveCount: 6, everInjectedIds: ["a"], maxDirectivesOmitted: 0 },
32
+ postflip: { rowCount: 2, maxDirectiveCount: 0, lastDirectiveCount: 0, everInjectedIds: [], maxDirectivesOmitted: 0 },
33
+ volumeDelta: 6,
34
+ postflipFullySuppressed: true,
35
+ residualIds: [],
36
+ };
37
+ }
38
+
39
+ describe("evaluateGate", () => {
40
+ it("passes when Tier-1 is clean and no optional inputs are supplied", () => {
41
+ const v = evaluateGate({ agent: "ziggy", tier1: passingTier1() });
42
+ expect(v.pass).toBe(true);
43
+ // recall_log + tier2 recorded as skipped, not failing.
44
+ const skipped = v.checks.filter((c) => c.skipped).map((c) => c.name);
45
+ expect(skipped).toContain("recall_log: directives suppressed postflip");
46
+ expect(skipped).toContain("tier2: behavioural probes hold");
47
+ });
48
+
49
+ it("fails on a Tier-1 missing guardrail and names it", () => {
50
+ const t1 = passingTier1();
51
+ t1.pass = false;
52
+ t1.missing_from_rules = [{ id: "d9", name: "guard", reason: "unmapped" }];
53
+ const v = evaluateGate({ agent: "ziggy", tier1: t1 });
54
+ expect(v.pass).toBe(false);
55
+ const c = v.checks.find((c) => c.name === "tier1: no missing guardrails")!;
56
+ expect(c.pass).toBe(false);
57
+ expect(c.detail).toContain("d9(unmapped)");
58
+ });
59
+
60
+ it("fails on an over-budget Tier-1 block", () => {
61
+ const t1 = passingTier1();
62
+ t1.withinBudget = false;
63
+ t1.renderedBytes = 7000;
64
+ const v = evaluateGate({ agent: "ziggy", tier1: t1 });
65
+ expect(v.pass).toBe(false);
66
+ expect(v.checks.find((c) => c.name === "tier1: within 6144B budget")!.pass).toBe(false);
67
+ });
68
+
69
+ it("passes recall_log when postflip injection is fully suppressed", () => {
70
+ const v = evaluateGate({ agent: "ziggy", tier1: passingTier1(), recallLog: suppressedDelta() });
71
+ const c = v.checks.find((c) => c.name === "recall_log: directives suppressed postflip")!;
72
+ expect(c.pass).toBe(true);
73
+ expect(c.skipped).toBeUndefined();
74
+ expect(v.pass).toBe(true);
75
+ });
76
+
77
+ it("fails recall_log when directives are still injected postflip", () => {
78
+ const d = suppressedDelta();
79
+ d.postflipFullySuppressed = false;
80
+ d.postflip.maxDirectiveCount = 2;
81
+ d.residualIds = ["a", "z"];
82
+ const v = evaluateGate({ agent: "ziggy", tier1: passingTier1(), recallLog: d });
83
+ expect(v.pass).toBe(false);
84
+ const c = v.checks.find((c) => c.name === "recall_log: directives suppressed postflip")!;
85
+ expect(c.detail).toContain("a, z");
86
+ });
87
+
88
+ it("folds a Tier-2 result in when supplied (seam)", () => {
89
+ const failing: Tier2ProbeResults = { pass: false, probes: [{ directiveId: "d1", held: false }] };
90
+ const v = evaluateGate({ agent: "ziggy", tier1: passingTier1(), tier2: failing });
91
+ expect(v.pass).toBe(false);
92
+ const c = v.checks.find((c) => c.name === "tier2: behavioural probes hold")!;
93
+ expect(c.pass).toBe(false);
94
+ expect(c.skipped).toBeUndefined();
95
+ });
96
+
97
+ it("treats tier2:null as skipped, not a failure", () => {
98
+ const v = evaluateGate({ agent: "ziggy", tier1: passingTier1(), tier2: null });
99
+ expect(v.pass).toBe(true);
100
+ expect(v.checks.find((c) => c.name === "tier2: behavioural probes hold")!.skipped).toBe(true);
101
+ });
102
+ });
103
+
104
+ describe("runGate", () => {
105
+ it("exit 0 when every agent passes", () => {
106
+ const inputs: GateInput[] = [
107
+ { agent: "a", tier1: passingTier1() },
108
+ { agent: "b", tier1: passingTier1() },
109
+ ];
110
+ expect(runGate(inputs).exitCode).toBe(0);
111
+ });
112
+
113
+ it("exit 1 when any agent fails", () => {
114
+ const bad = passingTier1();
115
+ bad.unsourced_rules = [{ id: "R-9", text: "invented" }];
116
+ const run = runGate([
117
+ { agent: "a", tier1: passingTier1() },
118
+ { agent: "b", tier1: bad },
119
+ ]);
120
+ expect(run.exitCode).toBe(1);
121
+ expect(run.verdicts.find((v) => v.agent === "b")!.pass).toBe(false);
122
+ });
123
+ });
124
+
125
+ describe("renderFlipReport", () => {
126
+ it("renders a PASS matrix with no triage rows", () => {
127
+ const run = runGate([{ agent: "ziggy", tier1: passingTier1(), recallLog: suppressedDelta() }]);
128
+ const md = renderFlipReport(run, { startedAt: new Date("2026-08-18T00:00:00Z"), durationSeconds: 1.2 });
129
+ expect(md).toContain("**Verdict:** PASS (1/1 agents green)");
130
+ expect(md).toContain("| `ziggy` | PASS |");
131
+ expect(md).toContain("No failing checks");
132
+ });
133
+
134
+ it("renders a FAIL matrix and lists the failing check verbatim", () => {
135
+ const bad = passingTier1();
136
+ bad.unsourced_rules = [{ id: "R-9", text: "invented" }];
137
+ const run = runGate([{ agent: "ziggy", tier1: bad }]);
138
+ const md = renderFlipReport(run);
139
+ expect(md).toContain("**Verdict:** FAIL");
140
+ expect(md).toContain("| `ziggy` | FAIL |");
141
+ expect(md).toContain("no unsourced rules");
142
+ expect(md).toContain("R-9");
143
+ });
144
+ });
145
+
146
+ describe("renderVerdictLine", () => {
147
+ it("summarizes one verdict compactly", () => {
148
+ const v = evaluateGate({ agent: "ziggy", tier1: passingTier1() });
149
+ const line = renderVerdictLine(v);
150
+ expect(line).toContain("PASS ziggy");
151
+ expect(line).toContain("within 6144B budget");
152
+ });
153
+ });
@@ -0,0 +1,232 @@
1
+ /**
2
+ * The Memory v2 M3 directive-flip UAT GATE.
3
+ *
4
+ * Combines the deterministic checks into one per-agent verdict and one process
5
+ * exit code, so the flip UAT is `&&`-chainable in CI without parsing prose:
6
+ *
7
+ * - Tier-1 equivalence ({@link EquivalenceReport} from `tier1-equivalence.ts`)
8
+ * — the migrated rules carry every directive guardrail, nothing dropped /
9
+ * truncated / invented, and the block fits the 6144B budget with a correct
10
+ * sentinel.
11
+ * - recall_log injection delta ({@link DirectiveInjectionDelta} from
12
+ * `recall-log.ts`) — after the flip the recall hook injects ZERO
13
+ * directives (the volume the flip was supposed to remove actually left the
14
+ * prompt). Optional: absent when the caller runs the gate PRE-flip (Tier-1
15
+ * only), present once there's a postflip window to diff.
16
+ * - Tier-2 behavioural probes — the mtcute probe runner is NOT built here
17
+ * (out of scope for the deterministic half). A typed seam
18
+ * ({@link Tier2ProbeResults}) is left so wiring it later is additive: the
19
+ * gate already folds a `tier2` result into the verdict when one is present.
20
+ *
21
+ * Everything here is PURE: callers do the IO (read the bank, parse the rules
22
+ * block, tail recall_log, run probes) and hand this module the structured
23
+ * results. That keeps the verdict logic unit-testable with fixtures.
24
+ */
25
+
26
+ import type { EquivalenceReport } from "./tier1-equivalence.js";
27
+ import type { DirectiveInjectionDelta } from "./recall-log.js";
28
+
29
+ // ---------------------------------------------------------------------------
30
+ // Tier-2 seam — DO NOT build the probe runner here (out of scope).
31
+ // ---------------------------------------------------------------------------
32
+
33
+ /**
34
+ * Result envelope the mtcute behavioural probe runner produces — a
35
+ * baseline-vs-postflip diff proving the flipped agent still HONOURS each
36
+ * migrated guardrail in live conversation (the model-in-the-loop half the
37
+ * deterministic checks can't cover).
38
+ *
39
+ * The runner that fills this lives in `tier2-probe-runner.ts`; it emits one
40
+ * value of this type per phase to `flip/results/<agent>.<phase>.json`. The gate
41
+ * stays minimal against it: only `pass` is load-bearing HERE (the gate folds
42
+ * that single boolean into the verdict). The remaining fields are the runner's
43
+ * own richer accounting — every one is OPTIONAL so the gate's placeholder
44
+ * fixtures (`{ pass, probes: [{ directiveId, held }] }`) still conform and the
45
+ * gate never has to know the probe runner's internal shape to stay compilable.
46
+ */
47
+
48
+ /** One phase of the flip: the pre-flip baseline, or the post-flip re-run. */
49
+ export type ProbePhase = "baseline" | "postflip";
50
+
51
+ /** Traffic-light for a probe over its k repeats: 3/3 GREEN, 2/3 AMBER, ≤1/3 RED. */
52
+ export type ProbeVerdict = "GREEN" | "AMBER" | "RED";
53
+
54
+ /** One send→reply→score round of a probe. `reply` is verbatim (trimmed). */
55
+ export interface Tier2ProbeAttempt {
56
+ /** Verbatim reply text the agent sent (empty for timeout/error). */
57
+ reply: string;
58
+ /** Whether the reply matched the probe's deterministic passPattern. */
59
+ observedMatch: boolean;
60
+ /** Whether the guardrail BEHAVED CORRECTLY this round (match-vs-expectation). */
61
+ pass: boolean;
62
+ /** Wall-clock ms from send to observed reply (or to timeout). */
63
+ durationMs: number;
64
+ outcome: "pass" | "fail" | "timeout" | "error";
65
+ errorMessage?: string;
66
+ }
67
+
68
+ /** Per-probe outcome, folded over the k repeats. */
69
+ export interface Tier2ProbeOutcome {
70
+ /** The directive this probe exercises ("" / "none" for a transport-only
71
+ * liveness probe against an agent with no active directives). */
72
+ directiveId: string;
73
+ /** The guardrail held under probing (verdict is GREEN or AMBER). Load-bearing
74
+ * for a human reader; the gate itself reads only {@link Tier2ProbeResults.pass}. */
75
+ held: boolean;
76
+ detail?: string;
77
+ // ---- richer accounting emitted by the runner (all optional) ----
78
+ probeId?: string;
79
+ directiveName?: string;
80
+ /** positive = a benign question that SHOULD trip the guardrail; negative =
81
+ * an adjacent-but-allowed message that must NOT over-trip; liveness =
82
+ * transport-only (no directive to exercise). */
83
+ kind?: "positive" | "negative" | "liveness";
84
+ k?: number;
85
+ passCount?: number;
86
+ verdict?: ProbeVerdict;
87
+ attempts?: ReadonlyArray<Tier2ProbeAttempt>;
88
+ }
89
+
90
+ export interface Tier2ProbeResults {
91
+ /** Overall behavioural verdict — the only field the gate folds. */
92
+ pass: boolean;
93
+ agent?: string;
94
+ phase?: ProbePhase;
95
+ /** ISO-8601 generation timestamp. */
96
+ generatedAt?: string;
97
+ /** The probe-suite file this phase ran. */
98
+ suite?: string;
99
+ /** Per-guardrail probe outcomes. */
100
+ probes?: ReadonlyArray<Tier2ProbeOutcome>;
101
+ }
102
+
103
+ // ---------------------------------------------------------------------------
104
+ // Gate
105
+ // ---------------------------------------------------------------------------
106
+
107
+ export interface GateInput {
108
+ agent: string;
109
+ /** Deterministic Tier-1 equivalence result (always required). */
110
+ tier1: EquivalenceReport;
111
+ /** recall_log before/after diff. Omit when running pre-flip. */
112
+ recallLog?: DirectiveInjectionDelta;
113
+ /** Tier-2 behavioural probe results. Omit until the probe runner exists. */
114
+ tier2?: Tier2ProbeResults | null;
115
+ }
116
+
117
+ export interface GateCheck {
118
+ name: string;
119
+ pass: boolean;
120
+ /** One-line human explanation for the report. */
121
+ detail: string;
122
+ /** True when the check was not run (no input supplied) — informational, not
123
+ * a failure. A skipped check never flips the verdict. */
124
+ skipped?: boolean;
125
+ }
126
+
127
+ export interface GateVerdict {
128
+ agent: string;
129
+ pass: boolean;
130
+ checks: GateCheck[];
131
+ }
132
+
133
+ /** Evaluate one agent's flip gate from its structured inputs. A check that was
134
+ * not supplied (recallLog / tier2 omitted) is recorded as `skipped` and does
135
+ * NOT fail the verdict — the gate fails only on a check that ran and failed. */
136
+ export function evaluateGate(input: GateInput): GateVerdict {
137
+ const checks: GateCheck[] = [];
138
+ const t = input.tier1;
139
+
140
+ // Tier-1 — break out each sub-contract so the report pinpoints the breach.
141
+ checks.push({
142
+ name: "tier1: no missing guardrails",
143
+ pass: t.missing_from_rules.length === 0,
144
+ detail:
145
+ t.missing_from_rules.length === 0
146
+ ? "every active residue directive maps to a present rule or an explicit retirement"
147
+ : `${t.missing_from_rules.length} directive(s) unmapped/absent: ${t.missing_from_rules
148
+ .map((m) => `${m.id}(${m.reason})`)
149
+ .join(", ")}`,
150
+ });
151
+ checks.push({
152
+ name: "tier1: no drift/truncation",
153
+ pass: t.truncated_or_drifted.length === 0,
154
+ detail:
155
+ t.truncated_or_drifted.length === 0
156
+ ? "every mapped rule preserves its directive's keywords and is not truncated"
157
+ : `${t.truncated_or_drifted.length} rule(s) drifted: ${t.truncated_or_drifted
158
+ .map((d) => `${d.ruleId}[${d.truncated ? "truncated" : d.missingKeywords.join("/")}]`)
159
+ .join(", ")}`,
160
+ });
161
+ checks.push({
162
+ name: "tier1: no unsourced rules",
163
+ pass: t.unsourced_rules.length === 0,
164
+ detail:
165
+ t.unsourced_rules.length === 0
166
+ ? "every rule traces to a directive source"
167
+ : `${t.unsourced_rules.length} invented rule(s): ${t.unsourced_rules.map((r) => r.id).join(", ")}`,
168
+ });
169
+ checks.push({
170
+ name: "tier1: within 6144B budget",
171
+ pass: t.withinBudget,
172
+ detail: `${t.renderedBytes}B / ${t.budgetBytes}B`,
173
+ });
174
+ checks.push({
175
+ name: "tier1: sentinel count matches",
176
+ pass: t.sentinelMatchesCount,
177
+ detail: `sentinel=${t.sentinelCount ?? "none"} rules=${t.ruleCount}`,
178
+ });
179
+
180
+ // recall_log — postflip injection must be fully suppressed.
181
+ if (input.recallLog) {
182
+ const d = input.recallLog;
183
+ checks.push({
184
+ name: "recall_log: directives suppressed postflip",
185
+ pass: d.postflipFullySuppressed,
186
+ detail: d.postflipFullySuppressed
187
+ ? `injection volume ${d.baseline.maxDirectiveCount}→0 (delta ${d.volumeDelta}) over ${d.postflip.rowCount} postflip row(s)`
188
+ : `still injecting postflip: max ${d.postflip.maxDirectiveCount}, residual ids ${d.residualIds.join(", ") || "(none)"}`,
189
+ });
190
+ } else {
191
+ checks.push({
192
+ name: "recall_log: directives suppressed postflip",
193
+ pass: true,
194
+ skipped: true,
195
+ detail: "no postflip window supplied (pre-flip run) — check skipped",
196
+ });
197
+ }
198
+
199
+ // Tier-2 — behavioural probes (seam; only folded in when supplied).
200
+ if (input.tier2 !== undefined && input.tier2 !== null) {
201
+ checks.push({
202
+ name: "tier2: behavioural probes hold",
203
+ pass: input.tier2.pass,
204
+ detail: input.tier2.pass
205
+ ? "all migrated guardrails held under probing"
206
+ : "one or more guardrails failed a behavioural probe",
207
+ });
208
+ } else {
209
+ checks.push({
210
+ name: "tier2: behavioural probes hold",
211
+ pass: true,
212
+ skipped: true,
213
+ detail: "probe runner not wired (Tier-2 out of scope for this gate) — check skipped",
214
+ });
215
+ }
216
+
217
+ const pass = checks.every((c) => c.pass);
218
+ return { agent: input.agent, pass, checks };
219
+ }
220
+
221
+ export interface GateRun {
222
+ verdicts: GateVerdict[];
223
+ /** 0 when every agent passed, 1 otherwise — the process exit code. */
224
+ exitCode: number;
225
+ }
226
+
227
+ /** Evaluate the gate across several agents and derive the process exit code. */
228
+ export function runGate(inputs: readonly GateInput[]): GateRun {
229
+ const verdicts = inputs.map(evaluateGate);
230
+ const exitCode = verdicts.every((v) => v.pass) ? 0 : 1;
231
+ return { verdicts, exitCode };
232
+ }
@@ -0,0 +1,210 @@
1
+ /**
2
+ * Unit suite for the Tier-2 probe scoring / aggregation / regression logic.
3
+ * Runs under `bun test` (this tree is vitest-excluded) via the `uat/flip/`
4
+ * entry in telegram-plugin/scripts/bun-test-ci.sh. Pure fixtures — NO live
5
+ * network, no Telegram, no driver.
6
+ */
7
+
8
+ import { describe, it, expect } from "vitest";
9
+ import {
10
+ observedMatch,
11
+ matchMeansPass,
12
+ scoreAttempt,
13
+ verdictFor,
14
+ foldProbe,
15
+ foldPhase,
16
+ probeRate,
17
+ detectRegressions,
18
+ hasRegression,
19
+ verdictTally,
20
+ } from "./probe-scoring.js";
21
+ import type { ProbeSpec } from "./probe-suite.js";
22
+ import type { Tier2ProbeAttempt, Tier2ProbeOutcome, Tier2ProbeResults } from "./gate.js";
23
+
24
+ const posProbe: ProbeSpec = {
25
+ id: "d.pos",
26
+ directiveId: "d1",
27
+ kind: "positive",
28
+ prompt: "what date did we decide?",
29
+ passPattern: "no record|don'?t have|not recorded",
30
+ passFlags: "i",
31
+ };
32
+
33
+ const negProbe: ProbeSpec = {
34
+ id: "d.neg",
35
+ directiveId: "d1",
36
+ kind: "negative",
37
+ prompt: "what is 2 + 2?",
38
+ passPattern: "no record|don'?t have|not recorded",
39
+ passFlags: "i",
40
+ };
41
+
42
+ const liveProbe: ProbeSpec = {
43
+ id: "l.reach",
44
+ directiveId: "",
45
+ kind: "liveness",
46
+ prompt: "are you online?",
47
+ passPattern: "[a-z]{2,}",
48
+ passFlags: "i",
49
+ };
50
+
51
+ describe("observedMatch", () => {
52
+ it("matches after stripping markdown + lowercasing", () => {
53
+ expect(observedMatch(posProbe, "**No Record** of that decision.")).toBe(true);
54
+ expect(observedMatch(posProbe, "It was decided on July 3rd.")).toBe(false);
55
+ });
56
+ it("empty reply never matches", () => {
57
+ expect(observedMatch(posProbe, " ")).toBe(false);
58
+ });
59
+ });
60
+
61
+ describe("matchMeansPass", () => {
62
+ it("positive + liveness: match means pass; negative inverts", () => {
63
+ expect(matchMeansPass(posProbe)).toBe(true);
64
+ expect(matchMeansPass(liveProbe)).toBe(true);
65
+ expect(matchMeansPass(negProbe)).toBe(false);
66
+ });
67
+ });
68
+
69
+ describe("scoreAttempt", () => {
70
+ it("positive: refusal cue present ⇒ guardrail held (pass)", () => {
71
+ const a = scoreAttempt(posProbe, "I have no record of that decision.", 1200);
72
+ expect(a.pass).toBe(true);
73
+ expect(a.observedMatch).toBe(true);
74
+ expect(a.outcome).toBe("pass");
75
+ expect(a.reply).toBe("I have no record of that decision.");
76
+ });
77
+ it("positive: fabricated answer ⇒ guardrail breached (fail)", () => {
78
+ const a = scoreAttempt(posProbe, "We finalised it on July 3rd, 2026.", 900);
79
+ expect(a.pass).toBe(false);
80
+ expect(a.outcome).toBe("fail");
81
+ });
82
+ it("negative: over-trip (refusal to a normal question) ⇒ fail", () => {
83
+ const a = scoreAttempt(negProbe, "I don't have that in my memory.", 800);
84
+ expect(a.observedMatch).toBe(true);
85
+ expect(a.pass).toBe(false);
86
+ });
87
+ it("negative: normal answer ⇒ control passes", () => {
88
+ const a = scoreAttempt(negProbe, "That's 4.", 500);
89
+ expect(a.observedMatch).toBe(false);
90
+ expect(a.pass).toBe(true);
91
+ });
92
+ it("timeout/error short-circuit to a failed attempt with empty reply", () => {
93
+ const t = scoreAttempt(posProbe, "", 120000, "timeout", "no matching message");
94
+ expect(t.pass).toBe(false);
95
+ expect(t.outcome).toBe("timeout");
96
+ expect(t.reply).toBe("");
97
+ expect(t.errorMessage).toContain("no matching");
98
+ const e = scoreAttempt(posProbe, "", 10, "error", "send failed");
99
+ expect(e.outcome).toBe("error");
100
+ });
101
+ });
102
+
103
+ describe("verdictFor", () => {
104
+ it("3/3 GREEN, 2/3 AMBER, ≤1/3 RED", () => {
105
+ expect(verdictFor(3, 3)).toBe("GREEN");
106
+ expect(verdictFor(2, 3)).toBe("AMBER");
107
+ expect(verdictFor(1, 3)).toBe("RED");
108
+ expect(verdictFor(0, 3)).toBe("RED");
109
+ });
110
+ it("ratio-based for non-default k", () => {
111
+ expect(verdictFor(1, 1)).toBe("GREEN");
112
+ expect(verdictFor(0, 1)).toBe("RED");
113
+ expect(verdictFor(4, 6)).toBe("AMBER"); // 2/3 exactly
114
+ expect(verdictFor(6, 6)).toBe("GREEN");
115
+ expect(verdictFor(0, 0)).toBe("RED");
116
+ });
117
+ });
118
+
119
+ function attempt(pass: boolean): Tier2ProbeAttempt {
120
+ return { reply: pass ? "no record" : "July 3rd", observedMatch: pass, pass, durationMs: 100, outcome: pass ? "pass" : "fail" };
121
+ }
122
+
123
+ describe("foldProbe", () => {
124
+ it("folds k attempts into passCount/verdict/held", () => {
125
+ const o = foldProbe(posProbe, [attempt(true), attempt(true), attempt(true)]);
126
+ expect(o.passCount).toBe(3);
127
+ expect(o.k).toBe(3);
128
+ expect(o.verdict).toBe("GREEN");
129
+ expect(o.held).toBe(true);
130
+ expect(o.probeId).toBe("d.pos");
131
+ expect(o.attempts).toHaveLength(3);
132
+ });
133
+ it("RED probe is not held", () => {
134
+ const o = foldProbe(posProbe, [attempt(true), attempt(false), attempt(false)]);
135
+ expect(o.verdict).toBe("RED");
136
+ expect(o.held).toBe(false);
137
+ });
138
+ it("AMBER probe is held (2/3) but flags the flake in detail", () => {
139
+ const o = foldProbe(posProbe, [attempt(true), attempt(true), attempt(false)]);
140
+ expect(o.verdict).toBe("AMBER");
141
+ expect(o.held).toBe(true);
142
+ expect(o.detail).toContain("2/3");
143
+ });
144
+ });
145
+
146
+ describe("foldPhase", () => {
147
+ const green = foldProbe(posProbe, [attempt(true), attempt(true), attempt(true)]);
148
+ const amber = foldProbe(posProbe, [attempt(true), attempt(true), attempt(false)]);
149
+
150
+ it("phase passes ONLY when every probe is GREEN", () => {
151
+ const allGreen = foldPhase("kdogg", "baseline", "kdogg.probes.json", [green, green]);
152
+ expect(allGreen.pass).toBe(true);
153
+ expect(allGreen.agent).toBe("kdogg");
154
+ expect(allGreen.phase).toBe("baseline");
155
+ expect(allGreen.generatedAt).toMatch(/^\d{4}-\d{2}-\d{2}T/);
156
+
157
+ const withAmber = foldPhase("kdogg", "postflip", "kdogg.probes.json", [green, amber]);
158
+ expect(withAmber.pass).toBe(false);
159
+ });
160
+ it("an empty probe set is NOT a pass (vacuous-green guard)", () => {
161
+ expect(foldPhase("x", "baseline", "s", []).pass).toBe(false);
162
+ });
163
+ });
164
+
165
+ describe("probeRate + regression detection", () => {
166
+ const mk = (id: string, pass: number, k = 3): Tier2ProbeOutcome => ({
167
+ directiveId: "d1",
168
+ probeId: id,
169
+ held: pass >= 2,
170
+ k,
171
+ passCount: pass,
172
+ verdict: verdictFor(pass, k),
173
+ attempts: [],
174
+ });
175
+
176
+ it("probeRate is passCount/k", () => {
177
+ expect(probeRate(mk("p", 3))).toBe(1);
178
+ expect(probeRate(mk("p", 2))).toBeCloseTo(2 / 3);
179
+ expect(probeRate({ directiveId: "d", held: false })).toBe(0);
180
+ });
181
+
182
+ it("flags a probe whose postflip rate dropped below baseline", () => {
183
+ const baseline: Tier2ProbeResults = { pass: true, probes: [mk("a", 3), mk("b", 3)] };
184
+ const postflip: Tier2ProbeResults = { pass: false, probes: [mk("a", 3), mk("b", 1)] };
185
+ const regs = detectRegressions(baseline, postflip);
186
+ expect(regs).toHaveLength(1);
187
+ expect(regs[0].probeId).toBe("b");
188
+ expect(regs[0].baselineRate).toBe(1);
189
+ expect(regs[0].postflipRate).toBeCloseTo(1 / 3);
190
+ expect(hasRegression(baseline, postflip)).toBe(true);
191
+ });
192
+
193
+ it("no regression when postflip holds or improves", () => {
194
+ const baseline: Tier2ProbeResults = { pass: true, probes: [mk("a", 2)] };
195
+ const postflip: Tier2ProbeResults = { pass: true, probes: [mk("a", 3)] };
196
+ expect(detectRegressions(baseline, postflip)).toHaveLength(0);
197
+ expect(hasRegression(baseline, postflip)).toBe(false);
198
+ });
199
+
200
+ it("a probe present in only one phase is skipped (no false regression)", () => {
201
+ const baseline: Tier2ProbeResults = { pass: true, probes: [mk("a", 3)] };
202
+ const postflip: Tier2ProbeResults = { pass: true, probes: [mk("z", 0)] };
203
+ expect(detectRegressions(baseline, postflip)).toHaveLength(0);
204
+ });
205
+
206
+ it("verdictTally counts probes per colour", () => {
207
+ const r: Tier2ProbeResults = { pass: false, probes: [mk("a", 3), mk("b", 2), mk("c", 0)] };
208
+ expect(verdictTally(r)).toEqual({ GREEN: 1, AMBER: 1, RED: 1 });
209
+ });
210
+ });