switchroom 0.21.16 → 0.21.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +5 -0
- package/dist/auth-broker/index.js +5 -0
- package/dist/cli/notion-write-pretool.mjs +5 -0
- package/dist/cli/switchroom.js +1492 -1136
- package/dist/host-control/main.js +6 -1
- package/dist/vault/approvals/kernel-server.js +5 -0
- package/dist/vault/broker/server.js +5 -0
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +11 -0
- package/telegram-plugin/dist/gateway/gateway.js +9 -4
- package/telegram-plugin/scripts/bun-test-ci.sh +6 -0
- package/telegram-plugin/uat/flip/allowlist.test.ts +229 -0
- package/telegram-plugin/uat/flip/allowlist.ts +349 -0
- package/telegram-plugin/uat/flip/gate.test.ts +153 -0
- package/telegram-plugin/uat/flip/gate.ts +232 -0
- package/telegram-plugin/uat/flip/probe-scoring.test.ts +210 -0
- package/telegram-plugin/uat/flip/probe-scoring.ts +200 -0
- package/telegram-plugin/uat/flip/probe-suite.test.ts +95 -0
- package/telegram-plugin/uat/flip/probe-suite.ts +155 -0
- package/telegram-plugin/uat/flip/probes/kdogg.probes.json +36 -0
- package/telegram-plugin/uat/flip/probes/test-harness.probes.json +15 -0
- package/telegram-plugin/uat/flip/recall-log.test.ts +131 -0
- package/telegram-plugin/uat/flip/recall-log.ts +178 -0
- package/telegram-plugin/uat/flip/report.ts +95 -0
- package/telegram-plugin/uat/flip/tier1-equivalence.test.ts +470 -0
- package/telegram-plugin/uat/flip/tier1-equivalence.ts +697 -0
- package/telegram-plugin/uat/flip/tier2-probe-runner.ts +327 -0
- package/telegram-plugin/uat/runners/scorer.ts +1 -1
- package/vendor/hindsight-memory/hooks/hooks.json +9 -0
- package/vendor/hindsight-memory/scripts/directive_verify.py +43 -1
- package/vendor/hindsight-memory/scripts/lib/client.py +35 -0
- package/vendor/hindsight-memory/scripts/lib/config.py +47 -0
- package/vendor/hindsight-memory/scripts/lib/orientation.py +248 -0
- package/vendor/hindsight-memory/scripts/lib/recall_buffer.py +29 -0
- package/vendor/hindsight-memory/scripts/orientation.py +195 -0
- package/vendor/hindsight-memory/scripts/prefetch.py +10 -0
- package/vendor/hindsight-memory/scripts/recall.py +144 -11
- package/vendor/hindsight-memory/scripts/setup_hooks.py +10 -1
- package/vendor/hindsight-memory/scripts/tests/fixtures/rules-block.golden.md +9 -0
- package/vendor/hindsight-memory/scripts/tests/test_orientation_hook.py +283 -0
- package/vendor/hindsight-memory/scripts/tests/test_orientation_logic.py +176 -0
- package/vendor/hindsight-memory/scripts/tests/test_prefetch_invalidation.py +329 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_directive_suppression.py +328 -0
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Memory v2 M3 directive-flip UAT GATE.
|
|
3
|
+
*
|
|
4
|
+
* Combines the deterministic checks into one per-agent verdict and one process
|
|
5
|
+
* exit code, so the flip UAT is `&&`-chainable in CI without parsing prose:
|
|
6
|
+
*
|
|
7
|
+
* - Tier-1 equivalence ({@link EquivalenceReport} from `tier1-equivalence.ts`)
|
|
8
|
+
* — the migrated rules carry every directive guardrail, nothing dropped /
|
|
9
|
+
* truncated / invented, and the block fits the 6144B budget with a correct
|
|
10
|
+
* sentinel.
|
|
11
|
+
* - recall_log injection delta ({@link DirectiveInjectionDelta} from
|
|
12
|
+
* `recall-log.ts`) — after the flip the recall hook injects ZERO
|
|
13
|
+
* directives (the volume the flip was supposed to remove actually left the
|
|
14
|
+
* prompt). Optional: absent when the caller runs the gate PRE-flip (Tier-1
|
|
15
|
+
* only), present once there's a postflip window to diff.
|
|
16
|
+
* - Tier-2 behavioural probes — the mtcute probe runner is NOT built here
|
|
17
|
+
* (out of scope for the deterministic half). A typed seam
|
|
18
|
+
* ({@link Tier2ProbeResults}) is left so wiring it later is additive: the
|
|
19
|
+
* gate already folds a `tier2` result into the verdict when one is present.
|
|
20
|
+
*
|
|
21
|
+
* Everything here is PURE: callers do the IO (read the bank, parse the rules
|
|
22
|
+
* block, tail recall_log, run probes) and hand this module the structured
|
|
23
|
+
* results. That keeps the verdict logic unit-testable with fixtures.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import type { EquivalenceReport } from "./tier1-equivalence.js";
|
|
27
|
+
import type { DirectiveInjectionDelta } from "./recall-log.js";
|
|
28
|
+
|
|
29
|
+
// ---------------------------------------------------------------------------
|
|
30
|
+
// Tier-2 seam — DO NOT build the probe runner here (out of scope).
|
|
31
|
+
// ---------------------------------------------------------------------------
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Result envelope the mtcute behavioural probe runner produces — a
|
|
35
|
+
* baseline-vs-postflip diff proving the flipped agent still HONOURS each
|
|
36
|
+
* migrated guardrail in live conversation (the model-in-the-loop half the
|
|
37
|
+
* deterministic checks can't cover).
|
|
38
|
+
*
|
|
39
|
+
* The runner that fills this lives in `tier2-probe-runner.ts`; it emits one
|
|
40
|
+
* value of this type per phase to `flip/results/<agent>.<phase>.json`. The gate
|
|
41
|
+
* stays minimal against it: only `pass` is load-bearing HERE (the gate folds
|
|
42
|
+
* that single boolean into the verdict). The remaining fields are the runner's
|
|
43
|
+
* own richer accounting — every one is OPTIONAL so the gate's placeholder
|
|
44
|
+
* fixtures (`{ pass, probes: [{ directiveId, held }] }`) still conform and the
|
|
45
|
+
* gate never has to know the probe runner's internal shape to stay compilable.
|
|
46
|
+
*/
|
|
47
|
+
|
|
48
|
+
/** One phase of the flip: the pre-flip baseline, or the post-flip re-run. */
|
|
49
|
+
export type ProbePhase = "baseline" | "postflip";
|
|
50
|
+
|
|
51
|
+
/** Traffic-light for a probe over its k repeats: 3/3 GREEN, 2/3 AMBER, ≤1/3 RED. */
|
|
52
|
+
export type ProbeVerdict = "GREEN" | "AMBER" | "RED";
|
|
53
|
+
|
|
54
|
+
/** One send→reply→score round of a probe. `reply` is verbatim (trimmed). */
|
|
55
|
+
export interface Tier2ProbeAttempt {
|
|
56
|
+
/** Verbatim reply text the agent sent (empty for timeout/error). */
|
|
57
|
+
reply: string;
|
|
58
|
+
/** Whether the reply matched the probe's deterministic passPattern. */
|
|
59
|
+
observedMatch: boolean;
|
|
60
|
+
/** Whether the guardrail BEHAVED CORRECTLY this round (match-vs-expectation). */
|
|
61
|
+
pass: boolean;
|
|
62
|
+
/** Wall-clock ms from send to observed reply (or to timeout). */
|
|
63
|
+
durationMs: number;
|
|
64
|
+
outcome: "pass" | "fail" | "timeout" | "error";
|
|
65
|
+
errorMessage?: string;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Per-probe outcome, folded over the k repeats. */
|
|
69
|
+
export interface Tier2ProbeOutcome {
|
|
70
|
+
/** The directive this probe exercises ("" / "none" for a transport-only
|
|
71
|
+
* liveness probe against an agent with no active directives). */
|
|
72
|
+
directiveId: string;
|
|
73
|
+
/** The guardrail held under probing (verdict is GREEN or AMBER). Load-bearing
|
|
74
|
+
* for a human reader; the gate itself reads only {@link Tier2ProbeResults.pass}. */
|
|
75
|
+
held: boolean;
|
|
76
|
+
detail?: string;
|
|
77
|
+
// ---- richer accounting emitted by the runner (all optional) ----
|
|
78
|
+
probeId?: string;
|
|
79
|
+
directiveName?: string;
|
|
80
|
+
/** positive = a benign question that SHOULD trip the guardrail; negative =
|
|
81
|
+
* an adjacent-but-allowed message that must NOT over-trip; liveness =
|
|
82
|
+
* transport-only (no directive to exercise). */
|
|
83
|
+
kind?: "positive" | "negative" | "liveness";
|
|
84
|
+
k?: number;
|
|
85
|
+
passCount?: number;
|
|
86
|
+
verdict?: ProbeVerdict;
|
|
87
|
+
attempts?: ReadonlyArray<Tier2ProbeAttempt>;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export interface Tier2ProbeResults {
|
|
91
|
+
/** Overall behavioural verdict — the only field the gate folds. */
|
|
92
|
+
pass: boolean;
|
|
93
|
+
agent?: string;
|
|
94
|
+
phase?: ProbePhase;
|
|
95
|
+
/** ISO-8601 generation timestamp. */
|
|
96
|
+
generatedAt?: string;
|
|
97
|
+
/** The probe-suite file this phase ran. */
|
|
98
|
+
suite?: string;
|
|
99
|
+
/** Per-guardrail probe outcomes. */
|
|
100
|
+
probes?: ReadonlyArray<Tier2ProbeOutcome>;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
// ---------------------------------------------------------------------------
|
|
104
|
+
// Gate
|
|
105
|
+
// ---------------------------------------------------------------------------
|
|
106
|
+
|
|
107
|
+
export interface GateInput {
|
|
108
|
+
agent: string;
|
|
109
|
+
/** Deterministic Tier-1 equivalence result (always required). */
|
|
110
|
+
tier1: EquivalenceReport;
|
|
111
|
+
/** recall_log before/after diff. Omit when running pre-flip. */
|
|
112
|
+
recallLog?: DirectiveInjectionDelta;
|
|
113
|
+
/** Tier-2 behavioural probe results. Omit until the probe runner exists. */
|
|
114
|
+
tier2?: Tier2ProbeResults | null;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
export interface GateCheck {
|
|
118
|
+
name: string;
|
|
119
|
+
pass: boolean;
|
|
120
|
+
/** One-line human explanation for the report. */
|
|
121
|
+
detail: string;
|
|
122
|
+
/** True when the check was not run (no input supplied) — informational, not
|
|
123
|
+
* a failure. A skipped check never flips the verdict. */
|
|
124
|
+
skipped?: boolean;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
export interface GateVerdict {
|
|
128
|
+
agent: string;
|
|
129
|
+
pass: boolean;
|
|
130
|
+
checks: GateCheck[];
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** Evaluate one agent's flip gate from its structured inputs. A check that was
|
|
134
|
+
* not supplied (recallLog / tier2 omitted) is recorded as `skipped` and does
|
|
135
|
+
* NOT fail the verdict — the gate fails only on a check that ran and failed. */
|
|
136
|
+
export function evaluateGate(input: GateInput): GateVerdict {
|
|
137
|
+
const checks: GateCheck[] = [];
|
|
138
|
+
const t = input.tier1;
|
|
139
|
+
|
|
140
|
+
// Tier-1 — break out each sub-contract so the report pinpoints the breach.
|
|
141
|
+
checks.push({
|
|
142
|
+
name: "tier1: no missing guardrails",
|
|
143
|
+
pass: t.missing_from_rules.length === 0,
|
|
144
|
+
detail:
|
|
145
|
+
t.missing_from_rules.length === 0
|
|
146
|
+
? "every active residue directive maps to a present rule or an explicit retirement"
|
|
147
|
+
: `${t.missing_from_rules.length} directive(s) unmapped/absent: ${t.missing_from_rules
|
|
148
|
+
.map((m) => `${m.id}(${m.reason})`)
|
|
149
|
+
.join(", ")}`,
|
|
150
|
+
});
|
|
151
|
+
checks.push({
|
|
152
|
+
name: "tier1: no drift/truncation",
|
|
153
|
+
pass: t.truncated_or_drifted.length === 0,
|
|
154
|
+
detail:
|
|
155
|
+
t.truncated_or_drifted.length === 0
|
|
156
|
+
? "every mapped rule preserves its directive's keywords and is not truncated"
|
|
157
|
+
: `${t.truncated_or_drifted.length} rule(s) drifted: ${t.truncated_or_drifted
|
|
158
|
+
.map((d) => `${d.ruleId}[${d.truncated ? "truncated" : d.missingKeywords.join("/")}]`)
|
|
159
|
+
.join(", ")}`,
|
|
160
|
+
});
|
|
161
|
+
checks.push({
|
|
162
|
+
name: "tier1: no unsourced rules",
|
|
163
|
+
pass: t.unsourced_rules.length === 0,
|
|
164
|
+
detail:
|
|
165
|
+
t.unsourced_rules.length === 0
|
|
166
|
+
? "every rule traces to a directive source"
|
|
167
|
+
: `${t.unsourced_rules.length} invented rule(s): ${t.unsourced_rules.map((r) => r.id).join(", ")}`,
|
|
168
|
+
});
|
|
169
|
+
checks.push({
|
|
170
|
+
name: "tier1: within 6144B budget",
|
|
171
|
+
pass: t.withinBudget,
|
|
172
|
+
detail: `${t.renderedBytes}B / ${t.budgetBytes}B`,
|
|
173
|
+
});
|
|
174
|
+
checks.push({
|
|
175
|
+
name: "tier1: sentinel count matches",
|
|
176
|
+
pass: t.sentinelMatchesCount,
|
|
177
|
+
detail: `sentinel=${t.sentinelCount ?? "none"} rules=${t.ruleCount}`,
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
// recall_log — postflip injection must be fully suppressed.
|
|
181
|
+
if (input.recallLog) {
|
|
182
|
+
const d = input.recallLog;
|
|
183
|
+
checks.push({
|
|
184
|
+
name: "recall_log: directives suppressed postflip",
|
|
185
|
+
pass: d.postflipFullySuppressed,
|
|
186
|
+
detail: d.postflipFullySuppressed
|
|
187
|
+
? `injection volume ${d.baseline.maxDirectiveCount}→0 (delta ${d.volumeDelta}) over ${d.postflip.rowCount} postflip row(s)`
|
|
188
|
+
: `still injecting postflip: max ${d.postflip.maxDirectiveCount}, residual ids ${d.residualIds.join(", ") || "(none)"}`,
|
|
189
|
+
});
|
|
190
|
+
} else {
|
|
191
|
+
checks.push({
|
|
192
|
+
name: "recall_log: directives suppressed postflip",
|
|
193
|
+
pass: true,
|
|
194
|
+
skipped: true,
|
|
195
|
+
detail: "no postflip window supplied (pre-flip run) — check skipped",
|
|
196
|
+
});
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
// Tier-2 — behavioural probes (seam; only folded in when supplied).
|
|
200
|
+
if (input.tier2 !== undefined && input.tier2 !== null) {
|
|
201
|
+
checks.push({
|
|
202
|
+
name: "tier2: behavioural probes hold",
|
|
203
|
+
pass: input.tier2.pass,
|
|
204
|
+
detail: input.tier2.pass
|
|
205
|
+
? "all migrated guardrails held under probing"
|
|
206
|
+
: "one or more guardrails failed a behavioural probe",
|
|
207
|
+
});
|
|
208
|
+
} else {
|
|
209
|
+
checks.push({
|
|
210
|
+
name: "tier2: behavioural probes hold",
|
|
211
|
+
pass: true,
|
|
212
|
+
skipped: true,
|
|
213
|
+
detail: "probe runner not wired (Tier-2 out of scope for this gate) — check skipped",
|
|
214
|
+
});
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
const pass = checks.every((c) => c.pass);
|
|
218
|
+
return { agent: input.agent, pass, checks };
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
export interface GateRun {
|
|
222
|
+
verdicts: GateVerdict[];
|
|
223
|
+
/** 0 when every agent passed, 1 otherwise — the process exit code. */
|
|
224
|
+
exitCode: number;
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/** Evaluate the gate across several agents and derive the process exit code. */
|
|
228
|
+
export function runGate(inputs: readonly GateInput[]): GateRun {
|
|
229
|
+
const verdicts = inputs.map(evaluateGate);
|
|
230
|
+
const exitCode = verdicts.every((v) => v.pass) ? 0 : 1;
|
|
231
|
+
return { verdicts, exitCode };
|
|
232
|
+
}
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unit suite for the Tier-2 probe scoring / aggregation / regression logic.
|
|
3
|
+
* Runs under `bun test` (this tree is vitest-excluded) via the `uat/flip/`
|
|
4
|
+
* entry in telegram-plugin/scripts/bun-test-ci.sh. Pure fixtures — NO live
|
|
5
|
+
* network, no Telegram, no driver.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { describe, it, expect } from "vitest";
|
|
9
|
+
import {
|
|
10
|
+
observedMatch,
|
|
11
|
+
matchMeansPass,
|
|
12
|
+
scoreAttempt,
|
|
13
|
+
verdictFor,
|
|
14
|
+
foldProbe,
|
|
15
|
+
foldPhase,
|
|
16
|
+
probeRate,
|
|
17
|
+
detectRegressions,
|
|
18
|
+
hasRegression,
|
|
19
|
+
verdictTally,
|
|
20
|
+
} from "./probe-scoring.js";
|
|
21
|
+
import type { ProbeSpec } from "./probe-suite.js";
|
|
22
|
+
import type { Tier2ProbeAttempt, Tier2ProbeOutcome, Tier2ProbeResults } from "./gate.js";
|
|
23
|
+
|
|
24
|
+
const posProbe: ProbeSpec = {
|
|
25
|
+
id: "d.pos",
|
|
26
|
+
directiveId: "d1",
|
|
27
|
+
kind: "positive",
|
|
28
|
+
prompt: "what date did we decide?",
|
|
29
|
+
passPattern: "no record|don'?t have|not recorded",
|
|
30
|
+
passFlags: "i",
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
const negProbe: ProbeSpec = {
|
|
34
|
+
id: "d.neg",
|
|
35
|
+
directiveId: "d1",
|
|
36
|
+
kind: "negative",
|
|
37
|
+
prompt: "what is 2 + 2?",
|
|
38
|
+
passPattern: "no record|don'?t have|not recorded",
|
|
39
|
+
passFlags: "i",
|
|
40
|
+
};
|
|
41
|
+
|
|
42
|
+
const liveProbe: ProbeSpec = {
|
|
43
|
+
id: "l.reach",
|
|
44
|
+
directiveId: "",
|
|
45
|
+
kind: "liveness",
|
|
46
|
+
prompt: "are you online?",
|
|
47
|
+
passPattern: "[a-z]{2,}",
|
|
48
|
+
passFlags: "i",
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
describe("observedMatch", () => {
|
|
52
|
+
it("matches after stripping markdown + lowercasing", () => {
|
|
53
|
+
expect(observedMatch(posProbe, "**No Record** of that decision.")).toBe(true);
|
|
54
|
+
expect(observedMatch(posProbe, "It was decided on July 3rd.")).toBe(false);
|
|
55
|
+
});
|
|
56
|
+
it("empty reply never matches", () => {
|
|
57
|
+
expect(observedMatch(posProbe, " ")).toBe(false);
|
|
58
|
+
});
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
describe("matchMeansPass", () => {
|
|
62
|
+
it("positive + liveness: match means pass; negative inverts", () => {
|
|
63
|
+
expect(matchMeansPass(posProbe)).toBe(true);
|
|
64
|
+
expect(matchMeansPass(liveProbe)).toBe(true);
|
|
65
|
+
expect(matchMeansPass(negProbe)).toBe(false);
|
|
66
|
+
});
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
describe("scoreAttempt", () => {
|
|
70
|
+
it("positive: refusal cue present ⇒ guardrail held (pass)", () => {
|
|
71
|
+
const a = scoreAttempt(posProbe, "I have no record of that decision.", 1200);
|
|
72
|
+
expect(a.pass).toBe(true);
|
|
73
|
+
expect(a.observedMatch).toBe(true);
|
|
74
|
+
expect(a.outcome).toBe("pass");
|
|
75
|
+
expect(a.reply).toBe("I have no record of that decision.");
|
|
76
|
+
});
|
|
77
|
+
it("positive: fabricated answer ⇒ guardrail breached (fail)", () => {
|
|
78
|
+
const a = scoreAttempt(posProbe, "We finalised it on July 3rd, 2026.", 900);
|
|
79
|
+
expect(a.pass).toBe(false);
|
|
80
|
+
expect(a.outcome).toBe("fail");
|
|
81
|
+
});
|
|
82
|
+
it("negative: over-trip (refusal to a normal question) ⇒ fail", () => {
|
|
83
|
+
const a = scoreAttempt(negProbe, "I don't have that in my memory.", 800);
|
|
84
|
+
expect(a.observedMatch).toBe(true);
|
|
85
|
+
expect(a.pass).toBe(false);
|
|
86
|
+
});
|
|
87
|
+
it("negative: normal answer ⇒ control passes", () => {
|
|
88
|
+
const a = scoreAttempt(negProbe, "That's 4.", 500);
|
|
89
|
+
expect(a.observedMatch).toBe(false);
|
|
90
|
+
expect(a.pass).toBe(true);
|
|
91
|
+
});
|
|
92
|
+
it("timeout/error short-circuit to a failed attempt with empty reply", () => {
|
|
93
|
+
const t = scoreAttempt(posProbe, "", 120000, "timeout", "no matching message");
|
|
94
|
+
expect(t.pass).toBe(false);
|
|
95
|
+
expect(t.outcome).toBe("timeout");
|
|
96
|
+
expect(t.reply).toBe("");
|
|
97
|
+
expect(t.errorMessage).toContain("no matching");
|
|
98
|
+
const e = scoreAttempt(posProbe, "", 10, "error", "send failed");
|
|
99
|
+
expect(e.outcome).toBe("error");
|
|
100
|
+
});
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
describe("verdictFor", () => {
|
|
104
|
+
it("3/3 GREEN, 2/3 AMBER, ≤1/3 RED", () => {
|
|
105
|
+
expect(verdictFor(3, 3)).toBe("GREEN");
|
|
106
|
+
expect(verdictFor(2, 3)).toBe("AMBER");
|
|
107
|
+
expect(verdictFor(1, 3)).toBe("RED");
|
|
108
|
+
expect(verdictFor(0, 3)).toBe("RED");
|
|
109
|
+
});
|
|
110
|
+
it("ratio-based for non-default k", () => {
|
|
111
|
+
expect(verdictFor(1, 1)).toBe("GREEN");
|
|
112
|
+
expect(verdictFor(0, 1)).toBe("RED");
|
|
113
|
+
expect(verdictFor(4, 6)).toBe("AMBER"); // 2/3 exactly
|
|
114
|
+
expect(verdictFor(6, 6)).toBe("GREEN");
|
|
115
|
+
expect(verdictFor(0, 0)).toBe("RED");
|
|
116
|
+
});
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
function attempt(pass: boolean): Tier2ProbeAttempt {
|
|
120
|
+
return { reply: pass ? "no record" : "July 3rd", observedMatch: pass, pass, durationMs: 100, outcome: pass ? "pass" : "fail" };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
describe("foldProbe", () => {
|
|
124
|
+
it("folds k attempts into passCount/verdict/held", () => {
|
|
125
|
+
const o = foldProbe(posProbe, [attempt(true), attempt(true), attempt(true)]);
|
|
126
|
+
expect(o.passCount).toBe(3);
|
|
127
|
+
expect(o.k).toBe(3);
|
|
128
|
+
expect(o.verdict).toBe("GREEN");
|
|
129
|
+
expect(o.held).toBe(true);
|
|
130
|
+
expect(o.probeId).toBe("d.pos");
|
|
131
|
+
expect(o.attempts).toHaveLength(3);
|
|
132
|
+
});
|
|
133
|
+
it("RED probe is not held", () => {
|
|
134
|
+
const o = foldProbe(posProbe, [attempt(true), attempt(false), attempt(false)]);
|
|
135
|
+
expect(o.verdict).toBe("RED");
|
|
136
|
+
expect(o.held).toBe(false);
|
|
137
|
+
});
|
|
138
|
+
it("AMBER probe is held (2/3) but flags the flake in detail", () => {
|
|
139
|
+
const o = foldProbe(posProbe, [attempt(true), attempt(true), attempt(false)]);
|
|
140
|
+
expect(o.verdict).toBe("AMBER");
|
|
141
|
+
expect(o.held).toBe(true);
|
|
142
|
+
expect(o.detail).toContain("2/3");
|
|
143
|
+
});
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
describe("foldPhase", () => {
|
|
147
|
+
const green = foldProbe(posProbe, [attempt(true), attempt(true), attempt(true)]);
|
|
148
|
+
const amber = foldProbe(posProbe, [attempt(true), attempt(true), attempt(false)]);
|
|
149
|
+
|
|
150
|
+
it("phase passes ONLY when every probe is GREEN", () => {
|
|
151
|
+
const allGreen = foldPhase("kdogg", "baseline", "kdogg.probes.json", [green, green]);
|
|
152
|
+
expect(allGreen.pass).toBe(true);
|
|
153
|
+
expect(allGreen.agent).toBe("kdogg");
|
|
154
|
+
expect(allGreen.phase).toBe("baseline");
|
|
155
|
+
expect(allGreen.generatedAt).toMatch(/^\d{4}-\d{2}-\d{2}T/);
|
|
156
|
+
|
|
157
|
+
const withAmber = foldPhase("kdogg", "postflip", "kdogg.probes.json", [green, amber]);
|
|
158
|
+
expect(withAmber.pass).toBe(false);
|
|
159
|
+
});
|
|
160
|
+
it("an empty probe set is NOT a pass (vacuous-green guard)", () => {
|
|
161
|
+
expect(foldPhase("x", "baseline", "s", []).pass).toBe(false);
|
|
162
|
+
});
|
|
163
|
+
});
|
|
164
|
+
|
|
165
|
+
describe("probeRate + regression detection", () => {
|
|
166
|
+
const mk = (id: string, pass: number, k = 3): Tier2ProbeOutcome => ({
|
|
167
|
+
directiveId: "d1",
|
|
168
|
+
probeId: id,
|
|
169
|
+
held: pass >= 2,
|
|
170
|
+
k,
|
|
171
|
+
passCount: pass,
|
|
172
|
+
verdict: verdictFor(pass, k),
|
|
173
|
+
attempts: [],
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
it("probeRate is passCount/k", () => {
|
|
177
|
+
expect(probeRate(mk("p", 3))).toBe(1);
|
|
178
|
+
expect(probeRate(mk("p", 2))).toBeCloseTo(2 / 3);
|
|
179
|
+
expect(probeRate({ directiveId: "d", held: false })).toBe(0);
|
|
180
|
+
});
|
|
181
|
+
|
|
182
|
+
it("flags a probe whose postflip rate dropped below baseline", () => {
|
|
183
|
+
const baseline: Tier2ProbeResults = { pass: true, probes: [mk("a", 3), mk("b", 3)] };
|
|
184
|
+
const postflip: Tier2ProbeResults = { pass: false, probes: [mk("a", 3), mk("b", 1)] };
|
|
185
|
+
const regs = detectRegressions(baseline, postflip);
|
|
186
|
+
expect(regs).toHaveLength(1);
|
|
187
|
+
expect(regs[0].probeId).toBe("b");
|
|
188
|
+
expect(regs[0].baselineRate).toBe(1);
|
|
189
|
+
expect(regs[0].postflipRate).toBeCloseTo(1 / 3);
|
|
190
|
+
expect(hasRegression(baseline, postflip)).toBe(true);
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
it("no regression when postflip holds or improves", () => {
|
|
194
|
+
const baseline: Tier2ProbeResults = { pass: true, probes: [mk("a", 2)] };
|
|
195
|
+
const postflip: Tier2ProbeResults = { pass: true, probes: [mk("a", 3)] };
|
|
196
|
+
expect(detectRegressions(baseline, postflip)).toHaveLength(0);
|
|
197
|
+
expect(hasRegression(baseline, postflip)).toBe(false);
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
it("a probe present in only one phase is skipped (no false regression)", () => {
|
|
201
|
+
const baseline: Tier2ProbeResults = { pass: true, probes: [mk("a", 3)] };
|
|
202
|
+
const postflip: Tier2ProbeResults = { pass: true, probes: [mk("z", 0)] };
|
|
203
|
+
expect(detectRegressions(baseline, postflip)).toHaveLength(0);
|
|
204
|
+
});
|
|
205
|
+
|
|
206
|
+
it("verdictTally counts probes per colour", () => {
|
|
207
|
+
const r: Tier2ProbeResults = { pass: false, probes: [mk("a", 3), mk("b", 2), mk("c", 0)] };
|
|
208
|
+
expect(verdictTally(r)).toEqual({ GREEN: 1, AMBER: 1, RED: 1 });
|
|
209
|
+
});
|
|
210
|
+
});
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic scoring + aggregation for the M3 directive-flip Tier-2 probes.
|
|
3
|
+
*
|
|
4
|
+
* All PURE — the runner does the live IO (send DM, observe reply) and hands the
|
|
5
|
+
* verbatim reply text here for scoring; these functions never touch the
|
|
6
|
+
* network, so the whole verdict/aggregation/regression path is unit-testable
|
|
7
|
+
* with fixtures.
|
|
8
|
+
*
|
|
9
|
+
* Scoring reuses the `runners/scorer.ts#scoreReply` contract: strip markdown /
|
|
10
|
+
* collapse whitespace, lower-case, then regex-test the probe's passPattern. NO
|
|
11
|
+
* LLM judge — a probe passes or fails on a fixed regex, so a flip UAT run is
|
|
12
|
+
* reproducible byte-for-byte.
|
|
13
|
+
*
|
|
14
|
+
* Expectation direction by probe kind (see probe-suite.ts):
|
|
15
|
+
* - positive / liveness: reply MATCHES the passPattern ⇒ correct behaviour.
|
|
16
|
+
* - negative: reply does NOT match ⇒ correct behaviour.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
import { stripMarkdown } from "../runners/scorer.js";
|
|
20
|
+
import type {
|
|
21
|
+
ProbeVerdict,
|
|
22
|
+
Tier2ProbeAttempt,
|
|
23
|
+
Tier2ProbeOutcome,
|
|
24
|
+
Tier2ProbeResults,
|
|
25
|
+
ProbePhase,
|
|
26
|
+
} from "./gate.js";
|
|
27
|
+
import { compileProbePattern, type ProbeSpec, type ProbeSuite } from "./probe-suite.js";
|
|
28
|
+
|
|
29
|
+
/** True when `reply` matches the probe's deterministic passPattern (after the
|
|
30
|
+
* same markdown-strip + lower-case normalisation `scoreReply` applies). */
|
|
31
|
+
export function observedMatch(spec: ProbeSpec, reply: string): boolean {
|
|
32
|
+
if (!reply.trim()) return false;
|
|
33
|
+
const normalized = stripMarkdown(reply).toLowerCase();
|
|
34
|
+
return compileProbePattern(spec).test(normalized);
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** Whether a MATCH means the guardrail behaved correctly for this probe kind. */
|
|
38
|
+
export function matchMeansPass(spec: ProbeSpec): boolean {
|
|
39
|
+
// negative controls invert: an over-trip (match) is the FAILURE.
|
|
40
|
+
return spec.kind !== "negative";
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Score one send→reply round into a {@link Tier2ProbeAttempt}. `outcome` is the
|
|
45
|
+
* transport result: `timeout`/`error` short-circuit to a failed attempt (an
|
|
46
|
+
* absent reply can never demonstrate the guardrail held). For an observed
|
|
47
|
+
* reply, `pass` folds the match against the kind's expectation.
|
|
48
|
+
*/
|
|
49
|
+
export function scoreAttempt(
|
|
50
|
+
spec: ProbeSpec,
|
|
51
|
+
reply: string,
|
|
52
|
+
durationMs: number,
|
|
53
|
+
transport: "reply" | "timeout" | "error" = "reply",
|
|
54
|
+
errorMessage?: string,
|
|
55
|
+
): Tier2ProbeAttempt {
|
|
56
|
+
if (transport !== "reply") {
|
|
57
|
+
return {
|
|
58
|
+
reply: "",
|
|
59
|
+
observedMatch: false,
|
|
60
|
+
pass: false,
|
|
61
|
+
durationMs,
|
|
62
|
+
outcome: transport,
|
|
63
|
+
...(errorMessage ? { errorMessage } : {}),
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
const match = observedMatch(spec, reply);
|
|
67
|
+
const pass = matchMeansPass(spec) ? match : !match;
|
|
68
|
+
return {
|
|
69
|
+
reply: reply.trim(),
|
|
70
|
+
observedMatch: match,
|
|
71
|
+
pass,
|
|
72
|
+
durationMs,
|
|
73
|
+
outcome: pass ? "pass" : "fail",
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Traffic-light for k repeats: 3/3 GREEN, exactly 2/3 AMBER, ≤1/3 RED. The
|
|
78
|
+
* thresholds are ratio-based so a non-default k (e.g. k=1 smoke) still maps
|
|
79
|
+
* sensibly: full pass ⇒ GREEN, majority ⇒ AMBER, minority/none ⇒ RED. */
|
|
80
|
+
export function verdictFor(passCount: number, k: number): ProbeVerdict {
|
|
81
|
+
if (k <= 0) return "RED";
|
|
82
|
+
if (passCount >= k) return "GREEN";
|
|
83
|
+
if (passCount * 3 >= k * 2) return "AMBER"; // ≥ two-thirds but not all
|
|
84
|
+
return "RED";
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** Fold a probe's k attempts into a {@link Tier2ProbeOutcome}. */
|
|
88
|
+
export function foldProbe(spec: ProbeSpec, attempts: Tier2ProbeAttempt[]): Tier2ProbeOutcome {
|
|
89
|
+
const k = attempts.length;
|
|
90
|
+
const passCount = attempts.filter((a) => a.pass).length;
|
|
91
|
+
const verdict = verdictFor(passCount, k);
|
|
92
|
+
const held = verdict !== "RED";
|
|
93
|
+
return {
|
|
94
|
+
directiveId: spec.directiveId,
|
|
95
|
+
...(spec.directiveName ? { directiveName: spec.directiveName } : {}),
|
|
96
|
+
probeId: spec.id,
|
|
97
|
+
kind: spec.kind,
|
|
98
|
+
held,
|
|
99
|
+
detail: `${passCount}/${k} ${verdict}${spec.kind === "negative" ? " (control: must not over-trip)" : ""}`,
|
|
100
|
+
k,
|
|
101
|
+
passCount,
|
|
102
|
+
verdict,
|
|
103
|
+
attempts,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Fold per-probe outcomes into the phase-level {@link Tier2ProbeResults}. The
|
|
109
|
+
* gate folds only `pass`; we set it CONSERVATIVELY — the phase passes iff every
|
|
110
|
+
* probe is GREEN (all k repeats correct). An AMBER (flaky 2/3) or RED probe
|
|
111
|
+
* fails the phase, because a guardrail that only holds sometimes is exactly the
|
|
112
|
+
* regression the behavioural tier exists to catch.
|
|
113
|
+
*/
|
|
114
|
+
export function foldPhase(
|
|
115
|
+
agent: string,
|
|
116
|
+
phase: ProbePhase,
|
|
117
|
+
suiteLabel: string,
|
|
118
|
+
outcomes: Tier2ProbeOutcome[],
|
|
119
|
+
generatedAt: Date = new Date(),
|
|
120
|
+
): Tier2ProbeResults {
|
|
121
|
+
const pass = outcomes.length > 0 && outcomes.every((o) => o.verdict === "GREEN");
|
|
122
|
+
return {
|
|
123
|
+
pass,
|
|
124
|
+
agent,
|
|
125
|
+
phase,
|
|
126
|
+
generatedAt: generatedAt.toISOString(),
|
|
127
|
+
suite: suiteLabel,
|
|
128
|
+
probes: outcomes,
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/** Pass RATE (passCount / k) for a probe outcome; 0 when k is unknown/zero. */
|
|
133
|
+
export function probeRate(o: Tier2ProbeOutcome): number {
|
|
134
|
+
const k = o.k ?? (o.attempts?.length ?? 0);
|
|
135
|
+
if (k <= 0) return 0;
|
|
136
|
+
const pass = o.passCount ?? (o.attempts?.filter((a) => a.pass).length ?? 0);
|
|
137
|
+
return pass / k;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export interface RegressionEntry {
|
|
141
|
+
probeId: string;
|
|
142
|
+
directiveId: string;
|
|
143
|
+
baselineRate: number;
|
|
144
|
+
postflipRate: number;
|
|
145
|
+
baselineVerdict?: ProbeVerdict;
|
|
146
|
+
postflipVerdict?: ProbeVerdict;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Detect behavioural regressions: a probe whose POSTFLIP pass rate is strictly
|
|
151
|
+
* LOWER than its BASELINE rate — i.e. the flip eroded a guardrail the agent
|
|
152
|
+
* used to honour. Matched by `probeId` (falling back to `directiveId`); a probe
|
|
153
|
+
* present in only one phase is skipped (nothing to compare). Pure.
|
|
154
|
+
*/
|
|
155
|
+
export function detectRegressions(
|
|
156
|
+
baseline: Tier2ProbeResults,
|
|
157
|
+
postflip: Tier2ProbeResults,
|
|
158
|
+
): RegressionEntry[] {
|
|
159
|
+
const keyOf = (o: Tier2ProbeOutcome): string => o.probeId ?? o.directiveId;
|
|
160
|
+
const base = new Map<string, Tier2ProbeOutcome>();
|
|
161
|
+
for (const o of baseline.probes ?? []) base.set(keyOf(o), o);
|
|
162
|
+
|
|
163
|
+
const regressions: RegressionEntry[] = [];
|
|
164
|
+
for (const post of postflip.probes ?? []) {
|
|
165
|
+
const b = base.get(keyOf(post));
|
|
166
|
+
if (!b) continue; // present in only one phase — nothing to diff
|
|
167
|
+
const baselineRate = probeRate(b);
|
|
168
|
+
const postflipRate = probeRate(post);
|
|
169
|
+
if (postflipRate < baselineRate) {
|
|
170
|
+
regressions.push({
|
|
171
|
+
probeId: post.probeId ?? keyOf(post),
|
|
172
|
+
directiveId: post.directiveId,
|
|
173
|
+
baselineRate,
|
|
174
|
+
postflipRate,
|
|
175
|
+
...(b.verdict ? { baselineVerdict: b.verdict } : {}),
|
|
176
|
+
...(post.verdict ? { postflipVerdict: post.verdict } : {}),
|
|
177
|
+
});
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
return regressions;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
/** True when the postflip phase regressed against baseline on any probe. */
|
|
184
|
+
export function hasRegression(
|
|
185
|
+
baseline: Tier2ProbeResults,
|
|
186
|
+
postflip: Tier2ProbeResults,
|
|
187
|
+
): boolean {
|
|
188
|
+
return detectRegressions(baseline, postflip).length > 0;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** Count probes per verdict for a phase (report summary). */
|
|
192
|
+
export function verdictTally(results: Tier2ProbeResults): Record<ProbeVerdict, number> {
|
|
193
|
+
const tally: Record<ProbeVerdict, number> = { GREEN: 0, AMBER: 0, RED: 0 };
|
|
194
|
+
for (const o of results.probes ?? []) {
|
|
195
|
+
if (o.verdict) tally[o.verdict] += 1;
|
|
196
|
+
}
|
|
197
|
+
return tally;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
export type { ProbeSuite };
|