switchroom 0.21.16 → 0.21.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +5 -0
- package/dist/auth-broker/index.js +5 -0
- package/dist/cli/notion-write-pretool.mjs +5 -0
- package/dist/cli/switchroom.js +1492 -1136
- package/dist/host-control/main.js +6 -1
- package/dist/vault/approvals/kernel-server.js +5 -0
- package/dist/vault/broker/server.js +5 -0
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +11 -0
- package/telegram-plugin/dist/gateway/gateway.js +9 -4
- package/telegram-plugin/scripts/bun-test-ci.sh +6 -0
- package/telegram-plugin/uat/flip/allowlist.test.ts +229 -0
- package/telegram-plugin/uat/flip/allowlist.ts +349 -0
- package/telegram-plugin/uat/flip/gate.test.ts +153 -0
- package/telegram-plugin/uat/flip/gate.ts +232 -0
- package/telegram-plugin/uat/flip/probe-scoring.test.ts +210 -0
- package/telegram-plugin/uat/flip/probe-scoring.ts +200 -0
- package/telegram-plugin/uat/flip/probe-suite.test.ts +95 -0
- package/telegram-plugin/uat/flip/probe-suite.ts +155 -0
- package/telegram-plugin/uat/flip/probes/kdogg.probes.json +36 -0
- package/telegram-plugin/uat/flip/probes/test-harness.probes.json +15 -0
- package/telegram-plugin/uat/flip/recall-log.test.ts +131 -0
- package/telegram-plugin/uat/flip/recall-log.ts +178 -0
- package/telegram-plugin/uat/flip/report.ts +95 -0
- package/telegram-plugin/uat/flip/tier1-equivalence.test.ts +470 -0
- package/telegram-plugin/uat/flip/tier1-equivalence.ts +697 -0
- package/telegram-plugin/uat/flip/tier2-probe-runner.ts +327 -0
- package/telegram-plugin/uat/runners/scorer.ts +1 -1
- package/vendor/hindsight-memory/hooks/hooks.json +9 -0
- package/vendor/hindsight-memory/scripts/directive_verify.py +43 -1
- package/vendor/hindsight-memory/scripts/lib/client.py +35 -0
- package/vendor/hindsight-memory/scripts/lib/config.py +47 -0
- package/vendor/hindsight-memory/scripts/lib/orientation.py +248 -0
- package/vendor/hindsight-memory/scripts/lib/recall_buffer.py +29 -0
- package/vendor/hindsight-memory/scripts/orientation.py +195 -0
- package/vendor/hindsight-memory/scripts/prefetch.py +10 -0
- package/vendor/hindsight-memory/scripts/recall.py +144 -11
- package/vendor/hindsight-memory/scripts/setup_hooks.py +10 -1
- package/vendor/hindsight-memory/scripts/tests/fixtures/rules-block.golden.md +9 -0
- package/vendor/hindsight-memory/scripts/tests/test_orientation_hook.py +283 -0
- package/vendor/hindsight-memory/scripts/tests/test_orientation_logic.py +176 -0
- package/vendor/hindsight-memory/scripts/tests/test_prefetch_invalidation.py +329 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_directive_suppression.py +328 -0
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Markdown report renderer for the M3 directive-flip UAT gate.
|
|
3
|
+
*
|
|
4
|
+
* Same shape as the agent-self-sufficiency runner's report
|
|
5
|
+
* (`telegram-plugin/uat/runners/report.ts`): a headline verdict, a per-agent ×
|
|
6
|
+
* per-check pass/fail matrix an operator reads in one glance, then a verbatim
|
|
7
|
+
* triage list of every failing check so a PR reviewer can diff without
|
|
8
|
+
* re-running. Pure — takes the gate verdicts, returns a string.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import type { GateRun, GateVerdict, GateCheck } from "./gate.js";
|
|
12
|
+
|
|
13
|
+
export interface FlipReportOptions {
|
|
14
|
+
startedAt?: Date;
|
|
15
|
+
durationSeconds?: number;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
function mark(c: GateCheck): string {
|
|
19
|
+
if (c.skipped) return "—";
|
|
20
|
+
return c.pass ? "✅" : "❌";
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export function renderFlipReport(run: GateRun, opts: FlipReportOptions = {}): string {
|
|
24
|
+
const { verdicts } = run;
|
|
25
|
+
const passed = verdicts.filter((v) => v.pass).length;
|
|
26
|
+
const lines: string[] = [];
|
|
27
|
+
|
|
28
|
+
lines.push("# M3 directive-flip UAT gate report");
|
|
29
|
+
lines.push("");
|
|
30
|
+
if (opts.startedAt) lines.push(`- **Run start:** ${opts.startedAt.toISOString()}`);
|
|
31
|
+
if (typeof opts.durationSeconds === "number")
|
|
32
|
+
lines.push(`- **Duration:** ${opts.durationSeconds.toFixed(1)}s`);
|
|
33
|
+
lines.push(`- **Agents:** ${verdicts.map((v) => v.agent).join(", ") || "(none)"}`);
|
|
34
|
+
lines.push(`- **Verdict:** ${run.exitCode === 0 ? "PASS" : "FAIL"} (${passed}/${verdicts.length} agents green)`);
|
|
35
|
+
lines.push("");
|
|
36
|
+
|
|
37
|
+
// Per-agent × per-check matrix. Check names are stable across agents, so use
|
|
38
|
+
// the first verdict's check order as the column set.
|
|
39
|
+
const checkNames = verdicts[0]?.checks.map((c) => c.name) ?? [];
|
|
40
|
+
if (checkNames.length > 0 && verdicts.length > 0) {
|
|
41
|
+
lines.push("## Check matrix");
|
|
42
|
+
lines.push("");
|
|
43
|
+
lines.push(`| Agent | Verdict | ${checkNames.map((n) => shortName(n)).join(" | ")} |`);
|
|
44
|
+
lines.push(`|---|---|${checkNames.map(() => "---").join("|")}|`);
|
|
45
|
+
for (const v of verdicts) {
|
|
46
|
+
const byName = new Map(v.checks.map((c) => [c.name, c]));
|
|
47
|
+
const cells = checkNames.map((n) => {
|
|
48
|
+
const c = byName.get(n);
|
|
49
|
+
return c ? mark(c) : "?";
|
|
50
|
+
});
|
|
51
|
+
lines.push(`| \`${v.agent}\` | ${v.pass ? "PASS" : "FAIL"} | ${cells.join(" | ")} |`);
|
|
52
|
+
}
|
|
53
|
+
lines.push("");
|
|
54
|
+
lines.push("_Legend: ✅ pass · ❌ fail · — skipped (input not supplied)._");
|
|
55
|
+
lines.push("");
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// Triage — every failing check, verbatim detail.
|
|
59
|
+
const failing: Array<{ agent: string; check: GateCheck }> = [];
|
|
60
|
+
for (const v of verdicts) {
|
|
61
|
+
for (const c of v.checks) {
|
|
62
|
+
if (!c.pass && !c.skipped) failing.push({ agent: v.agent, check: c });
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
lines.push("## Triage — failing checks");
|
|
66
|
+
lines.push("");
|
|
67
|
+
if (failing.length === 0) {
|
|
68
|
+
lines.push("No failing checks. Every ran check passed.");
|
|
69
|
+
} else {
|
|
70
|
+
lines.push("| Agent | Check | Detail |");
|
|
71
|
+
lines.push("|---|---|---|");
|
|
72
|
+
for (const f of failing) {
|
|
73
|
+
lines.push(`| \`${f.agent}\` | ${escapeCell(f.check.name)} | ${escapeCell(f.check.detail)} |`);
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
lines.push("");
|
|
77
|
+
|
|
78
|
+
return lines.join("\n");
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** Render just one agent's verdict as a compact block (for a per-agent log). */
|
|
82
|
+
export function renderVerdictLine(v: GateVerdict): string {
|
|
83
|
+
const parts = v.checks.map((c) => `${mark(c)} ${shortName(c.name)}`);
|
|
84
|
+
return `${v.pass ? "PASS" : "FAIL"} ${v.agent}: ${parts.join(" · ")}`;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** Drop the `tierN: ` / `recall_log: ` prefix for a compact column header. */
|
|
88
|
+
function shortName(name: string): string {
|
|
89
|
+
const idx = name.indexOf(": ");
|
|
90
|
+
return idx === -1 ? name : name.slice(idx + 2);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
function escapeCell(s: string): string {
|
|
94
|
+
return s.replace(/\|/g, "\\|").replace(/\n/g, " ").replace(/`/g, "ʼ");
|
|
95
|
+
}
|
|
@@ -0,0 +1,470 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unit suite for the Tier-1 directive⇄rules equivalence check. Runs under
|
|
3
|
+
* `bun test` (this tree is vitest-excluded — see vitest.config.ts) via the
|
|
4
|
+
* `uat/flip/` entry in telegram-plugin/scripts/bun-test-ci.sh; imports the
|
|
5
|
+
* shared `vitest` describe/it/expect that Bun's test runner understands, same
|
|
6
|
+
* as the sibling `uat/runners/*.test.ts`.
|
|
7
|
+
*
|
|
8
|
+
* Each failure mode gets its own case, and the fixtures are built by RENDERING
|
|
9
|
+
* a rule set with the real `renderRulesBlock` then parsing it back with the
|
|
10
|
+
* real `parseRulesBlock`, so the sentinel is genuine and the parser under test
|
|
11
|
+
* is the production one.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { describe, it, expect } from "vitest";
|
|
15
|
+
import {
|
|
16
|
+
compareDirectivesToRules,
|
|
17
|
+
extractKeywords,
|
|
18
|
+
residueDirectives,
|
|
19
|
+
type FlipDirective,
|
|
20
|
+
type DirectiveRuleMapping,
|
|
21
|
+
} from "./tier1-equivalence.js";
|
|
22
|
+
import {
|
|
23
|
+
renderRulesBlock,
|
|
24
|
+
parseRulesBlock,
|
|
25
|
+
type Rule,
|
|
26
|
+
type ParsedRulesBlock,
|
|
27
|
+
} from "../../../src/memory/rules-block.js";
|
|
28
|
+
|
|
29
|
+
function rule(id: string, text: string): Rule {
|
|
30
|
+
return { id, text, source: "telegram", created_at: "2026-08-18T00:00:00.000Z" };
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** Render + parse a rule set so the sentinel count is real and correct. */
|
|
34
|
+
function parsed(rules: Rule[]): ParsedRulesBlock {
|
|
35
|
+
const block = renderRulesBlock(rules);
|
|
36
|
+
const p = parseRulesBlock(block);
|
|
37
|
+
if (!p) throw new Error("fixture rules block failed to parse");
|
|
38
|
+
return p;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function directive(id: string, name: string, content: string, extra: Partial<FlipDirective> = {}): FlipDirective {
|
|
42
|
+
return { id, name, content, priority: 5, ...extra };
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
describe("compareDirectivesToRules — PASS", () => {
|
|
46
|
+
it("all residue directives mapped to present rules that preserve keywords", () => {
|
|
47
|
+
const directives = [
|
|
48
|
+
directive("d1", "no-exfil", 'Never exfiltrate secrets to "Telegram" chat.'),
|
|
49
|
+
directive("d2", "reply-tool", "You must always call the reply tool."),
|
|
50
|
+
];
|
|
51
|
+
const rules = [
|
|
52
|
+
rule("R-01", 'Never exfiltrate secrets to "Telegram" chat, ever.'),
|
|
53
|
+
rule("R-02", "You must always call the reply tool to answer."),
|
|
54
|
+
];
|
|
55
|
+
const mapping: DirectiveRuleMapping = { d1: "R-01", d2: "R-02" };
|
|
56
|
+
const rep = compareDirectivesToRules(directives, parsed(rules), mapping);
|
|
57
|
+
expect(rep.pass).toBe(true);
|
|
58
|
+
expect(rep.missing_from_rules).toEqual([]);
|
|
59
|
+
expect(rep.truncated_or_drifted).toEqual([]);
|
|
60
|
+
expect(rep.unsourced_rules).toEqual([]);
|
|
61
|
+
expect(rep.withinBudget).toBe(true);
|
|
62
|
+
expect(rep.sentinelMatchesCount).toBe(true);
|
|
63
|
+
expect(rep.residueDirectiveCount).toBe(2);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
it("an explicit retired:<reason> satisfies the obligation with no rule", () => {
|
|
67
|
+
const directives = [directive("d1", "stale", "Only use the old API endpoint.")];
|
|
68
|
+
const rules: Rule[] = [];
|
|
69
|
+
const mapping: DirectiveRuleMapping = { d1: "retired: superseded by the new gateway rule R-09" };
|
|
70
|
+
const rep = compareDirectivesToRules(directives, parsed(rules), mapping);
|
|
71
|
+
expect(rep.missing_from_rules).toEqual([]);
|
|
72
|
+
// No rules ⇒ nothing unsourced, budget fine, sentinel count 0 == 0.
|
|
73
|
+
expect(rep.pass).toBe(true);
|
|
74
|
+
});
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
describe("compareDirectivesToRules — (a) missing_from_rules", () => {
|
|
78
|
+
it("flags an unmapped active residue directive", () => {
|
|
79
|
+
const directives = [directive("d1", "guard", "Never delete production data.")];
|
|
80
|
+
const rep = compareDirectivesToRules(directives, parsed([]), {});
|
|
81
|
+
expect(rep.pass).toBe(false);
|
|
82
|
+
expect(rep.missing_from_rules).toEqual([{ id: "d1", name: "guard", reason: "unmapped" }]);
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
it("flags a directive mapped to a rule id that is not present", () => {
|
|
86
|
+
const directives = [directive("d1", "guard", "Never force-push main.")];
|
|
87
|
+
const rep = compareDirectivesToRules(directives, parsed([rule("R-01", "Never force-push main.")]), {
|
|
88
|
+
d1: "R-99",
|
|
89
|
+
});
|
|
90
|
+
expect(rep.pass).toBe(false);
|
|
91
|
+
expect(rep.missing_from_rules[0]).toMatchObject({ id: "d1", reason: "absent-rule", mappedTo: "R-99" });
|
|
92
|
+
// R-01 is present but nothing sources it ⇒ also unsourced.
|
|
93
|
+
expect(rep.unsourced_rules).toEqual([{ id: "R-01", text: "Never force-push main." }]);
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
it("flags an empty retirement reason as missing", () => {
|
|
97
|
+
const directives = [directive("d1", "guard", "Always branch off origin/main.")];
|
|
98
|
+
const rep = compareDirectivesToRules(directives, parsed([]), { d1: "retired: " });
|
|
99
|
+
expect(rep.pass).toBe(false);
|
|
100
|
+
expect(rep.missing_from_rules[0]).toMatchObject({ id: "d1", reason: "empty-retire-reason" });
|
|
101
|
+
});
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
describe("compareDirectivesToRules — (b) truncated_or_drifted", () => {
|
|
105
|
+
it("flags a rule that drops a load-bearing named fact (ALL-CAPS party name)", () => {
|
|
106
|
+
// Calibration: an incidental quoted proper noun is illustrative and NOT
|
|
107
|
+
// demanded (see the sibling test below) — but a SHOUTED party name is a
|
|
108
|
+
// guardrail a condensed rule cannot silently drop.
|
|
109
|
+
const directives = [directive("d1", "exec", "The executor is GARY DAVID BROWN, not Ian.")];
|
|
110
|
+
const rep = compareDirectivesToRules(
|
|
111
|
+
directives,
|
|
112
|
+
parsed([rule("R-01", "The executor is the estate trustee, not Ian.")]),
|
|
113
|
+
{ d1: "R-01" },
|
|
114
|
+
);
|
|
115
|
+
expect(rep.pass).toBe(false);
|
|
116
|
+
expect(rep.truncated_or_drifted[0]).toMatchObject({ id: "d1", ruleId: "R-01", truncated: false });
|
|
117
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("GARY DAVID BROWN");
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
it("does NOT flag a dropped illustrative token (incidental quoted proper noun)", () => {
|
|
121
|
+
// The OLD gate false-flagged this ("Twitter" dropped); the calibrated gate
|
|
122
|
+
// treats an un-cued quoted proper noun as an illustrative sample. The
|
|
123
|
+
// negation guardrail ("Never") is preserved, so this is a clean condense.
|
|
124
|
+
const directives = [directive("d1", "scope", 'Never post to "Twitter" without approval.')];
|
|
125
|
+
const rep = compareDirectivesToRules(directives, parsed([rule("R-01", "Never post without approval.")]), {
|
|
126
|
+
d1: "R-01",
|
|
127
|
+
});
|
|
128
|
+
expect(rep.truncated_or_drifted).toEqual([]);
|
|
129
|
+
expect(rep.pass).toBe(true);
|
|
130
|
+
});
|
|
131
|
+
|
|
132
|
+
it("flags a rule that drops a modal (never)", () => {
|
|
133
|
+
const directives = [directive("d1", "scope", "Never send emails automatically.")];
|
|
134
|
+
const rep = compareDirectivesToRules(directives, parsed([rule("R-01", "Send emails automatically.")]), {
|
|
135
|
+
d1: "R-01",
|
|
136
|
+
});
|
|
137
|
+
expect(rep.pass).toBe(false);
|
|
138
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("never");
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
it("flags a rule truncated with an ellipsis", () => {
|
|
142
|
+
const directives = [directive("d1", "scope", "Never delete production data without a backup first.")];
|
|
143
|
+
const rep = compareDirectivesToRules(
|
|
144
|
+
directives,
|
|
145
|
+
parsed([rule("R-01", "Never delete production data without a backup first…")]),
|
|
146
|
+
{ d1: "R-01" },
|
|
147
|
+
);
|
|
148
|
+
expect(rep.pass).toBe(false);
|
|
149
|
+
expect(rep.truncated_or_drifted[0].truncated).toBe(true);
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
it("flags a rule that is a strict prefix of the directive (cut short)", () => {
|
|
153
|
+
const directives = [directive("d1", "scope", "Always confirm intent before a destructive action")];
|
|
154
|
+
const rep = compareDirectivesToRules(
|
|
155
|
+
directives,
|
|
156
|
+
parsed([rule("R-01", "Always confirm intent before")]),
|
|
157
|
+
{ d1: "R-01" },
|
|
158
|
+
);
|
|
159
|
+
expect(rep.pass).toBe(false);
|
|
160
|
+
expect(rep.truncated_or_drifted[0].truncated).toBe(true);
|
|
161
|
+
});
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
describe("compareDirectivesToRules — (c) budget + sentinel", () => {
|
|
165
|
+
it("fails when the rendered block exceeds the 6144B budget", () => {
|
|
166
|
+
// One rule with a ~7000-char body blows the budget on its own.
|
|
167
|
+
const big = rule("R-01", "Never " + "x".repeat(7000));
|
|
168
|
+
const directives = [directive("d1", "big", big.text)];
|
|
169
|
+
const rep = compareDirectivesToRules(directives, parsed([big]), { d1: "R-01" });
|
|
170
|
+
expect(rep.withinBudget).toBe(false);
|
|
171
|
+
expect(rep.renderedBytes).toBeGreaterThan(6144);
|
|
172
|
+
expect(rep.pass).toBe(false);
|
|
173
|
+
});
|
|
174
|
+
|
|
175
|
+
it("fails when the sentinel count disagrees with the rule count", () => {
|
|
176
|
+
// Hand-forge a parsed block whose sentinel lies about the count.
|
|
177
|
+
const good = parsed([rule("R-01", "Never force-push main.")]);
|
|
178
|
+
const forged: ParsedRulesBlock = {
|
|
179
|
+
rules: good.rules,
|
|
180
|
+
sentinel: { hash: good.sentinel!.hash, count: 5 },
|
|
181
|
+
};
|
|
182
|
+
const directives = [directive("d1", "guard", "Never force-push main.")];
|
|
183
|
+
const rep = compareDirectivesToRules(directives, forged, { d1: "R-01" });
|
|
184
|
+
expect(rep.sentinelMatchesCount).toBe(false);
|
|
185
|
+
expect(rep.pass).toBe(false);
|
|
186
|
+
});
|
|
187
|
+
|
|
188
|
+
it("fails when the block has no sentinel at all", () => {
|
|
189
|
+
const noSentinel: ParsedRulesBlock = { rules: [rule("R-01", "Never x.")], sentinel: null };
|
|
190
|
+
const rep = compareDirectivesToRules([directive("d1", "g", "Never x.")], noSentinel, { d1: "R-01" });
|
|
191
|
+
expect(rep.sentinelCount).toBeNull();
|
|
192
|
+
expect(rep.sentinelMatchesCount).toBe(false);
|
|
193
|
+
expect(rep.pass).toBe(false);
|
|
194
|
+
});
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
describe("compareDirectivesToRules — (d) unsourced_rules", () => {
|
|
198
|
+
it("flags a rule with no directive mapping to it", () => {
|
|
199
|
+
const directives = [directive("d1", "g", "Never x.")];
|
|
200
|
+
const rules = [rule("R-01", "Never x."), rule("R-02", "An invented rule with no directive source.")];
|
|
201
|
+
const rep = compareDirectivesToRules(directives, parsed(rules), { d1: "R-01" });
|
|
202
|
+
expect(rep.pass).toBe(false);
|
|
203
|
+
expect(rep.unsourced_rules).toEqual([
|
|
204
|
+
{ id: "R-02", text: "An invented rule with no directive source." },
|
|
205
|
+
]);
|
|
206
|
+
});
|
|
207
|
+
});
|
|
208
|
+
|
|
209
|
+
describe("residue filtering", () => {
|
|
210
|
+
it("excludes non-residue categories and inactive directives from the obligation", () => {
|
|
211
|
+
const directives = [
|
|
212
|
+
directive("d1", "keep", "Never x.", { category: "reflect-directive" }),
|
|
213
|
+
directive("d2", "drop", "y.", { category: "retain-as-memory" }),
|
|
214
|
+
directive("d3", "gone", "z.", { category: "rules-block", isActive: false }),
|
|
215
|
+
];
|
|
216
|
+
expect(residueDirectives(directives).map((d) => d.id)).toEqual(["d1"]);
|
|
217
|
+
// d2/d3 need no rule; only d1 must be mapped.
|
|
218
|
+
const rep = compareDirectivesToRules(directives, parsed([rule("R-01", "Never x.")]), { d1: "R-01" });
|
|
219
|
+
expect(rep.pass).toBe(true);
|
|
220
|
+
expect(rep.residueDirectiveCount).toBe(1);
|
|
221
|
+
});
|
|
222
|
+
|
|
223
|
+
it("treats a directive with no category as active residue (caller pre-filtered)", () => {
|
|
224
|
+
const rep = compareDirectivesToRules([directive("d1", "g", "Never x.")], parsed([]), {});
|
|
225
|
+
expect(rep.missing_from_rules[0].reason).toBe("unmapped");
|
|
226
|
+
});
|
|
227
|
+
});
|
|
228
|
+
|
|
229
|
+
describe("extractKeywords — the fixed tokenizer", () => {
|
|
230
|
+
it("classes negation as a guardrail; quoted samples & proper nouns as illustrative", () => {
|
|
231
|
+
const kws = extractKeywords('Never post to "the public channel" on Twitter — always ask Ken first.');
|
|
232
|
+
const byVal = new Map(kws.map((k) => [k.value.toLowerCase(), k]));
|
|
233
|
+
// negation is a guardrail modal (synonym class `neg`).
|
|
234
|
+
expect(byVal.get("never")).toMatchObject({ kind: "modal", klass: "guardrail", modalClass: "neg" });
|
|
235
|
+
// the quoted phrase is present but illustrative (no verbatim cue precedes it).
|
|
236
|
+
expect(byVal.get("the public channel")).toMatchObject({ kind: "quote", klass: "illustrative" });
|
|
237
|
+
// incidental proper nouns are present but illustrative — never demanded.
|
|
238
|
+
expect(byVal.get("twitter")).toMatchObject({ klass: "illustrative" });
|
|
239
|
+
expect(byVal.get("ken")).toMatchObject({ klass: "illustrative" });
|
|
240
|
+
// bare "always" is emphasis, not an explicit scope phrase ⇒ no universal guardrail.
|
|
241
|
+
expect(kws.some((k) => k.kind === "modal" && k.modalClass === "universal")).toBe(false);
|
|
242
|
+
// "Never" at sentence start is the modal, not double-counted as a proper noun.
|
|
243
|
+
expect(kws.filter((k) => k.value.toLowerCase() === "never").length).toBe(1);
|
|
244
|
+
});
|
|
245
|
+
|
|
246
|
+
it("captures the don't modal with its apostrophe stem", () => {
|
|
247
|
+
const kws = extractKeywords("Don't ever run destructive commands.");
|
|
248
|
+
expect(kws.some((k) => k.kind === "modal" && k.value.toLowerCase().startsWith("don"))).toBe(true);
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
it("is stateless across calls — a prior call cannot make a later don't be missed", () => {
|
|
252
|
+
// Regression: a module-level /g regex used with .test() persists lastIndex,
|
|
253
|
+
// so a first call that matches "don't" mid-string would advance past
|
|
254
|
+
// position 0 and make this second call (with "Don't" at position 0) miss.
|
|
255
|
+
extractKeywords("You should never do this; don't do it either.");
|
|
256
|
+
const kws = extractKeywords("Don't run destructive commands.");
|
|
257
|
+
expect(kws.some((k) => k.kind === "modal" && k.value.toLowerCase().startsWith("don"))).toBe(true);
|
|
258
|
+
});
|
|
259
|
+
|
|
260
|
+
it("word-boundary matches modals so a substring does not satisfy", () => {
|
|
261
|
+
// The negation "cannot" is NOT satisfied by the substring "cannon" in the
|
|
262
|
+
// rule — the neg synonym class is word-boundary matched, not substring.
|
|
263
|
+
const directives = [directive("d1", "g", "You cannot deploy on Fridays.")];
|
|
264
|
+
const rep = compareDirectivesToRules(directives, parsed([rule("R-01", "Fire the cannon, deploy on Fridays.")]), {
|
|
265
|
+
d1: "R-01",
|
|
266
|
+
});
|
|
267
|
+
expect(rep.pass).toBe(false);
|
|
268
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("cannot");
|
|
269
|
+
});
|
|
270
|
+
});
|
|
271
|
+
|
|
272
|
+
// ---------------------------------------------------------------------------
|
|
273
|
+
// ADVERSARIAL DROP DETECTION — the acceptance criterion for the calibration.
|
|
274
|
+
//
|
|
275
|
+
// The calibration exists to stop condensation FALSE POSITIVES. The failure
|
|
276
|
+
// mode we refuse to introduce is the inverse: a gate so loose it greenlights a
|
|
277
|
+
// genuinely dropped guardrail. These tests pin that boundary with fixtures
|
|
278
|
+
// modelled on the real triage directives (lawgpt's executor / deferral-line /
|
|
279
|
+
// caveat facts). Each DROP fixture MUST FAIL; each synonym-preserving condense
|
|
280
|
+
// MUST PASS. If a future loosening regresses detection, one of these goes red.
|
|
281
|
+
// ---------------------------------------------------------------------------
|
|
282
|
+
|
|
283
|
+
// The lawgpt "defer-to-fiona" deferral line the directive says to emit verbatim.
|
|
284
|
+
const DEFERRAL_LINE =
|
|
285
|
+
"This is a Fiona question — raise it with Fiona Jessep at LHPW before acting on any of the above.";
|
|
286
|
+
const DEFER_DIRECTIVE =
|
|
287
|
+
`ENFORCEMENT: any legal-analysis response MUST end with exactly: "${DEFERRAL_LINE}" ` +
|
|
288
|
+
"All output channels; without that line the response is failed.";
|
|
289
|
+
|
|
290
|
+
describe("tier1 calibration — STILL FAILS on a genuinely dropped guardrail", () => {
|
|
291
|
+
it("drops a negation entirely: 'never send without approval' → 'send after review' ⇒ FAIL", () => {
|
|
292
|
+
const directives = [directive("d1", "confirm-external", "Never send letters to solicitors without Ken's approval.")];
|
|
293
|
+
const rep = compareDirectivesToRules(
|
|
294
|
+
directives,
|
|
295
|
+
parsed([rule("R-01", "Send letters to solicitors after review.")]),
|
|
296
|
+
{ d1: "R-01" },
|
|
297
|
+
);
|
|
298
|
+
expect(rep.pass).toBe(false);
|
|
299
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("never");
|
|
300
|
+
});
|
|
301
|
+
|
|
302
|
+
it("drops a named load-bearing fact: executor 'GARY DAVID BROWN' ⇒ FAIL", () => {
|
|
303
|
+
const directives = [
|
|
304
|
+
directive("d1", "gary-executor", "Executor of George's estate is GARY DAVID BROWN via the s17 chain."),
|
|
305
|
+
];
|
|
306
|
+
const rep = compareDirectivesToRules(
|
|
307
|
+
directives,
|
|
308
|
+
parsed([rule("R-01", "Executor of George's estate is the appointed representative via the s17 chain.")]),
|
|
309
|
+
{ d1: "R-01" },
|
|
310
|
+
);
|
|
311
|
+
expect(rep.pass).toBe(false);
|
|
312
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("GARY DAVID BROWN");
|
|
313
|
+
});
|
|
314
|
+
|
|
315
|
+
it("drops the required-verbatim deferral line ⇒ FAIL", () => {
|
|
316
|
+
const directives = [directive("d1", "defer-to-fiona", DEFER_DIRECTIVE)];
|
|
317
|
+
const rep = compareDirectivesToRules(
|
|
318
|
+
directives,
|
|
319
|
+
// rule paraphrases the obligation but omits the exact required wording.
|
|
320
|
+
parsed([rule("R-01", "Never give legal advice; end legal-analysis by pointing to the solicitor.")]),
|
|
321
|
+
{ d1: "R-01" },
|
|
322
|
+
);
|
|
323
|
+
expect(rep.pass).toBe(false);
|
|
324
|
+
expect(rep.truncated_or_drifted[0].missingKeywords.some((k) => k.includes("Fiona Jessep at LHPW"))).toBe(true);
|
|
325
|
+
});
|
|
326
|
+
|
|
327
|
+
it("drops a case-caveat number 'S CAV 2026 00037' ⇒ FAIL", () => {
|
|
328
|
+
const directives = [
|
|
329
|
+
directive("d1", "caveat", "Ian filed probate caveat S CAV 2026 00037, now cancelled; he holds no office."),
|
|
330
|
+
];
|
|
331
|
+
const rep = compareDirectivesToRules(
|
|
332
|
+
directives,
|
|
333
|
+
parsed([rule("R-01", "Ian filed a probate caveat, now cancelled; he holds no office.")]),
|
|
334
|
+
{ d1: "R-01" },
|
|
335
|
+
);
|
|
336
|
+
expect(rep.pass).toBe(false);
|
|
337
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("CAV 2026 00037");
|
|
338
|
+
});
|
|
339
|
+
|
|
340
|
+
it("truncates mid-guardrail (strict prefix cut) ⇒ FAIL", () => {
|
|
341
|
+
const directives = [
|
|
342
|
+
directive("d1", "no-ian", "Never call Ian the executor without a documentary grant of probate."),
|
|
343
|
+
];
|
|
344
|
+
const rep = compareDirectivesToRules(
|
|
345
|
+
directives,
|
|
346
|
+
parsed([rule("R-01", "Never call Ian the executor without a documentary")]),
|
|
347
|
+
{ d1: "R-01" },
|
|
348
|
+
);
|
|
349
|
+
expect(rep.pass).toBe(false);
|
|
350
|
+
expect(rep.truncated_or_drifted[0].truncated).toBe(true);
|
|
351
|
+
});
|
|
352
|
+
|
|
353
|
+
it("truncates mid-guardrail (ellipsis) ⇒ FAIL", () => {
|
|
354
|
+
const directives = [directive("d1", "defer-to-fiona", DEFER_DIRECTIVE)];
|
|
355
|
+
const rep = compareDirectivesToRules(
|
|
356
|
+
directives,
|
|
357
|
+
parsed([rule("R-01", `Never give legal advice; end with: "${DEFERRAL_LINE}"…`)]),
|
|
358
|
+
{ d1: "R-01" },
|
|
359
|
+
);
|
|
360
|
+
expect(rep.pass).toBe(false);
|
|
361
|
+
expect(rep.truncated_or_drifted[0].truncated).toBe(true);
|
|
362
|
+
});
|
|
363
|
+
});
|
|
364
|
+
|
|
365
|
+
describe("tier1 calibration — adversarial false-negatives (PR #4771 review)", () => {
|
|
366
|
+
// MAJOR 1 (review of PR #4771): a modal class must not be satisfied by an
|
|
367
|
+
// unrelated same-class survivor token. The directive NEGATES the proposition
|
|
368
|
+
// ("Never call Ian the executor …"); the condensed rule INVERTS it ("Ian is
|
|
369
|
+
// the executor …") while an unrelated "without" survives. Before the fix the
|
|
370
|
+
// neg class matched "without" anywhere in the rule and the inversion PASSED —
|
|
371
|
+
// the exact drop-through the gate exists to catch. It MUST FAIL.
|
|
372
|
+
it("negation dropped, meaning inverted, unrelated 'without' survives ⇒ FAIL", () => {
|
|
373
|
+
const directives = [
|
|
374
|
+
directive("d1", "no-ian", "Never call Ian the executor without a documentary grant of probate."),
|
|
375
|
+
];
|
|
376
|
+
const rep = compareDirectivesToRules(
|
|
377
|
+
directives,
|
|
378
|
+
parsed([rule("R-01", "Ian is the executor without a documentary grant.")]),
|
|
379
|
+
{ d1: "R-01" },
|
|
380
|
+
);
|
|
381
|
+
expect(rep.pass).toBe(false);
|
|
382
|
+
expect(rep.truncated_or_drifted[0]).toMatchObject({ id: "d1", ruleId: "R-01" });
|
|
383
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("never");
|
|
384
|
+
});
|
|
385
|
+
|
|
386
|
+
// MAJOR 2 (review of PR #4771): EXAMPLE_CONTEXT_RE included `like`/`including`,
|
|
387
|
+
// which reclassified a load-bearing fact sitting after those words as an
|
|
388
|
+
// illustrative sample and stopped demanding it. A rule that DROPS the caveat
|
|
389
|
+
// code then PASSED. With `like`/`including` removed from the example markers
|
|
390
|
+
// the code stays a guardrail and the drop MUST FAIL.
|
|
391
|
+
it("load-bearing case code after 'including' dropped by the rule ⇒ FAIL", () => {
|
|
392
|
+
const directives = [
|
|
393
|
+
directive("d1", "caveat", "Handle probate cases including S CAV 2026 00037 with the sealed-file protocol."),
|
|
394
|
+
];
|
|
395
|
+
const rep = compareDirectivesToRules(
|
|
396
|
+
directives,
|
|
397
|
+
parsed([rule("R-01", "Handle probate cases with the sealed-file protocol.")]),
|
|
398
|
+
{ d1: "R-01" },
|
|
399
|
+
);
|
|
400
|
+
expect(rep.pass).toBe(false);
|
|
401
|
+
expect(rep.truncated_or_drifted[0].missingKeywords).toContain("CAV 2026 00037");
|
|
402
|
+
});
|
|
403
|
+
});
|
|
404
|
+
|
|
405
|
+
describe("tier1 calibration — now PASSES real synonym-preserving condensations", () => {
|
|
406
|
+
it("negation synonym: directive 'don't' → rule 'NEVER' preserves the guardrail ⇒ PASS", () => {
|
|
407
|
+
const directives = [directive("d1", "confirm-external", "Don't send letters to solicitors without Ken's approval.")];
|
|
408
|
+
const rep = compareDirectivesToRules(
|
|
409
|
+
directives,
|
|
410
|
+
parsed([rule("R-01", "NEVER send letters to solicitors without approval; draft for review only.")]),
|
|
411
|
+
{ d1: "R-01" },
|
|
412
|
+
);
|
|
413
|
+
expect(rep.truncated_or_drifted).toEqual([]);
|
|
414
|
+
expect(rep.pass).toBe(true);
|
|
415
|
+
});
|
|
416
|
+
|
|
417
|
+
it("exclusivity synonym: directive 'only' → rule 'sole/alone' preserves scope ⇒ PASS", () => {
|
|
418
|
+
const directives = [directive("d1", "fiona-facts", "Include only documented facts visible on paper.")];
|
|
419
|
+
const rep = compareDirectivesToRules(
|
|
420
|
+
directives,
|
|
421
|
+
parsed([rule("R-01", "Include documented facts visible on paper exclusively.")]),
|
|
422
|
+
{ d1: "R-01" },
|
|
423
|
+
);
|
|
424
|
+
expect(rep.truncated_or_drifted).toEqual([]);
|
|
425
|
+
expect(rep.pass).toBe(true);
|
|
426
|
+
});
|
|
427
|
+
|
|
428
|
+
it("preserves the deferral line + named facts verbatim ⇒ PASS", () => {
|
|
429
|
+
const directives = [
|
|
430
|
+
directive("d1", "defer-to-fiona", DEFER_DIRECTIVE),
|
|
431
|
+
directive("d2", "gary-executor", "Executor is GARY DAVID BROWN; caveat S CAV 2026 00037 cancelled."),
|
|
432
|
+
];
|
|
433
|
+
const rep = compareDirectivesToRules(
|
|
434
|
+
directives,
|
|
435
|
+
parsed([
|
|
436
|
+
rule("R-01", `Never give legal advice. End any legal-analysis with exactly: "${DEFERRAL_LINE}"`),
|
|
437
|
+
rule("R-02", "Executor is GARY DAVID BROWN; the caveat S CAV 2026 00037 is cancelled — Ian holds no office."),
|
|
438
|
+
]),
|
|
439
|
+
{ d1: "R-01", d2: "R-02" },
|
|
440
|
+
);
|
|
441
|
+
expect(rep.truncated_or_drifted).toEqual([]);
|
|
442
|
+
expect(rep.pass).toBe(true);
|
|
443
|
+
});
|
|
444
|
+
|
|
445
|
+
it("drops an ILLUSTRATIVE example code (e.g. AG779131P) without flagging ⇒ PASS", () => {
|
|
446
|
+
const directives = [
|
|
447
|
+
directive("d1", "evidence", "Cite an instrument or ledger number (e.g. AG779131P, TR10399) inline with each claim."),
|
|
448
|
+
];
|
|
449
|
+
const rep = compareDirectivesToRules(
|
|
450
|
+
directives,
|
|
451
|
+
parsed([rule("R-01", "Cite an instrument or ledger number inline with every claim.")]),
|
|
452
|
+
{ d1: "R-01" },
|
|
453
|
+
);
|
|
454
|
+
expect(rep.truncated_or_drifted).toEqual([]);
|
|
455
|
+
expect(rep.pass).toBe(true);
|
|
456
|
+
});
|
|
457
|
+
|
|
458
|
+
it("drops an ILLUSTRATIVE sample toast (un-cued quote) without flagging ⇒ PASS", () => {
|
|
459
|
+
const directives = [
|
|
460
|
+
directive("d1", "toast", 'Add a per-button toast, e.g. ack_text "✓ Code fix — starting", for instant feedback.'),
|
|
461
|
+
];
|
|
462
|
+
const rep = compareDirectivesToRules(
|
|
463
|
+
directives,
|
|
464
|
+
parsed([rule("R-01", "Add a per-button ack_text toast for instant feedback on every callback button.")]),
|
|
465
|
+
{ d1: "R-01" },
|
|
466
|
+
);
|
|
467
|
+
expect(rep.truncated_or_drifted).toEqual([]);
|
|
468
|
+
expect(rep.pass).toBe(true);
|
|
469
|
+
});
|
|
470
|
+
});
|