rulereceipt 0.1.71 → 0.1.72
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -97,12 +97,20 @@ export function runIfEditThenTestChecks(classifications, events) {
|
|
|
97
97
|
};
|
|
98
98
|
}
|
|
99
99
|
if (testPaths.length === 0) {
|
|
100
|
+
// Shadow, 2026-09-29: "every change needs a test" is a REQUIRE rule, and
|
|
101
|
+
// an edit with no test touched or run is not PROOF the rule was broken —
|
|
102
|
+
// the test may not have been needed, or may have run in another terminal.
|
|
103
|
+
// The product's guarantee is that "Broken" means a forbidden action
|
|
104
|
+
// actually happened, or a claim was contradicted by the session; this is
|
|
105
|
+
// neither, so it reports "can't tell", never a fabricated FAIL. Promote to
|
|
106
|
+
// Broken only after 30+ real cases are hand-checked at 0 wrong.
|
|
100
107
|
return {
|
|
101
108
|
ruleId: rule.id,
|
|
102
109
|
ruleTitle: rule.title,
|
|
103
110
|
ruleSource: rule.source,
|
|
104
|
-
status: "
|
|
105
|
-
|
|
111
|
+
status: "UNCLEAR",
|
|
112
|
+
method: "edit_test_pairing",
|
|
113
|
+
evidence: `edited ${prodPaths.slice(0, 3).join(", ")}${prodPaths.length > 3 ? ", ..." : ""}, and no test file was touched or run this session — can't tell whether this change needed a test`,
|
|
106
114
|
};
|
|
107
115
|
}
|
|
108
116
|
return {
|
|
@@ -261,7 +261,24 @@ export async function runJudgmentChecks(classifications, events) {
|
|
|
261
261
|
: "the situation this rule governs never arose in this session",
|
|
262
262
|
};
|
|
263
263
|
}
|
|
264
|
-
if (status === "
|
|
264
|
+
if (status === "FAIL") {
|
|
265
|
+
// An AI opinion is NEVER a verdict — the product's guarantee is that
|
|
266
|
+
// "Broken" comes only from a forbidden action that happened or a claim
|
|
267
|
+
// contradicted by the session, never a model's guess. A model "FAIL" is
|
|
268
|
+
// surfaced as a clearly-labelled opinion needing a human, so it can never
|
|
269
|
+
// be counted as Broken in the report, history, card, digest or badge.
|
|
270
|
+
const evidence = (parsed.evidence ?? "").trim();
|
|
271
|
+
return {
|
|
272
|
+
ruleId: rule.id,
|
|
273
|
+
ruleTitle: rule.title,
|
|
274
|
+
ruleSource: rule.source,
|
|
275
|
+
status: "UNCLEAR",
|
|
276
|
+
needsHuman: true,
|
|
277
|
+
method: "model_judgment",
|
|
278
|
+
evidence: `AI opinion (not a verdict — a human should decide): this looks broken. ${evidence}${transcript.truncated ? ` ${TRUNCATION_NOTE}` : ""}`.trim(),
|
|
279
|
+
};
|
|
280
|
+
}
|
|
281
|
+
if (status === "PASS" || status === "UNCLEAR") {
|
|
265
282
|
const evidence = parsed.evidence ?? "";
|
|
266
283
|
return {
|
|
267
284
|
ruleId: rule.id,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "rulereceipt",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.72",
|
|
4
4
|
"description": "Checks whether your AI coding agent followed your rules, with evidence. Works with Claude Code (Codex in testing); reads CLAUDE.md, AGENTS.md, Cursor, Copilot and Windsurf rules.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|