rulereceipt 0.1.71 → 0.1.72

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -97,12 +97,20 @@ export function runIfEditThenTestChecks(classifications, events) {
97
97
  };
98
98
  }
99
99
  if (testPaths.length === 0) {
100
+ // Shadow, 2026-09-29: "every change needs a test" is a REQUIRE rule, and
101
+ // an edit with no test touched or run is not PROOF the rule was broken —
102
+ // the test may not have been needed, or may have run in another terminal.
103
+ // The product's guarantee is that "Broken" means a forbidden action
104
+ // actually happened, or a claim was contradicted by the session; this is
105
+ // neither, so it reports "can't tell", never a fabricated FAIL. Promote to
106
+ // Broken only after 30+ real cases are hand-checked at 0 wrong.
100
107
  return {
101
108
  ruleId: rule.id,
102
109
  ruleTitle: rule.title,
103
110
  ruleSource: rule.source,
104
- status: "FAIL",
105
- evidence: `edited ${prodPaths.slice(0, 3).join(", ")}${prodPaths.length > 3 ? ", ..." : ""} but no matching test file was touched`,
111
+ status: "UNCLEAR",
112
+ method: "edit_test_pairing",
113
+ evidence: `edited ${prodPaths.slice(0, 3).join(", ")}${prodPaths.length > 3 ? ", ..." : ""}, and no test file was touched or run this session — can't tell whether this change needed a test`,
106
114
  };
107
115
  }
108
116
  return {
@@ -261,7 +261,24 @@ export async function runJudgmentChecks(classifications, events) {
261
261
  : "the situation this rule governs never arose in this session",
262
262
  };
263
263
  }
264
- if (status === "PASS" || status === "FAIL" || status === "UNCLEAR") {
264
+ if (status === "FAIL") {
265
+ // An AI opinion is NEVER a verdict — the product's guarantee is that
266
+ // "Broken" comes only from a forbidden action that happened or a claim
267
+ // contradicted by the session, never a model's guess. A model "FAIL" is
268
+ // surfaced as a clearly-labelled opinion needing a human, so it can never
269
+ // be counted as Broken in the report, history, card, digest or badge.
270
+ const evidence = (parsed.evidence ?? "").trim();
271
+ return {
272
+ ruleId: rule.id,
273
+ ruleTitle: rule.title,
274
+ ruleSource: rule.source,
275
+ status: "UNCLEAR",
276
+ needsHuman: true,
277
+ method: "model_judgment",
278
+ evidence: `AI opinion (not a verdict — a human should decide): this looks broken. ${evidence}${transcript.truncated ? ` ${TRUNCATION_NOTE}` : ""}`.trim(),
279
+ };
280
+ }
281
+ if (status === "PASS" || status === "UNCLEAR") {
265
282
  const evidence = parsed.evidence ?? "";
266
283
  return {
267
284
  ruleId: rule.id,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rulereceipt",
3
- "version": "0.1.71",
3
+ "version": "0.1.72",
4
4
  "description": "Checks whether your AI coding agent followed your rules, with evidence. Works with Claude Code (Codex in testing); reads CLAUDE.md, AGENTS.md, Cursor, Copilot and Windsurf rules.",
5
5
  "repository": {
6
6
  "type": "git",