rulereceipt 0.1.32 → 0.1.34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/checks/claimEvidence.d.ts +12 -0
- package/dist/checks/claimEvidence.js +334 -0
- package/dist/checks/classify.d.ts +20 -1
- package/dist/checks/classify.js +158 -12
- package/dist/checks/codeContent.js +13 -9
- package/dist/checks/deterministicChecks.js +19 -3
- package/dist/checks/fileLifecycle.js +35 -9
- package/dist/checks/gitBranchPolicy.js +28 -9
- package/dist/checks/ifEditThenTest.js +37 -1
- package/dist/checks/judgmentChecks.js +70 -8
- package/dist/checks/projectPaths.d.ts +16 -0
- package/dist/checks/projectPaths.js +18 -0
- package/dist/checks/testCommands.d.ts +23 -0
- package/dist/checks/testCommands.js +32 -0
- package/dist/cli.js +3 -0
- package/dist/report/generateReport.js +60 -11
- package/dist/types.d.ts +71 -0
- package/dist/types.js +22 -1
- package/package.json +2 -2
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { violation } from "../types.js";
|
|
1
2
|
/**
|
|
2
3
|
* Second structured-check primitive: only scans the actual content of
|
|
3
4
|
* real file edits for a code-construct pattern (e.g. `print(`,
|
|
@@ -35,7 +36,7 @@ export function runCodeContentChecks(classifications, events) {
|
|
|
35
36
|
if (content)
|
|
36
37
|
editedContents.push(content);
|
|
37
38
|
}
|
|
38
|
-
return classifications.map(({ rule, patterns, polarity }) => {
|
|
39
|
+
return classifications.map(({ rule, patterns, polarity, polarityInferred }) => {
|
|
39
40
|
let foundPattern;
|
|
40
41
|
let foundContent;
|
|
41
42
|
for (const content of editedContents) {
|
|
@@ -51,19 +52,22 @@ export function runCodeContentChecks(classifications, events) {
|
|
|
51
52
|
}
|
|
52
53
|
if (polarity === "forbid") {
|
|
53
54
|
if (foundPattern && foundContent) {
|
|
54
|
-
return {
|
|
55
|
-
ruleId: rule.id,
|
|
56
|
-
ruleTitle: rule.title,
|
|
57
|
-
ruleSource: rule.source,
|
|
58
|
-
status: "FAIL",
|
|
59
|
-
evidence: `found "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`,
|
|
60
|
-
};
|
|
55
|
+
return violation(rule, polarity, `found "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`, { method: "code_content", polarityInferred });
|
|
61
56
|
}
|
|
57
|
+
// Trigger evaluated and absent: the rule never applied. Not
|
|
58
|
+
// "followed" — that word claims something the check cannot show.
|
|
62
59
|
return {
|
|
63
60
|
ruleId: rule.id,
|
|
64
61
|
ruleTitle: rule.title,
|
|
65
62
|
ruleSource: rule.source,
|
|
66
|
-
status:
|
|
63
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
64
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
65
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
66
|
+
// vocabulary exists to stop, one field further down.
|
|
67
|
+
status: "UNCLEAR",
|
|
68
|
+
outcome: "not_applicable",
|
|
69
|
+
method: "code_content",
|
|
70
|
+
ceiling: "a scan of content written through Write/Edit — it does not see content written by a shell command",
|
|
67
71
|
evidence: `no file edit actually contained ${patterns.map((p) => `"${p}"`).join(" or ")} this session`,
|
|
68
72
|
};
|
|
69
73
|
}
|
|
@@ -81,7 +81,12 @@ function matchesPattern(haystack, pattern) {
|
|
|
81
81
|
const lastChar = pattern[pattern.length - 1];
|
|
82
82
|
const needsTrailingBoundary = /[\w-]/.test(lastChar);
|
|
83
83
|
const suffix = needsTrailingBoundary ? "(?![\\w-])" : "";
|
|
84
|
-
|
|
84
|
+
// There was a trailing boundary and no leading one, so a short real
|
|
85
|
+
// pattern like `rm` matched inside "form", "storm" and "performance".
|
|
86
|
+
// Found 2026-09-12 running 559 rules files against 5 real sessions.
|
|
87
|
+
const firstChar = pattern[0];
|
|
88
|
+
const prefix = /[\w]/.test(firstChar) ? "(?<![\\w-])" : "";
|
|
89
|
+
const regex = new RegExp(prefix + escapeRegex(pattern) + suffix);
|
|
85
90
|
return regex.test(haystack);
|
|
86
91
|
}
|
|
87
92
|
/**
|
|
@@ -130,12 +135,23 @@ export function runDeterministicChecks(classifications, events) {
|
|
|
130
135
|
evidence: `"${foundPattern}" appears in a ${foundEvent.kind === "tool_use" ? foundEvent.toolName + " call" : foundEvent.kind}, but a text match alone can't tell an actual violation from a mention (a search for it, a quote, an explanation) — needs a human look: ${haystack.slice(0, 160)}`,
|
|
131
136
|
};
|
|
132
137
|
}
|
|
138
|
+
// For a prohibition the trigger IS the forbidden act. Evaluated and
|
|
139
|
+
// absent means the rule never applied — not that it was followed.
|
|
140
|
+
// Reporting it as followed is how an empty transcript produced 2,770
|
|
141
|
+
// green ticks across the 559-file corpus.
|
|
133
142
|
return {
|
|
134
143
|
ruleId: rule.id,
|
|
135
144
|
ruleTitle: rule.title,
|
|
136
145
|
ruleSource: rule.source,
|
|
137
|
-
status:
|
|
138
|
-
|
|
146
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
147
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
148
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
149
|
+
// vocabulary exists to stop, one field further down.
|
|
150
|
+
status: "UNCLEAR",
|
|
151
|
+
outcome: "not_applicable",
|
|
152
|
+
method: "text_scan",
|
|
153
|
+
ceiling: "a text scan of the recorded commands and messages — evidence, not proof the act did not happen, since a spelling this checker does not know would not be caught",
|
|
154
|
+
evidence: `no occurrence of ${patterns.map((p) => `"${p}"`).join(" or ")} in the commands and messages recorded this session`,
|
|
139
155
|
};
|
|
140
156
|
}
|
|
141
157
|
// polarity === "require": absence is the failure, not presence. But
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { violation } from "../types.js";
|
|
2
|
+
import { isProjectPath } from "./projectPaths.js";
|
|
1
3
|
/**
|
|
2
4
|
* Third structured-check primitive: only counts real MUTATIONS of a
|
|
3
5
|
* protected file, never reads of it. Real false-positive this fixes
|
|
@@ -36,6 +38,11 @@ function escapeRegex(literal) {
|
|
|
36
38
|
function pathPattern(filePath) {
|
|
37
39
|
return `(?:^|[\\s'"=/])${escapeRegex(filePath.replace(/^\.\//, ""))}(?=$|[\\s'";)])`;
|
|
38
40
|
}
|
|
41
|
+
/**
|
|
42
|
+
* The command moves into a throwaway tree before doing anything. Anything
|
|
43
|
+
* it mutates after that is a scratch file, not the project's.
|
|
44
|
+
*/
|
|
45
|
+
const CD_INTO_TEMP = /\bcd\s+["']?(?:\/private)?\/(?:tmp|var\/folders)\b|\bcd\s+["']?[^\s"'&|;]*\/(?:scratchpad|node_modules)\b/;
|
|
39
46
|
function mutatesPathInBash(command, filePath) {
|
|
40
47
|
const p = pathPattern(filePath);
|
|
41
48
|
const mutations = [
|
|
@@ -61,6 +68,12 @@ function findMutation(events, filePath) {
|
|
|
61
68
|
const input = event.input;
|
|
62
69
|
if (typeof input?.file_path === "string") {
|
|
63
70
|
const actual = input.file_path.replace(/^\.\//, "");
|
|
71
|
+
// A throwaway copy is not the project's file. A rule saying
|
|
72
|
+
// "CHANGELOG.md is release-only" fired on a scratchpad CHANGELOG.md
|
|
73
|
+
// written during a probe and deleted minutes later — the basename
|
|
74
|
+
// matched and nothing else was checked.
|
|
75
|
+
if (!isProjectPath(actual))
|
|
76
|
+
continue;
|
|
64
77
|
if (actual === normalized || actual.endsWith(`/${normalized}`)) {
|
|
65
78
|
return `${event.toolName} on ${input.file_path}`;
|
|
66
79
|
}
|
|
@@ -70,6 +83,16 @@ function findMutation(events, filePath) {
|
|
|
70
83
|
if (event.toolName === "Bash") {
|
|
71
84
|
const input = event.input;
|
|
72
85
|
if (typeof input?.command === "string" && mutatesPathInBash(input.command, filePath)) {
|
|
86
|
+
// A command whose working directory is a temp tree is operating on
|
|
87
|
+
// throwaway files, however the paths inside it are spelled.
|
|
88
|
+
//
|
|
89
|
+
// The first version required the temp prefix to sit next to the
|
|
90
|
+
// filename, which cannot work: the real command that exposed this
|
|
91
|
+
// does `cd /tmp` on its first line and writes `.claude/CLAUDE.md`
|
|
92
|
+
// three lines later. A shortened one-line fixture passed while the
|
|
93
|
+
// real command kept failing.
|
|
94
|
+
if (CD_INTO_TEMP.test(input.command))
|
|
95
|
+
continue;
|
|
73
96
|
return input.command;
|
|
74
97
|
}
|
|
75
98
|
}
|
|
@@ -77,23 +100,26 @@ function findMutation(events, filePath) {
|
|
|
77
100
|
return null;
|
|
78
101
|
}
|
|
79
102
|
export function runFileLifecycleChecks(classifications, events) {
|
|
80
|
-
return classifications.map(({ rule, filePath, polarity }) => {
|
|
103
|
+
return classifications.map(({ rule, filePath, polarity, polarityInferred }) => {
|
|
81
104
|
const mutation = findMutation(events, filePath);
|
|
82
105
|
if (polarity === "forbid") {
|
|
83
106
|
if (mutation) {
|
|
84
|
-
return {
|
|
85
|
-
ruleId: rule.id,
|
|
86
|
-
ruleTitle: rule.title,
|
|
87
|
-
ruleSource: rule.source,
|
|
88
|
-
status: "FAIL",
|
|
89
|
-
evidence: `"${filePath}" was actually modified: ${mutation.slice(0, 160)}`,
|
|
90
|
-
};
|
|
107
|
+
return violation(rule, polarity, `"${filePath}" was actually modified: ${mutation.slice(0, 160)}`, { method: "file_events", polarityInferred });
|
|
91
108
|
}
|
|
109
|
+
// Trigger evaluated and absent: the rule never applied. Not
|
|
110
|
+
// "followed" — that word claims something the check cannot show.
|
|
92
111
|
return {
|
|
93
112
|
ruleId: rule.id,
|
|
94
113
|
ruleTitle: rule.title,
|
|
95
114
|
ruleSource: rule.source,
|
|
96
|
-
status:
|
|
115
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
116
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
117
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
118
|
+
// vocabulary exists to stop, one field further down.
|
|
119
|
+
status: "UNCLEAR",
|
|
120
|
+
outcome: "not_applicable",
|
|
121
|
+
method: "file_events",
|
|
122
|
+
ceiling: "a check of file-mutation events — it shows no mutation of this path was recorded, not that the file is untouched on disk",
|
|
97
123
|
evidence: `"${filePath}" was never written to, deleted, or moved this session (reading it does not count)`,
|
|
98
124
|
};
|
|
99
125
|
}
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { violation } from "../types.js";
|
|
1
2
|
/**
|
|
2
3
|
* First real structured-check primitive: parses actual git command
|
|
3
4
|
* arguments instead of searching prose for a branch name as a substring.
|
|
@@ -86,26 +87,44 @@ export function runGitBranchPolicyChecks(classifications, events) {
|
|
|
86
87
|
allTargets.push({ branch, command });
|
|
87
88
|
}
|
|
88
89
|
}
|
|
89
|
-
|
|
90
|
+
const anyGitCommand = events.some((e) => e.kind === "tool_use" && /\bgit\s/.test(JSON.stringify(e.input ?? "")));
|
|
91
|
+
return classifications.map(({ rule, branchName, polarity, polarityInferred }) => {
|
|
92
|
+
// No git command ran, so a git rule never had a situation to govern.
|
|
93
|
+
// Calling that "followed" is how an empty session produced 2,770 green
|
|
94
|
+
// ticks across the 559-file corpus — every one true, none meaningful.
|
|
95
|
+
if (!anyGitCommand) {
|
|
96
|
+
return {
|
|
97
|
+
ruleId: rule.id,
|
|
98
|
+
ruleTitle: rule.title,
|
|
99
|
+
ruleSource: rule.source,
|
|
100
|
+
status: "UNCLEAR",
|
|
101
|
+
outcome: "not_applicable",
|
|
102
|
+
method: "git_events",
|
|
103
|
+
evidence: "no git command ran this session, so this rule never applied",
|
|
104
|
+
};
|
|
105
|
+
}
|
|
90
106
|
const pushOrCreateHit = allTargets.find((t) => t.branch === branchName &&
|
|
91
107
|
(GIT_PUSH.test(t.command) || GIT_BRANCH_CREATE.test(t.command) || GIT_CHECKOUT_CREATE.test(t.command)));
|
|
92
108
|
const commitViolationCommand = findCheckoutCommitViolation(commands, branchName);
|
|
93
109
|
const hit = pushOrCreateHit ?? (commitViolationCommand ? { branch: branchName, command: commitViolationCommand } : undefined);
|
|
94
110
|
if (polarity === "forbid") {
|
|
95
111
|
if (hit) {
|
|
96
|
-
return {
|
|
97
|
-
ruleId: rule.id,
|
|
98
|
-
ruleTitle: rule.title,
|
|
99
|
-
ruleSource: rule.source,
|
|
100
|
-
status: "FAIL",
|
|
101
|
-
evidence: `a git command actually targeted the "${branchName}" branch: ${hit.command}`,
|
|
102
|
-
};
|
|
112
|
+
return violation(rule, polarity, `a git command actually targeted the "${branchName}" branch: ${hit.command}`, { method: "git_events", polarityInferred });
|
|
103
113
|
}
|
|
114
|
+
// Trigger evaluated and absent: the rule never applied. Not
|
|
115
|
+
// "followed" — that word claims something the check cannot show.
|
|
104
116
|
return {
|
|
105
117
|
ruleId: rule.id,
|
|
106
118
|
ruleTitle: rule.title,
|
|
107
119
|
ruleSource: rule.source,
|
|
108
|
-
status:
|
|
120
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
121
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
122
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
123
|
+
// vocabulary exists to stop, one field further down.
|
|
124
|
+
status: "UNCLEAR",
|
|
125
|
+
outcome: "not_applicable",
|
|
126
|
+
method: "git_events",
|
|
127
|
+
ceiling: "a scan of recorded git commands — it does not see commands run outside this session",
|
|
109
128
|
evidence: `no git command targeted the "${branchName}" branch this session`,
|
|
110
129
|
};
|
|
111
130
|
}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { findTestRun } from "./testCommands.js";
|
|
2
|
+
import { isProjectPath } from "./projectPaths.js";
|
|
1
3
|
const TEST_FILE_PATTERN = /(\.test\.|\.spec\.|__tests__\/|_test\.|\/tests?\/)/i;
|
|
2
4
|
// Real false-positive found 2026-08-30 on an actual session: editing a
|
|
3
5
|
// markdown documentation file flagged "no test file touched" four
|
|
@@ -5,6 +7,18 @@ const TEST_FILE_PATTERN = /(\.test\.|\.spec\.|__tests__\/|_test\.|\/tests?\/)/i;
|
|
|
5
7
|
// companion under any reasonable reading of an "add tests for every
|
|
6
8
|
// change" rule. Excluded from prodPaths entirely, same as test files.
|
|
7
9
|
const NON_TESTABLE_FILE_PATTERN = /\.(md|mdx|txt|rst|json|ya?ml|toml|lock|csv|log)$/i;
|
|
10
|
+
/**
|
|
11
|
+
* Paths that are not this project's source, whatever their extension.
|
|
12
|
+
*
|
|
13
|
+
* Found 2026-09-12 running 559 real rules files against 5 real sessions: 24
|
|
14
|
+
* FAILs said "edited /private/tmp/.../scratchpad/probe.mjs but no matching
|
|
15
|
+
* test file was touched". That is a throwaway probe, written to inspect
|
|
16
|
+
* something and deleted minutes later. Demanding a test for it is nonsense;
|
|
17
|
+
* demanding one as a FAIL is a false accusation.
|
|
18
|
+
*
|
|
19
|
+
* Only file extensions were excluded before, so any temp file that happened
|
|
20
|
+
* to end in .ts or .mjs counted as production code.
|
|
21
|
+
*/
|
|
8
22
|
const WRITE_LIKE_TOOLS = new Set(["Write", "Edit", "NotebookEdit"]);
|
|
9
23
|
function extractEditedPaths(events) {
|
|
10
24
|
const paths = [];
|
|
@@ -40,8 +54,21 @@ function extractEditedPaths(events) {
|
|
|
40
54
|
*/
|
|
41
55
|
export function runIfEditThenTestChecks(classifications, events) {
|
|
42
56
|
const editedPaths = extractEditedPaths(events);
|
|
57
|
+
// Running the suite honours "add tests for every change" as much as
|
|
58
|
+
// touching a test file does. Without this, the most ordinary workflow
|
|
59
|
+
// there is — change code, run the tests, commit — produced a FAIL saying
|
|
60
|
+
// no test file was touched. Twelve rules across the 559-file corpus hit
|
|
61
|
+
// it on one synthetic session (2026-09-11).
|
|
62
|
+
//
|
|
63
|
+
// Any run in the session counts, not only one after the edit. Requiring
|
|
64
|
+
// the stricter ordering would buy a little precision and risk the
|
|
65
|
+
// expensive direction of error, and in this project a wrong FAIL costs
|
|
66
|
+
// more than a missed detection.
|
|
67
|
+
const testRun = findTestRun(events);
|
|
43
68
|
const testPaths = editedPaths.filter((p) => TEST_FILE_PATTERN.test(p));
|
|
44
|
-
const prodPaths = editedPaths.filter((p) => !TEST_FILE_PATTERN.test(p) &&
|
|
69
|
+
const prodPaths = editedPaths.filter((p) => !TEST_FILE_PATTERN.test(p) &&
|
|
70
|
+
!NON_TESTABLE_FILE_PATTERN.test(p) &&
|
|
71
|
+
isProjectPath(p));
|
|
45
72
|
return classifications.map(({ rule }) => {
|
|
46
73
|
if (prodPaths.length === 0) {
|
|
47
74
|
return {
|
|
@@ -54,6 +81,15 @@ export function runIfEditThenTestChecks(classifications, events) {
|
|
|
54
81
|
: "no code file was edited this session (only test/doc/config files, if any) — the rule never had a chance to apply",
|
|
55
82
|
};
|
|
56
83
|
}
|
|
84
|
+
if (testPaths.length === 0 && testRun !== null) {
|
|
85
|
+
return {
|
|
86
|
+
ruleId: rule.id,
|
|
87
|
+
ruleTitle: rule.title,
|
|
88
|
+
ruleSource: rule.source,
|
|
89
|
+
status: "PASS",
|
|
90
|
+
evidence: `edited ${prodPaths[0]} and ran the suite: \`${testRun}\` (no test file was edited, but the code was exercised)`,
|
|
91
|
+
};
|
|
92
|
+
}
|
|
57
93
|
if (testPaths.length === 0) {
|
|
58
94
|
return {
|
|
59
95
|
ruleId: rule.id,
|
|
@@ -31,6 +31,21 @@ function modelId() {
|
|
|
31
31
|
const MAX_TRANSCRIPT_CHARS = 120_000;
|
|
32
32
|
const HEAD_CHARS = 40_000;
|
|
33
33
|
const TAIL_CHARS = MAX_TRANSCRIPT_CHARS - HEAD_CHARS;
|
|
34
|
+
/**
|
|
35
|
+
* How much of a rule's own text goes into the prompt.
|
|
36
|
+
*
|
|
37
|
+
* The transcript was capped and the rule body was not. One rule in the
|
|
38
|
+
* 559-file corpus is 122,000 characters — a section heading whose body is
|
|
39
|
+
* an entire architecture document, parsed as a single rule — and it would
|
|
40
|
+
* have been sent whole on top of a 120,000-character transcript. Roughly
|
|
41
|
+
* 60k tokens for one verdict, with nothing bounding it.
|
|
42
|
+
*
|
|
43
|
+
* A rule that does not fit in 8,000 characters is not really one rule, and
|
|
44
|
+
* the model does not need the rest to judge it. The prompt says when it was
|
|
45
|
+
* cut, so a verdict is never formed from a fragment the model believes is
|
|
46
|
+
* whole.
|
|
47
|
+
*/
|
|
48
|
+
const MAX_RULE_CHARS = 8_000;
|
|
34
49
|
/** How many judgment calls may be in flight at once. */
|
|
35
50
|
const MAX_CONCURRENT_CALLS = 4;
|
|
36
51
|
const TRUNCATION_NOTE = "[judged on a truncated transcript — the middle of this session was not shown to the model]";
|
|
@@ -71,23 +86,46 @@ function summarizeEvents(events) {
|
|
|
71
86
|
*/
|
|
72
87
|
const RESULT_TOOL = {
|
|
73
88
|
name: "report_result",
|
|
74
|
-
description: "Report
|
|
89
|
+
description: "Report whether this one rule was followed, violated, unclear, or never applicable, with a verbatim line of evidence from the session.",
|
|
75
90
|
input_schema: {
|
|
76
91
|
type: "object",
|
|
77
92
|
properties: {
|
|
78
|
-
status: { type: "string", enum: ["PASS", "FAIL", "UNCLEAR"] },
|
|
93
|
+
status: { type: "string", enum: ["PASS", "FAIL", "UNCLEAR", "NOT_APPLICABLE"] },
|
|
79
94
|
evidence: {
|
|
80
95
|
type: "string",
|
|
81
|
-
description: "A short VERBATIM extract
|
|
96
|
+
description: "A short VERBATIM extract FROM THE SESSION TRANSCRIPT — copied exactly as it appears there, not reworded, not summarised, and not taken from the rule text. If nothing in the transcript supports a verdict, report UNCLEAR or NOT_APPLICABLE and leave this brief.",
|
|
82
97
|
},
|
|
83
98
|
},
|
|
84
99
|
required: ["status", "evidence"],
|
|
85
100
|
},
|
|
86
101
|
};
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
102
|
+
/**
|
|
103
|
+
* The four verdicts, and the balance between them.
|
|
104
|
+
*
|
|
105
|
+
* The previous version offered three and told the model only "never guess
|
|
106
|
+
* PASS when you are not sure". With no NOT_APPLICABLE, a rule that simply
|
|
107
|
+
* never came up had to be forced into one of pass, fail or unclear — and the
|
|
108
|
+
* single stated pressure pointed at the accusing one. Measured against the
|
|
109
|
+
* four-event example session on 2026-09-14: 30 rules, 10 FAILs, on a session
|
|
110
|
+
* containing one genuine issue. One failure cited the prompt itself as
|
|
111
|
+
* evidence.
|
|
112
|
+
*
|
|
113
|
+
* Most rules do not apply to most sessions. Saying so is the correction.
|
|
114
|
+
*/
|
|
115
|
+
const INSTRUCTIONS = "You judge whether one rule from a CLAUDE.md/AGENTS.md file was actually followed during a Claude Code session.\n\n" +
|
|
116
|
+
"MOST RULES WILL NOT APPLY. A session is usually a few minutes of work, and a rules file covers everything a project " +
|
|
117
|
+
"might ever do. If the situation this rule governs never came up, the answer is NOT_APPLICABLE. That is the common " +
|
|
118
|
+
"case and it is not a failure of any kind.\n\n" +
|
|
119
|
+
"PASS only when the transcript clearly shows the rule was followed.\n" +
|
|
120
|
+
"FAIL only when the transcript clearly shows it was violated.\n" +
|
|
121
|
+
"UNCLEAR when the situation arose but the transcript does not settle what happened.\n" +
|
|
122
|
+
"NOT_APPLICABLE when the situation the rule governs never arose.\n\n" +
|
|
123
|
+
"Do not guess in either direction. Guessing PASS invents compliance; guessing FAIL accuses someone of something they " +
|
|
124
|
+
"may not have done, which is the more expensive mistake and the harder one to recover from. If a rule is only loosely " +
|
|
125
|
+
"related to something in the session, that is NOT_APPLICABLE, not FAIL.\n\n" +
|
|
126
|
+
"Your evidence must be copied verbatim from the SESSION TRANSCRIPT. Never quote the rule back as evidence, and never " +
|
|
127
|
+
"quote these instructions. If you cannot find a line in the transcript that supports your verdict, you do not have a " +
|
|
128
|
+
"verdict.";
|
|
91
129
|
/**
|
|
92
130
|
* A rule the check never actually ran against.
|
|
93
131
|
*
|
|
@@ -99,6 +137,13 @@ const INSTRUCTIONS = "You judge whether one rule from a CLAUDE.md/AGENTS.md file
|
|
|
99
137
|
* about 13 rules it had never examined. Found 2026-09-08 by running the
|
|
100
138
|
* published package; nothing in 167 lines of tests here asserted the field.
|
|
101
139
|
*/
|
|
140
|
+
function ruleBody(text) {
|
|
141
|
+
if (text.length <= MAX_RULE_CHARS)
|
|
142
|
+
return text;
|
|
143
|
+
return (text.slice(0, MAX_RULE_CHARS) +
|
|
144
|
+
`\n\n...[rule text truncated at ${MAX_RULE_CHARS} characters — this rule's body is ` +
|
|
145
|
+
`${text.length} characters long and is probably a whole document parsed as one rule]`);
|
|
146
|
+
}
|
|
102
147
|
function didNotRun(rule, reason) {
|
|
103
148
|
return {
|
|
104
149
|
ruleId: rule.id,
|
|
@@ -106,6 +151,10 @@ function didNotRun(rule, reason) {
|
|
|
106
151
|
ruleSource: rule.source,
|
|
107
152
|
status: "UNCLEAR",
|
|
108
153
|
needsHuman: true,
|
|
154
|
+
// The check did not happen. Rendering that as a judgment call, or worse
|
|
155
|
+
// as "couldn't tell", was the original defect in this file.
|
|
156
|
+
outcome: "not_run",
|
|
157
|
+
method: "none",
|
|
109
158
|
evidence: reason,
|
|
110
159
|
};
|
|
111
160
|
}
|
|
@@ -177,7 +226,7 @@ export async function runJudgmentChecks(classifications, events) {
|
|
|
177
226
|
text: `SESSION TRANSCRIPT:\n${transcript.text}`,
|
|
178
227
|
cache_control: { type: "ephemeral" },
|
|
179
228
|
},
|
|
180
|
-
{ type: "text", text: `RULE — ${rule.title}\n${rule.text}` },
|
|
229
|
+
{ type: "text", text: `RULE — ${rule.title}\n${ruleBody(rule.text)}` },
|
|
181
230
|
],
|
|
182
231
|
},
|
|
183
232
|
],
|
|
@@ -193,6 +242,19 @@ export async function runJudgmentChecks(classifications, events) {
|
|
|
193
242
|
}
|
|
194
243
|
const parsed = toolUseBlock.input;
|
|
195
244
|
const status = parsed.status;
|
|
245
|
+
if (status === "NOT_APPLICABLE") {
|
|
246
|
+
return {
|
|
247
|
+
ruleId: rule.id,
|
|
248
|
+
ruleTitle: rule.title,
|
|
249
|
+
ruleSource: rule.source,
|
|
250
|
+
status: "UNCLEAR",
|
|
251
|
+
outcome: "not_applicable",
|
|
252
|
+
method: "model_judgment",
|
|
253
|
+
evidence: parsed.evidence?.trim()
|
|
254
|
+
? parsed.evidence
|
|
255
|
+
: "the situation this rule governs never arose in this session",
|
|
256
|
+
};
|
|
257
|
+
}
|
|
196
258
|
if (status === "PASS" || status === "FAIL" || status === "UNCLEAR") {
|
|
197
259
|
const evidence = parsed.evidence ?? "";
|
|
198
260
|
return {
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Paths that are not the project's own files, whatever their name.
|
|
3
|
+
*
|
|
4
|
+
* One definition, shared. Found the hard way twice: ifEditThenTest demanded
|
|
5
|
+
* a test for /private/tmp/.../scratchpad/probe.mjs, and fileLifecycle fired
|
|
6
|
+
* a "CHANGELOG.md is release-only" rule on a scratchpad copy of that file
|
|
7
|
+
* written during a probe and deleted minutes later. Two checkers needed the
|
|
8
|
+
* same exclusion and only one had it, which is exactly how the two copies
|
|
9
|
+
* of TEST_COMMAND would have drifted.
|
|
10
|
+
*
|
|
11
|
+
* Both callers use this to decide whether to accuse someone, so the cost of
|
|
12
|
+
* a miss here is a false FAIL on a throwaway file.
|
|
13
|
+
*/
|
|
14
|
+
export declare const NON_PROJECT_PATH: RegExp;
|
|
15
|
+
/** True when this path is somewhere the project's own rules should govern. */
|
|
16
|
+
export declare function isProjectPath(path: string): boolean;
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Paths that are not the project's own files, whatever their name.
|
|
3
|
+
*
|
|
4
|
+
* One definition, shared. Found the hard way twice: ifEditThenTest demanded
|
|
5
|
+
* a test for /private/tmp/.../scratchpad/probe.mjs, and fileLifecycle fired
|
|
6
|
+
* a "CHANGELOG.md is release-only" rule on a scratchpad copy of that file
|
|
7
|
+
* written during a probe and deleted minutes later. Two checkers needed the
|
|
8
|
+
* same exclusion and only one had it, which is exactly how the two copies
|
|
9
|
+
* of TEST_COMMAND would have drifted.
|
|
10
|
+
*
|
|
11
|
+
* Both callers use this to decide whether to accuse someone, so the cost of
|
|
12
|
+
* a miss here is a false FAIL on a throwaway file.
|
|
13
|
+
*/
|
|
14
|
+
export const NON_PROJECT_PATH = /(?:^|\/)(?:tmp|temp|scratch|scratchpad|node_modules|dist|build|out|coverage|\.git|\.next|\.cache|vendor|__pycache__)(?:\/|$)|^\/(?:private\/)?(?:tmp|var)\//i;
|
|
15
|
+
/** True when this path is somewhere the project's own rules should govern. */
|
|
16
|
+
export function isProjectPath(path) {
|
|
17
|
+
return !NON_PROJECT_PATH.test(path);
|
|
18
|
+
}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import type { TranscriptEvent } from "../types.js";
|
|
2
|
+
/**
|
|
3
|
+
* Commands that run a project's test suite.
|
|
4
|
+
*
|
|
5
|
+
* One definition, imported by every checker that needs it. Two checkers
|
|
6
|
+
* now ask "did the tests run" — claimEvidence, to see whether a claim of a
|
|
7
|
+
* passing suite had anything behind it, and ifEditThenTest, to see whether
|
|
8
|
+
* changed code was exercised. Two copies of this list would drift, and the
|
|
9
|
+
* drift would show up as one checker contradicting the other in the same
|
|
10
|
+
* report.
|
|
11
|
+
*
|
|
12
|
+
* It can never be complete — projects wire their suite to whatever script
|
|
13
|
+
* name they like — so callers must never let a miss become an accusation.
|
|
14
|
+
* See UNKNOWN_SCRIPT_RUNNER in claimEvidence.ts.
|
|
15
|
+
*
|
|
16
|
+
* Deliberately a known list rather than anything test-shaped. Both callers
|
|
17
|
+
* use it to decide whether to report a FAIL, and a rule that fires because
|
|
18
|
+
* someone ran a script with "test" in its name is the expensive kind of
|
|
19
|
+
* wrong.
|
|
20
|
+
*/
|
|
21
|
+
export declare const TEST_COMMAND: RegExp;
|
|
22
|
+
/** The first test command run in this session, or null if none ran. */
|
|
23
|
+
export declare function findTestRun(events: TranscriptEvent[]): string | null;
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Commands that run a project's test suite.
|
|
3
|
+
*
|
|
4
|
+
* One definition, imported by every checker that needs it. Two checkers
|
|
5
|
+
* now ask "did the tests run" — claimEvidence, to see whether a claim of a
|
|
6
|
+
* passing suite had anything behind it, and ifEditThenTest, to see whether
|
|
7
|
+
* changed code was exercised. Two copies of this list would drift, and the
|
|
8
|
+
* drift would show up as one checker contradicting the other in the same
|
|
9
|
+
* report.
|
|
10
|
+
*
|
|
11
|
+
* It can never be complete — projects wire their suite to whatever script
|
|
12
|
+
* name they like — so callers must never let a miss become an accusation.
|
|
13
|
+
* See UNKNOWN_SCRIPT_RUNNER in claimEvidence.ts.
|
|
14
|
+
*
|
|
15
|
+
* Deliberately a known list rather than anything test-shaped. Both callers
|
|
16
|
+
* use it to decide whether to report a FAIL, and a rule that fires because
|
|
17
|
+
* someone ran a script with "test" in its name is the expensive kind of
|
|
18
|
+
* wrong.
|
|
19
|
+
*/
|
|
20
|
+
export const TEST_COMMAND = /\b(?:npm|pnpm|yarn|bun)\s+(?:run\s+)?(?:test|verify|check|ci)\b|\bnpx\s+(?:vitest|jest|mocha|ava)\b|\b(?:vitest|jest|mocha|pytest|phpunit|rspec|tox)\b|\bcargo\s+test\b|\bgo\s+test\b|\bmvn\s+(?:test|verify)\b|\bgradle\s+test\b|\bdotnet\s+test\b|\bpython\s+-m\s+(?:pytest|unittest)\b/i;
|
|
21
|
+
/** The first test command run in this session, or null if none ran. */
|
|
22
|
+
export function findTestRun(events) {
|
|
23
|
+
for (const event of events) {
|
|
24
|
+
if (event.kind !== "tool_use")
|
|
25
|
+
continue;
|
|
26
|
+
const input = event.input;
|
|
27
|
+
const command = input && typeof input.command === "string" ? input.command : "";
|
|
28
|
+
if (command && TEST_COMMAND.test(command))
|
|
29
|
+
return command;
|
|
30
|
+
}
|
|
31
|
+
return null;
|
|
32
|
+
}
|
package/dist/cli.js
CHANGED
|
@@ -15,6 +15,7 @@ import { runIfEditThenTestChecks } from "./checks/ifEditThenTest.js";
|
|
|
15
15
|
import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
|
|
16
16
|
import { runCodeContentChecks } from "./checks/codeContent.js";
|
|
17
17
|
import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
|
|
18
|
+
import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
|
|
18
19
|
import { runJudgmentChecks } from "./checks/judgmentChecks.js";
|
|
19
20
|
import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
|
|
20
21
|
import { generateHtmlReport } from "./report/generateHtmlReport.js";
|
|
@@ -212,6 +213,7 @@ async function runCheck(opts) {
|
|
|
212
213
|
const gitBranchPolicy = classifications.filter((c) => c.kind === "gitBranchPolicy");
|
|
213
214
|
const codeContent = classifications.filter((c) => c.kind === "codeContent");
|
|
214
215
|
const fileLifecycle = classifications.filter((c) => c.kind === "fileLifecycle");
|
|
216
|
+
const claimEvidence = classifications.filter((c) => c.kind === "claimEvidence");
|
|
215
217
|
const judgment = classifications.filter((c) => c.kind === "judgment");
|
|
216
218
|
// Not rules at all — documentation, glossary entries, reference tables,
|
|
217
219
|
// URLs, directory listings, code examples.
|
|
@@ -236,6 +238,7 @@ async function runCheck(opts) {
|
|
|
236
238
|
...runGitBranchPolicyChecks(gitBranchPolicy, events),
|
|
237
239
|
...runCodeContentChecks(codeContent, events),
|
|
238
240
|
...runFileLifecycleChecks(fileLifecycle, events),
|
|
241
|
+
...runClaimEvidenceChecks(claimEvidence, events),
|
|
239
242
|
];
|
|
240
243
|
// Deterministic checks run by default, always, with no key — judgment
|
|
241
244
|
// rules only call out to an LLM with an explicit --llm on THIS run, never
|
|
@@ -60,18 +60,46 @@ function ruleLabel(r, results) {
|
|
|
60
60
|
* as a division of labour.
|
|
61
61
|
*/
|
|
62
62
|
function summaryLine(results) {
|
|
63
|
-
const
|
|
64
|
-
const
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
63
|
+
const n = (b) => results.filter((r) => bucketOf(r) === b).length;
|
|
64
|
+
const parts = [`${n("PASS")} followed`, `${n("FAIL")} not followed`];
|
|
65
|
+
if (n("UNCLEAR_EVIDENCE") > 0)
|
|
66
|
+
parts.push(`${n("UNCLEAR_EVIDENCE")} couldn't tell`);
|
|
67
|
+
if (n("NOT_RUN") > 0)
|
|
68
|
+
parts.push(`${n("NOT_RUN")} not run`);
|
|
69
|
+
if (n("NOT_APPLICABLE") > 0)
|
|
70
|
+
parts.push(`${n("NOT_APPLICABLE")} didn't apply`);
|
|
71
|
+
if (n("UNCLEAR_JUDGMENT") > 0)
|
|
72
|
+
parts.push(`${n("UNCLEAR_JUDGMENT")} need your judgment`);
|
|
72
73
|
return parts.join(" · ");
|
|
73
74
|
}
|
|
75
|
+
/**
|
|
76
|
+
* Six buckets, because "the tool looked and could not decide", "the tool
|
|
77
|
+
* never ran", and "the situation never arose" are three different things
|
|
78
|
+
* that were all rendering as one.
|
|
79
|
+
*
|
|
80
|
+
* From anthropics/claude-code#90542. With no API key the report printed
|
|
81
|
+
* "13 couldn't tell" — a phrase defined in this file as the tool having
|
|
82
|
+
* looked — about thirteen rules it had never examined. And an empty
|
|
83
|
+
* transcript produced 2,770 green ticks across the 559-file corpus, every
|
|
84
|
+
* one of them true and none of them meaning anything, because a rule whose
|
|
85
|
+
* situation never arose was being counted as followed.
|
|
86
|
+
*
|
|
87
|
+
* `outcome` is preferred where a checker sets it; `status` remains the
|
|
88
|
+
* fallback while the rest are migrated.
|
|
89
|
+
*/
|
|
74
90
|
function bucketOf(result) {
|
|
91
|
+
switch (result.outcome) {
|
|
92
|
+
case "fail":
|
|
93
|
+
return "FAIL";
|
|
94
|
+
case "pass":
|
|
95
|
+
return "PASS";
|
|
96
|
+
case "not_run":
|
|
97
|
+
return "NOT_RUN";
|
|
98
|
+
case "not_applicable":
|
|
99
|
+
return "NOT_APPLICABLE";
|
|
100
|
+
case "inconclusive":
|
|
101
|
+
return result.needsHuman ? "UNCLEAR_JUDGMENT" : "UNCLEAR_EVIDENCE";
|
|
102
|
+
}
|
|
75
103
|
if (result.status === "FAIL")
|
|
76
104
|
return "FAIL";
|
|
77
105
|
if (result.status === "PASS")
|
|
@@ -90,10 +118,25 @@ function bucketOf(result) {
|
|
|
90
118
|
const BUCKET_LABEL = {
|
|
91
119
|
FAIL: "Not followed",
|
|
92
120
|
UNCLEAR_EVIDENCE: "Couldn't tell",
|
|
93
|
-
|
|
121
|
+
NOT_RUN: "Not run",
|
|
94
122
|
PASS: "Followed",
|
|
123
|
+
NOT_APPLICABLE: "Didn't apply this session",
|
|
124
|
+
UNCLEAR_JUDGMENT: "Needs your judgment",
|
|
95
125
|
};
|
|
96
|
-
|
|
126
|
+
/**
|
|
127
|
+
* Failures, then the two kinds of gap, then what held, then what never came
|
|
128
|
+
* up, then the human's half. "Not run" sits near the top on purpose: a
|
|
129
|
+
* check that did not happen is closer to a gap than to a result, and
|
|
130
|
+
* burying it is how it got mistaken for one.
|
|
131
|
+
*/
|
|
132
|
+
const BUCKET_ORDER = [
|
|
133
|
+
"FAIL",
|
|
134
|
+
"UNCLEAR_EVIDENCE",
|
|
135
|
+
"NOT_RUN",
|
|
136
|
+
"PASS",
|
|
137
|
+
"NOT_APPLICABLE",
|
|
138
|
+
"UNCLEAR_JUDGMENT",
|
|
139
|
+
];
|
|
97
140
|
/**
|
|
98
141
|
* The explanation shared by every rule in a section, or null when they
|
|
99
142
|
* differ.
|
|
@@ -142,6 +185,12 @@ export function generateReport(results, meta) {
|
|
|
142
185
|
lines.push(`${MARK[r.status]} ${r.status.padEnd(7)} ${ruleLabel(r, clean)}`);
|
|
143
186
|
if (r.evidence)
|
|
144
187
|
lines.push(` evidence: ${r.evidence}`);
|
|
188
|
+
// What this method was ALLOWED to conclude, travelling with the
|
|
189
|
+
// verdict. A text scan may say it saw no occurrence of a spelling; it
|
|
190
|
+
// may not say the act did not happen. That distinction shipped for six
|
|
191
|
+
// versions as a PASS on a session that ran `git push -f`.
|
|
192
|
+
if (r.ceiling)
|
|
193
|
+
lines.push(` this means: ${r.ceiling}`);
|
|
145
194
|
}
|
|
146
195
|
}
|
|
147
196
|
lines.push("");
|