rulereceipt 0.1.32 → 0.1.33
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/checks/claimEvidence.d.ts +12 -0
- package/dist/checks/claimEvidence.js +334 -0
- package/dist/checks/classify.d.ts +20 -1
- package/dist/checks/classify.js +158 -12
- package/dist/checks/codeContent.js +13 -9
- package/dist/checks/deterministicChecks.js +19 -3
- package/dist/checks/fileLifecycle.js +35 -9
- package/dist/checks/gitBranchPolicy.js +28 -9
- package/dist/checks/ifEditThenTest.js +37 -1
- package/dist/checks/judgmentChecks.js +27 -1
- package/dist/checks/projectPaths.d.ts +16 -0
- package/dist/checks/projectPaths.js +18 -0
- package/dist/checks/testCommands.d.ts +23 -0
- package/dist/checks/testCommands.js +32 -0
- package/dist/cli.js +3 -0
- package/dist/report/generateReport.js +60 -11
- package/dist/types.d.ts +71 -0
- package/dist/types.js +22 -1
- package/package.json +2 -2
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { violation } from "../types.js";
|
|
1
2
|
/**
|
|
2
3
|
* Second structured-check primitive: only scans the actual content of
|
|
3
4
|
* real file edits for a code-construct pattern (e.g. `print(`,
|
|
@@ -35,7 +36,7 @@ export function runCodeContentChecks(classifications, events) {
|
|
|
35
36
|
if (content)
|
|
36
37
|
editedContents.push(content);
|
|
37
38
|
}
|
|
38
|
-
return classifications.map(({ rule, patterns, polarity }) => {
|
|
39
|
+
return classifications.map(({ rule, patterns, polarity, polarityInferred }) => {
|
|
39
40
|
let foundPattern;
|
|
40
41
|
let foundContent;
|
|
41
42
|
for (const content of editedContents) {
|
|
@@ -51,19 +52,22 @@ export function runCodeContentChecks(classifications, events) {
|
|
|
51
52
|
}
|
|
52
53
|
if (polarity === "forbid") {
|
|
53
54
|
if (foundPattern && foundContent) {
|
|
54
|
-
return {
|
|
55
|
-
ruleId: rule.id,
|
|
56
|
-
ruleTitle: rule.title,
|
|
57
|
-
ruleSource: rule.source,
|
|
58
|
-
status: "FAIL",
|
|
59
|
-
evidence: `found "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`,
|
|
60
|
-
};
|
|
55
|
+
return violation(rule, polarity, `found "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`, { method: "code_content", polarityInferred });
|
|
61
56
|
}
|
|
57
|
+
// Trigger evaluated and absent: the rule never applied. Not
|
|
58
|
+
// "followed" — that word claims something the check cannot show.
|
|
62
59
|
return {
|
|
63
60
|
ruleId: rule.id,
|
|
64
61
|
ruleTitle: rule.title,
|
|
65
62
|
ruleSource: rule.source,
|
|
66
|
-
status:
|
|
63
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
64
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
65
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
66
|
+
// vocabulary exists to stop, one field further down.
|
|
67
|
+
status: "UNCLEAR",
|
|
68
|
+
outcome: "not_applicable",
|
|
69
|
+
method: "code_content",
|
|
70
|
+
ceiling: "a scan of content written through Write/Edit — it does not see content written by a shell command",
|
|
67
71
|
evidence: `no file edit actually contained ${patterns.map((p) => `"${p}"`).join(" or ")} this session`,
|
|
68
72
|
};
|
|
69
73
|
}
|
|
@@ -81,7 +81,12 @@ function matchesPattern(haystack, pattern) {
|
|
|
81
81
|
const lastChar = pattern[pattern.length - 1];
|
|
82
82
|
const needsTrailingBoundary = /[\w-]/.test(lastChar);
|
|
83
83
|
const suffix = needsTrailingBoundary ? "(?![\\w-])" : "";
|
|
84
|
-
|
|
84
|
+
// There was a trailing boundary and no leading one, so a short real
|
|
85
|
+
// pattern like `rm` matched inside "form", "storm" and "performance".
|
|
86
|
+
// Found 2026-09-12 running 559 rules files against 5 real sessions.
|
|
87
|
+
const firstChar = pattern[0];
|
|
88
|
+
const prefix = /[\w]/.test(firstChar) ? "(?<![\\w-])" : "";
|
|
89
|
+
const regex = new RegExp(prefix + escapeRegex(pattern) + suffix);
|
|
85
90
|
return regex.test(haystack);
|
|
86
91
|
}
|
|
87
92
|
/**
|
|
@@ -130,12 +135,23 @@ export function runDeterministicChecks(classifications, events) {
|
|
|
130
135
|
evidence: `"${foundPattern}" appears in a ${foundEvent.kind === "tool_use" ? foundEvent.toolName + " call" : foundEvent.kind}, but a text match alone can't tell an actual violation from a mention (a search for it, a quote, an explanation) — needs a human look: ${haystack.slice(0, 160)}`,
|
|
131
136
|
};
|
|
132
137
|
}
|
|
138
|
+
// For a prohibition the trigger IS the forbidden act. Evaluated and
|
|
139
|
+
// absent means the rule never applied — not that it was followed.
|
|
140
|
+
// Reporting it as followed is how an empty transcript produced 2,770
|
|
141
|
+
// green ticks across the 559-file corpus.
|
|
133
142
|
return {
|
|
134
143
|
ruleId: rule.id,
|
|
135
144
|
ruleTitle: rule.title,
|
|
136
145
|
ruleSource: rule.source,
|
|
137
|
-
status:
|
|
138
|
-
|
|
146
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
147
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
148
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
149
|
+
// vocabulary exists to stop, one field further down.
|
|
150
|
+
status: "UNCLEAR",
|
|
151
|
+
outcome: "not_applicable",
|
|
152
|
+
method: "text_scan",
|
|
153
|
+
ceiling: "a text scan of the recorded commands and messages — evidence, not proof the act did not happen, since a spelling this checker does not know would not be caught",
|
|
154
|
+
evidence: `no occurrence of ${patterns.map((p) => `"${p}"`).join(" or ")} in the commands and messages recorded this session`,
|
|
139
155
|
};
|
|
140
156
|
}
|
|
141
157
|
// polarity === "require": absence is the failure, not presence. But
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { violation } from "../types.js";
|
|
2
|
+
import { isProjectPath } from "./projectPaths.js";
|
|
1
3
|
/**
|
|
2
4
|
* Third structured-check primitive: only counts real MUTATIONS of a
|
|
3
5
|
* protected file, never reads of it. Real false-positive this fixes
|
|
@@ -36,6 +38,11 @@ function escapeRegex(literal) {
|
|
|
36
38
|
function pathPattern(filePath) {
|
|
37
39
|
return `(?:^|[\\s'"=/])${escapeRegex(filePath.replace(/^\.\//, ""))}(?=$|[\\s'";)])`;
|
|
38
40
|
}
|
|
41
|
+
/**
|
|
42
|
+
* The command moves into a throwaway tree before doing anything. Anything
|
|
43
|
+
* it mutates after that is a scratch file, not the project's.
|
|
44
|
+
*/
|
|
45
|
+
const CD_INTO_TEMP = /\bcd\s+["']?(?:\/private)?\/(?:tmp|var\/folders)\b|\bcd\s+["']?[^\s"'&|;]*\/(?:scratchpad|node_modules)\b/;
|
|
39
46
|
function mutatesPathInBash(command, filePath) {
|
|
40
47
|
const p = pathPattern(filePath);
|
|
41
48
|
const mutations = [
|
|
@@ -61,6 +68,12 @@ function findMutation(events, filePath) {
|
|
|
61
68
|
const input = event.input;
|
|
62
69
|
if (typeof input?.file_path === "string") {
|
|
63
70
|
const actual = input.file_path.replace(/^\.\//, "");
|
|
71
|
+
// A throwaway copy is not the project's file. A rule saying
|
|
72
|
+
// "CHANGELOG.md is release-only" fired on a scratchpad CHANGELOG.md
|
|
73
|
+
// written during a probe and deleted minutes later — the basename
|
|
74
|
+
// matched and nothing else was checked.
|
|
75
|
+
if (!isProjectPath(actual))
|
|
76
|
+
continue;
|
|
64
77
|
if (actual === normalized || actual.endsWith(`/${normalized}`)) {
|
|
65
78
|
return `${event.toolName} on ${input.file_path}`;
|
|
66
79
|
}
|
|
@@ -70,6 +83,16 @@ function findMutation(events, filePath) {
|
|
|
70
83
|
if (event.toolName === "Bash") {
|
|
71
84
|
const input = event.input;
|
|
72
85
|
if (typeof input?.command === "string" && mutatesPathInBash(input.command, filePath)) {
|
|
86
|
+
// A command whose working directory is a temp tree is operating on
|
|
87
|
+
// throwaway files, however the paths inside it are spelled.
|
|
88
|
+
//
|
|
89
|
+
// The first version required the temp prefix to sit next to the
|
|
90
|
+
// filename, which cannot work: the real command that exposed this
|
|
91
|
+
// does `cd /tmp` on its first line and writes `.claude/CLAUDE.md`
|
|
92
|
+
// three lines later. A shortened one-line fixture passed while the
|
|
93
|
+
// real command kept failing.
|
|
94
|
+
if (CD_INTO_TEMP.test(input.command))
|
|
95
|
+
continue;
|
|
73
96
|
return input.command;
|
|
74
97
|
}
|
|
75
98
|
}
|
|
@@ -77,23 +100,26 @@ function findMutation(events, filePath) {
|
|
|
77
100
|
return null;
|
|
78
101
|
}
|
|
79
102
|
export function runFileLifecycleChecks(classifications, events) {
|
|
80
|
-
return classifications.map(({ rule, filePath, polarity }) => {
|
|
103
|
+
return classifications.map(({ rule, filePath, polarity, polarityInferred }) => {
|
|
81
104
|
const mutation = findMutation(events, filePath);
|
|
82
105
|
if (polarity === "forbid") {
|
|
83
106
|
if (mutation) {
|
|
84
|
-
return {
|
|
85
|
-
ruleId: rule.id,
|
|
86
|
-
ruleTitle: rule.title,
|
|
87
|
-
ruleSource: rule.source,
|
|
88
|
-
status: "FAIL",
|
|
89
|
-
evidence: `"${filePath}" was actually modified: ${mutation.slice(0, 160)}`,
|
|
90
|
-
};
|
|
107
|
+
return violation(rule, polarity, `"${filePath}" was actually modified: ${mutation.slice(0, 160)}`, { method: "file_events", polarityInferred });
|
|
91
108
|
}
|
|
109
|
+
// Trigger evaluated and absent: the rule never applied. Not
|
|
110
|
+
// "followed" — that word claims something the check cannot show.
|
|
92
111
|
return {
|
|
93
112
|
ruleId: rule.id,
|
|
94
113
|
ruleTitle: rule.title,
|
|
95
114
|
ruleSource: rule.source,
|
|
96
|
-
status:
|
|
115
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
116
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
117
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
118
|
+
// vocabulary exists to stop, one field further down.
|
|
119
|
+
status: "UNCLEAR",
|
|
120
|
+
outcome: "not_applicable",
|
|
121
|
+
method: "file_events",
|
|
122
|
+
ceiling: "a check of file-mutation events — it shows no mutation of this path was recorded, not that the file is untouched on disk",
|
|
97
123
|
evidence: `"${filePath}" was never written to, deleted, or moved this session (reading it does not count)`,
|
|
98
124
|
};
|
|
99
125
|
}
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { violation } from "../types.js";
|
|
1
2
|
/**
|
|
2
3
|
* First real structured-check primitive: parses actual git command
|
|
3
4
|
* arguments instead of searching prose for a branch name as a substring.
|
|
@@ -86,26 +87,44 @@ export function runGitBranchPolicyChecks(classifications, events) {
|
|
|
86
87
|
allTargets.push({ branch, command });
|
|
87
88
|
}
|
|
88
89
|
}
|
|
89
|
-
|
|
90
|
+
const anyGitCommand = events.some((e) => e.kind === "tool_use" && /\bgit\s/.test(JSON.stringify(e.input ?? "")));
|
|
91
|
+
return classifications.map(({ rule, branchName, polarity, polarityInferred }) => {
|
|
92
|
+
// No git command ran, so a git rule never had a situation to govern.
|
|
93
|
+
// Calling that "followed" is how an empty session produced 2,770 green
|
|
94
|
+
// ticks across the 559-file corpus — every one true, none meaningful.
|
|
95
|
+
if (!anyGitCommand) {
|
|
96
|
+
return {
|
|
97
|
+
ruleId: rule.id,
|
|
98
|
+
ruleTitle: rule.title,
|
|
99
|
+
ruleSource: rule.source,
|
|
100
|
+
status: "UNCLEAR",
|
|
101
|
+
outcome: "not_applicable",
|
|
102
|
+
method: "git_events",
|
|
103
|
+
evidence: "no git command ran this session, so this rule never applied",
|
|
104
|
+
};
|
|
105
|
+
}
|
|
90
106
|
const pushOrCreateHit = allTargets.find((t) => t.branch === branchName &&
|
|
91
107
|
(GIT_PUSH.test(t.command) || GIT_BRANCH_CREATE.test(t.command) || GIT_CHECKOUT_CREATE.test(t.command)));
|
|
92
108
|
const commitViolationCommand = findCheckoutCommitViolation(commands, branchName);
|
|
93
109
|
const hit = pushOrCreateHit ?? (commitViolationCommand ? { branch: branchName, command: commitViolationCommand } : undefined);
|
|
94
110
|
if (polarity === "forbid") {
|
|
95
111
|
if (hit) {
|
|
96
|
-
return {
|
|
97
|
-
ruleId: rule.id,
|
|
98
|
-
ruleTitle: rule.title,
|
|
99
|
-
ruleSource: rule.source,
|
|
100
|
-
status: "FAIL",
|
|
101
|
-
evidence: `a git command actually targeted the "${branchName}" branch: ${hit.command}`,
|
|
102
|
-
};
|
|
112
|
+
return violation(rule, polarity, `a git command actually targeted the "${branchName}" branch: ${hit.command}`, { method: "git_events", polarityInferred });
|
|
103
113
|
}
|
|
114
|
+
// Trigger evaluated and absent: the rule never applied. Not
|
|
115
|
+
// "followed" — that word claims something the check cannot show.
|
|
104
116
|
return {
|
|
105
117
|
ruleId: rule.id,
|
|
106
118
|
ruleTitle: rule.title,
|
|
107
119
|
ruleSource: rule.source,
|
|
108
|
-
status:
|
|
120
|
+
// status stays UNCLEAR: a legacy reader must not see a green
|
|
121
|
+
// tick for a rule that never applied. Setting PASS here while the
|
|
122
|
+
// outcome said not_applicable was the same word-borrowing this
|
|
123
|
+
// vocabulary exists to stop, one field further down.
|
|
124
|
+
status: "UNCLEAR",
|
|
125
|
+
outcome: "not_applicable",
|
|
126
|
+
method: "git_events",
|
|
127
|
+
ceiling: "a scan of recorded git commands — it does not see commands run outside this session",
|
|
109
128
|
evidence: `no git command targeted the "${branchName}" branch this session`,
|
|
110
129
|
};
|
|
111
130
|
}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { findTestRun } from "./testCommands.js";
|
|
2
|
+
import { isProjectPath } from "./projectPaths.js";
|
|
1
3
|
const TEST_FILE_PATTERN = /(\.test\.|\.spec\.|__tests__\/|_test\.|\/tests?\/)/i;
|
|
2
4
|
// Real false-positive found 2026-08-30 on an actual session: editing a
|
|
3
5
|
// markdown documentation file flagged "no test file touched" four
|
|
@@ -5,6 +7,18 @@ const TEST_FILE_PATTERN = /(\.test\.|\.spec\.|__tests__\/|_test\.|\/tests?\/)/i;
|
|
|
5
7
|
// companion under any reasonable reading of an "add tests for every
|
|
6
8
|
// change" rule. Excluded from prodPaths entirely, same as test files.
|
|
7
9
|
const NON_TESTABLE_FILE_PATTERN = /\.(md|mdx|txt|rst|json|ya?ml|toml|lock|csv|log)$/i;
|
|
10
|
+
/**
|
|
11
|
+
* Paths that are not this project's source, whatever their extension.
|
|
12
|
+
*
|
|
13
|
+
* Found 2026-09-12 running 559 real rules files against 5 real sessions: 24
|
|
14
|
+
* FAILs said "edited /private/tmp/.../scratchpad/probe.mjs but no matching
|
|
15
|
+
* test file was touched". That is a throwaway probe, written to inspect
|
|
16
|
+
* something and deleted minutes later. Demanding a test for it is nonsense;
|
|
17
|
+
* demanding one as a FAIL is a false accusation.
|
|
18
|
+
*
|
|
19
|
+
* Only file extensions were excluded before, so any temp file that happened
|
|
20
|
+
* to end in .ts or .mjs counted as production code.
|
|
21
|
+
*/
|
|
8
22
|
const WRITE_LIKE_TOOLS = new Set(["Write", "Edit", "NotebookEdit"]);
|
|
9
23
|
function extractEditedPaths(events) {
|
|
10
24
|
const paths = [];
|
|
@@ -40,8 +54,21 @@ function extractEditedPaths(events) {
|
|
|
40
54
|
*/
|
|
41
55
|
export function runIfEditThenTestChecks(classifications, events) {
|
|
42
56
|
const editedPaths = extractEditedPaths(events);
|
|
57
|
+
// Running the suite honours "add tests for every change" as much as
|
|
58
|
+
// touching a test file does. Without this, the most ordinary workflow
|
|
59
|
+
// there is — change code, run the tests, commit — produced a FAIL saying
|
|
60
|
+
// no test file was touched. Twelve rules across the 559-file corpus hit
|
|
61
|
+
// it on one synthetic session (2026-09-11).
|
|
62
|
+
//
|
|
63
|
+
// Any run in the session counts, not only one after the edit. Requiring
|
|
64
|
+
// the stricter ordering would buy a little precision and risk the
|
|
65
|
+
// expensive direction of error, and in this project a wrong FAIL costs
|
|
66
|
+
// more than a missed detection.
|
|
67
|
+
const testRun = findTestRun(events);
|
|
43
68
|
const testPaths = editedPaths.filter((p) => TEST_FILE_PATTERN.test(p));
|
|
44
|
-
const prodPaths = editedPaths.filter((p) => !TEST_FILE_PATTERN.test(p) &&
|
|
69
|
+
const prodPaths = editedPaths.filter((p) => !TEST_FILE_PATTERN.test(p) &&
|
|
70
|
+
!NON_TESTABLE_FILE_PATTERN.test(p) &&
|
|
71
|
+
isProjectPath(p));
|
|
45
72
|
return classifications.map(({ rule }) => {
|
|
46
73
|
if (prodPaths.length === 0) {
|
|
47
74
|
return {
|
|
@@ -54,6 +81,15 @@ export function runIfEditThenTestChecks(classifications, events) {
|
|
|
54
81
|
: "no code file was edited this session (only test/doc/config files, if any) — the rule never had a chance to apply",
|
|
55
82
|
};
|
|
56
83
|
}
|
|
84
|
+
if (testPaths.length === 0 && testRun !== null) {
|
|
85
|
+
return {
|
|
86
|
+
ruleId: rule.id,
|
|
87
|
+
ruleTitle: rule.title,
|
|
88
|
+
ruleSource: rule.source,
|
|
89
|
+
status: "PASS",
|
|
90
|
+
evidence: `edited ${prodPaths[0]} and ran the suite: \`${testRun}\` (no test file was edited, but the code was exercised)`,
|
|
91
|
+
};
|
|
92
|
+
}
|
|
57
93
|
if (testPaths.length === 0) {
|
|
58
94
|
return {
|
|
59
95
|
ruleId: rule.id,
|
|
@@ -31,6 +31,21 @@ function modelId() {
|
|
|
31
31
|
const MAX_TRANSCRIPT_CHARS = 120_000;
|
|
32
32
|
const HEAD_CHARS = 40_000;
|
|
33
33
|
const TAIL_CHARS = MAX_TRANSCRIPT_CHARS - HEAD_CHARS;
|
|
34
|
+
/**
|
|
35
|
+
* How much of a rule's own text goes into the prompt.
|
|
36
|
+
*
|
|
37
|
+
* The transcript was capped and the rule body was not. One rule in the
|
|
38
|
+
* 559-file corpus is 122,000 characters — a section heading whose body is
|
|
39
|
+
* an entire architecture document, parsed as a single rule — and it would
|
|
40
|
+
* have been sent whole on top of a 120,000-character transcript. Roughly
|
|
41
|
+
* 60k tokens for one verdict, with nothing bounding it.
|
|
42
|
+
*
|
|
43
|
+
* A rule that does not fit in 8,000 characters is not really one rule, and
|
|
44
|
+
* the model does not need the rest to judge it. The prompt says when it was
|
|
45
|
+
* cut, so a verdict is never formed from a fragment the model believes is
|
|
46
|
+
* whole.
|
|
47
|
+
*/
|
|
48
|
+
const MAX_RULE_CHARS = 8_000;
|
|
34
49
|
/** How many judgment calls may be in flight at once. */
|
|
35
50
|
const MAX_CONCURRENT_CALLS = 4;
|
|
36
51
|
const TRUNCATION_NOTE = "[judged on a truncated transcript — the middle of this session was not shown to the model]";
|
|
@@ -99,6 +114,13 @@ const INSTRUCTIONS = "You judge whether one rule from a CLAUDE.md/AGENTS.md file
|
|
|
99
114
|
* about 13 rules it had never examined. Found 2026-09-08 by running the
|
|
100
115
|
* published package; nothing in 167 lines of tests here asserted the field.
|
|
101
116
|
*/
|
|
117
|
+
function ruleBody(text) {
|
|
118
|
+
if (text.length <= MAX_RULE_CHARS)
|
|
119
|
+
return text;
|
|
120
|
+
return (text.slice(0, MAX_RULE_CHARS) +
|
|
121
|
+
`\n\n...[rule text truncated at ${MAX_RULE_CHARS} characters — this rule's body is ` +
|
|
122
|
+
`${text.length} characters long and is probably a whole document parsed as one rule]`);
|
|
123
|
+
}
|
|
102
124
|
function didNotRun(rule, reason) {
|
|
103
125
|
return {
|
|
104
126
|
ruleId: rule.id,
|
|
@@ -106,6 +128,10 @@ function didNotRun(rule, reason) {
|
|
|
106
128
|
ruleSource: rule.source,
|
|
107
129
|
status: "UNCLEAR",
|
|
108
130
|
needsHuman: true,
|
|
131
|
+
// The check did not happen. Rendering that as a judgment call, or worse
|
|
132
|
+
// as "couldn't tell", was the original defect in this file.
|
|
133
|
+
outcome: "not_run",
|
|
134
|
+
method: "none",
|
|
109
135
|
evidence: reason,
|
|
110
136
|
};
|
|
111
137
|
}
|
|
@@ -177,7 +203,7 @@ export async function runJudgmentChecks(classifications, events) {
|
|
|
177
203
|
text: `SESSION TRANSCRIPT:\n${transcript.text}`,
|
|
178
204
|
cache_control: { type: "ephemeral" },
|
|
179
205
|
},
|
|
180
|
-
{ type: "text", text: `RULE — ${rule.title}\n${rule.text}` },
|
|
206
|
+
{ type: "text", text: `RULE — ${rule.title}\n${ruleBody(rule.text)}` },
|
|
181
207
|
],
|
|
182
208
|
},
|
|
183
209
|
],
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Paths that are not the project's own files, whatever their name.
|
|
3
|
+
*
|
|
4
|
+
* One definition, shared. Found the hard way twice: ifEditThenTest demanded
|
|
5
|
+
* a test for /private/tmp/.../scratchpad/probe.mjs, and fileLifecycle fired
|
|
6
|
+
* a "CHANGELOG.md is release-only" rule on a scratchpad copy of that file
|
|
7
|
+
* written during a probe and deleted minutes later. Two checkers needed the
|
|
8
|
+
* same exclusion and only one had it, which is exactly how the two copies
|
|
9
|
+
* of TEST_COMMAND would have drifted.
|
|
10
|
+
*
|
|
11
|
+
* Both callers use this to decide whether to accuse someone, so the cost of
|
|
12
|
+
* a miss here is a false FAIL on a throwaway file.
|
|
13
|
+
*/
|
|
14
|
+
export declare const NON_PROJECT_PATH: RegExp;
|
|
15
|
+
/** True when this path is somewhere the project's own rules should govern. */
|
|
16
|
+
export declare function isProjectPath(path: string): boolean;
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Paths that are not the project's own files, whatever their name.
|
|
3
|
+
*
|
|
4
|
+
* One definition, shared. Found the hard way twice: ifEditThenTest demanded
|
|
5
|
+
* a test for /private/tmp/.../scratchpad/probe.mjs, and fileLifecycle fired
|
|
6
|
+
* a "CHANGELOG.md is release-only" rule on a scratchpad copy of that file
|
|
7
|
+
* written during a probe and deleted minutes later. Two checkers needed the
|
|
8
|
+
* same exclusion and only one had it, which is exactly how the two copies
|
|
9
|
+
* of TEST_COMMAND would have drifted.
|
|
10
|
+
*
|
|
11
|
+
* Both callers use this to decide whether to accuse someone, so the cost of
|
|
12
|
+
* a miss here is a false FAIL on a throwaway file.
|
|
13
|
+
*/
|
|
14
|
+
export const NON_PROJECT_PATH = /(?:^|\/)(?:tmp|temp|scratch|scratchpad|node_modules|dist|build|out|coverage|\.git|\.next|\.cache|vendor|__pycache__)(?:\/|$)|^\/(?:private\/)?(?:tmp|var)\//i;
|
|
15
|
+
/** True when this path is somewhere the project's own rules should govern. */
|
|
16
|
+
export function isProjectPath(path) {
|
|
17
|
+
return !NON_PROJECT_PATH.test(path);
|
|
18
|
+
}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import type { TranscriptEvent } from "../types.js";
|
|
2
|
+
/**
|
|
3
|
+
* Commands that run a project's test suite.
|
|
4
|
+
*
|
|
5
|
+
* One definition, imported by every checker that needs it. Two checkers
|
|
6
|
+
* now ask "did the tests run" — claimEvidence, to see whether a claim of a
|
|
7
|
+
* passing suite had anything behind it, and ifEditThenTest, to see whether
|
|
8
|
+
* changed code was exercised. Two copies of this list would drift, and the
|
|
9
|
+
* drift would show up as one checker contradicting the other in the same
|
|
10
|
+
* report.
|
|
11
|
+
*
|
|
12
|
+
* It can never be complete — projects wire their suite to whatever script
|
|
13
|
+
* name they like — so callers must never let a miss become an accusation.
|
|
14
|
+
* See UNKNOWN_SCRIPT_RUNNER in claimEvidence.ts.
|
|
15
|
+
*
|
|
16
|
+
* Deliberately a known list rather than anything test-shaped. Both callers
|
|
17
|
+
* use it to decide whether to report a FAIL, and a rule that fires because
|
|
18
|
+
* someone ran a script with "test" in its name is the expensive kind of
|
|
19
|
+
* wrong.
|
|
20
|
+
*/
|
|
21
|
+
export declare const TEST_COMMAND: RegExp;
|
|
22
|
+
/** The first test command run in this session, or null if none ran. */
|
|
23
|
+
export declare function findTestRun(events: TranscriptEvent[]): string | null;
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Commands that run a project's test suite.
|
|
3
|
+
*
|
|
4
|
+
* One definition, imported by every checker that needs it. Two checkers
|
|
5
|
+
* now ask "did the tests run" — claimEvidence, to see whether a claim of a
|
|
6
|
+
* passing suite had anything behind it, and ifEditThenTest, to see whether
|
|
7
|
+
* changed code was exercised. Two copies of this list would drift, and the
|
|
8
|
+
* drift would show up as one checker contradicting the other in the same
|
|
9
|
+
* report.
|
|
10
|
+
*
|
|
11
|
+
* It can never be complete — projects wire their suite to whatever script
|
|
12
|
+
* name they like — so callers must never let a miss become an accusation.
|
|
13
|
+
* See UNKNOWN_SCRIPT_RUNNER in claimEvidence.ts.
|
|
14
|
+
*
|
|
15
|
+
* Deliberately a known list rather than anything test-shaped. Both callers
|
|
16
|
+
* use it to decide whether to report a FAIL, and a rule that fires because
|
|
17
|
+
* someone ran a script with "test" in its name is the expensive kind of
|
|
18
|
+
* wrong.
|
|
19
|
+
*/
|
|
20
|
+
export const TEST_COMMAND = /\b(?:npm|pnpm|yarn|bun)\s+(?:run\s+)?(?:test|verify|check|ci)\b|\bnpx\s+(?:vitest|jest|mocha|ava)\b|\b(?:vitest|jest|mocha|pytest|phpunit|rspec|tox)\b|\bcargo\s+test\b|\bgo\s+test\b|\bmvn\s+(?:test|verify)\b|\bgradle\s+test\b|\bdotnet\s+test\b|\bpython\s+-m\s+(?:pytest|unittest)\b/i;
|
|
21
|
+
/** The first test command run in this session, or null if none ran. */
|
|
22
|
+
export function findTestRun(events) {
|
|
23
|
+
for (const event of events) {
|
|
24
|
+
if (event.kind !== "tool_use")
|
|
25
|
+
continue;
|
|
26
|
+
const input = event.input;
|
|
27
|
+
const command = input && typeof input.command === "string" ? input.command : "";
|
|
28
|
+
if (command && TEST_COMMAND.test(command))
|
|
29
|
+
return command;
|
|
30
|
+
}
|
|
31
|
+
return null;
|
|
32
|
+
}
|
package/dist/cli.js
CHANGED
|
@@ -15,6 +15,7 @@ import { runIfEditThenTestChecks } from "./checks/ifEditThenTest.js";
|
|
|
15
15
|
import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
|
|
16
16
|
import { runCodeContentChecks } from "./checks/codeContent.js";
|
|
17
17
|
import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
|
|
18
|
+
import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
|
|
18
19
|
import { runJudgmentChecks } from "./checks/judgmentChecks.js";
|
|
19
20
|
import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
|
|
20
21
|
import { generateHtmlReport } from "./report/generateHtmlReport.js";
|
|
@@ -212,6 +213,7 @@ async function runCheck(opts) {
|
|
|
212
213
|
const gitBranchPolicy = classifications.filter((c) => c.kind === "gitBranchPolicy");
|
|
213
214
|
const codeContent = classifications.filter((c) => c.kind === "codeContent");
|
|
214
215
|
const fileLifecycle = classifications.filter((c) => c.kind === "fileLifecycle");
|
|
216
|
+
const claimEvidence = classifications.filter((c) => c.kind === "claimEvidence");
|
|
215
217
|
const judgment = classifications.filter((c) => c.kind === "judgment");
|
|
216
218
|
// Not rules at all — documentation, glossary entries, reference tables,
|
|
217
219
|
// URLs, directory listings, code examples.
|
|
@@ -236,6 +238,7 @@ async function runCheck(opts) {
|
|
|
236
238
|
...runGitBranchPolicyChecks(gitBranchPolicy, events),
|
|
237
239
|
...runCodeContentChecks(codeContent, events),
|
|
238
240
|
...runFileLifecycleChecks(fileLifecycle, events),
|
|
241
|
+
...runClaimEvidenceChecks(claimEvidence, events),
|
|
239
242
|
];
|
|
240
243
|
// Deterministic checks run by default, always, with no key — judgment
|
|
241
244
|
// rules only call out to an LLM with an explicit --llm on THIS run, never
|
|
@@ -60,18 +60,46 @@ function ruleLabel(r, results) {
|
|
|
60
60
|
* as a division of labour.
|
|
61
61
|
*/
|
|
62
62
|
function summaryLine(results) {
|
|
63
|
-
const
|
|
64
|
-
const
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
63
|
+
const n = (b) => results.filter((r) => bucketOf(r) === b).length;
|
|
64
|
+
const parts = [`${n("PASS")} followed`, `${n("FAIL")} not followed`];
|
|
65
|
+
if (n("UNCLEAR_EVIDENCE") > 0)
|
|
66
|
+
parts.push(`${n("UNCLEAR_EVIDENCE")} couldn't tell`);
|
|
67
|
+
if (n("NOT_RUN") > 0)
|
|
68
|
+
parts.push(`${n("NOT_RUN")} not run`);
|
|
69
|
+
if (n("NOT_APPLICABLE") > 0)
|
|
70
|
+
parts.push(`${n("NOT_APPLICABLE")} didn't apply`);
|
|
71
|
+
if (n("UNCLEAR_JUDGMENT") > 0)
|
|
72
|
+
parts.push(`${n("UNCLEAR_JUDGMENT")} need your judgment`);
|
|
72
73
|
return parts.join(" · ");
|
|
73
74
|
}
|
|
75
|
+
/**
|
|
76
|
+
* Six buckets, because "the tool looked and could not decide", "the tool
|
|
77
|
+
* never ran", and "the situation never arose" are three different things
|
|
78
|
+
* that were all rendering as one.
|
|
79
|
+
*
|
|
80
|
+
* From anthropics/claude-code#90542. With no API key the report printed
|
|
81
|
+
* "13 couldn't tell" — a phrase defined in this file as the tool having
|
|
82
|
+
* looked — about thirteen rules it had never examined. And an empty
|
|
83
|
+
* transcript produced 2,770 green ticks across the 559-file corpus, every
|
|
84
|
+
* one of them true and none of them meaning anything, because a rule whose
|
|
85
|
+
* situation never arose was being counted as followed.
|
|
86
|
+
*
|
|
87
|
+
* `outcome` is preferred where a checker sets it; `status` remains the
|
|
88
|
+
* fallback while the rest are migrated.
|
|
89
|
+
*/
|
|
74
90
|
function bucketOf(result) {
|
|
91
|
+
switch (result.outcome) {
|
|
92
|
+
case "fail":
|
|
93
|
+
return "FAIL";
|
|
94
|
+
case "pass":
|
|
95
|
+
return "PASS";
|
|
96
|
+
case "not_run":
|
|
97
|
+
return "NOT_RUN";
|
|
98
|
+
case "not_applicable":
|
|
99
|
+
return "NOT_APPLICABLE";
|
|
100
|
+
case "inconclusive":
|
|
101
|
+
return result.needsHuman ? "UNCLEAR_JUDGMENT" : "UNCLEAR_EVIDENCE";
|
|
102
|
+
}
|
|
75
103
|
if (result.status === "FAIL")
|
|
76
104
|
return "FAIL";
|
|
77
105
|
if (result.status === "PASS")
|
|
@@ -90,10 +118,25 @@ function bucketOf(result) {
|
|
|
90
118
|
const BUCKET_LABEL = {
|
|
91
119
|
FAIL: "Not followed",
|
|
92
120
|
UNCLEAR_EVIDENCE: "Couldn't tell",
|
|
93
|
-
|
|
121
|
+
NOT_RUN: "Not run",
|
|
94
122
|
PASS: "Followed",
|
|
123
|
+
NOT_APPLICABLE: "Didn't apply this session",
|
|
124
|
+
UNCLEAR_JUDGMENT: "Needs your judgment",
|
|
95
125
|
};
|
|
96
|
-
|
|
126
|
+
/**
|
|
127
|
+
* Failures, then the two kinds of gap, then what held, then what never came
|
|
128
|
+
* up, then the human's half. "Not run" sits near the top on purpose: a
|
|
129
|
+
* check that did not happen is closer to a gap than to a result, and
|
|
130
|
+
* burying it is how it got mistaken for one.
|
|
131
|
+
*/
|
|
132
|
+
const BUCKET_ORDER = [
|
|
133
|
+
"FAIL",
|
|
134
|
+
"UNCLEAR_EVIDENCE",
|
|
135
|
+
"NOT_RUN",
|
|
136
|
+
"PASS",
|
|
137
|
+
"NOT_APPLICABLE",
|
|
138
|
+
"UNCLEAR_JUDGMENT",
|
|
139
|
+
];
|
|
97
140
|
/**
|
|
98
141
|
* The explanation shared by every rule in a section, or null when they
|
|
99
142
|
* differ.
|
|
@@ -142,6 +185,12 @@ export function generateReport(results, meta) {
|
|
|
142
185
|
lines.push(`${MARK[r.status]} ${r.status.padEnd(7)} ${ruleLabel(r, clean)}`);
|
|
143
186
|
if (r.evidence)
|
|
144
187
|
lines.push(` evidence: ${r.evidence}`);
|
|
188
|
+
// What this method was ALLOWED to conclude, travelling with the
|
|
189
|
+
// verdict. A text scan may say it saw no occurrence of a spelling; it
|
|
190
|
+
// may not say the act did not happen. That distinction shipped for six
|
|
191
|
+
// versions as a PASS on a session that ran `git push -f`.
|
|
192
|
+
if (r.ceiling)
|
|
193
|
+
lines.push(` this means: ${r.ceiling}`);
|
|
145
194
|
}
|
|
146
195
|
}
|
|
147
196
|
lines.push("");
|
package/dist/types.d.ts
CHANGED
|
@@ -26,6 +26,32 @@ export interface TranscriptToolResultEvent {
|
|
|
26
26
|
}
|
|
27
27
|
export type TranscriptEvent = TranscriptTextEvent | TranscriptToolUseEvent | TranscriptToolResultEvent;
|
|
28
28
|
export type CheckStatus = "PASS" | "FAIL" | "UNCLEAR";
|
|
29
|
+
/**
|
|
30
|
+
* Five outcomes, not three.
|
|
31
|
+
*
|
|
32
|
+
* From anthropics/claude-code#90542. The failure that started this was not a
|
|
33
|
+
* bad matcher — it was vocabulary reuse. With no API key the tool printed
|
|
34
|
+
* "13 couldn't tell", a phrase this codebase defines as "the tool looked and
|
|
35
|
+
* the evidence was ambiguous", about thirteen rules it had never examined.
|
|
36
|
+
* Once `not_run` is its own outcome that lie has nowhere to sit, however
|
|
37
|
+
* good or bad the matcher is.
|
|
38
|
+
*
|
|
39
|
+
* `not_applicable` earns its place the same way. A session that never
|
|
40
|
+
* touched git cannot have violated a git rule, and calling that "followed"
|
|
41
|
+
* is how an empty transcript produced 2,770 green ticks across the 559-file
|
|
42
|
+
* corpus — every one of them true and none of them meaning anything.
|
|
43
|
+
*/
|
|
44
|
+
export type CheckOutcome = "pass" | "fail"
|
|
45
|
+
/** It looked, and the evidence did not settle it. */
|
|
46
|
+
| "inconclusive"
|
|
47
|
+
/** No check happened: no key, an error, no ratified reading. */
|
|
48
|
+
| "not_run"
|
|
49
|
+
/** The trigger never fired, so there was nothing to judge. */
|
|
50
|
+
| "not_applicable";
|
|
51
|
+
/** How a verdict was reached. A verdict with no method is a verdict with no standing. */
|
|
52
|
+
export type CheckMethod = "text_scan" | "file_events" | "git_events" | "code_content" | "edit_test_pairing" | "claim_vs_evidence" | "model_judgment"
|
|
53
|
+
/** Nothing ran. */
|
|
54
|
+
| "none";
|
|
29
55
|
export interface CheckResult {
|
|
30
56
|
ruleId: string;
|
|
31
57
|
ruleTitle: string;
|
|
@@ -48,4 +74,49 @@ export interface CheckResult {
|
|
|
48
74
|
* not blur them.
|
|
49
75
|
*/
|
|
50
76
|
needsHuman?: boolean;
|
|
77
|
+
/**
|
|
78
|
+
* The outcome in the five-value vocabulary. Optional while the checkers
|
|
79
|
+
* are migrated one at a time; `status` remains the fallback.
|
|
80
|
+
*/
|
|
81
|
+
outcome?: CheckOutcome;
|
|
82
|
+
/** How this verdict was reached. */
|
|
83
|
+
method?: CheckMethod;
|
|
84
|
+
/**
|
|
85
|
+
* What this method is ALLOWED to claim.
|
|
86
|
+
*
|
|
87
|
+
* A text scan may say "no occurrence of these spellings in this scope".
|
|
88
|
+
* It may not say "the act did not happen" — that PASS shipped for six
|
|
89
|
+
* versions and passed a session that ran `git push -f`. The ceiling
|
|
90
|
+
* travels with the verdict so the report cannot overclaim on its behalf.
|
|
91
|
+
*/
|
|
92
|
+
ceiling?: string;
|
|
93
|
+
/** Why an inconclusive or not_run outcome came out that way, e.g. scope_incomplete. */
|
|
94
|
+
reason?: string;
|
|
95
|
+
/**
|
|
96
|
+
* Set when the rule's direction was inferred rather than read from an
|
|
97
|
+
* explicit signal word — a bare imperative like "Use `npm`" taken as a
|
|
98
|
+
* requirement.
|
|
99
|
+
*
|
|
100
|
+
* Named on the verdict so a measurement can split inferred rows from
|
|
101
|
+
* explicit ones and settle whether the leftover is coverage or noise,
|
|
102
|
+
* rather than the question being argued. Suggested on
|
|
103
|
+
* anthropics/claude-code#90542.
|
|
104
|
+
*/
|
|
105
|
+
polarityInferred?: boolean;
|
|
51
106
|
}
|
|
107
|
+
/**
|
|
108
|
+
* A FAIL may only be constructed from a forbidding rule.
|
|
109
|
+
*
|
|
110
|
+
* This is the "cannot accuse" property as a compile-time invariant rather
|
|
111
|
+
* than a convention. It held by inspection — every `status: "FAIL"` sat
|
|
112
|
+
* inside a `polarity === "forbid"` branch — and inspection is exactly what
|
|
113
|
+
* stops holding the day someone adds a require-FAIL path. Passing the
|
|
114
|
+
* polarity in means a require branch cannot call this: `"require"` is not
|
|
115
|
+
* assignable to `"forbid"`, and the build fails rather than a user being
|
|
116
|
+
* accused of not doing something the tool guessed they had to do.
|
|
117
|
+
*/
|
|
118
|
+
export declare function violation(rule: {
|
|
119
|
+
id: string;
|
|
120
|
+
title: string;
|
|
121
|
+
source: "global" | "project";
|
|
122
|
+
}, polarity: "forbid", evidence: string, extra?: Partial<CheckResult>): CheckResult;
|