rulereceipt 0.1.10 → 0.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -16
- package/dist/checks/classify.d.ts +69 -1
- package/dist/checks/classify.js +134 -6
- package/dist/checks/codeContent.d.ts +3 -0
- package/dist/checks/codeContent.js +89 -0
- package/dist/checks/deterministicChecks.js +15 -2
- package/dist/checks/fileLifecycle.d.ts +3 -0
- package/dist/checks/fileLifecycle.js +119 -0
- package/dist/checks/gitBranchPolicy.d.ts +3 -0
- package/dist/checks/gitBranchPolicy.js +131 -0
- package/dist/cli.js +30 -2
- package/dist/rules.js +43 -4
- package/package.json +4 -3
package/README.md
CHANGED
|
@@ -23,29 +23,38 @@ only (`rulereceipt check`). No automatic hooks, ever, in v1.
|
|
|
23
23
|
|
|
24
24
|
## Status
|
|
25
25
|
|
|
26
|
-
Published and live on npm
|
|
27
|
-
passing).
|
|
26
|
+
Published and live on npm, actively developed.
|
|
28
27
|
|
|
29
28
|
## How it works
|
|
30
29
|
|
|
31
30
|
1. Reads your CLAUDE.md / AGENTS.md and extracts individual rules —
|
|
32
|
-
from the current project directory and
|
|
33
|
-
2. Reads your most recent Claude Code session transcript
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
31
|
+
from the current project directory and your global rules file.
|
|
32
|
+
2. Reads your most recent Claude Code session transcript, wherever Claude
|
|
33
|
+
Code stored it — including hosted or enterprise variants that use a
|
|
34
|
+
different directory.
|
|
35
|
+
3. Routes each rule to the narrowest check that can actually answer it:
|
|
36
|
+
- **Structured checks** read what the session really did — an actual
|
|
37
|
+
git command's branch argument, actual file edits, actual file
|
|
38
|
+
operations. These are the only checks that report a confident FAIL,
|
|
39
|
+
because they can tell an action from a mention.
|
|
40
|
+
- **Literal checks** look for a specific string named in the rule.
|
|
41
|
+
Absence is real evidence, so a clean session PASSes. A match reports
|
|
42
|
+
UNCLEAR with the text quoted, because a text match alone cannot
|
|
43
|
+
distinguish doing the forbidden thing from grepping for it, quoting
|
|
44
|
+
it, or naming it in a commit message.
|
|
45
|
+
- **Judgment** rules need real understanding (e.g. "surface bad news
|
|
46
|
+
first"). With `--llm` each is graded individually using *your own*
|
|
47
|
+
Claude key; without it they report UNCLEAR rather than guessing.
|
|
48
|
+
- Lines containing no instruction at all — directory listings,
|
|
49
|
+
reference tables, examples — aren't rules, and are reported as such
|
|
50
|
+
instead of being checked.
|
|
43
51
|
4. Prints a report — terminal table by default, or `--markdown` for
|
|
44
52
|
pasting into a PR or Slack message — showing what passed, what
|
|
45
53
|
failed, and a quoted line of evidence for each. Every report includes
|
|
46
54
|
a SHA-256 hash of the session file it checked, so anyone with that
|
|
47
|
-
file can
|
|
48
|
-
|
|
55
|
+
file can confirm the report describes that exact file. (It proves the
|
|
56
|
+
report matches the file, not that the file is an unmodified record —
|
|
57
|
+
see SECURITY.md.)
|
|
49
58
|
|
|
50
59
|
## Try it with zero setup
|
|
51
60
|
|
|
@@ -61,7 +70,14 @@ report so you can see the output shape immediately.
|
|
|
61
70
|
```bash
|
|
62
71
|
rulereceipt check # check the latest session in this project
|
|
63
72
|
rulereceipt check --markdown # same, formatted for pasting into a PR/Slack
|
|
64
|
-
rulereceipt check --
|
|
73
|
+
rulereceipt check --llm # opt-in: grade judgment rules with your own Claude key
|
|
74
|
+
rulereceipt check --share # opt-in: send anonymous pass/fail/unclear counts
|
|
75
|
+
rulereceipt check --telemetry # opt-in: send one random per-machine ID
|
|
76
|
+
rulereceipt check --transcript <path> # check a specific session file
|
|
77
|
+
rulereceipt doctor # list hooks/auto-run tasks configured on this machine
|
|
78
|
+
rulereceipt lint # find contradictions between CLAUDE.md and AGENTS.md
|
|
79
|
+
rulereceipt digest # summarise recent checks; --email to send it
|
|
80
|
+
rulereceipt config # set up email sending (stays on your machine)
|
|
65
81
|
rulereceipt demo # sample output, no setup needed
|
|
66
82
|
rulereceipt demo --markdown
|
|
67
83
|
rulereceipt --version # print the installed version
|
|
@@ -19,7 +19,75 @@ export interface JudgmentClassification {
|
|
|
19
19
|
kind: "judgment";
|
|
20
20
|
rule: Rule;
|
|
21
21
|
}
|
|
22
|
-
|
|
22
|
+
/**
|
|
23
|
+
* First real structured-check primitive (2026-08-30), replacing keyword
|
|
24
|
+
* search for one whole rule category: git branch policy. A rule naming a
|
|
25
|
+
* branch (e.g. "never touch the `demo` branch") was previously checked by
|
|
26
|
+
* searching for the word "demo" ANYWHERE in the transcript — matching a
|
|
27
|
+
* repo name, a directory, a sentence, anything. This routes instead to a
|
|
28
|
+
* real parser (gitBranchPolicy.ts) that reads actual git command
|
|
29
|
+
* arguments and checks the literal branch name, not a substring search.
|
|
30
|
+
*/
|
|
31
|
+
export interface GitBranchPolicyClassification {
|
|
32
|
+
kind: "gitBranchPolicy";
|
|
33
|
+
rule: Rule;
|
|
34
|
+
branchName: string;
|
|
35
|
+
polarity: DeterministicPolarity;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Second structured-check primitive (2026-08-30): code content, for rules
|
|
39
|
+
* naming an actual code construct — a function/method call like `print(`
|
|
40
|
+
* or `analytics.track(` — rather than a CLI command. Real false-positive
|
|
41
|
+
* this fixes: even after excluding tool_result (see deterministicChecks.ts),
|
|
42
|
+
* a rule like "no `print(` statements" still matched when the agent's OWN
|
|
43
|
+
* Bash command merely MENTIONED the pattern as an argument (e.g. grepping
|
|
44
|
+
* for it), because generic deterministic matching scans the whole
|
|
45
|
+
* stringified tool_use input, commands included. This routes instead to
|
|
46
|
+
* codeContent.ts, which only looks at the actual content of real file
|
|
47
|
+
* edits (Write/Edit/NotebookEdit) — never a Bash command string, never
|
|
48
|
+
* prose, never a search argument.
|
|
49
|
+
*/
|
|
50
|
+
export interface CodeContentClassification {
|
|
51
|
+
kind: "codeContent";
|
|
52
|
+
rule: Rule;
|
|
53
|
+
patterns: string[];
|
|
54
|
+
polarity: DeterministicPolarity;
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* Third structured-check primitive (2026-08-30): file lifecycle, for
|
|
58
|
+
* rules protecting a specific file ("never modify `.claude/settings.json`",
|
|
59
|
+
* "don't delete `config.yaml`"). Real false-positive this fixes: the path
|
|
60
|
+
* was flagged as touched when the agent merely READ it (`cat
|
|
61
|
+
* .claude/settings.json` to verify its contents) — reading a protected
|
|
62
|
+
* file is not modifying it. Routes to fileLifecycle.ts, which only counts
|
|
63
|
+
* real mutations: Write/Edit on that path, or a Bash rm/mv/truncate-style
|
|
64
|
+
* command targeting it.
|
|
65
|
+
*/
|
|
66
|
+
export interface FileLifecycleClassification {
|
|
67
|
+
kind: "fileLifecycle";
|
|
68
|
+
rule: Rule;
|
|
69
|
+
filePath: string;
|
|
70
|
+
polarity: DeterministicPolarity;
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Not every line in a real CLAUDE.md is a rule. Measured against 40 real
|
|
74
|
+
* public rule files (1,441 parsed items, 2026-08-30): only ~20% contained
|
|
75
|
+
* any directive language at all. The other ~80% is documentation —
|
|
76
|
+
* directory listings ("`forge/llm/` - Multi-provider LLM integrations"),
|
|
77
|
+
* import examples, model-name tables, glob-syntax references. 521 of those
|
|
78
|
+
* were getting a literal keyword check run against them, which is the
|
|
79
|
+
* single largest source of false positives: checking whether a session
|
|
80
|
+
* "violated" a directory listing is meaningless, and any coincidental
|
|
81
|
+
* match is noise reported as a finding.
|
|
82
|
+
*
|
|
83
|
+
* These are reported as N/A and excluded from pass/fail entirely — not
|
|
84
|
+
* sent to the LLM either, since there's no rule to judge.
|
|
85
|
+
*/
|
|
86
|
+
export interface NotARuleClassification {
|
|
87
|
+
kind: "notARule";
|
|
88
|
+
rule: Rule;
|
|
89
|
+
}
|
|
90
|
+
export type Classification = DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | JudgmentClassification;
|
|
23
91
|
/**
|
|
24
92
|
* A rule is only treated as deterministic when it names a specific,
|
|
25
93
|
* literal, checkable token (a CLI flag, a command, an exact string) in
|
package/dist/checks/classify.js
CHANGED
|
@@ -1,16 +1,81 @@
|
|
|
1
|
+
// Normative language — the thing that makes a line a rule rather than a
|
|
2
|
+
// description. Deliberately broad on modals AND imperative verbs, because
|
|
3
|
+
// a wrongly-excluded rule is a silent miss.
|
|
4
|
+
const DIRECTIVE_LANGUAGE = /\b(never|always|must|should|shall|do not|don't|dont|cannot|can't|required?|requires|ensure|avoid|prefer|forbidden|prohibited|only|make sure|be sure|need|needs|needed|need to|has to|have to|expected to|responsible for)\b/i;
|
|
5
|
+
/**
|
|
6
|
+
* Imperative instruction — a bare command verb starting a clause ("Use
|
|
7
|
+
* `gh pr merge`", "Run the tests first", "Keep functions small"). This is
|
|
8
|
+
* the other way a real rule is written when it doesn't use a modal.
|
|
9
|
+
*
|
|
10
|
+
* Anchored to a clause start (line start, or after sentence/bullet
|
|
11
|
+
* punctuation) on purpose: the same verbs appear mid-sentence in pure
|
|
12
|
+
* documentation ("the CLI can run migrations"), where they describe a
|
|
13
|
+
* capability rather than instruct the agent.
|
|
14
|
+
*/
|
|
15
|
+
const IMPERATIVE_INSTRUCTION = /(?:^|[.;:!?]\s+|^\s*[-*+]\s*|\n\s*[-*+]\s*)(use|run|keep|write|add|remove|delete|check|verify|test|commit|document|update|create|follow|apply|include|exclude|handle|validate|escape|sanitize|log|report|raise|throw|return|call|invoke|split|group|sort|name|place|put|store|read|load|save|close|open|start|stop|restart|install|build|deploy|review|refactor|rename|move|copy|merge|rebase|squash|tag|branch|push|pull|fetch|clone|stage|stash|lead|state|explain|describe|list|show|surface|flag|mark|label|note|treat|assume|confirm|ask|wait|stick|limit|cap|batch|cache|mock|stub|assert|expect|measure|quantify|label)\b/i;
|
|
16
|
+
/**
|
|
17
|
+
* Deliberately inverted: tests for the presence of a DIRECTIVE, never for
|
|
18
|
+
* the shape of documentation.
|
|
19
|
+
*
|
|
20
|
+
* The first version of this enumerated documentation shapes (backtick
|
|
21
|
+
* glossary, bold-term definition, arrow mapping, label rows, "Reference:"
|
|
22
|
+
* prefixes). That approach cannot work: every new rules-file convention is
|
|
23
|
+
* a new shape, so the list grows forever and is always one format behind —
|
|
24
|
+
* measurably so, since `notARule` coverage fell from 17.4% on a 40-file
|
|
25
|
+
* sample to 7.1% on a 658-file one purely because the bigger corpus used
|
|
26
|
+
* shapes the list didn't have yet.
|
|
27
|
+
*
|
|
28
|
+
* "Does this contain an instruction" is a bounded question about English —
|
|
29
|
+
* modal verbs plus the closed class of imperative command verbs — and it
|
|
30
|
+
* doesn't change when someone invents a new markdown convention. A line
|
|
31
|
+
* with no instruction in it has nothing to check compliance against,
|
|
32
|
+
* whatever its punctuation.
|
|
33
|
+
*/
|
|
34
|
+
function isNotARule(rule) {
|
|
35
|
+
const combined = `${rule.title} ${rule.text}`;
|
|
36
|
+
if (DIRECTIVE_LANGUAGE.test(combined))
|
|
37
|
+
return false;
|
|
38
|
+
// Title and text are tested SEPARATELY: the imperative pattern is
|
|
39
|
+
// anchored to a clause start, and concatenating them pushes the text's
|
|
40
|
+
// opening verb into mid-string where the anchor can never match. That
|
|
41
|
+
// bug silently classified real rules ("Use `npm` for this project")
|
|
42
|
+
// as non-rules — caught by an existing test, not by inspection.
|
|
43
|
+
return !IMPERATIVE_INSTRUCTION.test(rule.title) && !IMPERATIVE_INSTRUCTION.test(rule.text);
|
|
44
|
+
}
|
|
45
|
+
const BRANCH_WORD = /\bbranch\b/i;
|
|
46
|
+
// A function/method-call shape ("print(", "analytics.track(") is a strong,
|
|
47
|
+
// simple signal that a backtick literal names actual CODE, not a CLI
|
|
48
|
+
// command or flag ("git push --force", "npm test" never look like this).
|
|
49
|
+
const CODE_CONSTRUCT_PATTERN = /\(/;
|
|
50
|
+
// A file-path shape: a known config/source extension, or a path with a
|
|
51
|
+
// directory separator. Deliberately requires no spaces — a real path
|
|
52
|
+
// literal ("`.claude/settings.json`", "`config.yaml`") never has one,
|
|
53
|
+
// while a command that happens to contain a slash ("`git push --force`")
|
|
54
|
+
// does. Checked AFTER the code-construct test, so "foo(" never lands here.
|
|
55
|
+
const FILE_PATH_PATTERN = /^[^\s]*(\.(json|ya?ml|toml|md|env|lock|ini|cfg|conf|xml|txt|js|ts|py|rb|go|rs|sh)$|\/)/i;
|
|
56
|
+
// Only route to fileLifecycle when the rule is actually about touching the
|
|
57
|
+
// file, not merely mentioning one (e.g. "read `config.yaml` before
|
|
58
|
+
// starting" names a path but isn't a protection rule).
|
|
59
|
+
const FILE_MUTATION_INTENT = /\b(modif|chang|edit|delet|remov|overwrit|touch|writ|creat|rename|mov)\w*\b/i;
|
|
1
60
|
// Catches rules like "add tests for every change" or "every new function
|
|
2
61
|
// needs a test" - no literal backtick token to pattern-match, so without
|
|
3
62
|
// this they'd fall all the way through to judgment (an LLM call) even
|
|
4
63
|
// though they're actually structurally checkable: did a production file
|
|
5
64
|
// get edited without a corresponding test file also being touched.
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
//
|
|
9
|
-
|
|
10
|
-
|
|
65
|
+
//
|
|
66
|
+
// Real false-positive found 2026-08-30 on an actual session: the earlier
|
|
67
|
+
// version of this heuristic (bare "test" word + any require-signal word,
|
|
68
|
+
// independently anywhere in the text) misclassified "Tests must be able
|
|
69
|
+
// to fail" - a rule about test QUALITY (a test must be demonstrated to
|
|
70
|
+
// fail on bad input before it counts) - as an edit-implies-test rule,
|
|
71
|
+
// producing nonsense like "you edited .env but no test file was touched"
|
|
72
|
+
// for a rule that was never about that at all. Fixed by requiring a
|
|
73
|
+
// specific phrase shape - a create/add/need verb close to the word
|
|
74
|
+
// "test" - not just independent word presence anywhere in the text.
|
|
75
|
+
const EDIT_IMPLIES_TEST_PHRASE = /\b(needs?\s+(a\s+)?(corresponding\s+)?test|add(?:ing|ed)?\s+tests?|write\s+tests?|include\s+tests?|corresponding\s+test|test\s+coverage\s+for)\b/i;
|
|
11
76
|
function isEditImpliesTestRule(rule) {
|
|
12
77
|
const text = `${rule.title} ${rule.text}`;
|
|
13
|
-
return
|
|
78
|
+
return EDIT_IMPLIES_TEST_PHRASE.test(text);
|
|
14
79
|
}
|
|
15
80
|
const BACKTICK_TOKEN = /`([^`]+)`/g;
|
|
16
81
|
// Keyword signal near the rule text that this is a mandatory action, not a
|
|
@@ -21,6 +86,44 @@ const BACKTICK_TOKEN = /`([^`]+)`/g;
|
|
|
21
86
|
// only an additive one for rules that clearly ask for a required action.
|
|
22
87
|
const REQUIRE_SIGNAL = /\b(always|must|required|require|ensure|need to)\b/i;
|
|
23
88
|
const FORBID_SIGNAL = /\b(never|don't|do not|forbidden|banned|must not)\b/i;
|
|
89
|
+
/**
|
|
90
|
+
* A rule can prohibit one thing AND prescribe another in the same breath:
|
|
91
|
+
* "NEVER squash when merging PRs. Use `gh pr merge --merge --admin`".
|
|
92
|
+
* Literal pattern matching cannot handle this — it extracts `--merge` from
|
|
93
|
+
* the PRESCRIBED command, then applies the rule's forbid polarity to it,
|
|
94
|
+
* so doing exactly what the rule demands gets reported as violating it.
|
|
95
|
+
* Real false positive found 2026-08-30 on a real rules file.
|
|
96
|
+
*
|
|
97
|
+
* There's no honest way to split which literal belongs to which half by
|
|
98
|
+
* pattern alone, so a mixed-polarity rule goes to judgment instead of
|
|
99
|
+
* being guessed at. Better an "needs --llm" than a confident wrong FAIL.
|
|
100
|
+
*/
|
|
101
|
+
// Prescriptive language beyond the modal verbs in REQUIRE_SIGNAL. A rule
|
|
102
|
+
// often prescribes with a bare imperative ("Use `gh pr merge --merge`")
|
|
103
|
+
// rather than "you must use" — and it's exactly those imperative clauses
|
|
104
|
+
// whose literals get misattributed to the prohibiting half.
|
|
105
|
+
const PRESCRIPTIVE_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to)\b/i;
|
|
106
|
+
function hasMixedPolarity(rule) {
|
|
107
|
+
const text = `${rule.title} ${rule.text}`;
|
|
108
|
+
if (!FORBID_SIGNAL.test(text))
|
|
109
|
+
return false;
|
|
110
|
+
// The prescription must live in a DIFFERENT clause than the prohibition.
|
|
111
|
+
// "Never run `git push --force`" is one clause: "run" belongs to the
|
|
112
|
+
// forbidden action, and treating that as mixed polarity would send the
|
|
113
|
+
// most basic literal rule there is to the LLM. "NEVER squash when
|
|
114
|
+
// merging PRs. Use `gh pr merge --merge`" is two clauses, and only the
|
|
115
|
+
// second one's literals would be misattributed.
|
|
116
|
+
//
|
|
117
|
+
// Code spans are masked before splitting. Real bug caught by the
|
|
118
|
+
// end-to-end violation tests: "Never leave a `console.log(` call in
|
|
119
|
+
// committed code" was split on the period INSIDE the code span, leaving
|
|
120
|
+
// a fragment ("log(` call in committed code") with no forbid word but a
|
|
121
|
+
// prescriptive one, so a plain prohibition was misread as mixed
|
|
122
|
+
// polarity and sent to judgment instead of being checked.
|
|
123
|
+
const masked = text.replace(/`[^`]*`/g, (m) => "`" + "x".repeat(Math.max(m.length - 2, 0)) + "`");
|
|
124
|
+
const clauses = masked.split(/[.;\n]|(?:\s+-\s+)/).filter((c) => c.trim().length > 0);
|
|
125
|
+
return clauses.some((clause) => !FORBID_SIGNAL.test(clause) && (REQUIRE_SIGNAL.test(clause) || PRESCRIPTIVE_VERB.test(clause)));
|
|
126
|
+
}
|
|
24
127
|
function detectPolarity(rule) {
|
|
25
128
|
const text = `${rule.title} ${rule.text}`;
|
|
26
129
|
// An explicit forbid word anywhere wins over a require word — "you must
|
|
@@ -39,6 +142,11 @@ function detectPolarity(rule) {
|
|
|
39
142
|
* guessing it's safe to pattern-match.
|
|
40
143
|
*/
|
|
41
144
|
export function classifyRule(rule) {
|
|
145
|
+
// Checked first: if this isn't a rule at all, no check of any kind
|
|
146
|
+
// should run against it — not a keyword match, not an LLM call.
|
|
147
|
+
if (isNotARule(rule)) {
|
|
148
|
+
return { kind: "notARule", rule };
|
|
149
|
+
}
|
|
42
150
|
const patterns = new Set();
|
|
43
151
|
for (const match of rule.text.matchAll(BACKTICK_TOKEN)) {
|
|
44
152
|
const token = match[1].trim();
|
|
@@ -50,12 +158,32 @@ export function classifyRule(rule) {
|
|
|
50
158
|
if (token.length > 0)
|
|
51
159
|
patterns.add(token);
|
|
52
160
|
}
|
|
161
|
+
// A rule that both forbids and prescribes can't be checked by literal
|
|
162
|
+
// matching without misattributing one half's tokens to the other.
|
|
163
|
+
if (patterns.size > 0 && hasMixedPolarity(rule)) {
|
|
164
|
+
return { kind: "judgment", rule };
|
|
165
|
+
}
|
|
53
166
|
if (patterns.size === 0) {
|
|
54
167
|
if (isEditImpliesTestRule(rule)) {
|
|
55
168
|
return { kind: "ifEditThenTest", rule };
|
|
56
169
|
}
|
|
57
170
|
return { kind: "judgment", rule };
|
|
58
171
|
}
|
|
172
|
+
const text = `${rule.title} ${rule.text}`;
|
|
173
|
+
if (BRANCH_WORD.test(text)) {
|
|
174
|
+
// first backtick literal is treated as the branch name — real rules
|
|
175
|
+
// this targets name exactly one branch ("the `demo` branch", "never
|
|
176
|
+
// push to `main`"), not a set of them
|
|
177
|
+
const [branchName] = patterns;
|
|
178
|
+
return { kind: "gitBranchPolicy", rule, branchName, polarity: detectPolarity(rule) };
|
|
179
|
+
}
|
|
180
|
+
if ([...patterns].some((p) => CODE_CONSTRUCT_PATTERN.test(p))) {
|
|
181
|
+
return { kind: "codeContent", rule, patterns: [...patterns], polarity: detectPolarity(rule) };
|
|
182
|
+
}
|
|
183
|
+
const filePath = [...patterns].find((p) => FILE_PATH_PATTERN.test(p));
|
|
184
|
+
if (filePath && FILE_MUTATION_INTENT.test(text)) {
|
|
185
|
+
return { kind: "fileLifecycle", rule, filePath, polarity: detectPolarity(rule) };
|
|
186
|
+
}
|
|
59
187
|
return { kind: "deterministic", rule, patterns: [...patterns], polarity: detectPolarity(rule) };
|
|
60
188
|
}
|
|
61
189
|
export function classifyRules(rules) {
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Second structured-check primitive: only scans the actual content of
|
|
3
|
+
* real file edits for a code-construct pattern (e.g. `print(`,
|
|
4
|
+
* `analytics.track(`) — never a Bash command string, never prose, never
|
|
5
|
+
* tool_result. Real false-positive this fixes (found 2026-08-30, on the
|
|
6
|
+
* same real session as the git-branch bug): even after excluding
|
|
7
|
+
* tool_result from the generic deterministic check, a rule like "no
|
|
8
|
+
* `print(` statements" still matched because the agent's own Bash
|
|
9
|
+
* command MENTIONED the pattern as a search argument (e.g. `grep -rn
|
|
10
|
+
* "print(" src/`) — no print statement was ever written into a file.
|
|
11
|
+
*
|
|
12
|
+
* Known, stated limitation: `content`/`new_string` are the confirmed
|
|
13
|
+
* real field names for Write/Edit tool_use input; NotebookEdit's field
|
|
14
|
+
* name is included as best-effort (not independently confirmed against a
|
|
15
|
+
* real NotebookEdit transcript event before shipping this) — a session
|
|
16
|
+
* that only writes matching code via NotebookEdit could under-report,
|
|
17
|
+
* which fails toward UNCLEAR/PASS, not a fabricated FAIL.
|
|
18
|
+
*/
|
|
19
|
+
function editedContentFromEvent(event) {
|
|
20
|
+
if (event.kind !== "tool_use")
|
|
21
|
+
return null;
|
|
22
|
+
const input = event.input;
|
|
23
|
+
if (event.toolName === "Write" && typeof input?.content === "string")
|
|
24
|
+
return input.content;
|
|
25
|
+
if (event.toolName === "Edit" && typeof input?.new_string === "string")
|
|
26
|
+
return input.new_string;
|
|
27
|
+
if (event.toolName === "NotebookEdit" && typeof input?.new_source === "string")
|
|
28
|
+
return input.new_source;
|
|
29
|
+
return null;
|
|
30
|
+
}
|
|
31
|
+
export function runCodeContentChecks(classifications, events) {
|
|
32
|
+
const editedContents = [];
|
|
33
|
+
for (const event of events) {
|
|
34
|
+
const content = editedContentFromEvent(event);
|
|
35
|
+
if (content)
|
|
36
|
+
editedContents.push(content);
|
|
37
|
+
}
|
|
38
|
+
return classifications.map(({ rule, patterns, polarity }) => {
|
|
39
|
+
let foundPattern;
|
|
40
|
+
let foundContent;
|
|
41
|
+
for (const content of editedContents) {
|
|
42
|
+
for (const pattern of patterns) {
|
|
43
|
+
if (content.includes(pattern)) {
|
|
44
|
+
foundPattern = pattern;
|
|
45
|
+
foundContent = content;
|
|
46
|
+
break;
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
if (foundPattern)
|
|
50
|
+
break;
|
|
51
|
+
}
|
|
52
|
+
if (polarity === "forbid") {
|
|
53
|
+
if (foundPattern && foundContent) {
|
|
54
|
+
return {
|
|
55
|
+
ruleId: rule.id,
|
|
56
|
+
ruleTitle: rule.title,
|
|
57
|
+
ruleSource: rule.source,
|
|
58
|
+
status: "FAIL",
|
|
59
|
+
evidence: `found "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`,
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
return {
|
|
63
|
+
ruleId: rule.id,
|
|
64
|
+
ruleTitle: rule.title,
|
|
65
|
+
ruleSource: rule.source,
|
|
66
|
+
status: "PASS",
|
|
67
|
+
evidence: `no file edit actually contained ${patterns.map((p) => `"${p}"`).join(" or ")} this session`,
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
// require: absence is UNCLEAR, not a fabricated FAIL — same reasoning
|
|
71
|
+
// as deterministicChecks.ts's require-polarity handling
|
|
72
|
+
if (foundPattern && foundContent) {
|
|
73
|
+
return {
|
|
74
|
+
ruleId: rule.id,
|
|
75
|
+
ruleTitle: rule.title,
|
|
76
|
+
ruleSource: rule.source,
|
|
77
|
+
status: "PASS",
|
|
78
|
+
evidence: `found required "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
return {
|
|
82
|
+
ruleId: rule.id,
|
|
83
|
+
ruleTitle: rule.title,
|
|
84
|
+
ruleSource: rule.source,
|
|
85
|
+
status: "UNCLEAR",
|
|
86
|
+
evidence: `no file edit contained the required ${patterns.map((p) => `"${p}"`).join(" or ")} this session — can't tell if the rule didn't apply, or applied and was skipped`,
|
|
87
|
+
};
|
|
88
|
+
});
|
|
89
|
+
}
|
|
@@ -70,12 +70,25 @@ export function runDeterministicChecks(classifications, events) {
|
|
|
70
70
|
if (polarity === "forbid") {
|
|
71
71
|
if (foundEvent && foundPattern) {
|
|
72
72
|
const haystack = eventSearchText(foundEvent);
|
|
73
|
+
// Deliberately UNCLEAR, never FAIL. A bare literal match proves
|
|
74
|
+
// the string appeared somewhere; it cannot prove the agent DID
|
|
75
|
+
// the forbidden thing. The same match is produced by grepping
|
|
76
|
+
// for the pattern, quoting it in an explanation, or naming it in
|
|
77
|
+
// a commit message — all compliant. Every false positive found
|
|
78
|
+
// on 2026-08-30 was this exact confusion, and no amount of
|
|
79
|
+
// pattern tuning fixes it, because the information needed to
|
|
80
|
+
// tell action from mention is not in the string.
|
|
81
|
+
//
|
|
82
|
+
// Confident FAILs come only from the structured primitives
|
|
83
|
+
// (gitBranchPolicy, codeContent, fileLifecycle), which read what
|
|
84
|
+
// the agent actually executed or wrote. This path reports the
|
|
85
|
+
// evidence and says plainly that it can't confirm a violation.
|
|
73
86
|
return {
|
|
74
87
|
ruleId: rule.id,
|
|
75
88
|
ruleTitle: rule.title,
|
|
76
89
|
ruleSource: rule.source,
|
|
77
|
-
status: "
|
|
78
|
-
evidence: `
|
|
90
|
+
status: "UNCLEAR",
|
|
91
|
+
evidence: `"${foundPattern}" appears in a ${foundEvent.kind === "tool_use" ? foundEvent.toolName + " call" : foundEvent.kind}, but a text match alone can't tell an actual violation from a mention (a search for it, a quote, an explanation) — needs a human look: ${haystack.slice(0, 160)}`,
|
|
79
92
|
};
|
|
80
93
|
}
|
|
81
94
|
return {
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Third structured-check primitive: only counts real MUTATIONS of a
|
|
3
|
+
* protected file, never reads of it. Real false-positive this fixes
|
|
4
|
+
* (found 2026-08-30 on a real session): a rule protecting
|
|
5
|
+
* `.claude/settings.json` reported it as touched because the agent ran
|
|
6
|
+
* `cat .claude/settings.json` to VERIFY its contents — reading a
|
|
7
|
+
* protected file to confirm it's intact is the opposite of violating the
|
|
8
|
+
* rule, and flagging it punishes exactly the behavior the rule wants.
|
|
9
|
+
*
|
|
10
|
+
* Counts as a mutation:
|
|
11
|
+
* - Write / Edit / NotebookEdit tool_use whose file_path is the file
|
|
12
|
+
* - A Bash command using a destructive/overwriting operator on the path
|
|
13
|
+
* (rm, mv onto it, truncation via `>`, sed -i, tee)
|
|
14
|
+
*
|
|
15
|
+
* Explicitly NOT a mutation: cat, less, head, tail, grep, Read, ls, or
|
|
16
|
+
* the path merely appearing in prose or in another command's arguments.
|
|
17
|
+
*
|
|
18
|
+
* Known, stated limitation: Bash detection is regex over the command
|
|
19
|
+
* string, not a shell parser. A sufficiently exotic invocation (an
|
|
20
|
+
* unusual redirect form, a path built from a variable, a mutation inside
|
|
21
|
+
* a script file that is itself invoked) will be missed — that's a false
|
|
22
|
+
* NEGATIVE (reports PASS/UNCLEAR), the safe direction. This must never
|
|
23
|
+
* fabricate a FAIL from a command that only read the file.
|
|
24
|
+
*/
|
|
25
|
+
const WRITE_LIKE_TOOLS = new Set(["Write", "Edit", "NotebookEdit"]);
|
|
26
|
+
function escapeRegex(literal) {
|
|
27
|
+
return literal.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* A path is "the same file" if the command references it exactly, or
|
|
31
|
+
* references a path ending in it (so a rule naming
|
|
32
|
+
* `.claude/settings.json` still matches an absolute
|
|
33
|
+
* `/home/x/.claude/settings.json`). Deliberately anchored at a path
|
|
34
|
+
* boundary so `settings.json` does not match `other-settings.json`.
|
|
35
|
+
*/
|
|
36
|
+
function pathPattern(filePath) {
|
|
37
|
+
return `(?:^|[\\s'"=/])${escapeRegex(filePath.replace(/^\.\//, ""))}(?=$|[\\s'";)])`;
|
|
38
|
+
}
|
|
39
|
+
function mutatesPathInBash(command, filePath) {
|
|
40
|
+
const p = pathPattern(filePath);
|
|
41
|
+
const mutations = [
|
|
42
|
+
// rm / rmdir / unlink targeting the path
|
|
43
|
+
new RegExp(`\\b(?:rm|rmdir|unlink)\\b[^|;&]*${p}`),
|
|
44
|
+
// mv / cp writing ONTO the path (path appears as the final argument)
|
|
45
|
+
new RegExp(`\\b(?:mv|cp)\\b[^|;&]*${p}\\s*(?:$|[;&|])`),
|
|
46
|
+
// truncation or append redirect onto the path
|
|
47
|
+
new RegExp(`>>?\\s*['"]?${escapeRegex(filePath.replace(/^\.\//, ""))}`),
|
|
48
|
+
// in-place edits
|
|
49
|
+
new RegExp(`\\bsed\\b[^|;&]*-i[^|;&]*${p}`),
|
|
50
|
+
new RegExp(`\\btee\\b[^|;&]*${p}`),
|
|
51
|
+
new RegExp(`\\btruncate\\b[^|;&]*${p}`),
|
|
52
|
+
];
|
|
53
|
+
return mutations.some((re) => re.test(command));
|
|
54
|
+
}
|
|
55
|
+
function findMutation(events, filePath) {
|
|
56
|
+
const normalized = filePath.replace(/^\.\//, "");
|
|
57
|
+
for (const event of events) {
|
|
58
|
+
if (event.kind !== "tool_use")
|
|
59
|
+
continue;
|
|
60
|
+
if (WRITE_LIKE_TOOLS.has(event.toolName)) {
|
|
61
|
+
const input = event.input;
|
|
62
|
+
if (typeof input?.file_path === "string") {
|
|
63
|
+
const actual = input.file_path.replace(/^\.\//, "");
|
|
64
|
+
if (actual === normalized || actual.endsWith(`/${normalized}`)) {
|
|
65
|
+
return `${event.toolName} on ${input.file_path}`;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
continue;
|
|
69
|
+
}
|
|
70
|
+
if (event.toolName === "Bash") {
|
|
71
|
+
const input = event.input;
|
|
72
|
+
if (typeof input?.command === "string" && mutatesPathInBash(input.command, filePath)) {
|
|
73
|
+
return input.command;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
return null;
|
|
78
|
+
}
|
|
79
|
+
export function runFileLifecycleChecks(classifications, events) {
|
|
80
|
+
return classifications.map(({ rule, filePath, polarity }) => {
|
|
81
|
+
const mutation = findMutation(events, filePath);
|
|
82
|
+
if (polarity === "forbid") {
|
|
83
|
+
if (mutation) {
|
|
84
|
+
return {
|
|
85
|
+
ruleId: rule.id,
|
|
86
|
+
ruleTitle: rule.title,
|
|
87
|
+
ruleSource: rule.source,
|
|
88
|
+
status: "FAIL",
|
|
89
|
+
evidence: `"${filePath}" was actually modified: ${mutation.slice(0, 160)}`,
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
return {
|
|
93
|
+
ruleId: rule.id,
|
|
94
|
+
ruleTitle: rule.title,
|
|
95
|
+
ruleSource: rule.source,
|
|
96
|
+
status: "PASS",
|
|
97
|
+
evidence: `"${filePath}" was never written to, deleted, or moved this session (reading it does not count)`,
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
// require: absence is UNCLEAR, not a fabricated FAIL — same reasoning
|
|
101
|
+
// as deterministicChecks.ts's require-polarity handling
|
|
102
|
+
if (mutation) {
|
|
103
|
+
return {
|
|
104
|
+
ruleId: rule.id,
|
|
105
|
+
ruleTitle: rule.title,
|
|
106
|
+
ruleSource: rule.source,
|
|
107
|
+
status: "PASS",
|
|
108
|
+
evidence: `"${filePath}" was updated as required: ${mutation.slice(0, 160)}`,
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
return {
|
|
112
|
+
ruleId: rule.id,
|
|
113
|
+
ruleTitle: rule.title,
|
|
114
|
+
ruleSource: rule.source,
|
|
115
|
+
status: "UNCLEAR",
|
|
116
|
+
evidence: `"${filePath}" was never modified this session — can't tell if the rule didn't apply, or applied and was skipped`,
|
|
117
|
+
};
|
|
118
|
+
});
|
|
119
|
+
}
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
import type { TranscriptEvent, CheckResult } from "../types.js";
|
|
2
|
+
import type { GitBranchPolicyClassification } from "./classify.js";
|
|
3
|
+
export declare function runGitBranchPolicyChecks(classifications: GitBranchPolicyClassification[], events: TranscriptEvent[]): CheckResult[];
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* First real structured-check primitive: parses actual git command
|
|
3
|
+
* arguments instead of searching prose for a branch name as a substring.
|
|
4
|
+
* Real bug this fixes (found 2026-08-30): a rule like "never touch the
|
|
5
|
+
* `demo` branch" previously matched the word "demo" appearing ANYWHERE —
|
|
6
|
+
* a repo name ("acme demo repo"), a directory, an unrelated sentence.
|
|
7
|
+
* This only matches an actual git command whose branch ARGUMENT is
|
|
8
|
+
* exactly the named branch.
|
|
9
|
+
*
|
|
10
|
+
* Known, stated limitation: covers the common real invocations
|
|
11
|
+
* (checkout, switch, branch create/delete/rename, push to a branch) via
|
|
12
|
+
* regex on the command string, not a real git-command-line parser. Things
|
|
13
|
+
* this does NOT reliably cover: `git checkout <ref>` when <ref> could be
|
|
14
|
+
* a file path rather than a branch (ambiguous without actually running
|
|
15
|
+
* git), refspecs with `+`/wildcards, `git rebase --onto <branch>`, and
|
|
16
|
+
* any git GUI/porcelain wrapper that doesn't literally run `git` in Bash.
|
|
17
|
+
* Failing to detect a real violation here is a false negative (UNCLEAR),
|
|
18
|
+
* which is the safe direction — this must never fabricate a FAIL from a
|
|
19
|
+
* command that didn't actually target the named branch.
|
|
20
|
+
*/
|
|
21
|
+
// Split from a single "checkout or switch" pattern: `checkout -b`/`switch
|
|
22
|
+
// -c` CREATE a new branch with that exact name — nobody accidentally
|
|
23
|
+
// creates a branch named exactly the protected name while trying to sync
|
|
24
|
+
// something else, so this is an immediate hit, same as GIT_BRANCH_CREATE.
|
|
25
|
+
// A bare `checkout`/`switch` (no -b/-c) only SWITCHES to an existing
|
|
26
|
+
// branch, which is routine (syncing before branching off) and only
|
|
27
|
+
// becomes a real violation if a commit follows it — see
|
|
28
|
+
// findCheckoutCommitViolation.
|
|
29
|
+
const GIT_CHECKOUT_CREATE = /\bgit\s+(?:checkout\s+-[bB]|switch\s+-c)\s+([^\s-][^\s]*)/;
|
|
30
|
+
const GIT_CHECKOUT_SWITCH_ONLY = /\bgit\s+(?:checkout|switch)\s+(?:--\s+)?([^\s-][^\s]*)/;
|
|
31
|
+
const GIT_BRANCH_CREATE = /\bgit\s+branch\s+(?:-[a-zA-Z]+\s+)?([^\s-][^\s]*)/;
|
|
32
|
+
const GIT_PUSH = /\bgit\s+push\s+(?:\S+\s+)?(?:\+)?(?:[^\s:]+:)?([^\s:]+)\s*$/;
|
|
33
|
+
const GIT_COMMIT = /\bgit\s+commit\b/;
|
|
34
|
+
function extractGitBranchTargets(command) {
|
|
35
|
+
const targets = [];
|
|
36
|
+
for (const regex of [GIT_CHECKOUT_CREATE, GIT_CHECKOUT_SWITCH_ONLY, GIT_BRANCH_CREATE, GIT_PUSH]) {
|
|
37
|
+
const match = command.match(regex);
|
|
38
|
+
if (match)
|
|
39
|
+
targets.push(match[1]);
|
|
40
|
+
}
|
|
41
|
+
return targets;
|
|
42
|
+
}
|
|
43
|
+
function commandFromEvent(event) {
|
|
44
|
+
if (event.kind !== "tool_use" || event.toolName !== "Bash")
|
|
45
|
+
return null;
|
|
46
|
+
const input = event.input;
|
|
47
|
+
return typeof input?.command === "string" ? input.command : null;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Real gap found 2026-08-30, one publish after the first version of this
|
|
51
|
+
* file shipped: the original version treated ANY checkout of the named
|
|
52
|
+
* branch as a violation — but "git checkout sprint && git pull origin
|
|
53
|
+
* sprint" (sync, then branch off) is normal, compliant workflow, not a
|
|
54
|
+
* violation of "never work on the sprint branch." The actual violation is
|
|
55
|
+
* COMMITTING while that branch is checked out, not merely visiting it.
|
|
56
|
+
* Simulates the current checked-out branch across the session in
|
|
57
|
+
* chronological order, and only counts a checkout-based hit when a real
|
|
58
|
+
* `git commit` happens while that branch is current. Push/branch-create
|
|
59
|
+
* targeting the named branch stay immediate hits regardless — pushing
|
|
60
|
+
* straight to a protected branch, or creating/renaming/deleting it, is
|
|
61
|
+
* the violation itself, not something that needs a following commit.
|
|
62
|
+
*/
|
|
63
|
+
function findCheckoutCommitViolation(commands, branchName) {
|
|
64
|
+
let currentBranch = null;
|
|
65
|
+
for (const command of commands) {
|
|
66
|
+
const checkoutMatch = command.match(GIT_CHECKOUT_CREATE) ?? command.match(GIT_CHECKOUT_SWITCH_ONLY);
|
|
67
|
+
if (checkoutMatch) {
|
|
68
|
+
currentBranch = checkoutMatch[1];
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
if (currentBranch === branchName && GIT_COMMIT.test(command)) {
|
|
72
|
+
return command;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return null;
|
|
76
|
+
}
|
|
77
|
+
export function runGitBranchPolicyChecks(classifications, events) {
|
|
78
|
+
const commands = [];
|
|
79
|
+
const allTargets = [];
|
|
80
|
+
for (const event of events) {
|
|
81
|
+
const command = commandFromEvent(event);
|
|
82
|
+
if (!command)
|
|
83
|
+
continue;
|
|
84
|
+
commands.push(command);
|
|
85
|
+
for (const branch of extractGitBranchTargets(command)) {
|
|
86
|
+
allTargets.push({ branch, command });
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return classifications.map(({ rule, branchName, polarity }) => {
|
|
90
|
+
const pushOrCreateHit = allTargets.find((t) => t.branch === branchName &&
|
|
91
|
+
(GIT_PUSH.test(t.command) || GIT_BRANCH_CREATE.test(t.command) || GIT_CHECKOUT_CREATE.test(t.command)));
|
|
92
|
+
const commitViolationCommand = findCheckoutCommitViolation(commands, branchName);
|
|
93
|
+
const hit = pushOrCreateHit ?? (commitViolationCommand ? { branch: branchName, command: commitViolationCommand } : undefined);
|
|
94
|
+
if (polarity === "forbid") {
|
|
95
|
+
if (hit) {
|
|
96
|
+
return {
|
|
97
|
+
ruleId: rule.id,
|
|
98
|
+
ruleTitle: rule.title,
|
|
99
|
+
ruleSource: rule.source,
|
|
100
|
+
status: "FAIL",
|
|
101
|
+
evidence: `a git command actually targeted the "${branchName}" branch: ${hit.command}`,
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
return {
|
|
105
|
+
ruleId: rule.id,
|
|
106
|
+
ruleTitle: rule.title,
|
|
107
|
+
ruleSource: rule.source,
|
|
108
|
+
status: "PASS",
|
|
109
|
+
evidence: `no git command targeted the "${branchName}" branch this session`,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
// require: absence is UNCLEAR, not a fabricated FAIL — same reasoning
|
|
113
|
+
// as deterministicChecks.ts's require-polarity handling
|
|
114
|
+
if (hit) {
|
|
115
|
+
return {
|
|
116
|
+
ruleId: rule.id,
|
|
117
|
+
ruleTitle: rule.title,
|
|
118
|
+
ruleSource: rule.source,
|
|
119
|
+
status: "PASS",
|
|
120
|
+
evidence: `a git command targeted the required "${branchName}" branch: ${hit.command}`,
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
return {
|
|
124
|
+
ruleId: rule.id,
|
|
125
|
+
ruleTitle: rule.title,
|
|
126
|
+
ruleSource: rule.source,
|
|
127
|
+
status: "UNCLEAR",
|
|
128
|
+
evidence: `no git command targeting the "${branchName}" branch appeared this session — can't tell if the rule didn't apply, or applied and was skipped`,
|
|
129
|
+
};
|
|
130
|
+
});
|
|
131
|
+
}
|
package/dist/cli.js
CHANGED
|
@@ -11,6 +11,9 @@ import { loadRules } from "./rules.js";
|
|
|
11
11
|
import { classifyRules } from "./checks/classify.js";
|
|
12
12
|
import { runDeterministicChecks } from "./checks/deterministicChecks.js";
|
|
13
13
|
import { runIfEditThenTestChecks } from "./checks/ifEditThenTest.js";
|
|
14
|
+
import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
|
|
15
|
+
import { runCodeContentChecks } from "./checks/codeContent.js";
|
|
16
|
+
import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
|
|
14
17
|
import { runJudgmentChecks } from "./checks/judgmentChecks.js";
|
|
15
18
|
import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
|
|
16
19
|
import { verifySessionHash } from "./verifyHash.js";
|
|
@@ -72,13 +75,23 @@ async function emailResults(reportText) {
|
|
|
72
75
|
console.log(`\n(--email failed to send: ${result.error} — report above is unaffected)`);
|
|
73
76
|
}
|
|
74
77
|
}
|
|
78
|
+
/**
|
|
79
|
+
* A rule that genuinely requires judgment ("surface bad news first",
|
|
80
|
+
* "write readable code") has no mechanical answer. Reporting that as a
|
|
81
|
+
* deficiency of the tool ("run with --llm") frames the honest answer as a
|
|
82
|
+
* missing feature; it isn't. Deciding a subjective rule was followed is a
|
|
83
|
+
* human call, and saying so plainly is the product working correctly.
|
|
84
|
+
*
|
|
85
|
+
* --llm is offered as what it is — a second opinion from a model, still
|
|
86
|
+
* not a substitute for the reader's judgment.
|
|
87
|
+
*/
|
|
75
88
|
function needsLlmResult(rule) {
|
|
76
89
|
return {
|
|
77
90
|
ruleId: rule.id,
|
|
78
91
|
ruleTitle: rule.title,
|
|
79
92
|
ruleSource: rule.source,
|
|
80
93
|
status: "UNCLEAR",
|
|
81
|
-
evidence: "this rule
|
|
94
|
+
evidence: "NEEDS HUMAN REVIEW — this rule is a judgment call, not something that can be settled by looking at what commands ran. Read the session and decide for yourself. (`--llm` will give you a model's opinion on it, using your own Anthropic key — an opinion, not a verdict.)",
|
|
82
95
|
};
|
|
83
96
|
}
|
|
84
97
|
async function runCheck(markdown, share, email, emailAlways, llm, telemetry, transcriptOverride) {
|
|
@@ -105,10 +118,22 @@ async function runCheck(markdown, share, email, emailAlways, llm, telemetry, tra
|
|
|
105
118
|
const classifications = classifyRules(rules);
|
|
106
119
|
const deterministic = classifications.filter((c) => c.kind === "deterministic");
|
|
107
120
|
const ifEditThenTest = classifications.filter((c) => c.kind === "ifEditThenTest");
|
|
121
|
+
const gitBranchPolicy = classifications.filter((c) => c.kind === "gitBranchPolicy");
|
|
122
|
+
const codeContent = classifications.filter((c) => c.kind === "codeContent");
|
|
123
|
+
const fileLifecycle = classifications.filter((c) => c.kind === "fileLifecycle");
|
|
108
124
|
const judgment = classifications.filter((c) => c.kind === "judgment");
|
|
125
|
+
// Not rules at all — documentation, glossary entries, reference tables.
|
|
126
|
+
// Measured on 40 real public rule files: ~17% of parsed items. Reported
|
|
127
|
+
// as a count so nothing is silently dropped, but never checked, since
|
|
128
|
+
// "did the session violate a directory listing" has no meaningful answer
|
|
129
|
+
// and any coincidental match is pure noise.
|
|
130
|
+
const notARule = classifications.filter((c) => c.kind === "notARule");
|
|
109
131
|
const deterministicResults = [
|
|
110
132
|
...runDeterministicChecks(deterministic, events),
|
|
111
133
|
...runIfEditThenTestChecks(ifEditThenTest, events),
|
|
134
|
+
...runGitBranchPolicyChecks(gitBranchPolicy, events),
|
|
135
|
+
...runCodeContentChecks(codeContent, events),
|
|
136
|
+
...runFileLifecycleChecks(fileLifecycle, events),
|
|
112
137
|
];
|
|
113
138
|
// Deterministic checks run by default, always, with no key — judgment
|
|
114
139
|
// rules only call out to an LLM with an explicit --llm on THIS run, never
|
|
@@ -122,9 +147,12 @@ async function runCheck(markdown, share, email, emailAlways, llm, telemetry, tra
|
|
|
122
147
|
// regardless of --llm.
|
|
123
148
|
const judgmentResults = llm ? await runJudgmentChecks(judgment, events) : judgment.map(({ rule }) => needsLlmResult(rule));
|
|
124
149
|
const results = [...deterministicResults, ...judgmentResults];
|
|
125
|
-
const meta = { sessionFilePath, ruleCount:
|
|
150
|
+
const meta = { sessionFilePath, ruleCount: results.length };
|
|
126
151
|
const reportText = markdown ? generateMarkdownReport(results, meta) : generateReport(results, meta);
|
|
127
152
|
console.log(reportText);
|
|
153
|
+
if (notARule.length > 0) {
|
|
154
|
+
console.log(`\n(${notARule.length} item${notARule.length === 1 ? "" : "s"} in your rules file ${notARule.length === 1 ? "is" : "are"} documentation, not a rule — directory listings, reference tables, examples. Not checked, because there's nothing to check.)`);
|
|
155
|
+
}
|
|
128
156
|
appendHistory(results, sessionFilePath);
|
|
129
157
|
if (share) {
|
|
130
158
|
await shareResults(results);
|
package/dist/rules.js
CHANGED
|
@@ -1,7 +1,47 @@
|
|
|
1
1
|
import { homedir } from "node:os";
|
|
2
|
-
import { join } from "node:path";
|
|
2
|
+
import { dirname, join, parse } from "node:path";
|
|
3
|
+
import { existsSync } from "node:fs";
|
|
3
4
|
import { parseClaudeMd } from "./parsers/claudeMdParser.js";
|
|
4
5
|
import { findClaudeHomeDirNames } from "./parsers/transcriptParser.js";
|
|
6
|
+
const RULE_FILE_NAMES = ["CLAUDE.md", "AGENTS.md"];
|
|
7
|
+
/**
|
|
8
|
+
* Walks from the working directory up toward the repository root,
|
|
9
|
+
* collecting rules files at every level.
|
|
10
|
+
*
|
|
11
|
+
* Real gap: rules were only read from the exact directory the command ran
|
|
12
|
+
* in. Claude Code itself applies a rules file to everything beneath it, so
|
|
13
|
+
* in a monorepo the root CLAUDE.md governs `packages/api/` — but running
|
|
14
|
+
* the check inside that package silently missed it, reporting on a subset
|
|
15
|
+
* of the rules that actually applied and never saying so.
|
|
16
|
+
*
|
|
17
|
+
* Stops at the repository root (a directory containing `.git`) so an
|
|
18
|
+
* unrelated rules file further up the filesystem — in a parent workspace,
|
|
19
|
+
* or the home directory — is never pulled into an unrelated project.
|
|
20
|
+
* Global rules are handled separately, deliberately, below.
|
|
21
|
+
*/
|
|
22
|
+
function findProjectRuleFiles(cwd) {
|
|
23
|
+
const found = [];
|
|
24
|
+
const { root } = parse(cwd);
|
|
25
|
+
const home = homedir();
|
|
26
|
+
let dir = cwd;
|
|
27
|
+
for (;;) {
|
|
28
|
+
for (const name of RULE_FILE_NAMES) {
|
|
29
|
+
const p = join(dir, name);
|
|
30
|
+
if (existsSync(p))
|
|
31
|
+
found.push(p);
|
|
32
|
+
}
|
|
33
|
+
// stop AT the repo root (inclusive) — its rules do apply
|
|
34
|
+
if (existsSync(join(dir, ".git")))
|
|
35
|
+
break;
|
|
36
|
+
if (dir === root || dir === home)
|
|
37
|
+
break;
|
|
38
|
+
const parent = dirname(dir);
|
|
39
|
+
if (parent === dir)
|
|
40
|
+
break;
|
|
41
|
+
dir = parent;
|
|
42
|
+
}
|
|
43
|
+
return found;
|
|
44
|
+
}
|
|
5
45
|
/**
|
|
6
46
|
* Global rules come from every .claude*-prefixed home dir found, not just
|
|
7
47
|
* ~/.claude — a hosted/enterprise Claude Code variant can keep its own
|
|
@@ -11,9 +51,8 @@ import { findClaudeHomeDirNames } from "./parsers/transcriptParser.js";
|
|
|
11
51
|
*/
|
|
12
52
|
export function loadRules(cwd) {
|
|
13
53
|
const rules = findClaudeHomeDirNames().flatMap((dirName) => parseClaudeMd(join(homedir(), dirName, "CLAUDE.md"), "global"));
|
|
14
|
-
for (const
|
|
15
|
-
|
|
16
|
-
rules.push(...parseClaudeMd(projectPath, "project"));
|
|
54
|
+
for (const path of findProjectRuleFiles(cwd)) {
|
|
55
|
+
rules.push(...parseClaudeMd(path, "project"));
|
|
17
56
|
}
|
|
18
57
|
return rules;
|
|
19
58
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "rulereceipt",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.18",
|
|
4
4
|
"description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -18,7 +18,8 @@
|
|
|
18
18
|
"dev": "tsx src/cli.ts",
|
|
19
19
|
"test": "vitest run",
|
|
20
20
|
"test:watch": "vitest",
|
|
21
|
-
"lint": "eslint src tests"
|
|
21
|
+
"lint": "eslint src tests",
|
|
22
|
+
"prepublishOnly": "npm run build && npm test"
|
|
22
23
|
},
|
|
23
24
|
"engines": {
|
|
24
25
|
"node": ">=18"
|
|
@@ -36,6 +37,6 @@
|
|
|
36
37
|
"tsx": "^4.19.0",
|
|
37
38
|
"typescript": "^5.6.0",
|
|
38
39
|
"typescript-eslint": "^8.67.0",
|
|
39
|
-
"vitest": "^
|
|
40
|
+
"vitest": "^4.1.11"
|
|
40
41
|
}
|
|
41
42
|
}
|