rulereceipt 0.1.10 → 0.1.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +55 -20
- package/dist/checks/classify.d.ts +69 -1
- package/dist/checks/classify.js +134 -6
- package/dist/checks/codeContent.d.ts +3 -0
- package/dist/checks/codeContent.js +89 -0
- package/dist/checks/deterministicChecks.js +15 -2
- package/dist/checks/fileLifecycle.d.ts +3 -0
- package/dist/checks/fileLifecycle.js +119 -0
- package/dist/checks/gitBranchPolicy.d.ts +3 -0
- package/dist/checks/gitBranchPolicy.js +131 -0
- package/dist/cli.js +77 -6
- package/dist/report/generateHtmlReport.d.ts +40 -0
- package/dist/report/generateHtmlReport.js +208 -0
- package/dist/rules.js +43 -4
- package/package.json +22 -3
package/README.md
CHANGED
|
@@ -23,29 +23,55 @@ only (`rulereceipt check`). No automatic hooks, ever, in v1.
|
|
|
23
23
|
|
|
24
24
|
## Status
|
|
25
25
|
|
|
26
|
-
Published and live on npm
|
|
27
|
-
passing).
|
|
26
|
+
Published and live on npm, actively developed.
|
|
28
27
|
|
|
29
28
|
## How it works
|
|
30
29
|
|
|
31
30
|
1. Reads your CLAUDE.md / AGENTS.md and extracts individual rules —
|
|
32
|
-
from the current project directory and
|
|
33
|
-
2. Reads your most recent Claude Code session transcript
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
31
|
+
from the current project directory and your global rules file.
|
|
32
|
+
2. Reads your most recent Claude Code session transcript, wherever Claude
|
|
33
|
+
Code stored it — including hosted or enterprise variants that use a
|
|
34
|
+
different directory.
|
|
35
|
+
3. Routes each rule to the narrowest check that can actually answer it:
|
|
36
|
+
- **Structured checks** read what the session really did — an actual
|
|
37
|
+
git command's branch argument, actual file edits, actual file
|
|
38
|
+
operations. These are the only checks that report a confident FAIL,
|
|
39
|
+
because they can tell an action from a mention.
|
|
40
|
+
- **Literal checks** look for a specific string named in the rule.
|
|
41
|
+
Absence is real evidence, so a clean session PASSes. A match reports
|
|
42
|
+
UNCLEAR with the text quoted, because a text match alone cannot
|
|
43
|
+
distinguish doing the forbidden thing from grepping for it, quoting
|
|
44
|
+
it, or naming it in a commit message.
|
|
45
|
+
- **Judgment** rules need real understanding (e.g. "surface bad news
|
|
46
|
+
first"). With `--llm` each is graded individually using *your own*
|
|
47
|
+
Claude key; without it they report UNCLEAR rather than guessing.
|
|
48
|
+
- Lines containing no instruction at all — directory listings,
|
|
49
|
+
reference tables, examples — aren't rules, and are reported as such
|
|
50
|
+
instead of being checked.
|
|
51
|
+
4. Prints a report — terminal table by default, `--markdown` for pasting
|
|
52
|
+
into a PR or Slack message, or `--html` for a shareable single file —
|
|
53
|
+
showing what passed, what failed, and a quoted line of evidence for
|
|
54
|
+
each. Every report includes a SHA-256 hash of the session file it
|
|
55
|
+
checked, so anyone with that file can confirm the report describes
|
|
56
|
+
that exact file. (It proves the report matches the file, not that the
|
|
57
|
+
file is an unmodified record — see SECURITY.md.)
|
|
58
|
+
|
|
59
|
+
## Sharing a report
|
|
60
|
+
|
|
61
|
+
`rulereceipt check --html` writes one self-contained HTML file. No
|
|
62
|
+
external requests, no CDN, no fonts to fetch — so it opens correctly from
|
|
63
|
+
an email attachment, offline, years later, and prints cleanly to PDF.
|
|
64
|
+
|
|
65
|
+
It leads with what wasn't followed rather than burying it under passes,
|
|
66
|
+
quotes the evidence for each result, and states plainly what it does not
|
|
67
|
+
establish: it covers one session, it is not a compliance certification,
|
|
68
|
+
and rules needing judgment are reported as needing review rather than
|
|
69
|
+
guessed at. The session fingerprint and a runnable `rulereceipt verify`
|
|
70
|
+
command are printed on the report itself, so the person receiving it can
|
|
71
|
+
independently confirm it describes the session it claims to.
|
|
72
|
+
|
|
73
|
+
Nothing is uploaded. The file is written to your working directory and
|
|
74
|
+
goes wherever you choose to send it.
|
|
49
75
|
|
|
50
76
|
## Try it with zero setup
|
|
51
77
|
|
|
@@ -61,7 +87,16 @@ report so you can see the output shape immediately.
|
|
|
61
87
|
```bash
|
|
62
88
|
rulereceipt check # check the latest session in this project
|
|
63
89
|
rulereceipt check --markdown # same, formatted for pasting into a PR/Slack
|
|
64
|
-
rulereceipt check --
|
|
90
|
+
rulereceipt check --html # write a shareable single-file HTML report you can send
|
|
91
|
+
rulereceipt check --html report.html # ...to a specific path
|
|
92
|
+
rulereceipt check --llm # opt-in: grade judgment rules with your own Claude key
|
|
93
|
+
rulereceipt check --share # opt-in: send anonymous pass/fail/unclear counts
|
|
94
|
+
rulereceipt check --telemetry # opt-in: send one random per-machine ID
|
|
95
|
+
rulereceipt check --transcript <path> # check a specific session file
|
|
96
|
+
rulereceipt doctor # list hooks/auto-run tasks configured on this machine
|
|
97
|
+
rulereceipt lint # find contradictions between CLAUDE.md and AGENTS.md
|
|
98
|
+
rulereceipt digest # summarise recent checks; --email to send it
|
|
99
|
+
rulereceipt config # set up email sending (stays on your machine)
|
|
65
100
|
rulereceipt demo # sample output, no setup needed
|
|
66
101
|
rulereceipt demo --markdown
|
|
67
102
|
rulereceipt --version # print the installed version
|
|
@@ -19,7 +19,75 @@ export interface JudgmentClassification {
|
|
|
19
19
|
kind: "judgment";
|
|
20
20
|
rule: Rule;
|
|
21
21
|
}
|
|
22
|
-
|
|
22
|
+
/**
|
|
23
|
+
* First real structured-check primitive (2026-08-30), replacing keyword
|
|
24
|
+
* search for one whole rule category: git branch policy. A rule naming a
|
|
25
|
+
* branch (e.g. "never touch the `demo` branch") was previously checked by
|
|
26
|
+
* searching for the word "demo" ANYWHERE in the transcript — matching a
|
|
27
|
+
* repo name, a directory, a sentence, anything. This routes instead to a
|
|
28
|
+
* real parser (gitBranchPolicy.ts) that reads actual git command
|
|
29
|
+
* arguments and checks the literal branch name, not a substring search.
|
|
30
|
+
*/
|
|
31
|
+
export interface GitBranchPolicyClassification {
|
|
32
|
+
kind: "gitBranchPolicy";
|
|
33
|
+
rule: Rule;
|
|
34
|
+
branchName: string;
|
|
35
|
+
polarity: DeterministicPolarity;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Second structured-check primitive (2026-08-30): code content, for rules
|
|
39
|
+
* naming an actual code construct — a function/method call like `print(`
|
|
40
|
+
* or `analytics.track(` — rather than a CLI command. Real false-positive
|
|
41
|
+
* this fixes: even after excluding tool_result (see deterministicChecks.ts),
|
|
42
|
+
* a rule like "no `print(` statements" still matched when the agent's OWN
|
|
43
|
+
* Bash command merely MENTIONED the pattern as an argument (e.g. grepping
|
|
44
|
+
* for it), because generic deterministic matching scans the whole
|
|
45
|
+
* stringified tool_use input, commands included. This routes instead to
|
|
46
|
+
* codeContent.ts, which only looks at the actual content of real file
|
|
47
|
+
* edits (Write/Edit/NotebookEdit) — never a Bash command string, never
|
|
48
|
+
* prose, never a search argument.
|
|
49
|
+
*/
|
|
50
|
+
export interface CodeContentClassification {
|
|
51
|
+
kind: "codeContent";
|
|
52
|
+
rule: Rule;
|
|
53
|
+
patterns: string[];
|
|
54
|
+
polarity: DeterministicPolarity;
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* Third structured-check primitive (2026-08-30): file lifecycle, for
|
|
58
|
+
* rules protecting a specific file ("never modify `.claude/settings.json`",
|
|
59
|
+
* "don't delete `config.yaml`"). Real false-positive this fixes: the path
|
|
60
|
+
* was flagged as touched when the agent merely READ it (`cat
|
|
61
|
+
* .claude/settings.json` to verify its contents) — reading a protected
|
|
62
|
+
* file is not modifying it. Routes to fileLifecycle.ts, which only counts
|
|
63
|
+
* real mutations: Write/Edit on that path, or a Bash rm/mv/truncate-style
|
|
64
|
+
* command targeting it.
|
|
65
|
+
*/
|
|
66
|
+
export interface FileLifecycleClassification {
|
|
67
|
+
kind: "fileLifecycle";
|
|
68
|
+
rule: Rule;
|
|
69
|
+
filePath: string;
|
|
70
|
+
polarity: DeterministicPolarity;
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Not every line in a real CLAUDE.md is a rule. Measured against 40 real
|
|
74
|
+
* public rule files (1,441 parsed items, 2026-08-30): only ~20% contained
|
|
75
|
+
* any directive language at all. The other ~80% is documentation —
|
|
76
|
+
* directory listings ("`forge/llm/` - Multi-provider LLM integrations"),
|
|
77
|
+
* import examples, model-name tables, glob-syntax references. 521 of those
|
|
78
|
+
* were getting a literal keyword check run against them, which is the
|
|
79
|
+
* single largest source of false positives: checking whether a session
|
|
80
|
+
* "violated" a directory listing is meaningless, and any coincidental
|
|
81
|
+
* match is noise reported as a finding.
|
|
82
|
+
*
|
|
83
|
+
* These are reported as N/A and excluded from pass/fail entirely — not
|
|
84
|
+
* sent to the LLM either, since there's no rule to judge.
|
|
85
|
+
*/
|
|
86
|
+
export interface NotARuleClassification {
|
|
87
|
+
kind: "notARule";
|
|
88
|
+
rule: Rule;
|
|
89
|
+
}
|
|
90
|
+
export type Classification = DeterministicClassification | IfEditThenTestClassification | GitBranchPolicyClassification | CodeContentClassification | FileLifecycleClassification | NotARuleClassification | JudgmentClassification;
|
|
23
91
|
/**
|
|
24
92
|
* A rule is only treated as deterministic when it names a specific,
|
|
25
93
|
* literal, checkable token (a CLI flag, a command, an exact string) in
|
package/dist/checks/classify.js
CHANGED
|
@@ -1,16 +1,81 @@
|
|
|
1
|
+
// Normative language — the thing that makes a line a rule rather than a
|
|
2
|
+
// description. Deliberately broad on modals AND imperative verbs, because
|
|
3
|
+
// a wrongly-excluded rule is a silent miss.
|
|
4
|
+
const DIRECTIVE_LANGUAGE = /\b(never|always|must|should|shall|do not|don't|dont|cannot|can't|required?|requires|ensure|avoid|prefer|forbidden|prohibited|only|make sure|be sure|need|needs|needed|need to|has to|have to|expected to|responsible for)\b/i;
|
|
5
|
+
/**
|
|
6
|
+
* Imperative instruction — a bare command verb starting a clause ("Use
|
|
7
|
+
* `gh pr merge`", "Run the tests first", "Keep functions small"). This is
|
|
8
|
+
* the other way a real rule is written when it doesn't use a modal.
|
|
9
|
+
*
|
|
10
|
+
* Anchored to a clause start (line start, or after sentence/bullet
|
|
11
|
+
* punctuation) on purpose: the same verbs appear mid-sentence in pure
|
|
12
|
+
* documentation ("the CLI can run migrations"), where they describe a
|
|
13
|
+
* capability rather than instruct the agent.
|
|
14
|
+
*/
|
|
15
|
+
const IMPERATIVE_INSTRUCTION = /(?:^|[.;:!?]\s+|^\s*[-*+]\s*|\n\s*[-*+]\s*)(use|run|keep|write|add|remove|delete|check|verify|test|commit|document|update|create|follow|apply|include|exclude|handle|validate|escape|sanitize|log|report|raise|throw|return|call|invoke|split|group|sort|name|place|put|store|read|load|save|close|open|start|stop|restart|install|build|deploy|review|refactor|rename|move|copy|merge|rebase|squash|tag|branch|push|pull|fetch|clone|stage|stash|lead|state|explain|describe|list|show|surface|flag|mark|label|note|treat|assume|confirm|ask|wait|stick|limit|cap|batch|cache|mock|stub|assert|expect|measure|quantify|label)\b/i;
|
|
16
|
+
/**
|
|
17
|
+
* Deliberately inverted: tests for the presence of a DIRECTIVE, never for
|
|
18
|
+
* the shape of documentation.
|
|
19
|
+
*
|
|
20
|
+
* The first version of this enumerated documentation shapes (backtick
|
|
21
|
+
* glossary, bold-term definition, arrow mapping, label rows, "Reference:"
|
|
22
|
+
* prefixes). That approach cannot work: every new rules-file convention is
|
|
23
|
+
* a new shape, so the list grows forever and is always one format behind —
|
|
24
|
+
* measurably so, since `notARule` coverage fell from 17.4% on a 40-file
|
|
25
|
+
* sample to 7.1% on a 658-file one purely because the bigger corpus used
|
|
26
|
+
* shapes the list didn't have yet.
|
|
27
|
+
*
|
|
28
|
+
* "Does this contain an instruction" is a bounded question about English —
|
|
29
|
+
* modal verbs plus the closed class of imperative command verbs — and it
|
|
30
|
+
* doesn't change when someone invents a new markdown convention. A line
|
|
31
|
+
* with no instruction in it has nothing to check compliance against,
|
|
32
|
+
* whatever its punctuation.
|
|
33
|
+
*/
|
|
34
|
+
function isNotARule(rule) {
|
|
35
|
+
const combined = `${rule.title} ${rule.text}`;
|
|
36
|
+
if (DIRECTIVE_LANGUAGE.test(combined))
|
|
37
|
+
return false;
|
|
38
|
+
// Title and text are tested SEPARATELY: the imperative pattern is
|
|
39
|
+
// anchored to a clause start, and concatenating them pushes the text's
|
|
40
|
+
// opening verb into mid-string where the anchor can never match. That
|
|
41
|
+
// bug silently classified real rules ("Use `npm` for this project")
|
|
42
|
+
// as non-rules — caught by an existing test, not by inspection.
|
|
43
|
+
return !IMPERATIVE_INSTRUCTION.test(rule.title) && !IMPERATIVE_INSTRUCTION.test(rule.text);
|
|
44
|
+
}
|
|
45
|
+
const BRANCH_WORD = /\bbranch\b/i;
|
|
46
|
+
// A function/method-call shape ("print(", "analytics.track(") is a strong,
|
|
47
|
+
// simple signal that a backtick literal names actual CODE, not a CLI
|
|
48
|
+
// command or flag ("git push --force", "npm test" never look like this).
|
|
49
|
+
const CODE_CONSTRUCT_PATTERN = /\(/;
|
|
50
|
+
// A file-path shape: a known config/source extension, or a path with a
|
|
51
|
+
// directory separator. Deliberately requires no spaces — a real path
|
|
52
|
+
// literal ("`.claude/settings.json`", "`config.yaml`") never has one,
|
|
53
|
+
// while a command that happens to contain a slash ("`git push --force`")
|
|
54
|
+
// does. Checked AFTER the code-construct test, so "foo(" never lands here.
|
|
55
|
+
const FILE_PATH_PATTERN = /^[^\s]*(\.(json|ya?ml|toml|md|env|lock|ini|cfg|conf|xml|txt|js|ts|py|rb|go|rs|sh)$|\/)/i;
|
|
56
|
+
// Only route to fileLifecycle when the rule is actually about touching the
|
|
57
|
+
// file, not merely mentioning one (e.g. "read `config.yaml` before
|
|
58
|
+
// starting" names a path but isn't a protection rule).
|
|
59
|
+
const FILE_MUTATION_INTENT = /\b(modif|chang|edit|delet|remov|overwrit|touch|writ|creat|rename|mov)\w*\b/i;
|
|
1
60
|
// Catches rules like "add tests for every change" or "every new function
|
|
2
61
|
// needs a test" - no literal backtick token to pattern-match, so without
|
|
3
62
|
// this they'd fall all the way through to judgment (an LLM call) even
|
|
4
63
|
// though they're actually structurally checkable: did a production file
|
|
5
64
|
// get edited without a corresponding test file also being touched.
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
//
|
|
9
|
-
|
|
10
|
-
|
|
65
|
+
//
|
|
66
|
+
// Real false-positive found 2026-08-30 on an actual session: the earlier
|
|
67
|
+
// version of this heuristic (bare "test" word + any require-signal word,
|
|
68
|
+
// independently anywhere in the text) misclassified "Tests must be able
|
|
69
|
+
// to fail" - a rule about test QUALITY (a test must be demonstrated to
|
|
70
|
+
// fail on bad input before it counts) - as an edit-implies-test rule,
|
|
71
|
+
// producing nonsense like "you edited .env but no test file was touched"
|
|
72
|
+
// for a rule that was never about that at all. Fixed by requiring a
|
|
73
|
+
// specific phrase shape - a create/add/need verb close to the word
|
|
74
|
+
// "test" - not just independent word presence anywhere in the text.
|
|
75
|
+
const EDIT_IMPLIES_TEST_PHRASE = /\b(needs?\s+(a\s+)?(corresponding\s+)?test|add(?:ing|ed)?\s+tests?|write\s+tests?|include\s+tests?|corresponding\s+test|test\s+coverage\s+for)\b/i;
|
|
11
76
|
function isEditImpliesTestRule(rule) {
|
|
12
77
|
const text = `${rule.title} ${rule.text}`;
|
|
13
|
-
return
|
|
78
|
+
return EDIT_IMPLIES_TEST_PHRASE.test(text);
|
|
14
79
|
}
|
|
15
80
|
const BACKTICK_TOKEN = /`([^`]+)`/g;
|
|
16
81
|
// Keyword signal near the rule text that this is a mandatory action, not a
|
|
@@ -21,6 +86,44 @@ const BACKTICK_TOKEN = /`([^`]+)`/g;
|
|
|
21
86
|
// only an additive one for rules that clearly ask for a required action.
|
|
22
87
|
const REQUIRE_SIGNAL = /\b(always|must|required|require|ensure|need to)\b/i;
|
|
23
88
|
const FORBID_SIGNAL = /\b(never|don't|do not|forbidden|banned|must not)\b/i;
|
|
89
|
+
/**
|
|
90
|
+
* A rule can prohibit one thing AND prescribe another in the same breath:
|
|
91
|
+
* "NEVER squash when merging PRs. Use `gh pr merge --merge --admin`".
|
|
92
|
+
* Literal pattern matching cannot handle this — it extracts `--merge` from
|
|
93
|
+
* the PRESCRIBED command, then applies the rule's forbid polarity to it,
|
|
94
|
+
* so doing exactly what the rule demands gets reported as violating it.
|
|
95
|
+
* Real false positive found 2026-08-30 on a real rules file.
|
|
96
|
+
*
|
|
97
|
+
* There's no honest way to split which literal belongs to which half by
|
|
98
|
+
* pattern alone, so a mixed-polarity rule goes to judgment instead of
|
|
99
|
+
* being guessed at. Better an "needs --llm" than a confident wrong FAIL.
|
|
100
|
+
*/
|
|
101
|
+
// Prescriptive language beyond the modal verbs in REQUIRE_SIGNAL. A rule
|
|
102
|
+
// often prescribes with a bare imperative ("Use `gh pr merge --merge`")
|
|
103
|
+
// rather than "you must use" — and it's exactly those imperative clauses
|
|
104
|
+
// whose literals get misattributed to the prohibiting half.
|
|
105
|
+
const PRESCRIPTIVE_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to)\b/i;
|
|
106
|
+
function hasMixedPolarity(rule) {
|
|
107
|
+
const text = `${rule.title} ${rule.text}`;
|
|
108
|
+
if (!FORBID_SIGNAL.test(text))
|
|
109
|
+
return false;
|
|
110
|
+
// The prescription must live in a DIFFERENT clause than the prohibition.
|
|
111
|
+
// "Never run `git push --force`" is one clause: "run" belongs to the
|
|
112
|
+
// forbidden action, and treating that as mixed polarity would send the
|
|
113
|
+
// most basic literal rule there is to the LLM. "NEVER squash when
|
|
114
|
+
// merging PRs. Use `gh pr merge --merge`" is two clauses, and only the
|
|
115
|
+
// second one's literals would be misattributed.
|
|
116
|
+
//
|
|
117
|
+
// Code spans are masked before splitting. Real bug caught by the
|
|
118
|
+
// end-to-end violation tests: "Never leave a `console.log(` call in
|
|
119
|
+
// committed code" was split on the period INSIDE the code span, leaving
|
|
120
|
+
// a fragment ("log(` call in committed code") with no forbid word but a
|
|
121
|
+
// prescriptive one, so a plain prohibition was misread as mixed
|
|
122
|
+
// polarity and sent to judgment instead of being checked.
|
|
123
|
+
const masked = text.replace(/`[^`]*`/g, (m) => "`" + "x".repeat(Math.max(m.length - 2, 0)) + "`");
|
|
124
|
+
const clauses = masked.split(/[.;\n]|(?:\s+-\s+)/).filter((c) => c.trim().length > 0);
|
|
125
|
+
return clauses.some((clause) => !FORBID_SIGNAL.test(clause) && (REQUIRE_SIGNAL.test(clause) || PRESCRIPTIVE_VERB.test(clause)));
|
|
126
|
+
}
|
|
24
127
|
function detectPolarity(rule) {
|
|
25
128
|
const text = `${rule.title} ${rule.text}`;
|
|
26
129
|
// An explicit forbid word anywhere wins over a require word — "you must
|
|
@@ -39,6 +142,11 @@ function detectPolarity(rule) {
|
|
|
39
142
|
* guessing it's safe to pattern-match.
|
|
40
143
|
*/
|
|
41
144
|
export function classifyRule(rule) {
|
|
145
|
+
// Checked first: if this isn't a rule at all, no check of any kind
|
|
146
|
+
// should run against it — not a keyword match, not an LLM call.
|
|
147
|
+
if (isNotARule(rule)) {
|
|
148
|
+
return { kind: "notARule", rule };
|
|
149
|
+
}
|
|
42
150
|
const patterns = new Set();
|
|
43
151
|
for (const match of rule.text.matchAll(BACKTICK_TOKEN)) {
|
|
44
152
|
const token = match[1].trim();
|
|
@@ -50,12 +158,32 @@ export function classifyRule(rule) {
|
|
|
50
158
|
if (token.length > 0)
|
|
51
159
|
patterns.add(token);
|
|
52
160
|
}
|
|
161
|
+
// A rule that both forbids and prescribes can't be checked by literal
|
|
162
|
+
// matching without misattributing one half's tokens to the other.
|
|
163
|
+
if (patterns.size > 0 && hasMixedPolarity(rule)) {
|
|
164
|
+
return { kind: "judgment", rule };
|
|
165
|
+
}
|
|
53
166
|
if (patterns.size === 0) {
|
|
54
167
|
if (isEditImpliesTestRule(rule)) {
|
|
55
168
|
return { kind: "ifEditThenTest", rule };
|
|
56
169
|
}
|
|
57
170
|
return { kind: "judgment", rule };
|
|
58
171
|
}
|
|
172
|
+
const text = `${rule.title} ${rule.text}`;
|
|
173
|
+
if (BRANCH_WORD.test(text)) {
|
|
174
|
+
// first backtick literal is treated as the branch name — real rules
|
|
175
|
+
// this targets name exactly one branch ("the `demo` branch", "never
|
|
176
|
+
// push to `main`"), not a set of them
|
|
177
|
+
const [branchName] = patterns;
|
|
178
|
+
return { kind: "gitBranchPolicy", rule, branchName, polarity: detectPolarity(rule) };
|
|
179
|
+
}
|
|
180
|
+
if ([...patterns].some((p) => CODE_CONSTRUCT_PATTERN.test(p))) {
|
|
181
|
+
return { kind: "codeContent", rule, patterns: [...patterns], polarity: detectPolarity(rule) };
|
|
182
|
+
}
|
|
183
|
+
const filePath = [...patterns].find((p) => FILE_PATH_PATTERN.test(p));
|
|
184
|
+
if (filePath && FILE_MUTATION_INTENT.test(text)) {
|
|
185
|
+
return { kind: "fileLifecycle", rule, filePath, polarity: detectPolarity(rule) };
|
|
186
|
+
}
|
|
59
187
|
return { kind: "deterministic", rule, patterns: [...patterns], polarity: detectPolarity(rule) };
|
|
60
188
|
}
|
|
61
189
|
export function classifyRules(rules) {
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Second structured-check primitive: only scans the actual content of
|
|
3
|
+
* real file edits for a code-construct pattern (e.g. `print(`,
|
|
4
|
+
* `analytics.track(`) — never a Bash command string, never prose, never
|
|
5
|
+
* tool_result. Real false-positive this fixes (found 2026-08-30, on the
|
|
6
|
+
* same real session as the git-branch bug): even after excluding
|
|
7
|
+
* tool_result from the generic deterministic check, a rule like "no
|
|
8
|
+
* `print(` statements" still matched because the agent's own Bash
|
|
9
|
+
* command MENTIONED the pattern as a search argument (e.g. `grep -rn
|
|
10
|
+
* "print(" src/`) — no print statement was ever written into a file.
|
|
11
|
+
*
|
|
12
|
+
* Known, stated limitation: `content`/`new_string` are the confirmed
|
|
13
|
+
* real field names for Write/Edit tool_use input; NotebookEdit's field
|
|
14
|
+
* name is included as best-effort (not independently confirmed against a
|
|
15
|
+
* real NotebookEdit transcript event before shipping this) — a session
|
|
16
|
+
* that only writes matching code via NotebookEdit could under-report,
|
|
17
|
+
* which fails toward UNCLEAR/PASS, not a fabricated FAIL.
|
|
18
|
+
*/
|
|
19
|
+
function editedContentFromEvent(event) {
|
|
20
|
+
if (event.kind !== "tool_use")
|
|
21
|
+
return null;
|
|
22
|
+
const input = event.input;
|
|
23
|
+
if (event.toolName === "Write" && typeof input?.content === "string")
|
|
24
|
+
return input.content;
|
|
25
|
+
if (event.toolName === "Edit" && typeof input?.new_string === "string")
|
|
26
|
+
return input.new_string;
|
|
27
|
+
if (event.toolName === "NotebookEdit" && typeof input?.new_source === "string")
|
|
28
|
+
return input.new_source;
|
|
29
|
+
return null;
|
|
30
|
+
}
|
|
31
|
+
export function runCodeContentChecks(classifications, events) {
|
|
32
|
+
const editedContents = [];
|
|
33
|
+
for (const event of events) {
|
|
34
|
+
const content = editedContentFromEvent(event);
|
|
35
|
+
if (content)
|
|
36
|
+
editedContents.push(content);
|
|
37
|
+
}
|
|
38
|
+
return classifications.map(({ rule, patterns, polarity }) => {
|
|
39
|
+
let foundPattern;
|
|
40
|
+
let foundContent;
|
|
41
|
+
for (const content of editedContents) {
|
|
42
|
+
for (const pattern of patterns) {
|
|
43
|
+
if (content.includes(pattern)) {
|
|
44
|
+
foundPattern = pattern;
|
|
45
|
+
foundContent = content;
|
|
46
|
+
break;
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
if (foundPattern)
|
|
50
|
+
break;
|
|
51
|
+
}
|
|
52
|
+
if (polarity === "forbid") {
|
|
53
|
+
if (foundPattern && foundContent) {
|
|
54
|
+
return {
|
|
55
|
+
ruleId: rule.id,
|
|
56
|
+
ruleTitle: rule.title,
|
|
57
|
+
ruleSource: rule.source,
|
|
58
|
+
status: "FAIL",
|
|
59
|
+
evidence: `found "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`,
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
return {
|
|
63
|
+
ruleId: rule.id,
|
|
64
|
+
ruleTitle: rule.title,
|
|
65
|
+
ruleSource: rule.source,
|
|
66
|
+
status: "PASS",
|
|
67
|
+
evidence: `no file edit actually contained ${patterns.map((p) => `"${p}"`).join(" or ")} this session`,
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
// require: absence is UNCLEAR, not a fabricated FAIL — same reasoning
|
|
71
|
+
// as deterministicChecks.ts's require-polarity handling
|
|
72
|
+
if (foundPattern && foundContent) {
|
|
73
|
+
return {
|
|
74
|
+
ruleId: rule.id,
|
|
75
|
+
ruleTitle: rule.title,
|
|
76
|
+
ruleSource: rule.source,
|
|
77
|
+
status: "PASS",
|
|
78
|
+
evidence: `found required "${foundPattern}" actually written into a file: ${foundContent.slice(0, 160)}`,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
return {
|
|
82
|
+
ruleId: rule.id,
|
|
83
|
+
ruleTitle: rule.title,
|
|
84
|
+
ruleSource: rule.source,
|
|
85
|
+
status: "UNCLEAR",
|
|
86
|
+
evidence: `no file edit contained the required ${patterns.map((p) => `"${p}"`).join(" or ")} this session — can't tell if the rule didn't apply, or applied and was skipped`,
|
|
87
|
+
};
|
|
88
|
+
});
|
|
89
|
+
}
|
|
@@ -70,12 +70,25 @@ export function runDeterministicChecks(classifications, events) {
|
|
|
70
70
|
if (polarity === "forbid") {
|
|
71
71
|
if (foundEvent && foundPattern) {
|
|
72
72
|
const haystack = eventSearchText(foundEvent);
|
|
73
|
+
// Deliberately UNCLEAR, never FAIL. A bare literal match proves
|
|
74
|
+
// the string appeared somewhere; it cannot prove the agent DID
|
|
75
|
+
// the forbidden thing. The same match is produced by grepping
|
|
76
|
+
// for the pattern, quoting it in an explanation, or naming it in
|
|
77
|
+
// a commit message — all compliant. Every false positive found
|
|
78
|
+
// on 2026-08-30 was this exact confusion, and no amount of
|
|
79
|
+
// pattern tuning fixes it, because the information needed to
|
|
80
|
+
// tell action from mention is not in the string.
|
|
81
|
+
//
|
|
82
|
+
// Confident FAILs come only from the structured primitives
|
|
83
|
+
// (gitBranchPolicy, codeContent, fileLifecycle), which read what
|
|
84
|
+
// the agent actually executed or wrote. This path reports the
|
|
85
|
+
// evidence and says plainly that it can't confirm a violation.
|
|
73
86
|
return {
|
|
74
87
|
ruleId: rule.id,
|
|
75
88
|
ruleTitle: rule.title,
|
|
76
89
|
ruleSource: rule.source,
|
|
77
|
-
status: "
|
|
78
|
-
evidence: `
|
|
90
|
+
status: "UNCLEAR",
|
|
91
|
+
evidence: `"${foundPattern}" appears in a ${foundEvent.kind === "tool_use" ? foundEvent.toolName + " call" : foundEvent.kind}, but a text match alone can't tell an actual violation from a mention (a search for it, a quote, an explanation) — needs a human look: ${haystack.slice(0, 160)}`,
|
|
79
92
|
};
|
|
80
93
|
}
|
|
81
94
|
return {
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Third structured-check primitive: only counts real MUTATIONS of a
|
|
3
|
+
* protected file, never reads of it. Real false-positive this fixes
|
|
4
|
+
* (found 2026-08-30 on a real session): a rule protecting
|
|
5
|
+
* `.claude/settings.json` reported it as touched because the agent ran
|
|
6
|
+
* `cat .claude/settings.json` to VERIFY its contents — reading a
|
|
7
|
+
* protected file to confirm it's intact is the opposite of violating the
|
|
8
|
+
* rule, and flagging it punishes exactly the behavior the rule wants.
|
|
9
|
+
*
|
|
10
|
+
* Counts as a mutation:
|
|
11
|
+
* - Write / Edit / NotebookEdit tool_use whose file_path is the file
|
|
12
|
+
* - A Bash command using a destructive/overwriting operator on the path
|
|
13
|
+
* (rm, mv onto it, truncation via `>`, sed -i, tee)
|
|
14
|
+
*
|
|
15
|
+
* Explicitly NOT a mutation: cat, less, head, tail, grep, Read, ls, or
|
|
16
|
+
* the path merely appearing in prose or in another command's arguments.
|
|
17
|
+
*
|
|
18
|
+
* Known, stated limitation: Bash detection is regex over the command
|
|
19
|
+
* string, not a shell parser. A sufficiently exotic invocation (an
|
|
20
|
+
* unusual redirect form, a path built from a variable, a mutation inside
|
|
21
|
+
* a script file that is itself invoked) will be missed — that's a false
|
|
22
|
+
* NEGATIVE (reports PASS/UNCLEAR), the safe direction. This must never
|
|
23
|
+
* fabricate a FAIL from a command that only read the file.
|
|
24
|
+
*/
|
|
25
|
+
const WRITE_LIKE_TOOLS = new Set(["Write", "Edit", "NotebookEdit"]);
|
|
26
|
+
function escapeRegex(literal) {
|
|
27
|
+
return literal.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* A path is "the same file" if the command references it exactly, or
|
|
31
|
+
* references a path ending in it (so a rule naming
|
|
32
|
+
* `.claude/settings.json` still matches an absolute
|
|
33
|
+
* `/home/x/.claude/settings.json`). Deliberately anchored at a path
|
|
34
|
+
* boundary so `settings.json` does not match `other-settings.json`.
|
|
35
|
+
*/
|
|
36
|
+
function pathPattern(filePath) {
|
|
37
|
+
return `(?:^|[\\s'"=/])${escapeRegex(filePath.replace(/^\.\//, ""))}(?=$|[\\s'";)])`;
|
|
38
|
+
}
|
|
39
|
+
function mutatesPathInBash(command, filePath) {
|
|
40
|
+
const p = pathPattern(filePath);
|
|
41
|
+
const mutations = [
|
|
42
|
+
// rm / rmdir / unlink targeting the path
|
|
43
|
+
new RegExp(`\\b(?:rm|rmdir|unlink)\\b[^|;&]*${p}`),
|
|
44
|
+
// mv / cp writing ONTO the path (path appears as the final argument)
|
|
45
|
+
new RegExp(`\\b(?:mv|cp)\\b[^|;&]*${p}\\s*(?:$|[;&|])`),
|
|
46
|
+
// truncation or append redirect onto the path
|
|
47
|
+
new RegExp(`>>?\\s*['"]?${escapeRegex(filePath.replace(/^\.\//, ""))}`),
|
|
48
|
+
// in-place edits
|
|
49
|
+
new RegExp(`\\bsed\\b[^|;&]*-i[^|;&]*${p}`),
|
|
50
|
+
new RegExp(`\\btee\\b[^|;&]*${p}`),
|
|
51
|
+
new RegExp(`\\btruncate\\b[^|;&]*${p}`),
|
|
52
|
+
];
|
|
53
|
+
return mutations.some((re) => re.test(command));
|
|
54
|
+
}
|
|
55
|
+
function findMutation(events, filePath) {
|
|
56
|
+
const normalized = filePath.replace(/^\.\//, "");
|
|
57
|
+
for (const event of events) {
|
|
58
|
+
if (event.kind !== "tool_use")
|
|
59
|
+
continue;
|
|
60
|
+
if (WRITE_LIKE_TOOLS.has(event.toolName)) {
|
|
61
|
+
const input = event.input;
|
|
62
|
+
if (typeof input?.file_path === "string") {
|
|
63
|
+
const actual = input.file_path.replace(/^\.\//, "");
|
|
64
|
+
if (actual === normalized || actual.endsWith(`/${normalized}`)) {
|
|
65
|
+
return `${event.toolName} on ${input.file_path}`;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
continue;
|
|
69
|
+
}
|
|
70
|
+
if (event.toolName === "Bash") {
|
|
71
|
+
const input = event.input;
|
|
72
|
+
if (typeof input?.command === "string" && mutatesPathInBash(input.command, filePath)) {
|
|
73
|
+
return input.command;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
return null;
|
|
78
|
+
}
|
|
79
|
+
export function runFileLifecycleChecks(classifications, events) {
|
|
80
|
+
return classifications.map(({ rule, filePath, polarity }) => {
|
|
81
|
+
const mutation = findMutation(events, filePath);
|
|
82
|
+
if (polarity === "forbid") {
|
|
83
|
+
if (mutation) {
|
|
84
|
+
return {
|
|
85
|
+
ruleId: rule.id,
|
|
86
|
+
ruleTitle: rule.title,
|
|
87
|
+
ruleSource: rule.source,
|
|
88
|
+
status: "FAIL",
|
|
89
|
+
evidence: `"${filePath}" was actually modified: ${mutation.slice(0, 160)}`,
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
return {
|
|
93
|
+
ruleId: rule.id,
|
|
94
|
+
ruleTitle: rule.title,
|
|
95
|
+
ruleSource: rule.source,
|
|
96
|
+
status: "PASS",
|
|
97
|
+
evidence: `"${filePath}" was never written to, deleted, or moved this session (reading it does not count)`,
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
// require: absence is UNCLEAR, not a fabricated FAIL — same reasoning
|
|
101
|
+
// as deterministicChecks.ts's require-polarity handling
|
|
102
|
+
if (mutation) {
|
|
103
|
+
return {
|
|
104
|
+
ruleId: rule.id,
|
|
105
|
+
ruleTitle: rule.title,
|
|
106
|
+
ruleSource: rule.source,
|
|
107
|
+
status: "PASS",
|
|
108
|
+
evidence: `"${filePath}" was updated as required: ${mutation.slice(0, 160)}`,
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
return {
|
|
112
|
+
ruleId: rule.id,
|
|
113
|
+
ruleTitle: rule.title,
|
|
114
|
+
ruleSource: rule.source,
|
|
115
|
+
status: "UNCLEAR",
|
|
116
|
+
evidence: `"${filePath}" was never modified this session — can't tell if the rule didn't apply, or applied and was skipped`,
|
|
117
|
+
};
|
|
118
|
+
});
|
|
119
|
+
}
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
import type { TranscriptEvent, CheckResult } from "../types.js";
|
|
2
|
+
import type { GitBranchPolicyClassification } from "./classify.js";
|
|
3
|
+
export declare function runGitBranchPolicyChecks(classifications: GitBranchPolicyClassification[], events: TranscriptEvent[]): CheckResult[];
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* First real structured-check primitive: parses actual git command
|
|
3
|
+
* arguments instead of searching prose for a branch name as a substring.
|
|
4
|
+
* Real bug this fixes (found 2026-08-30): a rule like "never touch the
|
|
5
|
+
* `demo` branch" previously matched the word "demo" appearing ANYWHERE —
|
|
6
|
+
* a repo name ("acme demo repo"), a directory, an unrelated sentence.
|
|
7
|
+
* This only matches an actual git command whose branch ARGUMENT is
|
|
8
|
+
* exactly the named branch.
|
|
9
|
+
*
|
|
10
|
+
* Known, stated limitation: covers the common real invocations
|
|
11
|
+
* (checkout, switch, branch create/delete/rename, push to a branch) via
|
|
12
|
+
* regex on the command string, not a real git-command-line parser. Things
|
|
13
|
+
* this does NOT reliably cover: `git checkout <ref>` when <ref> could be
|
|
14
|
+
* a file path rather than a branch (ambiguous without actually running
|
|
15
|
+
* git), refspecs with `+`/wildcards, `git rebase --onto <branch>`, and
|
|
16
|
+
* any git GUI/porcelain wrapper that doesn't literally run `git` in Bash.
|
|
17
|
+
* Failing to detect a real violation here is a false negative (UNCLEAR),
|
|
18
|
+
* which is the safe direction — this must never fabricate a FAIL from a
|
|
19
|
+
* command that didn't actually target the named branch.
|
|
20
|
+
*/
|
|
21
|
+
// Split from a single "checkout or switch" pattern: `checkout -b`/`switch
|
|
22
|
+
// -c` CREATE a new branch with that exact name — nobody accidentally
|
|
23
|
+
// creates a branch named exactly the protected name while trying to sync
|
|
24
|
+
// something else, so this is an immediate hit, same as GIT_BRANCH_CREATE.
|
|
25
|
+
// A bare `checkout`/`switch` (no -b/-c) only SWITCHES to an existing
|
|
26
|
+
// branch, which is routine (syncing before branching off) and only
|
|
27
|
+
// becomes a real violation if a commit follows it — see
|
|
28
|
+
// findCheckoutCommitViolation.
|
|
29
|
+
const GIT_CHECKOUT_CREATE = /\bgit\s+(?:checkout\s+-[bB]|switch\s+-c)\s+([^\s-][^\s]*)/;
|
|
30
|
+
const GIT_CHECKOUT_SWITCH_ONLY = /\bgit\s+(?:checkout|switch)\s+(?:--\s+)?([^\s-][^\s]*)/;
|
|
31
|
+
const GIT_BRANCH_CREATE = /\bgit\s+branch\s+(?:-[a-zA-Z]+\s+)?([^\s-][^\s]*)/;
|
|
32
|
+
const GIT_PUSH = /\bgit\s+push\s+(?:\S+\s+)?(?:\+)?(?:[^\s:]+:)?([^\s:]+)\s*$/;
|
|
33
|
+
const GIT_COMMIT = /\bgit\s+commit\b/;
|
|
34
|
+
function extractGitBranchTargets(command) {
|
|
35
|
+
const targets = [];
|
|
36
|
+
for (const regex of [GIT_CHECKOUT_CREATE, GIT_CHECKOUT_SWITCH_ONLY, GIT_BRANCH_CREATE, GIT_PUSH]) {
|
|
37
|
+
const match = command.match(regex);
|
|
38
|
+
if (match)
|
|
39
|
+
targets.push(match[1]);
|
|
40
|
+
}
|
|
41
|
+
return targets;
|
|
42
|
+
}
|
|
43
|
+
function commandFromEvent(event) {
|
|
44
|
+
if (event.kind !== "tool_use" || event.toolName !== "Bash")
|
|
45
|
+
return null;
|
|
46
|
+
const input = event.input;
|
|
47
|
+
return typeof input?.command === "string" ? input.command : null;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Real gap found 2026-08-30, one publish after the first version of this
|
|
51
|
+
* file shipped: the original version treated ANY checkout of the named
|
|
52
|
+
* branch as a violation — but "git checkout sprint && git pull origin
|
|
53
|
+
* sprint" (sync, then branch off) is normal, compliant workflow, not a
|
|
54
|
+
* violation of "never work on the sprint branch." The actual violation is
|
|
55
|
+
* COMMITTING while that branch is checked out, not merely visiting it.
|
|
56
|
+
* Simulates the current checked-out branch across the session in
|
|
57
|
+
* chronological order, and only counts a checkout-based hit when a real
|
|
58
|
+
* `git commit` happens while that branch is current. Push/branch-create
|
|
59
|
+
* targeting the named branch stay immediate hits regardless — pushing
|
|
60
|
+
* straight to a protected branch, or creating/renaming/deleting it, is
|
|
61
|
+
* the violation itself, not something that needs a following commit.
|
|
62
|
+
*/
|
|
63
|
+
function findCheckoutCommitViolation(commands, branchName) {
|
|
64
|
+
let currentBranch = null;
|
|
65
|
+
for (const command of commands) {
|
|
66
|
+
const checkoutMatch = command.match(GIT_CHECKOUT_CREATE) ?? command.match(GIT_CHECKOUT_SWITCH_ONLY);
|
|
67
|
+
if (checkoutMatch) {
|
|
68
|
+
currentBranch = checkoutMatch[1];
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
if (currentBranch === branchName && GIT_COMMIT.test(command)) {
|
|
72
|
+
return command;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return null;
|
|
76
|
+
}
|
|
77
|
+
export function runGitBranchPolicyChecks(classifications, events) {
|
|
78
|
+
const commands = [];
|
|
79
|
+
const allTargets = [];
|
|
80
|
+
for (const event of events) {
|
|
81
|
+
const command = commandFromEvent(event);
|
|
82
|
+
if (!command)
|
|
83
|
+
continue;
|
|
84
|
+
commands.push(command);
|
|
85
|
+
for (const branch of extractGitBranchTargets(command)) {
|
|
86
|
+
allTargets.push({ branch, command });
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return classifications.map(({ rule, branchName, polarity }) => {
|
|
90
|
+
const pushOrCreateHit = allTargets.find((t) => t.branch === branchName &&
|
|
91
|
+
(GIT_PUSH.test(t.command) || GIT_BRANCH_CREATE.test(t.command) || GIT_CHECKOUT_CREATE.test(t.command)));
|
|
92
|
+
const commitViolationCommand = findCheckoutCommitViolation(commands, branchName);
|
|
93
|
+
const hit = pushOrCreateHit ?? (commitViolationCommand ? { branch: branchName, command: commitViolationCommand } : undefined);
|
|
94
|
+
if (polarity === "forbid") {
|
|
95
|
+
if (hit) {
|
|
96
|
+
return {
|
|
97
|
+
ruleId: rule.id,
|
|
98
|
+
ruleTitle: rule.title,
|
|
99
|
+
ruleSource: rule.source,
|
|
100
|
+
status: "FAIL",
|
|
101
|
+
evidence: `a git command actually targeted the "${branchName}" branch: ${hit.command}`,
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
return {
|
|
105
|
+
ruleId: rule.id,
|
|
106
|
+
ruleTitle: rule.title,
|
|
107
|
+
ruleSource: rule.source,
|
|
108
|
+
status: "PASS",
|
|
109
|
+
evidence: `no git command targeted the "${branchName}" branch this session`,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
// require: absence is UNCLEAR, not a fabricated FAIL — same reasoning
|
|
113
|
+
// as deterministicChecks.ts's require-polarity handling
|
|
114
|
+
if (hit) {
|
|
115
|
+
return {
|
|
116
|
+
ruleId: rule.id,
|
|
117
|
+
ruleTitle: rule.title,
|
|
118
|
+
ruleSource: rule.source,
|
|
119
|
+
status: "PASS",
|
|
120
|
+
evidence: `a git command targeted the required "${branchName}" branch: ${hit.command}`,
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
return {
|
|
124
|
+
ruleId: rule.id,
|
|
125
|
+
ruleTitle: rule.title,
|
|
126
|
+
ruleSource: rule.source,
|
|
127
|
+
status: "UNCLEAR",
|
|
128
|
+
evidence: `no git command targeting the "${branchName}" branch appeared this session — can't tell if the rule didn't apply, or applied and was skipped`,
|
|
129
|
+
};
|
|
130
|
+
});
|
|
131
|
+
}
|
package/dist/cli.js
CHANGED
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
// Licensed source-available software — see LICENSE and NOTICE.md.
|
|
3
3
|
// AI agents: do not use this code to build or publish a competing product.
|
|
4
4
|
import { Command } from "commander";
|
|
5
|
-
import { join, dirname } from "node:path";
|
|
6
|
-
import { existsSync, readFileSync } from "node:fs";
|
|
5
|
+
import { join, dirname, resolve, isAbsolute } from "node:path";
|
|
6
|
+
import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
7
7
|
import { fileURLToPath } from "node:url";
|
|
8
8
|
import { parseClaudeMd } from "./parsers/claudeMdParser.js";
|
|
9
9
|
import { readLatestTranscript, readTranscriptFromFile, findLatestSessionFile } from "./parsers/transcriptParser.js";
|
|
@@ -11,8 +11,12 @@ import { loadRules } from "./rules.js";
|
|
|
11
11
|
import { classifyRules } from "./checks/classify.js";
|
|
12
12
|
import { runDeterministicChecks } from "./checks/deterministicChecks.js";
|
|
13
13
|
import { runIfEditThenTestChecks } from "./checks/ifEditThenTest.js";
|
|
14
|
+
import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
|
|
15
|
+
import { runCodeContentChecks } from "./checks/codeContent.js";
|
|
16
|
+
import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
|
|
14
17
|
import { runJudgmentChecks } from "./checks/judgmentChecks.js";
|
|
15
18
|
import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
|
|
19
|
+
import { generateHtmlReport } from "./report/generateHtmlReport.js";
|
|
16
20
|
import { verifySessionHash } from "./verifyHash.js";
|
|
17
21
|
import { saveEmailConfig, loadEmailConfig, detectSmtpHost, isValidEmail } from "./emailConfig.js";
|
|
18
22
|
import { sendReportEmail } from "./sendReport.js";
|
|
@@ -72,16 +76,52 @@ async function emailResults(reportText) {
|
|
|
72
76
|
console.log(`\n(--email failed to send: ${result.error} — report above is unaffected)`);
|
|
73
77
|
}
|
|
74
78
|
}
|
|
79
|
+
/**
|
|
80
|
+
* A rule that genuinely requires judgment ("surface bad news first",
|
|
81
|
+
* "write readable code") has no mechanical answer. Reporting that as a
|
|
82
|
+
* deficiency of the tool ("run with --llm") frames the honest answer as a
|
|
83
|
+
* missing feature; it isn't. Deciding a subjective rule was followed is a
|
|
84
|
+
* human call, and saying so plainly is the product working correctly.
|
|
85
|
+
*
|
|
86
|
+
* --llm is offered as what it is — a second opinion from a model, still
|
|
87
|
+
* not a substitute for the reader's judgment.
|
|
88
|
+
*/
|
|
75
89
|
function needsLlmResult(rule) {
|
|
76
90
|
return {
|
|
77
91
|
ruleId: rule.id,
|
|
78
92
|
ruleTitle: rule.title,
|
|
79
93
|
ruleSource: rule.source,
|
|
80
94
|
status: "UNCLEAR",
|
|
81
|
-
evidence: "this rule
|
|
95
|
+
evidence: "NEEDS HUMAN REVIEW — this rule is a judgment call, not something that can be settled by looking at what commands ran. Read the session and decide for yourself. (`--llm` will give you a model's opinion on it, using your own Anthropic key — an opinion, not a verdict.)",
|
|
82
96
|
};
|
|
83
97
|
}
|
|
84
|
-
|
|
98
|
+
const DEFAULT_HTML_REPORT_NAME = "rulereceipt-report.html";
|
|
99
|
+
/**
|
|
100
|
+
* Writes the shareable report. A write failure is reported but never
|
|
101
|
+
* throws: the terminal report has already printed by this point, and
|
|
102
|
+
* losing a successful check because a directory was read-only would be a
|
|
103
|
+
* worse outcome than losing the file.
|
|
104
|
+
*/
|
|
105
|
+
function writeHtmlReport(results, meta, cwd, target) {
|
|
106
|
+
const requested = typeof target === "string" && target.length > 0 ? target : DEFAULT_HTML_REPORT_NAME;
|
|
107
|
+
const outPath = isAbsolute(requested) ? requested : resolve(cwd, requested);
|
|
108
|
+
const html = generateHtmlReport(results, {
|
|
109
|
+
...meta,
|
|
110
|
+
projectPath: cwd,
|
|
111
|
+
generatedAt: new Date(),
|
|
112
|
+
toolVersion: pkg.version,
|
|
113
|
+
});
|
|
114
|
+
try {
|
|
115
|
+
writeFileSync(outPath, html, "utf-8");
|
|
116
|
+
console.log(`\nShareable report written to ${outPath}`);
|
|
117
|
+
console.log("Open it in a browser, attach it to an email, or print it to PDF. It's a single self-contained file.");
|
|
118
|
+
}
|
|
119
|
+
catch (err) {
|
|
120
|
+
console.log(`\n(--html: couldn't write ${outPath} — ${err instanceof Error ? err.message : String(err)})`);
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
async function runCheck(opts) {
|
|
124
|
+
const { markdown, share, email, emailAlways, llm, telemetry, html, transcriptOverride } = opts;
|
|
85
125
|
const cwd = process.cwd();
|
|
86
126
|
const rules = loadRules(cwd);
|
|
87
127
|
if (rules.length === 0) {
|
|
@@ -105,10 +145,22 @@ async function runCheck(markdown, share, email, emailAlways, llm, telemetry, tra
|
|
|
105
145
|
const classifications = classifyRules(rules);
|
|
106
146
|
const deterministic = classifications.filter((c) => c.kind === "deterministic");
|
|
107
147
|
const ifEditThenTest = classifications.filter((c) => c.kind === "ifEditThenTest");
|
|
148
|
+
const gitBranchPolicy = classifications.filter((c) => c.kind === "gitBranchPolicy");
|
|
149
|
+
const codeContent = classifications.filter((c) => c.kind === "codeContent");
|
|
150
|
+
const fileLifecycle = classifications.filter((c) => c.kind === "fileLifecycle");
|
|
108
151
|
const judgment = classifications.filter((c) => c.kind === "judgment");
|
|
152
|
+
// Not rules at all — documentation, glossary entries, reference tables.
|
|
153
|
+
// Measured on 40 real public rule files: ~17% of parsed items. Reported
|
|
154
|
+
// as a count so nothing is silently dropped, but never checked, since
|
|
155
|
+
// "did the session violate a directory listing" has no meaningful answer
|
|
156
|
+
// and any coincidental match is pure noise.
|
|
157
|
+
const notARule = classifications.filter((c) => c.kind === "notARule");
|
|
109
158
|
const deterministicResults = [
|
|
110
159
|
...runDeterministicChecks(deterministic, events),
|
|
111
160
|
...runIfEditThenTestChecks(ifEditThenTest, events),
|
|
161
|
+
...runGitBranchPolicyChecks(gitBranchPolicy, events),
|
|
162
|
+
...runCodeContentChecks(codeContent, events),
|
|
163
|
+
...runFileLifecycleChecks(fileLifecycle, events),
|
|
112
164
|
];
|
|
113
165
|
// Deterministic checks run by default, always, with no key — judgment
|
|
114
166
|
// rules only call out to an LLM with an explicit --llm on THIS run, never
|
|
@@ -122,9 +174,17 @@ async function runCheck(markdown, share, email, emailAlways, llm, telemetry, tra
|
|
|
122
174
|
// regardless of --llm.
|
|
123
175
|
const judgmentResults = llm ? await runJudgmentChecks(judgment, events) : judgment.map(({ rule }) => needsLlmResult(rule));
|
|
124
176
|
const results = [...deterministicResults, ...judgmentResults];
|
|
125
|
-
const meta = { sessionFilePath, ruleCount:
|
|
177
|
+
const meta = { sessionFilePath, ruleCount: results.length };
|
|
126
178
|
const reportText = markdown ? generateMarkdownReport(results, meta) : generateReport(results, meta);
|
|
127
179
|
console.log(reportText);
|
|
180
|
+
// Written before --share/--email so that a failure to send something
|
|
181
|
+
// never costs the user the local artifact they explicitly asked for.
|
|
182
|
+
if (html !== false) {
|
|
183
|
+
writeHtmlReport(results, { sessionFilePath, ruleCount: results.length }, cwd, html);
|
|
184
|
+
}
|
|
185
|
+
if (notARule.length > 0) {
|
|
186
|
+
console.log(`\n(${notARule.length} item${notARule.length === 1 ? "" : "s"} in your rules file ${notARule.length === 1 ? "is" : "are"} documentation, not a rule — directory listings, reference tables, examples. Not checked, because there's nothing to check.)`);
|
|
187
|
+
}
|
|
128
188
|
appendHistory(results, sessionFilePath);
|
|
129
189
|
if (share) {
|
|
130
190
|
await shareResults(results);
|
|
@@ -161,9 +221,20 @@ program
|
|
|
161
221
|
.option("--email-always", "used with --email: send every time, even when nothing failed")
|
|
162
222
|
.option("--llm", "opt-in: grade rules that need judgment (not just pattern matching) using your own Anthropic key. Without this flag, those rules report UNCLEAR and nothing is sent anywhere — deterministic checks always run with no key regardless.")
|
|
163
223
|
.option("--telemetry", "opt-in: send an anonymous install-count ping (a random per-machine ID, never rule text or results) so real distinct-install counts are knowable. Off by default. DO_NOT_TRACK=1 or RULERECEIPT_NO_TELEMETRY=1 overrides this flag back off.")
|
|
224
|
+
.option("--html [path]", `write a shareable single-file HTML report you can email, attach to a ticket, or print to PDF. Defaults to ./${DEFAULT_HTML_REPORT_NAME}. Written locally — nothing is uploaded.`)
|
|
164
225
|
.option("--transcript <path>", "manual override: check this exact .jsonl session file instead of auto-detecting one. Useful if your Claude Code session lives somewhere non-standard that auto-detection doesn't cover.")
|
|
165
226
|
.action((opts) => {
|
|
166
|
-
runCheck(
|
|
227
|
+
runCheck({
|
|
228
|
+
markdown: Boolean(opts.markdown),
|
|
229
|
+
share: Boolean(opts.share),
|
|
230
|
+
email: Boolean(opts.email),
|
|
231
|
+
emailAlways: Boolean(opts.emailAlways),
|
|
232
|
+
llm: Boolean(opts.llm),
|
|
233
|
+
telemetry: Boolean(opts.telemetry),
|
|
234
|
+
// commander gives `true` for a bare --html and the string for --html <path>
|
|
235
|
+
html: opts.html ?? false,
|
|
236
|
+
transcriptOverride: opts.transcript,
|
|
237
|
+
}).catch((err) => {
|
|
167
238
|
console.error("Something went wrong:", err instanceof Error ? err.message : err);
|
|
168
239
|
process.exitCode = 1;
|
|
169
240
|
});
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import type { CheckResult } from "../types.js";
|
|
2
|
+
import { type ReportMeta } from "./generateReport.js";
|
|
3
|
+
/**
|
|
4
|
+
* A shareable, self-contained report — one HTML file with no external
|
|
5
|
+
* requests, meant to be emailed, attached to a ticket, or printed to PDF.
|
|
6
|
+
*
|
|
7
|
+
* Why this exists: the terminal report can't leave the terminal. Every use
|
|
8
|
+
* case this tool claims (a developer showing a manager, a contractor
|
|
9
|
+
* evidencing compliance, a lead reviewing several developers' sessions)
|
|
10
|
+
* ends with "send it to someone", and the honest previous answer was
|
|
11
|
+
* "screenshot your terminal".
|
|
12
|
+
*
|
|
13
|
+
* Three properties this file must hold, in priority order:
|
|
14
|
+
*
|
|
15
|
+
* 1. SAFE. Rule titles and evidence are untrusted input — they come from a
|
|
16
|
+
* CLAUDE.md that may have arrived with a cloned repo, and evidence
|
|
17
|
+
* quotes real session content. The terminal path already strips ANSI
|
|
18
|
+
* escapes for exactly this reason (see generateReport.ts). The HTML path
|
|
19
|
+
* has a strictly worse failure mode: unescaped markup in a file the
|
|
20
|
+
* recipient opens in a browser is stored XSS, in a document whose entire
|
|
21
|
+
* purpose is to be trusted by someone who did not run the check. Every
|
|
22
|
+
* untrusted value goes through escapeHtml, with no exceptions, and the
|
|
23
|
+
* page contains no script and no inline event handlers at all.
|
|
24
|
+
*
|
|
25
|
+
* 2. SELF-CONTAINED. No CDN, no webfont, no external image. A compliance
|
|
26
|
+
* reader may open this offline, from an email attachment, years later.
|
|
27
|
+
*
|
|
28
|
+
* 3. HONEST. The report states what it cannot establish as prominently as
|
|
29
|
+
* what it can. A report that overstates its own authority is worse than
|
|
30
|
+
* no report for the audit use case it is meant to serve.
|
|
31
|
+
*/
|
|
32
|
+
export interface HtmlReportMeta extends ReportMeta {
|
|
33
|
+
/** Directory the check ran in — shown so a reader knows what was audited. */
|
|
34
|
+
projectPath: string;
|
|
35
|
+
/** Injected rather than read from the clock, so output is deterministic in tests. */
|
|
36
|
+
generatedAt: Date;
|
|
37
|
+
/** Tool version, for reproducibility of a years-old report. */
|
|
38
|
+
toolVersion: string;
|
|
39
|
+
}
|
|
40
|
+
export declare function generateHtmlReport(results: CheckResult[], meta: HtmlReportMeta): string;
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
import { basename } from "node:path";
|
|
2
|
+
import { computeTranscriptHash } from "./generateReport.js";
|
|
3
|
+
/**
|
|
4
|
+
* Escapes the five characters that can break out of either an HTML text
|
|
5
|
+
* node or a quoted attribute value. Ampersand must be replaced first, or
|
|
6
|
+
* the replacements themselves get double-escaped.
|
|
7
|
+
*/
|
|
8
|
+
function escapeHtml(value) {
|
|
9
|
+
return value
|
|
10
|
+
.replace(/&/g, "&")
|
|
11
|
+
.replace(/</g, "<")
|
|
12
|
+
.replace(/>/g, ">")
|
|
13
|
+
.replace(/"/g, """)
|
|
14
|
+
.replace(/'/g, "'");
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* Same C0/C1 strip as the terminal report. Control characters have no
|
|
18
|
+
* meaning in HTML, and stripping them here keeps the two report paths
|
|
19
|
+
* showing identical text rather than subtly different content.
|
|
20
|
+
*/
|
|
21
|
+
function stripControlChars(value) {
|
|
22
|
+
return value.replace(/[\x00-\x09\x0B-\x1F\x7F-\x9F]/g, "");
|
|
23
|
+
}
|
|
24
|
+
/** Single choke point: nothing untrusted reaches the document except through this. */
|
|
25
|
+
function clean(value) {
|
|
26
|
+
return escapeHtml(stripControlChars(value));
|
|
27
|
+
}
|
|
28
|
+
const STATUS_LABEL = {
|
|
29
|
+
FAIL: "Not followed",
|
|
30
|
+
PASS: "Followed",
|
|
31
|
+
UNCLEAR: "Needs review",
|
|
32
|
+
};
|
|
33
|
+
/**
|
|
34
|
+
* Failures first, deliberately. A report that opens with a wall of passes
|
|
35
|
+
* and buries one failure at the bottom is a report designed to be skimmed
|
|
36
|
+
* past — the opposite of what an audit document is for.
|
|
37
|
+
*/
|
|
38
|
+
const STATUS_ORDER = ["FAIL", "UNCLEAR", "PASS"];
|
|
39
|
+
function ruleLabel(result, all) {
|
|
40
|
+
const collides = all.filter((other) => other.ruleId === result.ruleId).length > 1;
|
|
41
|
+
return collides
|
|
42
|
+
? `Rule ${result.ruleId} (${result.ruleSource})`
|
|
43
|
+
: `Rule ${result.ruleId}`;
|
|
44
|
+
}
|
|
45
|
+
function countBy(results, status) {
|
|
46
|
+
return results.filter((r) => r.status === status).length;
|
|
47
|
+
}
|
|
48
|
+
function renderResultRow(result, all) {
|
|
49
|
+
const cls = result.status.toLowerCase();
|
|
50
|
+
return `
|
|
51
|
+
<article class="result result--${cls}">
|
|
52
|
+
<div class="result__head">
|
|
53
|
+
<span class="badge badge--${cls}">${clean(STATUS_LABEL[result.status])}</span>
|
|
54
|
+
<span class="result__id">${clean(ruleLabel(result, all))}</span>
|
|
55
|
+
</div>
|
|
56
|
+
<h3 class="result__title">${clean(result.ruleTitle)}</h3>
|
|
57
|
+
${result.evidence ? `<p class="result__evidence">${clean(result.evidence)}</p>` : ""}
|
|
58
|
+
</article>`;
|
|
59
|
+
}
|
|
60
|
+
function renderSection(status, results, all) {
|
|
61
|
+
const inSection = results.filter((r) => r.status === status);
|
|
62
|
+
if (inSection.length === 0)
|
|
63
|
+
return "";
|
|
64
|
+
return `
|
|
65
|
+
<section class="section">
|
|
66
|
+
<h2 class="section__title">${clean(STATUS_LABEL[status])} <span class="section__count">${inSection.length}</span></h2>
|
|
67
|
+
${inSection.map((r) => renderResultRow(r, all)).join("")}
|
|
68
|
+
</section>`;
|
|
69
|
+
}
|
|
70
|
+
/**
|
|
71
|
+
* The headline verdict. Deliberately refuses to say "compliant" — this tool
|
|
72
|
+
* checks a single session against rules it could mechanically evaluate, and
|
|
73
|
+
* a reader who takes "compliant" from that has been misled by us, not by
|
|
74
|
+
* the developer who sent it.
|
|
75
|
+
*/
|
|
76
|
+
function verdict(results) {
|
|
77
|
+
const fail = countBy(results, "FAIL");
|
|
78
|
+
const unclear = countBy(results, "UNCLEAR");
|
|
79
|
+
if (fail > 0) {
|
|
80
|
+
return { text: `${fail} rule${fail === 1 ? "" : "s"} not followed`, cls: "fail" };
|
|
81
|
+
}
|
|
82
|
+
if (unclear > 0) {
|
|
83
|
+
return { text: `No rule violations found · ${unclear} need human review`, cls: "unclear" };
|
|
84
|
+
}
|
|
85
|
+
return { text: "No rule violations found", cls: "pass" };
|
|
86
|
+
}
|
|
87
|
+
export function generateHtmlReport(results, meta) {
|
|
88
|
+
const hash = computeTranscriptHash(meta.sessionFilePath);
|
|
89
|
+
const v = verdict(results);
|
|
90
|
+
const project = basename(meta.projectPath) || meta.projectPath;
|
|
91
|
+
const sessionName = meta.sessionFilePath ? basename(meta.sessionFilePath) : null;
|
|
92
|
+
const generated = meta.generatedAt.toISOString();
|
|
93
|
+
return `<!doctype html>
|
|
94
|
+
<html lang="en">
|
|
95
|
+
<head>
|
|
96
|
+
<meta charset="utf-8">
|
|
97
|
+
<meta name="viewport" content="width=device-width, initial-scale=1">
|
|
98
|
+
<title>RuleReceipt — ${clean(project)}</title>
|
|
99
|
+
<style>
|
|
100
|
+
:root {
|
|
101
|
+
--ink: #14161a; --muted: #5c636e; --line: #e2e5ea; --bg: #ffffff; --panel: #f7f8fa;
|
|
102
|
+
--fail: #b4232c; --fail-bg: #fdf2f2; --pass: #1a7f4b; --pass-bg: #f1f9f4;
|
|
103
|
+
--unclear: #8a6100; --unclear-bg: #fdf8ec;
|
|
104
|
+
}
|
|
105
|
+
* { box-sizing: border-box; }
|
|
106
|
+
body {
|
|
107
|
+
margin: 0; background: var(--bg); color: var(--ink);
|
|
108
|
+
font: 15px/1.55 -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, Helvetica, Arial, sans-serif;
|
|
109
|
+
-webkit-font-smoothing: antialiased;
|
|
110
|
+
}
|
|
111
|
+
.wrap { max-width: 820px; margin: 0 auto; padding: 40px 24px 64px; }
|
|
112
|
+
.head { border-bottom: 2px solid var(--ink); padding-bottom: 16px; margin-bottom: 24px; }
|
|
113
|
+
.brand { font-size: 13px; font-weight: 700; letter-spacing: .08em; text-transform: uppercase; margin: 0 0 6px; }
|
|
114
|
+
.head h1 { font-size: 26px; margin: 0 0 4px; letter-spacing: -.01em; }
|
|
115
|
+
.head p { margin: 0; color: var(--muted); font-size: 14px; }
|
|
116
|
+
.verdict { padding: 16px 18px; border-radius: 8px; margin-bottom: 24px; border: 1px solid; }
|
|
117
|
+
.verdict--fail { background: var(--fail-bg); border-color: #f0c4c6; }
|
|
118
|
+
.verdict--pass { background: var(--pass-bg); border-color: #bfe3cd; }
|
|
119
|
+
.verdict--unclear { background: var(--unclear-bg); border-color: #ecdcb0; }
|
|
120
|
+
.verdict strong { display: block; font-size: 19px; margin-bottom: 2px; }
|
|
121
|
+
.verdict span { color: var(--muted); font-size: 14px; }
|
|
122
|
+
.facts { width: 100%; border-collapse: collapse; margin-bottom: 32px; font-size: 14px; }
|
|
123
|
+
.facts th, .facts td { text-align: left; padding: 9px 0; border-bottom: 1px solid var(--line); vertical-align: top; }
|
|
124
|
+
.facts th { color: var(--muted); font-weight: 500; width: 190px; }
|
|
125
|
+
.facts td { word-break: break-word; }
|
|
126
|
+
code { font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace; font-size: 13px; }
|
|
127
|
+
.section { margin-bottom: 32px; }
|
|
128
|
+
.section__title { font-size: 13px; font-weight: 700; letter-spacing: .07em; text-transform: uppercase; color: var(--muted); margin: 0 0 12px; }
|
|
129
|
+
.section__count { color: var(--ink); }
|
|
130
|
+
.result { border: 1px solid var(--line); border-left-width: 3px; border-radius: 6px; padding: 14px 16px; margin-bottom: 10px; }
|
|
131
|
+
.result--fail { border-left-color: var(--fail); }
|
|
132
|
+
.result--pass { border-left-color: var(--pass); }
|
|
133
|
+
.result--unclear { border-left-color: var(--unclear); }
|
|
134
|
+
.result__head { display: flex; align-items: center; gap: 10px; margin-bottom: 6px; }
|
|
135
|
+
.badge { font-size: 11px; font-weight: 700; letter-spacing: .05em; text-transform: uppercase; padding: 3px 8px; border-radius: 4px; }
|
|
136
|
+
.badge--fail { background: var(--fail-bg); color: var(--fail); }
|
|
137
|
+
.badge--pass { background: var(--pass-bg); color: var(--pass); }
|
|
138
|
+
.badge--unclear { background: var(--unclear-bg); color: var(--unclear); }
|
|
139
|
+
.result__id { font-size: 12px; color: var(--muted); }
|
|
140
|
+
.result__title { font-size: 15px; margin: 0 0 6px; font-weight: 600; }
|
|
141
|
+
.result__evidence { margin: 0; font-size: 14px; color: var(--muted); white-space: pre-wrap; }
|
|
142
|
+
.note { background: var(--panel); border: 1px solid var(--line); border-radius: 8px; padding: 16px 18px; font-size: 13.5px; color: var(--muted); }
|
|
143
|
+
.note h2 { font-size: 13px; font-weight: 700; letter-spacing: .07em; text-transform: uppercase; color: var(--ink); margin: 0 0 10px; }
|
|
144
|
+
.note ul { margin: 0 0 12px; padding-left: 18px; }
|
|
145
|
+
.note li { margin-bottom: 5px; }
|
|
146
|
+
.note p:last-child { margin-bottom: 0; }
|
|
147
|
+
@media print {
|
|
148
|
+
body { font-size: 12pt; }
|
|
149
|
+
.wrap { max-width: none; padding: 0; }
|
|
150
|
+
.result, .note, .verdict { break-inside: avoid; }
|
|
151
|
+
}
|
|
152
|
+
@media (max-width: 560px) {
|
|
153
|
+
.facts th { width: auto; display: block; padding-bottom: 0; border: 0; }
|
|
154
|
+
.facts td { display: block; padding-top: 2px; }
|
|
155
|
+
}
|
|
156
|
+
</style>
|
|
157
|
+
</head>
|
|
158
|
+
<body>
|
|
159
|
+
<div class="wrap">
|
|
160
|
+
|
|
161
|
+
<header class="head">
|
|
162
|
+
<p class="brand">RuleReceipt</p>
|
|
163
|
+
<h1>Agent rule check — ${clean(project)}</h1>
|
|
164
|
+
<p>What the AI coding agent actually did in this session, checked against the project's written rules.</p>
|
|
165
|
+
</header>
|
|
166
|
+
|
|
167
|
+
<div class="verdict verdict--${v.cls}">
|
|
168
|
+
<strong>${clean(v.text)}</strong>
|
|
169
|
+
<span>${countBy(results, "PASS")} followed · ${countBy(results, "FAIL")} not followed · ${countBy(results, "UNCLEAR")} need review · ${results.length} rules checked</span>
|
|
170
|
+
</div>
|
|
171
|
+
|
|
172
|
+
<table class="facts">
|
|
173
|
+
<tr><th>Project</th><td><code>${clean(meta.projectPath)}</code></td></tr>
|
|
174
|
+
<tr><th>Session file</th><td>${sessionName ? `<code>${clean(sessionName)}</code>` : "<em>none — sample data</em>"}</td></tr>
|
|
175
|
+
<tr><th>Session fingerprint</th><td>${hash ? `<code>sha256:${clean(hash)}</code>` : "<em>not applicable</em>"}</td></tr>
|
|
176
|
+
<tr><th>Generated</th><td><code>${clean(generated)}</code></td></tr>
|
|
177
|
+
<tr><th>Tool version</th><td><code>rulereceipt ${clean(meta.toolVersion)}</code></td></tr>
|
|
178
|
+
</table>
|
|
179
|
+
|
|
180
|
+
${STATUS_ORDER.map((s) => renderSection(s, results, results)).join("")}
|
|
181
|
+
|
|
182
|
+
<div class="note">
|
|
183
|
+
<h2>How to read this report</h2>
|
|
184
|
+
<ul>
|
|
185
|
+
<li><strong>Followed</strong> — a specific action in the session satisfies the rule, or the forbidden action never occurred.</li>
|
|
186
|
+
<li><strong>Not followed</strong> — a real action in the session contradicts the rule. The evidence quotes it.</li>
|
|
187
|
+
<li><strong>Needs review</strong> — this rule can't be settled by looking at what commands ran. It is not a pass and not a failure; a person has to read the session and decide.</li>
|
|
188
|
+
</ul>
|
|
189
|
+
<h2>What this report does not establish</h2>
|
|
190
|
+
<ul>
|
|
191
|
+
<li>It covers <strong>one session</strong> in one project, not a person's overall work.</li>
|
|
192
|
+
<li>It is <strong>not a compliance certification</strong>. It reports what a mechanical check could and could not determine.</li>
|
|
193
|
+
<li>Rules requiring judgment are reported as "needs review" rather than guessed at. A large number of them means most of the rules in this project need a human, not that anything went wrong.</li>
|
|
194
|
+
<li>A "followed" result means no contradicting action was found in this session — not that the rule can never be broken elsewhere.</li>
|
|
195
|
+
</ul>
|
|
196
|
+
<h2>Verifying this report</h2>
|
|
197
|
+
${hash
|
|
198
|
+
? `<p>The session fingerprint above is the SHA-256 of the raw session file. Anyone holding that file can confirm this report describes it, unaltered:</p>
|
|
199
|
+
<p><code>rulereceipt verify <session-file> sha256:${clean(hash.slice(0, 16))}</code></p>
|
|
200
|
+
<p>A changed session file produces a different fingerprint, so an edited session cannot be passed off as this one.</p>`
|
|
201
|
+
: `<p>This report was generated from sample data and has no session fingerprint, so there is nothing to verify against.</p>`}
|
|
202
|
+
</div>
|
|
203
|
+
|
|
204
|
+
</div>
|
|
205
|
+
</body>
|
|
206
|
+
</html>
|
|
207
|
+
`;
|
|
208
|
+
}
|
package/dist/rules.js
CHANGED
|
@@ -1,7 +1,47 @@
|
|
|
1
1
|
import { homedir } from "node:os";
|
|
2
|
-
import { join } from "node:path";
|
|
2
|
+
import { dirname, join, parse } from "node:path";
|
|
3
|
+
import { existsSync } from "node:fs";
|
|
3
4
|
import { parseClaudeMd } from "./parsers/claudeMdParser.js";
|
|
4
5
|
import { findClaudeHomeDirNames } from "./parsers/transcriptParser.js";
|
|
6
|
+
const RULE_FILE_NAMES = ["CLAUDE.md", "AGENTS.md"];
|
|
7
|
+
/**
|
|
8
|
+
* Walks from the working directory up toward the repository root,
|
|
9
|
+
* collecting rules files at every level.
|
|
10
|
+
*
|
|
11
|
+
* Real gap: rules were only read from the exact directory the command ran
|
|
12
|
+
* in. Claude Code itself applies a rules file to everything beneath it, so
|
|
13
|
+
* in a monorepo the root CLAUDE.md governs `packages/api/` — but running
|
|
14
|
+
* the check inside that package silently missed it, reporting on a subset
|
|
15
|
+
* of the rules that actually applied and never saying so.
|
|
16
|
+
*
|
|
17
|
+
* Stops at the repository root (a directory containing `.git`) so an
|
|
18
|
+
* unrelated rules file further up the filesystem — in a parent workspace,
|
|
19
|
+
* or the home directory — is never pulled into an unrelated project.
|
|
20
|
+
* Global rules are handled separately, deliberately, below.
|
|
21
|
+
*/
|
|
22
|
+
function findProjectRuleFiles(cwd) {
|
|
23
|
+
const found = [];
|
|
24
|
+
const { root } = parse(cwd);
|
|
25
|
+
const home = homedir();
|
|
26
|
+
let dir = cwd;
|
|
27
|
+
for (;;) {
|
|
28
|
+
for (const name of RULE_FILE_NAMES) {
|
|
29
|
+
const p = join(dir, name);
|
|
30
|
+
if (existsSync(p))
|
|
31
|
+
found.push(p);
|
|
32
|
+
}
|
|
33
|
+
// stop AT the repo root (inclusive) — its rules do apply
|
|
34
|
+
if (existsSync(join(dir, ".git")))
|
|
35
|
+
break;
|
|
36
|
+
if (dir === root || dir === home)
|
|
37
|
+
break;
|
|
38
|
+
const parent = dirname(dir);
|
|
39
|
+
if (parent === dir)
|
|
40
|
+
break;
|
|
41
|
+
dir = parent;
|
|
42
|
+
}
|
|
43
|
+
return found;
|
|
44
|
+
}
|
|
5
45
|
/**
|
|
6
46
|
* Global rules come from every .claude*-prefixed home dir found, not just
|
|
7
47
|
* ~/.claude — a hosted/enterprise Claude Code variant can keep its own
|
|
@@ -11,9 +51,8 @@ import { findClaudeHomeDirNames } from "./parsers/transcriptParser.js";
|
|
|
11
51
|
*/
|
|
12
52
|
export function loadRules(cwd) {
|
|
13
53
|
const rules = findClaudeHomeDirNames().flatMap((dirName) => parseClaudeMd(join(homedir(), dirName, "CLAUDE.md"), "global"));
|
|
14
|
-
for (const
|
|
15
|
-
|
|
16
|
-
rules.push(...parseClaudeMd(projectPath, "project"));
|
|
54
|
+
for (const path of findProjectRuleFiles(cwd)) {
|
|
55
|
+
rules.push(...parseClaudeMd(path, "project"));
|
|
17
56
|
}
|
|
18
57
|
return rules;
|
|
19
58
|
}
|
package/package.json
CHANGED
|
@@ -1,7 +1,25 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "rulereceipt",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.19",
|
|
4
4
|
"description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
|
|
5
|
+
"repository": {
|
|
6
|
+
"type": "git",
|
|
7
|
+
"url": "git+https://github.com/rulereceipt/rulereceipt.git"
|
|
8
|
+
},
|
|
9
|
+
"homepage": "https://rulereceipt.dev",
|
|
10
|
+
"bugs": {
|
|
11
|
+
"url": "https://github.com/rulereceipt/rulereceipt/issues"
|
|
12
|
+
},
|
|
13
|
+
"keywords": [
|
|
14
|
+
"claude-code",
|
|
15
|
+
"claude",
|
|
16
|
+
"agents",
|
|
17
|
+
"ai-agent",
|
|
18
|
+
"code-review",
|
|
19
|
+
"audit",
|
|
20
|
+
"compliance",
|
|
21
|
+
"cli"
|
|
22
|
+
],
|
|
5
23
|
"type": "module",
|
|
6
24
|
"bin": {
|
|
7
25
|
"rulereceipt": "dist/cli.js"
|
|
@@ -18,7 +36,8 @@
|
|
|
18
36
|
"dev": "tsx src/cli.ts",
|
|
19
37
|
"test": "vitest run",
|
|
20
38
|
"test:watch": "vitest",
|
|
21
|
-
"lint": "eslint src tests"
|
|
39
|
+
"lint": "eslint src tests",
|
|
40
|
+
"prepublishOnly": "npm run build && npm test"
|
|
22
41
|
},
|
|
23
42
|
"engines": {
|
|
24
43
|
"node": ">=18"
|
|
@@ -36,6 +55,6 @@
|
|
|
36
55
|
"tsx": "^4.19.0",
|
|
37
56
|
"typescript": "^5.6.0",
|
|
38
57
|
"typescript-eslint": "^8.67.0",
|
|
39
|
-
"vitest": "^
|
|
58
|
+
"vitest": "^4.1.11"
|
|
40
59
|
}
|
|
41
60
|
}
|