rulereceipt 0.1.28 → 0.1.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/checks/judgmentChecks.d.ts +14 -9
- package/dist/checks/judgmentChecks.js +150 -29
- package/dist/report/generateReport.js +68 -4
- package/dist/rules.js +27 -9
- package/package.json +1 -1
|
@@ -2,14 +2,19 @@ import type { TranscriptEvent, CheckResult } from "../types.js";
|
|
|
2
2
|
import type { JudgmentClassification } from "./classify.js";
|
|
3
3
|
/**
|
|
4
4
|
* One isolated API call per judgment rule, not one batched call covering
|
|
5
|
-
* all of them. Deliberate,
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
5
|
+
* all of them. Deliberate, and kept: a rule judged in a fresh context, with
|
|
6
|
+
* no other rules' text in the same prompt, can't have its verdict coloured
|
|
7
|
+
* by how the model just judged a neighbouring rule.
|
|
8
|
+
*
|
|
9
|
+
* What that cost, before this: the transcript was re-sent in full with
|
|
10
|
+
* every single call, so a 13-rule file paid thirteen times for identical
|
|
11
|
+
* tokens. The transcript is byte-identical across the calls, so it now
|
|
12
|
+
* carries a cache_control marker and sits ahead of the per-rule text —
|
|
13
|
+
* isolation kept, the repeat sends nearly free.
|
|
14
|
+
*
|
|
15
|
+
* Uses the user's OWN Anthropic API key (never ours, never proxied). Fails
|
|
16
|
+
* closed per rule: any problem (no key, API error, malformed response)
|
|
17
|
+
* reports that rule as needing a human — never silently PASS — and one
|
|
18
|
+
* rule's failure never blocks the others.
|
|
14
19
|
*/
|
|
15
20
|
export declare function runJudgmentChecks(classifications: JudgmentClassification[], events: TranscriptEvent[]): Promise<CheckResult[]>;
|
|
@@ -1,6 +1,39 @@
|
|
|
1
1
|
import Anthropic from "@anthropic-ai/sdk";
|
|
2
|
-
|
|
3
|
-
|
|
2
|
+
/**
|
|
3
|
+
* The model every judgment verdict comes from.
|
|
4
|
+
*
|
|
5
|
+
* Overridable by environment, deliberately. A hardcoded id means each new
|
|
6
|
+
* model needs a release to adopt, and — worse — a wrong id breaks `--llm`
|
|
7
|
+
* for everyone until that release ships. There is no way to verify an id
|
|
8
|
+
* without a live API key, so the failure mode has to be recoverable by the
|
|
9
|
+
* person hitting it rather than by a publish.
|
|
10
|
+
*/
|
|
11
|
+
const DEFAULT_MODEL = "claude-sonnet-5";
|
|
12
|
+
function modelId() {
|
|
13
|
+
const override = process.env.RULERECEIPT_MODEL;
|
|
14
|
+
return override && override.trim().length > 0 ? override.trim() : DEFAULT_MODEL;
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* How much session text one judgment call is allowed to carry.
|
|
18
|
+
*
|
|
19
|
+
* The previous version kept only the last 60,000 characters, on the
|
|
20
|
+
* reasoning that a rule like "surface bad news first" hinges on the end of
|
|
21
|
+
* a session. True for that rule, and false in a way that produced the
|
|
22
|
+
* worst possible outcome for the others: a rule broken early in a long
|
|
23
|
+
* session and clean at the end came back PASS, because the part where it
|
|
24
|
+
* broke was never sent. A false PASS caused by truncation is exactly the
|
|
25
|
+
* failure this project published a postmortem about.
|
|
26
|
+
*
|
|
27
|
+
* So: a larger budget, both ENDS kept when it still doesn't fit, and the
|
|
28
|
+
* verdict says it was working from a partial transcript. A qualified PASS
|
|
29
|
+
* is honest; a silent one is not.
|
|
30
|
+
*/
|
|
31
|
+
const MAX_TRANSCRIPT_CHARS = 120_000;
|
|
32
|
+
const HEAD_CHARS = 40_000;
|
|
33
|
+
const TAIL_CHARS = MAX_TRANSCRIPT_CHARS - HEAD_CHARS;
|
|
34
|
+
/** How many judgment calls may be in flight at once. */
|
|
35
|
+
const MAX_CONCURRENT_CALLS = 4;
|
|
36
|
+
const TRUNCATION_NOTE = "[judged on a truncated transcript — the middle of this session was not shown to the model]";
|
|
4
37
|
function summarizeEvents(events) {
|
|
5
38
|
const lines = [];
|
|
6
39
|
for (const event of events) {
|
|
@@ -16,76 +49,164 @@ function summarizeEvents(events) {
|
|
|
16
49
|
}
|
|
17
50
|
const full = lines.join("\n");
|
|
18
51
|
if (full.length <= MAX_TRANSCRIPT_CHARS)
|
|
19
|
-
return full;
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
|
|
52
|
+
return { text: full, truncated: false };
|
|
53
|
+
// Both ends. The start is where setup, instructions and early decisions
|
|
54
|
+
// live; the end is where the reporting happens. The middle is the part a
|
|
55
|
+
// rule is least often decided on, so it is the part to drop.
|
|
56
|
+
return {
|
|
57
|
+
text: full.slice(0, HEAD_CHARS) +
|
|
58
|
+
"\n\n...[middle of session omitted for length — this is NOT the whole session]...\n\n" +
|
|
59
|
+
full.slice(-TAIL_CHARS),
|
|
60
|
+
truncated: true,
|
|
61
|
+
};
|
|
23
62
|
}
|
|
63
|
+
/**
|
|
64
|
+
* The evidence must be a line that actually appears in the session.
|
|
65
|
+
*
|
|
66
|
+
* The previous schema accepted "one short quoted or paraphrased line".
|
|
67
|
+
* The product's stated promise is quoted evidence, and a paraphrase can't
|
|
68
|
+
* be checked against the transcript by the person reading the report —
|
|
69
|
+
* which matters most here, because this is the one code path where a line
|
|
70
|
+
* can be invented outright.
|
|
71
|
+
*/
|
|
24
72
|
const RESULT_TOOL = {
|
|
25
73
|
name: "report_result",
|
|
26
|
-
description: "Report PASS/FAIL/UNCLEAR for this one rule with a
|
|
74
|
+
description: "Report PASS/FAIL/UNCLEAR for this one rule, with a verbatim line of evidence from the session.",
|
|
27
75
|
input_schema: {
|
|
28
76
|
type: "object",
|
|
29
77
|
properties: {
|
|
30
78
|
status: { type: "string", enum: ["PASS", "FAIL", "UNCLEAR"] },
|
|
31
|
-
evidence: {
|
|
79
|
+
evidence: {
|
|
80
|
+
type: "string",
|
|
81
|
+
description: "A short VERBATIM extract from the session transcript above — copied exactly as it appears, not reworded or summarised. If no exact line supports a verdict, report UNCLEAR.",
|
|
82
|
+
},
|
|
32
83
|
},
|
|
33
84
|
required: ["status", "evidence"],
|
|
34
85
|
},
|
|
35
86
|
};
|
|
36
|
-
|
|
37
|
-
|
|
87
|
+
const INSTRUCTIONS = "You judge whether one rule from a CLAUDE.md/AGENTS.md file was actually followed during a Claude Code session. " +
|
|
88
|
+
"Report PASS only if the transcript clearly shows it was followed, FAIL only if it clearly shows it was violated, " +
|
|
89
|
+
"and UNCLEAR whenever the transcript does not settle it — never guess PASS when you are not sure. " +
|
|
90
|
+
"Your evidence must be copied verbatim from the transcript.";
|
|
91
|
+
/**
|
|
92
|
+
* A rule the check never actually ran against.
|
|
93
|
+
*
|
|
94
|
+
* needsHuman is set deliberately. The report separates two things a reader
|
|
95
|
+
* must not confuse: "couldn't tell" means the tool looked and the evidence
|
|
96
|
+
* was ambiguous, and "needs your judgment" means no verdict was reached and
|
|
97
|
+
* a person still has to decide. Without this flag every --llm result landed
|
|
98
|
+
* in the first bucket, so a run with no API key printed "13 couldn't tell"
|
|
99
|
+
* about 13 rules it had never examined. Found 2026-09-08 by running the
|
|
100
|
+
* published package; nothing in 167 lines of tests here asserted the field.
|
|
101
|
+
*/
|
|
102
|
+
function didNotRun(rule, reason) {
|
|
103
|
+
return {
|
|
104
|
+
ruleId: rule.id,
|
|
105
|
+
ruleTitle: rule.title,
|
|
106
|
+
ruleSource: rule.source,
|
|
107
|
+
status: "UNCLEAR",
|
|
108
|
+
needsHuman: true,
|
|
109
|
+
evidence: reason,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
/**
|
|
113
|
+
* Runs `fn` over `items` with at most `limit` in flight.
|
|
114
|
+
*
|
|
115
|
+
* Promise.all fired every judgment rule at once. A corpus file with 100+
|
|
116
|
+
* judgment rules opened 100+ simultaneous requests, hit rate limits, and
|
|
117
|
+
* returned "API call failed" for all of them — which the report then showed
|
|
118
|
+
* as the tool being unsure rather than throttled.
|
|
119
|
+
*/
|
|
120
|
+
async function mapWithLimit(items, limit, fn) {
|
|
121
|
+
const out = new Array(items.length);
|
|
122
|
+
let next = 0;
|
|
123
|
+
const workers = Array.from({ length: Math.min(limit, items.length) }, async () => {
|
|
124
|
+
for (;;) {
|
|
125
|
+
const index = next++;
|
|
126
|
+
if (index >= items.length)
|
|
127
|
+
return;
|
|
128
|
+
out[index] = await fn(items[index]);
|
|
129
|
+
}
|
|
130
|
+
});
|
|
131
|
+
await Promise.all(workers);
|
|
132
|
+
return out;
|
|
38
133
|
}
|
|
39
134
|
/**
|
|
40
135
|
* One isolated API call per judgment rule, not one batched call covering
|
|
41
|
-
* all of them. Deliberate,
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
136
|
+
* all of them. Deliberate, and kept: a rule judged in a fresh context, with
|
|
137
|
+
* no other rules' text in the same prompt, can't have its verdict coloured
|
|
138
|
+
* by how the model just judged a neighbouring rule.
|
|
139
|
+
*
|
|
140
|
+
* What that cost, before this: the transcript was re-sent in full with
|
|
141
|
+
* every single call, so a 13-rule file paid thirteen times for identical
|
|
142
|
+
* tokens. The transcript is byte-identical across the calls, so it now
|
|
143
|
+
* carries a cache_control marker and sits ahead of the per-rule text —
|
|
144
|
+
* isolation kept, the repeat sends nearly free.
|
|
145
|
+
*
|
|
146
|
+
* Uses the user's OWN Anthropic API key (never ours, never proxied). Fails
|
|
147
|
+
* closed per rule: any problem (no key, API error, malformed response)
|
|
148
|
+
* reports that rule as needing a human — never silently PASS — and one
|
|
149
|
+
* rule's failure never blocks the others.
|
|
50
150
|
*/
|
|
51
151
|
export async function runJudgmentChecks(classifications, events) {
|
|
52
152
|
if (classifications.length === 0)
|
|
53
153
|
return [];
|
|
54
154
|
const apiKey = process.env.ANTHROPIC_API_KEY;
|
|
55
155
|
if (!apiKey) {
|
|
56
|
-
return classifications.map(({ rule }) =>
|
|
156
|
+
return classifications.map(({ rule }) => didNotRun(rule, "no ANTHROPIC_API_KEY set in your environment — set it to the same key Claude Code already uses to run judgment checks"));
|
|
57
157
|
}
|
|
58
158
|
const client = new Anthropic({ apiKey });
|
|
59
|
-
const
|
|
60
|
-
return
|
|
159
|
+
const transcript = summarizeEvents(events);
|
|
160
|
+
return mapWithLimit(classifications, MAX_CONCURRENT_CALLS, async ({ rule }) => {
|
|
61
161
|
let response;
|
|
62
162
|
try {
|
|
63
163
|
response = await client.messages.create({
|
|
64
|
-
model:
|
|
164
|
+
model: modelId(),
|
|
65
165
|
max_tokens: 512,
|
|
166
|
+
system: [{ type: "text", text: INSTRUCTIONS }],
|
|
66
167
|
tools: [RESULT_TOOL],
|
|
67
168
|
tool_choice: { type: "tool", name: "report_result" },
|
|
68
169
|
messages: [
|
|
69
170
|
{
|
|
70
171
|
role: "user",
|
|
71
|
-
content:
|
|
172
|
+
content: [
|
|
173
|
+
// Identical across every rule in this run, so it is the
|
|
174
|
+
// cacheable prefix. The rule text below it is what varies.
|
|
175
|
+
{
|
|
176
|
+
type: "text",
|
|
177
|
+
text: `SESSION TRANSCRIPT:\n${transcript.text}`,
|
|
178
|
+
cache_control: { type: "ephemeral" },
|
|
179
|
+
},
|
|
180
|
+
{ type: "text", text: `RULE — ${rule.title}\n${rule.text}` },
|
|
181
|
+
],
|
|
72
182
|
},
|
|
73
183
|
],
|
|
74
184
|
});
|
|
75
185
|
}
|
|
76
186
|
catch (err) {
|
|
77
187
|
const message = err instanceof Error ? err.message : String(err);
|
|
78
|
-
return
|
|
188
|
+
return didNotRun(rule, `API call failed (${message}) — this check did not run`);
|
|
79
189
|
}
|
|
80
190
|
const toolUseBlock = response.content.find((block) => block.type === "tool_use");
|
|
81
191
|
if (!toolUseBlock) {
|
|
82
|
-
return
|
|
192
|
+
return didNotRun(rule, "model did not return a structured result — this check did not run");
|
|
83
193
|
}
|
|
84
194
|
const parsed = toolUseBlock.input;
|
|
85
195
|
const status = parsed.status;
|
|
86
196
|
if (status === "PASS" || status === "FAIL" || status === "UNCLEAR") {
|
|
87
|
-
|
|
197
|
+
const evidence = parsed.evidence ?? "";
|
|
198
|
+
return {
|
|
199
|
+
ruleId: rule.id,
|
|
200
|
+
ruleTitle: rule.title,
|
|
201
|
+
ruleSource: rule.source,
|
|
202
|
+
status,
|
|
203
|
+
// A model-returned UNCLEAR is the tool having looked and found the
|
|
204
|
+
// session genuinely ambiguous — "couldn't tell", not "needs your
|
|
205
|
+
// judgment". Marking it as unexamined would be the same lie in the
|
|
206
|
+
// other direction.
|
|
207
|
+
evidence: transcript.truncated ? `${evidence} ${TRUNCATION_NOTE}`.trim() : evidence,
|
|
208
|
+
};
|
|
88
209
|
}
|
|
89
|
-
return
|
|
90
|
-
})
|
|
210
|
+
return didNotRun(rule, "model response did not include a valid result for this rule");
|
|
211
|
+
});
|
|
91
212
|
}
|
|
@@ -71,16 +71,80 @@ function summaryLine(results) {
|
|
|
71
71
|
parts.push(`${needsHuman} need your judgment`);
|
|
72
72
|
return parts.join(" · ");
|
|
73
73
|
}
|
|
74
|
+
function bucketOf(result) {
|
|
75
|
+
if (result.status === "FAIL")
|
|
76
|
+
return "FAIL";
|
|
77
|
+
if (result.status === "PASS")
|
|
78
|
+
return "PASS";
|
|
79
|
+
return result.needsHuman ? "UNCLEAR_JUDGMENT" : "UNCLEAR_EVIDENCE";
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Deliberately identical to generateHtmlReport's buckets, order and labels.
|
|
83
|
+
* Two reports of the same session that disagree about what is worth
|
|
84
|
+
* showing first are two reports nobody can reconcile.
|
|
85
|
+
*
|
|
86
|
+
* Failures first: a report that opens with passes and buries a failure at
|
|
87
|
+
* the bottom is built to be skimmed past. Judgment calls last — they're
|
|
88
|
+
* expected, and they're the longest section.
|
|
89
|
+
*/
|
|
90
|
+
const BUCKET_LABEL = {
|
|
91
|
+
FAIL: "Not followed",
|
|
92
|
+
UNCLEAR_EVIDENCE: "Couldn't tell",
|
|
93
|
+
UNCLEAR_JUDGMENT: "Needs your judgment",
|
|
94
|
+
PASS: "Followed",
|
|
95
|
+
};
|
|
96
|
+
const BUCKET_ORDER = ["FAIL", "UNCLEAR_EVIDENCE", "PASS", "UNCLEAR_JUDGMENT"];
|
|
97
|
+
/**
|
|
98
|
+
* The explanation shared by every rule in a section, or null when they
|
|
99
|
+
* differ.
|
|
100
|
+
*
|
|
101
|
+
* Judgment rules all carry the same sentence, because the reason is the
|
|
102
|
+
* same one every time: nothing in a transcript settles them. Printed
|
|
103
|
+
* per-rule that produced 13 copies of one 300-character paragraph on a
|
|
104
|
+
* real 14-rule file, and the report read as though the tool had done
|
|
105
|
+
* nothing.
|
|
106
|
+
*
|
|
107
|
+
* Conditional on the text actually being identical, which matters: with
|
|
108
|
+
* `--llm` each judgment rule carries its own model opinion, and hoisting
|
|
109
|
+
* those would delete the only per-rule content the section has.
|
|
110
|
+
*/
|
|
111
|
+
function sharedEvidence(rs) {
|
|
112
|
+
if (rs.length < 2)
|
|
113
|
+
return null;
|
|
114
|
+
const first = rs[0].evidence;
|
|
115
|
+
if (!first)
|
|
116
|
+
return null;
|
|
117
|
+
return rs.every((r) => r.evidence === first) ? first : null;
|
|
118
|
+
}
|
|
74
119
|
export function generateReport(results, meta) {
|
|
75
120
|
const clean = results.map(sanitize);
|
|
76
121
|
const lines = [];
|
|
77
122
|
lines.push(`RuleReceipt · ${meta.ruleCount} rules checked`);
|
|
78
123
|
lines.push("─".repeat(40));
|
|
79
|
-
for (const
|
|
80
|
-
|
|
81
|
-
if (
|
|
82
|
-
|
|
124
|
+
for (const bucket of BUCKET_ORDER) {
|
|
125
|
+
const inBucket = clean.filter((r) => bucketOf(r) === bucket);
|
|
126
|
+
if (inBucket.length === 0)
|
|
127
|
+
continue;
|
|
128
|
+
lines.push("");
|
|
129
|
+
lines.push(`${BUCKET_LABEL[bucket]} (${inBucket.length})`);
|
|
130
|
+
// A hoisted section states its status once in the heading, so the rows
|
|
131
|
+
// carry only the rules. Repeating "? UNCLEAR" on all 13 lines beneath
|
|
132
|
+
// "Needs your judgment (13)" is the same repetition one size smaller.
|
|
133
|
+
const shared = sharedEvidence(inBucket);
|
|
134
|
+
if (shared) {
|
|
135
|
+
lines.push(` ${shared}`);
|
|
136
|
+
lines.push("");
|
|
137
|
+
for (const r of inBucket)
|
|
138
|
+
lines.push(` ${ruleLabel(r, clean)}`);
|
|
139
|
+
continue;
|
|
140
|
+
}
|
|
141
|
+
for (const r of inBucket) {
|
|
142
|
+
lines.push(`${MARK[r.status]} ${r.status.padEnd(7)} ${ruleLabel(r, clean)}`);
|
|
143
|
+
if (r.evidence)
|
|
144
|
+
lines.push(` evidence: ${r.evidence}`);
|
|
145
|
+
}
|
|
83
146
|
}
|
|
147
|
+
lines.push("");
|
|
84
148
|
lines.push("─".repeat(40));
|
|
85
149
|
lines.push(summaryLine(clean));
|
|
86
150
|
const hash = computeTranscriptHash(meta.sessionFilePath);
|
package/dist/rules.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { homedir } from "node:os";
|
|
2
|
-
import { dirname, join, parse } from "node:path";
|
|
2
|
+
import { dirname, join, parse, resolve } from "node:path";
|
|
3
3
|
import { existsSync, readdirSync, statSync } from "node:fs";
|
|
4
4
|
import { parseClaudeMd } from "./parsers/readClaudeMd.js";
|
|
5
5
|
import { findClaudeHomeDirNames } from "./parsers/transcriptParser.js";
|
|
@@ -78,11 +78,20 @@ function findProjectRuleFiles(cwd) {
|
|
|
78
78
|
const home = homedir();
|
|
79
79
|
let dir = cwd;
|
|
80
80
|
for (;;) {
|
|
81
|
+
// The home directory's rules are GLOBAL, and loadRules reads them as
|
|
82
|
+
// such. Guard BEFORE collecting: the original break sat below the push,
|
|
83
|
+
// so the home level was always collected first and every global rule was
|
|
84
|
+
// reported a second time as a project rule. Found 2026-09-08 — a machine
|
|
85
|
+
// with one 14-rule file reported "28 rules checked", doubling every
|
|
86
|
+
// figure on the report. Running from inside the home dir still collects
|
|
87
|
+
// it; loadRules dedupes that against the global pass.
|
|
88
|
+
if (dir === home && dir !== cwd)
|
|
89
|
+
break;
|
|
81
90
|
found.push(...ruleFilesAtLevel(dir));
|
|
82
91
|
// stop AT the repo root (inclusive) — its rules do apply
|
|
83
92
|
if (existsSync(join(dir, ".git")))
|
|
84
93
|
break;
|
|
85
|
-
if (dir === root
|
|
94
|
+
if (dir === root)
|
|
86
95
|
break;
|
|
87
96
|
const parent = dirname(dir);
|
|
88
97
|
if (parent === dir)
|
|
@@ -110,15 +119,24 @@ function findProjectRuleFiles(cwd) {
|
|
|
110
119
|
*/
|
|
111
120
|
export function loadRules(cwd) {
|
|
112
121
|
const rules = [];
|
|
122
|
+
// One file, one set of rules. Globals are read first, so a file reachable
|
|
123
|
+
// both ways keeps its "global" label. Without this, running the check from
|
|
124
|
+
// inside the home directory reported every global rule twice.
|
|
125
|
+
const seen = new Set();
|
|
126
|
+
const read = (path, source) => {
|
|
127
|
+
const key = resolve(path);
|
|
128
|
+
if (seen.has(key))
|
|
129
|
+
return;
|
|
130
|
+
seen.add(key);
|
|
131
|
+
rules.push(...parseClaudeMd(path, source));
|
|
132
|
+
};
|
|
113
133
|
for (const dirName of findClaudeHomeDirNames()) {
|
|
114
134
|
const base = join(homedir(), dirName);
|
|
115
|
-
|
|
116
|
-
for (const file of markdownFilesIn(join(base, "rules")))
|
|
117
|
-
|
|
118
|
-
}
|
|
119
|
-
}
|
|
120
|
-
for (const path of findProjectRuleFiles(cwd)) {
|
|
121
|
-
rules.push(...parseClaudeMd(path, "project"));
|
|
135
|
+
read(join(base, "CLAUDE.md"), "global");
|
|
136
|
+
for (const file of markdownFilesIn(join(base, "rules")))
|
|
137
|
+
read(file, "global");
|
|
122
138
|
}
|
|
139
|
+
for (const path of findProjectRuleFiles(cwd))
|
|
140
|
+
read(path, "project");
|
|
123
141
|
return rules;
|
|
124
142
|
}
|