rulereceipt 0.1.20 → 0.1.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +17 -5
- package/dist/report/generateHtmlReport.js +50 -15
- package/dist/report/generateReport.js +23 -2
- package/dist/types.d.ts +16 -0
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -92,6 +92,7 @@ function needsLlmResult(rule) {
|
|
|
92
92
|
ruleTitle: rule.title,
|
|
93
93
|
ruleSource: rule.source,
|
|
94
94
|
status: "UNCLEAR",
|
|
95
|
+
needsHuman: true,
|
|
95
96
|
evidence: "NEEDS HUMAN REVIEW — this rule is a judgment call, not something that can be settled by looking at what commands ran. Read the session and decide for yourself. (`--llm` will give you a model's opinion on it, using your own Anthropic key — an opinion, not a verdict.)",
|
|
96
97
|
};
|
|
97
98
|
}
|
|
@@ -174,11 +175,22 @@ async function runCheck(opts) {
|
|
|
174
175
|
const codeContent = classifications.filter((c) => c.kind === "codeContent");
|
|
175
176
|
const fileLifecycle = classifications.filter((c) => c.kind === "fileLifecycle");
|
|
176
177
|
const judgment = classifications.filter((c) => c.kind === "judgment");
|
|
177
|
-
// Not rules at all — documentation, glossary entries, reference tables
|
|
178
|
-
//
|
|
179
|
-
//
|
|
180
|
-
//
|
|
181
|
-
//
|
|
178
|
+
// Not rules at all — documentation, glossary entries, reference tables,
|
|
179
|
+
// URLs, directory listings, code examples.
|
|
180
|
+
//
|
|
181
|
+
// Re-measured 2026-08-31 across 40 real public rule files: 943 of 1,441
|
|
182
|
+
// parsed items, i.e. 65.4%. A previous comment here said ~17%, which was
|
|
183
|
+
// wrong and made the tool look like it was discarding far less than it
|
|
184
|
+
// is. Sampled and reviewed by hand before trusting the new figure: the
|
|
185
|
+
// classification is correct, real rules files simply are mostly prose.
|
|
186
|
+
//
|
|
187
|
+
// The number that actually matters is the one about RULES rather than
|
|
188
|
+
// lines: of the 498 genuine rules in that corpus, 238 (47.8%) are
|
|
189
|
+
// mechanically answerable and 260 (52.2%) are judgment calls.
|
|
190
|
+
//
|
|
191
|
+
// Reported as a count so nothing is silently dropped, but never checked,
|
|
192
|
+
// since "did the session violate a directory listing" has no meaningful
|
|
193
|
+
// answer and any coincidental match is pure noise.
|
|
182
194
|
const notARule = classifications.filter((c) => c.kind === "notARule");
|
|
183
195
|
const deterministicResults = [
|
|
184
196
|
...runDeterministicChecks(deterministic, events),
|
|
@@ -25,17 +25,32 @@ function stripControlChars(value) {
|
|
|
25
25
|
function clean(value) {
|
|
26
26
|
return escapeHtml(stripControlChars(value));
|
|
27
27
|
}
|
|
28
|
-
|
|
28
|
+
function bucketOf(result) {
|
|
29
|
+
if (result.status === "FAIL")
|
|
30
|
+
return "FAIL";
|
|
31
|
+
if (result.status === "PASS")
|
|
32
|
+
return "PASS";
|
|
33
|
+
return result.needsHuman ? "UNCLEAR_JUDGMENT" : "UNCLEAR_EVIDENCE";
|
|
34
|
+
}
|
|
35
|
+
const BUCKET_LABEL = {
|
|
29
36
|
FAIL: "Not followed",
|
|
37
|
+
UNCLEAR_EVIDENCE: "Couldn't tell",
|
|
38
|
+
UNCLEAR_JUDGMENT: "Needs your judgment",
|
|
30
39
|
PASS: "Followed",
|
|
31
|
-
|
|
40
|
+
};
|
|
41
|
+
const BUCKET_CLASS = {
|
|
42
|
+
FAIL: "fail",
|
|
43
|
+
UNCLEAR_EVIDENCE: "unclear",
|
|
44
|
+
UNCLEAR_JUDGMENT: "judgment",
|
|
45
|
+
PASS: "pass",
|
|
32
46
|
};
|
|
33
47
|
/**
|
|
34
48
|
* Failures first, deliberately. A report that opens with a wall of passes
|
|
35
49
|
* and buries one failure at the bottom is a report designed to be skimmed
|
|
36
|
-
* past — the opposite of what an audit document is for.
|
|
50
|
+
* past — the opposite of what an audit document is for. Judgment calls
|
|
51
|
+
* sit last: they're expected, and they're the longest section.
|
|
37
52
|
*/
|
|
38
|
-
const
|
|
53
|
+
const BUCKET_ORDER = ["FAIL", "UNCLEAR_EVIDENCE", "PASS", "UNCLEAR_JUDGMENT"];
|
|
39
54
|
function ruleLabel(result, all) {
|
|
40
55
|
const collides = all.filter((other) => other.ruleId === result.ruleId).length > 1;
|
|
41
56
|
return collides
|
|
@@ -46,24 +61,29 @@ function countBy(results, status) {
|
|
|
46
61
|
return results.filter((r) => r.status === status).length;
|
|
47
62
|
}
|
|
48
63
|
function renderResultRow(result, all) {
|
|
49
|
-
const
|
|
64
|
+
const bucket = bucketOf(result);
|
|
65
|
+
const cls = BUCKET_CLASS[bucket];
|
|
50
66
|
return `
|
|
51
67
|
<article class="result result--${cls}">
|
|
52
68
|
<div class="result__head">
|
|
53
|
-
<span class="badge badge--${cls}">${clean(
|
|
69
|
+
<span class="badge badge--${cls}">${clean(BUCKET_LABEL[bucket])}</span>
|
|
54
70
|
<span class="result__id">${clean(ruleLabel(result, all))}</span>
|
|
55
71
|
</div>
|
|
56
72
|
<h3 class="result__title">${clean(result.ruleTitle)}</h3>
|
|
57
73
|
${result.evidence ? `<p class="result__evidence">${clean(result.evidence)}</p>` : ""}
|
|
58
74
|
</article>`;
|
|
59
75
|
}
|
|
60
|
-
function renderSection(
|
|
61
|
-
const inSection = results.filter((r) => r
|
|
76
|
+
function renderSection(bucket, results, all) {
|
|
77
|
+
const inSection = results.filter((r) => bucketOf(r) === bucket);
|
|
62
78
|
if (inSection.length === 0)
|
|
63
79
|
return "";
|
|
80
|
+
const note = bucket === "UNCLEAR_JUDGMENT"
|
|
81
|
+
? `<p class="section__note">These were never questions a tool could settle — they need someone to read the session and decide. That is expected, not a gap in the check.</p>`
|
|
82
|
+
: "";
|
|
64
83
|
return `
|
|
65
84
|
<section class="section">
|
|
66
|
-
<h2 class="section__title">${clean(
|
|
85
|
+
<h2 class="section__title">${clean(BUCKET_LABEL[bucket])} <span class="section__count">${inSection.length}</span></h2>
|
|
86
|
+
${note}
|
|
67
87
|
${inSection.map((r) => renderResultRow(r, all)).join("")}
|
|
68
88
|
</section>`;
|
|
69
89
|
}
|
|
@@ -75,12 +95,22 @@ function renderSection(status, results, all) {
|
|
|
75
95
|
*/
|
|
76
96
|
function verdict(results) {
|
|
77
97
|
const fail = countBy(results, "FAIL");
|
|
78
|
-
const
|
|
98
|
+
const couldntTell = results.filter((r) => bucketOf(r) === "UNCLEAR_EVIDENCE").length;
|
|
99
|
+
const needsHuman = results.filter((r) => bucketOf(r) === "UNCLEAR_JUDGMENT").length;
|
|
79
100
|
if (fail > 0) {
|
|
80
101
|
return { text: `${fail} rule${fail === 1 ? "" : "s"} not followed`, cls: "fail" };
|
|
81
102
|
}
|
|
82
|
-
if (
|
|
83
|
-
return { text: `No rule violations found · ${
|
|
103
|
+
if (couldntTell > 0) {
|
|
104
|
+
return { text: `No rule violations found · ${couldntTell} couldn't be determined`, cls: "unclear" };
|
|
105
|
+
}
|
|
106
|
+
// Judgment rules still appear in the headline. They must not be styled
|
|
107
|
+
// as a problem — they aren't one — but they must not be omitted either.
|
|
108
|
+
// "No rule violations found" on its own, when half the rules were never
|
|
109
|
+
// evaluated mechanically, reads as full coverage. Naming the count is
|
|
110
|
+
// what keeps the headline from overclaiming, and it is the property the
|
|
111
|
+
// "does not claim a clean result" test exists to hold.
|
|
112
|
+
if (needsHuman > 0) {
|
|
113
|
+
return { text: `No rule violations found · ${needsHuman} still need human review`, cls: "judgment" };
|
|
84
114
|
}
|
|
85
115
|
return { text: "No rule violations found", cls: "pass" };
|
|
86
116
|
}
|
|
@@ -117,6 +147,7 @@ export function generateHtmlReport(results, meta) {
|
|
|
117
147
|
.verdict--fail { background: var(--fail-bg); border-color: #f0c4c6; }
|
|
118
148
|
.verdict--pass { background: var(--pass-bg); border-color: #bfe3cd; }
|
|
119
149
|
.verdict--unclear { background: var(--unclear-bg); border-color: #ecdcb0; }
|
|
150
|
+
.verdict--judgment { background: var(--panel); border-color: var(--line); }
|
|
120
151
|
.verdict strong { display: block; font-size: 19px; margin-bottom: 2px; }
|
|
121
152
|
.verdict span { color: var(--muted); font-size: 14px; }
|
|
122
153
|
.facts { width: 100%; border-collapse: collapse; margin-bottom: 32px; font-size: 14px; }
|
|
@@ -136,6 +167,9 @@ export function generateHtmlReport(results, meta) {
|
|
|
136
167
|
.badge--fail { background: var(--fail-bg); color: var(--fail); }
|
|
137
168
|
.badge--pass { background: var(--pass-bg); color: var(--pass); }
|
|
138
169
|
.badge--unclear { background: var(--unclear-bg); color: var(--unclear); }
|
|
170
|
+
.result--judgment { border-left-color: #6b6f76; }
|
|
171
|
+
.badge--judgment { background: #f2f3f5; color: #4a4e55; }
|
|
172
|
+
.section__note { font-size: 13px; color: var(--muted); margin: -4px 0 12px; }
|
|
139
173
|
.result__id { font-size: 12px; color: var(--muted); }
|
|
140
174
|
.result__title { font-size: 15px; margin: 0 0 6px; font-weight: 600; }
|
|
141
175
|
.result__evidence { margin: 0; font-size: 14px; color: var(--muted); white-space: pre-wrap; }
|
|
@@ -166,7 +200,7 @@ export function generateHtmlReport(results, meta) {
|
|
|
166
200
|
|
|
167
201
|
<div class="verdict verdict--${v.cls}">
|
|
168
202
|
<strong>${clean(v.text)}</strong>
|
|
169
|
-
<span>${countBy(results, "PASS")} followed · ${countBy(results, "FAIL")} not followed · ${
|
|
203
|
+
<span>${countBy(results, "PASS")} followed · ${countBy(results, "FAIL")} not followed · ${results.filter((r) => bucketOf(r) === "UNCLEAR_EVIDENCE").length} couldn't tell · ${results.filter((r) => bucketOf(r) === "UNCLEAR_JUDGMENT").length} need your judgment</span>
|
|
170
204
|
</div>
|
|
171
205
|
|
|
172
206
|
<table class="facts">
|
|
@@ -177,14 +211,15 @@ export function generateHtmlReport(results, meta) {
|
|
|
177
211
|
<tr><th>Tool version</th><td><code>rulereceipt ${clean(meta.toolVersion)}</code></td></tr>
|
|
178
212
|
</table>
|
|
179
213
|
|
|
180
|
-
${
|
|
214
|
+
${BUCKET_ORDER.map((b) => renderSection(b, results, results)).join("")}
|
|
181
215
|
|
|
182
216
|
<div class="note">
|
|
183
217
|
<h2>How to read this report</h2>
|
|
184
218
|
<ul>
|
|
185
219
|
<li><strong>Followed</strong> — a specific action in the session satisfies the rule, or the forbidden action never occurred.</li>
|
|
186
220
|
<li><strong>Not followed</strong> — a real action in the session contradicts the rule. The evidence quotes it.</li>
|
|
187
|
-
<li><strong>
|
|
221
|
+
<li><strong>Couldn't tell</strong> — the check ran and the evidence was ambiguous. A genuine gap.</li>
|
|
222
|
+
<li><strong>Needs your judgment</strong> — this rule never had a mechanical answer ("surface bad news first"). Not a pass, not a failure, and not a shortcoming of the check: it is the part that was always a person's call.</li>
|
|
188
223
|
</ul>
|
|
189
224
|
<h2>What this report does not establish</h2>
|
|
190
225
|
<ul>
|
|
@@ -44,11 +44,32 @@ function ruleLabel(r, results) {
|
|
|
44
44
|
const collides = results.filter((other) => other.ruleId === r.ruleId).length > 1;
|
|
45
45
|
return collides ? `Rule ${r.ruleId} (${r.ruleSource}) — ${r.ruleTitle}` : `Rule ${r.ruleId} — ${r.ruleTitle}`;
|
|
46
46
|
}
|
|
47
|
+
/**
|
|
48
|
+
* Separates the two very different things that were both being reported
|
|
49
|
+
* as "unclear":
|
|
50
|
+
*
|
|
51
|
+
* - couldn't tell — the tool looked and the evidence was ambiguous.
|
|
52
|
+
* - needs you — a judgment call that never had a mechanical
|
|
53
|
+
* answer ("surface bad news first").
|
|
54
|
+
*
|
|
55
|
+
* Collapsing them made the tool look like it failed on most rules. It
|
|
56
|
+
* doesn't: across 40 real rules files, 52.2% of the actual rules people
|
|
57
|
+
* write are judgment calls. Those aren't the tool falling short, they're
|
|
58
|
+
* the half of the work that was always a human's. Naming that honestly is
|
|
59
|
+
* the difference between a report that reads as broken and one that reads
|
|
60
|
+
* as a division of labour.
|
|
61
|
+
*/
|
|
47
62
|
function summaryLine(results) {
|
|
48
63
|
const pass = results.filter((r) => r.status === "PASS").length;
|
|
49
64
|
const fail = results.filter((r) => r.status === "FAIL").length;
|
|
50
|
-
const
|
|
51
|
-
|
|
65
|
+
const needsHuman = results.filter((r) => r.status === "UNCLEAR" && r.needsHuman).length;
|
|
66
|
+
const couldntTell = results.filter((r) => r.status === "UNCLEAR" && !r.needsHuman).length;
|
|
67
|
+
const parts = [`${pass} followed`, `${fail} not followed`];
|
|
68
|
+
if (couldntTell > 0)
|
|
69
|
+
parts.push(`${couldntTell} couldn't tell`);
|
|
70
|
+
if (needsHuman > 0)
|
|
71
|
+
parts.push(`${needsHuman} need your judgment`);
|
|
72
|
+
return parts.join(" · ");
|
|
52
73
|
}
|
|
53
74
|
export function generateReport(results, meta) {
|
|
54
75
|
const clean = results.map(sanitize);
|
package/dist/types.d.ts
CHANGED
|
@@ -32,4 +32,20 @@ export interface CheckResult {
|
|
|
32
32
|
ruleSource: "global" | "project";
|
|
33
33
|
status: CheckStatus;
|
|
34
34
|
evidence: string;
|
|
35
|
+
/**
|
|
36
|
+
* True when this rule was never mechanically answerable — a judgment
|
|
37
|
+
* call like "surface bad news first", which has no command to inspect.
|
|
38
|
+
*
|
|
39
|
+
* Exists because collapsing these into a single UNCLEAR count made the
|
|
40
|
+
* tool look broken. Measured across 40 real rules files: of the actual
|
|
41
|
+
* rules people write, 47.8% are mechanically answerable and 52.2% are
|
|
42
|
+
* judgment calls. Reporting "1 pass · 0 fail · 14 unclear" reads as
|
|
43
|
+
* fourteen failures, when most of those were a human's call from the
|
|
44
|
+
* start and the tool is working exactly as intended.
|
|
45
|
+
*
|
|
46
|
+
* "I could not determine this" and "this was always yours to decide"
|
|
47
|
+
* are different statements, and a tool about honest reporting should
|
|
48
|
+
* not blur them.
|
|
49
|
+
*/
|
|
50
|
+
needsHuman?: boolean;
|
|
35
51
|
}
|