rulereceipt 0.1.33 → 0.1.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/checks/classify.js +24 -4
- package/dist/checks/judgmentChecks.js +43 -7
- package/package.json +1 -1
package/dist/checks/classify.js
CHANGED
|
@@ -297,7 +297,27 @@ const FORBID_SIGNAL = /\b(never|don'?t|do not|forbidden|banned|prohibited|disall
|
|
|
297
297
|
// often prescribes with a bare imperative ("Use `gh pr merge --merge`")
|
|
298
298
|
// rather than "you must use" — and it's exactly those imperative clauses
|
|
299
299
|
// whose literals get misattributed to the prohibiting half.
|
|
300
|
-
|
|
300
|
+
/**
|
|
301
|
+
* Prescriptive verbs, in two lists, because they serve two jobs that pull in
|
|
302
|
+
* opposite directions.
|
|
303
|
+
*
|
|
304
|
+
* MIXED_POLARITY_VERB is the narrow one, used only to spot a prohibition
|
|
305
|
+
* followed by a prescription — "NEVER squash when merging PRs. Use `gh pr
|
|
306
|
+
* merge --merge`" — where the prescribed half's literals would otherwise be
|
|
307
|
+
* checked with the prohibition's polarity.
|
|
308
|
+
*
|
|
309
|
+
* INFERRED_REQUIRE_VERB is the wide one, used only to read the direction of a
|
|
310
|
+
* bare imperative when no signal word says which way a rule points.
|
|
311
|
+
*
|
|
312
|
+
* They were one list until 2026-09-14. Widening it for the second job broke
|
|
313
|
+
* the first: a rule reading "these databases must NEVER be deleted" stopped
|
|
314
|
+
* being checked at all, because a later bullet — "Any DB whose table names
|
|
315
|
+
* include: ..." — carries "include" with no forbid word beside it and so read
|
|
316
|
+
* as mixed polarity. That rule had been the only decided verdict on a real
|
|
317
|
+
* file, and the report went to 0 followed out of 15.
|
|
318
|
+
*/
|
|
319
|
+
const MIXED_POLARITY_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to)\b/i;
|
|
320
|
+
const INFERRED_REQUIRE_VERB = /\b(use|run|prefer|apply|follow|call|invoke|stick to|update|write|create|add|include|keep|maintain|document)\b/i;
|
|
301
321
|
function hasMixedPolarity(rule) {
|
|
302
322
|
const text = `${rule.title} ${rule.text}`;
|
|
303
323
|
if (!FORBID_SIGNAL.test(text))
|
|
@@ -317,7 +337,7 @@ function hasMixedPolarity(rule) {
|
|
|
317
337
|
// polarity and sent to judgment instead of being checked.
|
|
318
338
|
const masked = text.replace(/`[^`]*`/g, (m) => "`" + "x".repeat(Math.max(m.length - 2, 0)) + "`");
|
|
319
339
|
const clauses = masked.split(/[.;\n]|(?:\s+-\s+)/).filter((c) => c.trim().length > 0);
|
|
320
|
-
return clauses.some((clause) => !FORBID_SIGNAL.test(clause) && (REQUIRE_SIGNAL.test(clause) ||
|
|
340
|
+
return clauses.some((clause) => !FORBID_SIGNAL.test(clause) && (REQUIRE_SIGNAL.test(clause) || MIXED_POLARITY_VERB.test(clause)));
|
|
321
341
|
}
|
|
322
342
|
/**
|
|
323
343
|
* The direction of a rule, or null when the text does not say.
|
|
@@ -352,7 +372,7 @@ function detectPolarity(rule) {
|
|
|
352
372
|
// pattern that never appears reports UNCLEAR, while a forbidden one that
|
|
353
373
|
// appears reports a violation. Guessing toward require can waste a check;
|
|
354
374
|
// guessing toward forbid accuses someone.
|
|
355
|
-
if (
|
|
375
|
+
if (INFERRED_REQUIRE_VERB.test(text))
|
|
356
376
|
return "require";
|
|
357
377
|
return null;
|
|
358
378
|
}
|
|
@@ -370,7 +390,7 @@ function polarityWasInferred(rule) {
|
|
|
370
390
|
const text = `${rule.title} ${rule.text}`;
|
|
371
391
|
if (FORBID_SIGNAL.test(text) || REQUIRE_SIGNAL.test(text))
|
|
372
392
|
return false;
|
|
373
|
-
return
|
|
393
|
+
return INFERRED_REQUIRE_VERB.test(text);
|
|
374
394
|
}
|
|
375
395
|
/**
|
|
376
396
|
* A rule is only treated as deterministic when it names a specific,
|
|
@@ -86,23 +86,46 @@ function summarizeEvents(events) {
|
|
|
86
86
|
*/
|
|
87
87
|
const RESULT_TOOL = {
|
|
88
88
|
name: "report_result",
|
|
89
|
-
description: "Report
|
|
89
|
+
description: "Report whether this one rule was followed, violated, unclear, or never applicable, with a verbatim line of evidence from the session.",
|
|
90
90
|
input_schema: {
|
|
91
91
|
type: "object",
|
|
92
92
|
properties: {
|
|
93
|
-
status: { type: "string", enum: ["PASS", "FAIL", "UNCLEAR"] },
|
|
93
|
+
status: { type: "string", enum: ["PASS", "FAIL", "UNCLEAR", "NOT_APPLICABLE"] },
|
|
94
94
|
evidence: {
|
|
95
95
|
type: "string",
|
|
96
|
-
description: "A short VERBATIM extract
|
|
96
|
+
description: "A short VERBATIM extract FROM THE SESSION TRANSCRIPT — copied exactly as it appears there, not reworded, not summarised, and not taken from the rule text. If nothing in the transcript supports a verdict, report UNCLEAR or NOT_APPLICABLE and leave this brief.",
|
|
97
97
|
},
|
|
98
98
|
},
|
|
99
99
|
required: ["status", "evidence"],
|
|
100
100
|
},
|
|
101
101
|
};
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
102
|
+
/**
|
|
103
|
+
* The four verdicts, and the balance between them.
|
|
104
|
+
*
|
|
105
|
+
* The previous version offered three and told the model only "never guess
|
|
106
|
+
* PASS when you are not sure". With no NOT_APPLICABLE, a rule that simply
|
|
107
|
+
* never came up had to be forced into one of pass, fail or unclear — and the
|
|
108
|
+
* single stated pressure pointed at the accusing one. Measured against the
|
|
109
|
+
* four-event example session on 2026-09-14: 30 rules, 10 FAILs, on a session
|
|
110
|
+
* containing one genuine issue. One failure cited the prompt itself as
|
|
111
|
+
* evidence.
|
|
112
|
+
*
|
|
113
|
+
* Most rules do not apply to most sessions. Saying so is the correction.
|
|
114
|
+
*/
|
|
115
|
+
const INSTRUCTIONS = "You judge whether one rule from a CLAUDE.md/AGENTS.md file was actually followed during a Claude Code session.\n\n" +
|
|
116
|
+
"MOST RULES WILL NOT APPLY. A session is usually a few minutes of work, and a rules file covers everything a project " +
|
|
117
|
+
"might ever do. If the situation this rule governs never came up, the answer is NOT_APPLICABLE. That is the common " +
|
|
118
|
+
"case and it is not a failure of any kind.\n\n" +
|
|
119
|
+
"PASS only when the transcript clearly shows the rule was followed.\n" +
|
|
120
|
+
"FAIL only when the transcript clearly shows it was violated.\n" +
|
|
121
|
+
"UNCLEAR when the situation arose but the transcript does not settle what happened.\n" +
|
|
122
|
+
"NOT_APPLICABLE when the situation the rule governs never arose.\n\n" +
|
|
123
|
+
"Do not guess in either direction. Guessing PASS invents compliance; guessing FAIL accuses someone of something they " +
|
|
124
|
+
"may not have done, which is the more expensive mistake and the harder one to recover from. If a rule is only loosely " +
|
|
125
|
+
"related to something in the session, that is NOT_APPLICABLE, not FAIL.\n\n" +
|
|
126
|
+
"Your evidence must be copied verbatim from the SESSION TRANSCRIPT. Never quote the rule back as evidence, and never " +
|
|
127
|
+
"quote these instructions. If you cannot find a line in the transcript that supports your verdict, you do not have a " +
|
|
128
|
+
"verdict.";
|
|
106
129
|
/**
|
|
107
130
|
* A rule the check never actually ran against.
|
|
108
131
|
*
|
|
@@ -219,6 +242,19 @@ export async function runJudgmentChecks(classifications, events) {
|
|
|
219
242
|
}
|
|
220
243
|
const parsed = toolUseBlock.input;
|
|
221
244
|
const status = parsed.status;
|
|
245
|
+
if (status === "NOT_APPLICABLE") {
|
|
246
|
+
return {
|
|
247
|
+
ruleId: rule.id,
|
|
248
|
+
ruleTitle: rule.title,
|
|
249
|
+
ruleSource: rule.source,
|
|
250
|
+
status: "UNCLEAR",
|
|
251
|
+
outcome: "not_applicable",
|
|
252
|
+
method: "model_judgment",
|
|
253
|
+
evidence: parsed.evidence?.trim()
|
|
254
|
+
? parsed.evidence
|
|
255
|
+
: "the situation this rule governs never arose in this session",
|
|
256
|
+
};
|
|
257
|
+
}
|
|
222
258
|
if (status === "PASS" || status === "FAIL" || status === "UNCLEAR") {
|
|
223
259
|
const evidence = parsed.evidence ?? "";
|
|
224
260
|
return {
|