vigiles 12.7.0 → 12.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -15
- package/dist/audit-report.d.ts +38 -0
- package/dist/audit-report.js +13 -2
- package/dist/audit-report.template.html +44 -29
- package/dist/audit-score.js +9 -1
- package/dist/audit-verdict.d.ts +89 -0
- package/dist/audit-verdict.js +281 -0
- package/dist/cli.js +130 -0
- package/dist/eval-cost.d.ts +1 -1
- package/dist/eval.d.ts +2 -2
- package/dist/eval.js +1 -1
- package/dist/leaderboard.js +9 -3
- package/dist/rule-inventory.d.ts +90 -0
- package/dist/rule-inventory.js +327 -0
- package/dist/rule-routing.d.ts +46 -0
- package/dist/rule-routing.js +135 -0
- package/dist/scaffold-test.js +1 -1
- package/dist/segment.d.ts +33 -0
- package/dist/segment.js +454 -0
- package/package.json +1 -1
package/dist/audit-score.js
CHANGED
|
@@ -33,6 +33,11 @@ const leaderboard_js_1 = require("./leaderboard.js");
|
|
|
33
33
|
// category rings and the single health number can never drift. W_UNTESTED is
|
|
34
34
|
// audit-only — untested surfaces are advisory (shown, never scored into overall).
|
|
35
35
|
const W_UNTESTED = 3;
|
|
36
|
+
/** Resolve the terse "thing(s)" plural placeholder against a count:
|
|
37
|
+
* n===1 drops the "(s)" ("1 tool"); otherwise it becomes "s" ("3 tools"). */
|
|
38
|
+
function pluralizeLabel(n, label) {
|
|
39
|
+
return label.replace(/\(s\)/g, n === 1 ? "" : "s");
|
|
40
|
+
}
|
|
36
41
|
/** Apply deductions to a 100 base, clamped to [0,100], collecting non-zero labels. */
|
|
37
42
|
function scoreFrom(deductions) {
|
|
38
43
|
let penalty = 0;
|
|
@@ -41,7 +46,10 @@ function scoreFrom(deductions) {
|
|
|
41
46
|
if (d.n <= 0)
|
|
42
47
|
continue;
|
|
43
48
|
penalty += d.n * d.weight;
|
|
44
|
-
findings.push({
|
|
49
|
+
findings.push({
|
|
50
|
+
n: d.n,
|
|
51
|
+
text: `${String(d.n)} ${pluralizeLabel(d.n, d.label)}`,
|
|
52
|
+
});
|
|
45
53
|
}
|
|
46
54
|
findings.sort((a, b) => b.n - a.n);
|
|
47
55
|
return {
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The audit VERDICT engine — a header sentence + per-recommendation `pointsIfFixed`,
|
|
3
|
+
* both derived by RE-SCORING, never by a hardcoded template with a fake number.
|
|
4
|
+
*
|
|
5
|
+
* The audit scorer ({@link auditScore}) is pure and deterministic: `overall` is
|
|
6
|
+
* `100 − Σ(graded penalties)` (clamped to [0,100]) and the letter grade comes from
|
|
7
|
+
* fixed thresholds (A ≥90 … F <60). Because it's pure, we can answer two "what if"
|
|
8
|
+
* questions by RUNNING it again with a single recommendation's finding(s) removed
|
|
9
|
+
* and diffing the `overall`:
|
|
10
|
+
*
|
|
11
|
+
* 1. `pointsIfFixed` per recommendation — the exact number of overall points the
|
|
12
|
+
* grade gains if THAT one fix is applied (so a fix card can show `+N pts` and
|
|
13
|
+
* sort by it). Computed as `overall(report − thisFinding) − overall(report)`.
|
|
14
|
+
* 2. A verdict `sentence` for the report header — e.g. "Two one-line fixes away
|
|
15
|
+
* from an A." — where the COUNT is the minimal number of fixes whose COMBINED
|
|
16
|
+
* removal actually crosses the next grade threshold (a real cumulative
|
|
17
|
+
* re-score), and `pointsToNextGrade` is the real threshold gap.
|
|
18
|
+
*
|
|
19
|
+
* Every recommendation maps 1:1 to exactly one graded finding (see `optimize` /
|
|
20
|
+
* `explainScore`): each is a single deduction of a known weight, so removing it
|
|
21
|
+
* lowers the penalty by that weight (modulo the [0,100] clamp). We remove the
|
|
22
|
+
* finding by its detector + content (matched against the given recommendation),
|
|
23
|
+
* NOT by reproducing `explainScore`'s ordering — so this stays aligned with the
|
|
24
|
+
* `recommendations` array the caller passes, by index.
|
|
25
|
+
*
|
|
26
|
+
* Pure over its inputs (the same pieces `buildAuditReport` already has): no fs, no
|
|
27
|
+
* clock, no model, no mutation of the caller's report.
|
|
28
|
+
*
|
|
29
|
+
* LIMITATION (documented, not fabricated): some graded penalties have NO
|
|
30
|
+
* corresponding recommendation (a hard lethal-trifecta contract, an MCP server
|
|
31
|
+
* that can't start, a `disallowedTools` typo, an invalid model/color, unresolved
|
|
32
|
+
* skill resources, an invisible skill, a misplaced plugin dir, an ineffective /
|
|
33
|
+
* never-firing hook). Those cannot be "fixed" through a recommendation here, so
|
|
34
|
+
* they never contribute to `pointsIfFixed` and can make `fixesToNextGrade` null
|
|
35
|
+
* (the gap can't be closed by the deterministic fix list alone) — in which case
|
|
36
|
+
* the verdict leads with the dominant blocking finding instead. The
|
|
37
|
+
* fixes-to-next-grade count uses a greedy largest-delta-first ordering; that is
|
|
38
|
+
* provably minimal when penalties are additive (the common case, away from the
|
|
39
|
+
* score-0 clamp) and a close upper bound otherwise.
|
|
40
|
+
*/
|
|
41
|
+
import { type AuditScore } from "./audit-score.js";
|
|
42
|
+
import type { Recommendation } from "./optimize.js";
|
|
43
|
+
import { type PluginScore } from "./leaderboard.js";
|
|
44
|
+
import type { ScanReport } from "./scan.js";
|
|
45
|
+
/**
|
|
46
|
+
* The inputs the verdict needs — exactly the pieces {@link buildAuditReport}
|
|
47
|
+
* already holds. `score` is the authoritative base (`overall` + `grade`) the
|
|
48
|
+
* report displays; `report` is re-scored with a finding removed to diff against
|
|
49
|
+
* it; `recommendations` is the array whose indices `perRecommendation` aligns to.
|
|
50
|
+
*/
|
|
51
|
+
export interface VerdictInput {
|
|
52
|
+
readonly report: ScanReport;
|
|
53
|
+
readonly score: AuditScore;
|
|
54
|
+
readonly recommendations: readonly Recommendation[];
|
|
55
|
+
}
|
|
56
|
+
/** One recommendation's overall-points gain if its single fix is applied. */
|
|
57
|
+
export interface RecommendationPoints {
|
|
58
|
+
/** Index into the input `recommendations` array. */
|
|
59
|
+
readonly index: number;
|
|
60
|
+
/**
|
|
61
|
+
* `overall(report − thisFinding) − overall(report)` — always ≥ 0 (removing a
|
|
62
|
+
* penalty can only raise or hold the score). Can be 0 when the score is clamped
|
|
63
|
+
* at 0 (removing one weight still leaves the penalty ≥ 100).
|
|
64
|
+
*/
|
|
65
|
+
readonly pointsIfFixed: number;
|
|
66
|
+
}
|
|
67
|
+
export interface Verdict {
|
|
68
|
+
/** The header sentence — real numbers from the re-score + grade thresholds. */
|
|
69
|
+
readonly sentence: string;
|
|
70
|
+
/** The current letter grade (echoed from the base score). */
|
|
71
|
+
readonly grade: PluginScore["grade"];
|
|
72
|
+
/** Points to the next-higher grade band, or null when already an A. */
|
|
73
|
+
readonly pointsToNextGrade: number | null;
|
|
74
|
+
/**
|
|
75
|
+
* The minimal number of deterministic fixes whose COMBINED removal crosses the
|
|
76
|
+
* next grade threshold (real cumulative re-score, greedy largest-delta-first),
|
|
77
|
+
* or null when the deterministic fix list can't close the gap (non-recommendation
|
|
78
|
+
* penalties dominate) or the grade is already an A.
|
|
79
|
+
*/
|
|
80
|
+
readonly fixesToNextGrade: number | null;
|
|
81
|
+
readonly perRecommendation: readonly RecommendationPoints[];
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* Compute the audit verdict + per-recommendation `pointsIfFixed` by re-scoring.
|
|
85
|
+
* Pure and deterministic over its inputs. `perRecommendation` is index-aligned to
|
|
86
|
+
* the input `recommendations`.
|
|
87
|
+
*/
|
|
88
|
+
export declare function computeVerdict(input: VerdictInput): Verdict;
|
|
89
|
+
//# sourceMappingURL=audit-verdict.d.ts.map
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* The audit VERDICT engine — a header sentence + per-recommendation `pointsIfFixed`,
|
|
4
|
+
* both derived by RE-SCORING, never by a hardcoded template with a fake number.
|
|
5
|
+
*
|
|
6
|
+
* The audit scorer ({@link auditScore}) is pure and deterministic: `overall` is
|
|
7
|
+
* `100 − Σ(graded penalties)` (clamped to [0,100]) and the letter grade comes from
|
|
8
|
+
* fixed thresholds (A ≥90 … F <60). Because it's pure, we can answer two "what if"
|
|
9
|
+
* questions by RUNNING it again with a single recommendation's finding(s) removed
|
|
10
|
+
* and diffing the `overall`:
|
|
11
|
+
*
|
|
12
|
+
* 1. `pointsIfFixed` per recommendation — the exact number of overall points the
|
|
13
|
+
* grade gains if THAT one fix is applied (so a fix card can show `+N pts` and
|
|
14
|
+
* sort by it). Computed as `overall(report − thisFinding) − overall(report)`.
|
|
15
|
+
* 2. A verdict `sentence` for the report header — e.g. "Two one-line fixes away
|
|
16
|
+
* from an A." — where the COUNT is the minimal number of fixes whose COMBINED
|
|
17
|
+
* removal actually crosses the next grade threshold (a real cumulative
|
|
18
|
+
* re-score), and `pointsToNextGrade` is the real threshold gap.
|
|
19
|
+
*
|
|
20
|
+
* Every recommendation maps 1:1 to exactly one graded finding (see `optimize` /
|
|
21
|
+
* `explainScore`): each is a single deduction of a known weight, so removing it
|
|
22
|
+
* lowers the penalty by that weight (modulo the [0,100] clamp). We remove the
|
|
23
|
+
* finding by its detector + content (matched against the given recommendation),
|
|
24
|
+
* NOT by reproducing `explainScore`'s ordering — so this stays aligned with the
|
|
25
|
+
* `recommendations` array the caller passes, by index.
|
|
26
|
+
*
|
|
27
|
+
* Pure over its inputs (the same pieces `buildAuditReport` already has): no fs, no
|
|
28
|
+
* clock, no model, no mutation of the caller's report.
|
|
29
|
+
*
|
|
30
|
+
* LIMITATION (documented, not fabricated): some graded penalties have NO
|
|
31
|
+
* corresponding recommendation (a hard lethal-trifecta contract, an MCP server
|
|
32
|
+
* that can't start, a `disallowedTools` typo, an invalid model/color, unresolved
|
|
33
|
+
* skill resources, an invisible skill, a misplaced plugin dir, an ineffective /
|
|
34
|
+
* never-firing hook). Those cannot be "fixed" through a recommendation here, so
|
|
35
|
+
* they never contribute to `pointsIfFixed` and can make `fixesToNextGrade` null
|
|
36
|
+
* (the gap can't be closed by the deterministic fix list alone) — in which case
|
|
37
|
+
* the verdict leads with the dominant blocking finding instead. The
|
|
38
|
+
* fixes-to-next-grade count uses a greedy largest-delta-first ordering; that is
|
|
39
|
+
* provably minimal when penalties are additive (the common case, away from the
|
|
40
|
+
* score-0 clamp) and a close upper bound otherwise.
|
|
41
|
+
*/
|
|
42
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
43
|
+
exports.computeVerdict = computeVerdict;
|
|
44
|
+
const audit_score_js_1 = require("./audit-score.js");
|
|
45
|
+
const leaderboard_js_1 = require("./leaderboard.js");
|
|
46
|
+
// The grade-band FLOORS (A ≥90 … D ≥60; below 60 is F), mirroring gradeFor.
|
|
47
|
+
const GRADE_FLOORS = [60, 70, 80, 90];
|
|
48
|
+
/** The smallest band floor strictly above `overall`, or null when already an A. */
|
|
49
|
+
function nextGradeFloor(overall) {
|
|
50
|
+
for (const floor of GRADE_FLOORS) {
|
|
51
|
+
if (overall < floor)
|
|
52
|
+
return floor;
|
|
53
|
+
}
|
|
54
|
+
return null;
|
|
55
|
+
}
|
|
56
|
+
const NUMBER_WORDS = [
|
|
57
|
+
"zero",
|
|
58
|
+
"one",
|
|
59
|
+
"two",
|
|
60
|
+
"three",
|
|
61
|
+
"four",
|
|
62
|
+
"five",
|
|
63
|
+
"six",
|
|
64
|
+
"seven",
|
|
65
|
+
"eight",
|
|
66
|
+
"nine",
|
|
67
|
+
"ten",
|
|
68
|
+
];
|
|
69
|
+
/** Small counts read as words ("two"); larger ones fall back to digits. */
|
|
70
|
+
function numberWord(n) {
|
|
71
|
+
return n >= 0 && n < NUMBER_WORDS.length ? NUMBER_WORDS[n] : String(n);
|
|
72
|
+
}
|
|
73
|
+
function capitalize(s) {
|
|
74
|
+
return s.length === 0 ? s : s[0].toUpperCase() + s.slice(1);
|
|
75
|
+
}
|
|
76
|
+
/** Resolve the scorer's terse "thing(s)" plural placeholder against a count —
|
|
77
|
+
* mirrors audit-score's `pluralizeLabel` so the verdict sentence reads "1 unit"
|
|
78
|
+
* / "3 units", never "1 unit(s)". */
|
|
79
|
+
function pluralizeLabel(n, label) {
|
|
80
|
+
return label.replace(/\(s\)/g, n === 1 ? "" : "s");
|
|
81
|
+
}
|
|
82
|
+
function fixNoun(n) {
|
|
83
|
+
return n === 1 ? "fix" : "fixes";
|
|
84
|
+
}
|
|
85
|
+
function article(grade) {
|
|
86
|
+
return grade === "A" ? "an" : "a";
|
|
87
|
+
}
|
|
88
|
+
/** Remove the FIRST array element matching `pred`; returns a new array (or the
|
|
89
|
+
* same reference when nothing matches, so unrelated re-scores are byte-identical). */
|
|
90
|
+
function removeFirst(arr, pred) {
|
|
91
|
+
const i = arr.findIndex(pred);
|
|
92
|
+
if (i < 0)
|
|
93
|
+
return arr;
|
|
94
|
+
return [...arr.slice(0, i), ...arr.slice(i + 1)];
|
|
95
|
+
}
|
|
96
|
+
/**
|
|
97
|
+
* Remove the first agent-issue the `pick` callback rewrites (it returns the
|
|
98
|
+
* rewritten agent, or null when this agent has no matching issue). Stops after the
|
|
99
|
+
* first hit so exactly one graded unit is dropped per call.
|
|
100
|
+
*/
|
|
101
|
+
function removeAgentIssue(report, agentName, pick) {
|
|
102
|
+
let done = false;
|
|
103
|
+
const agents = report.agents.map((a) => {
|
|
104
|
+
if (done || a.name !== agentName)
|
|
105
|
+
return a;
|
|
106
|
+
const next = pick(a);
|
|
107
|
+
if (next !== null) {
|
|
108
|
+
done = true;
|
|
109
|
+
return next;
|
|
110
|
+
}
|
|
111
|
+
return a;
|
|
112
|
+
});
|
|
113
|
+
return done ? { ...report, agents } : report;
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Return a copy of `report` with the SINGLE finding behind `rec` neutralized —
|
|
117
|
+
* matched by the recommendation's detector + its content (surface / rationale),
|
|
118
|
+
* so it stays aligned with the caller's recommendation, not with `explainScore`'s
|
|
119
|
+
* internal ordering. Each removal reduces exactly one graded penalty unit; an
|
|
120
|
+
* unrecognized detector is a no-op (returns the report unchanged), so its
|
|
121
|
+
* `pointsIfFixed` is an honest 0 rather than a fabricated number.
|
|
122
|
+
*/
|
|
123
|
+
function withFindingRemoved(report, rec) {
|
|
124
|
+
switch (rec.detector) {
|
|
125
|
+
case "description-overlap": {
|
|
126
|
+
// surface is the "a ↔ b" pair; drop that overlap (feeds the W_OVERLAP count).
|
|
127
|
+
const descriptionOverlaps = removeFirst(report.descriptionOverlaps, (o) => `${o.a} ↔ ${o.b}` === rec.surface);
|
|
128
|
+
return { ...report, descriptionOverlaps };
|
|
129
|
+
}
|
|
130
|
+
case "skill-frontmatter": {
|
|
131
|
+
// Flip the matched skill's hasDescription — the only field the noDesc penalty
|
|
132
|
+
// reads — without dropping the skill (keeps Safety's assessable count intact).
|
|
133
|
+
let flipped = false;
|
|
134
|
+
const skills = report.skills.map((s) => {
|
|
135
|
+
if (!flipped && s.name === rec.surface && !s.hasDescription) {
|
|
136
|
+
flipped = true;
|
|
137
|
+
return { ...s, hasDescription: true };
|
|
138
|
+
}
|
|
139
|
+
return s;
|
|
140
|
+
});
|
|
141
|
+
return flipped ? { ...report, skills } : report;
|
|
142
|
+
}
|
|
143
|
+
case "subagent-tool-contract":
|
|
144
|
+
return removeAgentIssue(report, rec.surface, (a) => {
|
|
145
|
+
const toolIssues = removeFirst(a.toolIssues, (t) => t.message === rec.rationale);
|
|
146
|
+
return toolIssues !== a.toolIssues ? { ...a, toolIssues } : null;
|
|
147
|
+
});
|
|
148
|
+
case "mcp-tool-resolves":
|
|
149
|
+
return removeAgentIssue(report, rec.surface, (a) => {
|
|
150
|
+
const mcpToolIssues = removeFirst(a.mcpToolIssues, (m) => m.message === rec.rationale);
|
|
151
|
+
return mcpToolIssues !== a.mcpToolIssues
|
|
152
|
+
? { ...a, mcpToolIssues }
|
|
153
|
+
: null;
|
|
154
|
+
});
|
|
155
|
+
case "hook-events": {
|
|
156
|
+
const hookEventIssues = removeFirst(report.hookEventIssues, (h) => h.message === rec.rationale);
|
|
157
|
+
return hookEventIssues !== report.hookEventIssues
|
|
158
|
+
? { ...report, hookEventIssues }
|
|
159
|
+
: report;
|
|
160
|
+
}
|
|
161
|
+
case "hook-script-exists": {
|
|
162
|
+
// Flip the matched missing hook to "ok" (the missingHooks penalty reads status).
|
|
163
|
+
let flipped = false;
|
|
164
|
+
const hooks = report.hooks.map((h) => {
|
|
165
|
+
if (!flipped && h.script === rec.surface && h.status === "missing") {
|
|
166
|
+
flipped = true;
|
|
167
|
+
return { ...h, status: "ok" };
|
|
168
|
+
}
|
|
169
|
+
return h;
|
|
170
|
+
});
|
|
171
|
+
return flipped ? { ...report, hooks } : report;
|
|
172
|
+
}
|
|
173
|
+
case "subagent-frontmatter": {
|
|
174
|
+
const frontmatterIssues = removeFirst(report.frontmatterIssues, (f) => f.kind === "agent" && f.path === rec.surface);
|
|
175
|
+
return frontmatterIssues !== report.frontmatterIssues
|
|
176
|
+
? { ...report, frontmatterIssues }
|
|
177
|
+
: report;
|
|
178
|
+
}
|
|
179
|
+
default:
|
|
180
|
+
// Unknown detector — can't map it to a graded finding; no-op (honest 0 delta).
|
|
181
|
+
return report;
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
/** Re-score `report` with EVERY listed recommendation's finding removed at once. */
|
|
185
|
+
function overallWithout(report, recs) {
|
|
186
|
+
let cur = report;
|
|
187
|
+
for (const rec of recs)
|
|
188
|
+
cur = withFindingRemoved(cur, rec);
|
|
189
|
+
return (0, audit_score_js_1.auditScore)(cur).overall;
|
|
190
|
+
}
|
|
191
|
+
/**
|
|
192
|
+
* The single largest blocking deduction (max `n × weight`, `n > 0`), for the
|
|
193
|
+
* issue-forward verdict when the fix list can't close the grade gap. Tie-break by
|
|
194
|
+
* heavier per-item weight, then the report's own deduction order.
|
|
195
|
+
*/
|
|
196
|
+
function dominantDeduction(report) {
|
|
197
|
+
let best = null;
|
|
198
|
+
for (const d of (0, leaderboard_js_1.reportDeductions)(report)) {
|
|
199
|
+
if (d.n <= 0)
|
|
200
|
+
continue;
|
|
201
|
+
const cost = d.n * d.weight;
|
|
202
|
+
if (best === null ||
|
|
203
|
+
cost > best.cost ||
|
|
204
|
+
(cost === best.cost && d.weight > best.weight)) {
|
|
205
|
+
best = { n: d.n, label: d.label, cost, weight: d.weight };
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
return best ? { n: best.n, label: best.label } : null;
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* Minimal number of fixes whose COMBINED removal reaches `targetOverall`, applying
|
|
212
|
+
* recommendations largest-`pointsIfFixed`-first (index-asc tie-break) and
|
|
213
|
+
* re-scoring the growing set each step. Provably minimal when penalties are
|
|
214
|
+
* additive (away from the score-0 clamp); a tight upper bound otherwise. Null when
|
|
215
|
+
* removing every recommendation still doesn't reach the target.
|
|
216
|
+
*/
|
|
217
|
+
function fixesToReach(report, recommendations, perRec, targetOverall) {
|
|
218
|
+
const order = [...perRec].sort((a, b) => b.pointsIfFixed - a.pointsIfFixed || a.index - b.index);
|
|
219
|
+
const chosen = [];
|
|
220
|
+
for (const p of order) {
|
|
221
|
+
chosen.push(recommendations[p.index]);
|
|
222
|
+
if (overallWithout(report, chosen) >= targetOverall)
|
|
223
|
+
return chosen.length;
|
|
224
|
+
}
|
|
225
|
+
return null;
|
|
226
|
+
}
|
|
227
|
+
function buildSentence(input, pointsToNextGrade, fixesToNextGrade) {
|
|
228
|
+
const { score, recommendations, report } = input;
|
|
229
|
+
if (score.empty) {
|
|
230
|
+
return "No loadable harness surface — nothing to grade yet.";
|
|
231
|
+
}
|
|
232
|
+
// Already an A: nothing is blocking the grade.
|
|
233
|
+
if (pointsToNextGrade === null) {
|
|
234
|
+
if (recommendations.length === 0) {
|
|
235
|
+
return "A — nothing blocking; the harness is structurally clean.";
|
|
236
|
+
}
|
|
237
|
+
const n = recommendations.length;
|
|
238
|
+
return `A — nothing blocking the grade; ${numberWord(n)} deterministic ${fixNoun(n)} would harden it further.`;
|
|
239
|
+
}
|
|
240
|
+
// The next band's grade is gradeFor(its FLOOR); the floor is base + the gap.
|
|
241
|
+
const nextGrade = (0, leaderboard_js_1.gradeFor)(score.overall + pointsToNextGrade);
|
|
242
|
+
// Reachable by the deterministic fix list: fix-count-forward (the actionable framing).
|
|
243
|
+
if (fixesToNextGrade !== null) {
|
|
244
|
+
return `${capitalize(numberWord(fixesToNextGrade))} one-line ${fixNoun(fixesToNextGrade)} away from ${article(nextGrade)} ${nextGrade}.`;
|
|
245
|
+
}
|
|
246
|
+
// Not reachable by recommendations alone — lead with the dominant blocking finding.
|
|
247
|
+
const dom = dominantDeduction(report);
|
|
248
|
+
if (dom === null) {
|
|
249
|
+
return `${score.grade} — ${String(pointsToNextGrade)} points below ${article(nextGrade)} ${nextGrade}.`;
|
|
250
|
+
}
|
|
251
|
+
return `${score.grade} — ${String(dom.n)} ${pluralizeLabel(dom.n, dom.label)}; fixing every deterministic finding still lands below ${article(nextGrade)} ${nextGrade}.`;
|
|
252
|
+
}
|
|
253
|
+
/**
|
|
254
|
+
* Compute the audit verdict + per-recommendation `pointsIfFixed` by re-scoring.
|
|
255
|
+
* Pure and deterministic over its inputs. `perRecommendation` is index-aligned to
|
|
256
|
+
* the input `recommendations`.
|
|
257
|
+
*/
|
|
258
|
+
function computeVerdict(input) {
|
|
259
|
+
const { report, score, recommendations } = input;
|
|
260
|
+
const base = score.overall;
|
|
261
|
+
const perRecommendation = recommendations.map((rec, index) => {
|
|
262
|
+
const after = (0, audit_score_js_1.auditScore)(withFindingRemoved(report, rec)).overall;
|
|
263
|
+
// Removing a penalty can only raise or hold the score; clamp negatives to 0
|
|
264
|
+
// to defend against any future non-monotonic scorer change.
|
|
265
|
+
return { index, pointsIfFixed: Math.max(0, after - base) };
|
|
266
|
+
});
|
|
267
|
+
const nextFloor = nextGradeFloor(base);
|
|
268
|
+
const pointsToNextGrade = nextFloor === null ? null : nextFloor - base;
|
|
269
|
+
const fixesToNextGrade = nextFloor === null
|
|
270
|
+
? null
|
|
271
|
+
: fixesToReach(report, recommendations, perRecommendation, nextFloor);
|
|
272
|
+
const sentence = buildSentence(input, pointsToNextGrade, fixesToNextGrade);
|
|
273
|
+
return {
|
|
274
|
+
sentence,
|
|
275
|
+
grade: score.grade,
|
|
276
|
+
pointsToNextGrade,
|
|
277
|
+
fixesToNextGrade,
|
|
278
|
+
perRecommendation,
|
|
279
|
+
};
|
|
280
|
+
}
|
|
281
|
+
//# sourceMappingURL=audit-verdict.js.map
|
package/dist/cli.js
CHANGED
|
@@ -35,6 +35,8 @@ const audit_prompts_js_1 = require("./audit-prompts.js");
|
|
|
35
35
|
const audit_html_js_1 = require("./audit-html.js");
|
|
36
36
|
const audit_serve_js_1 = require("./audit-serve.js");
|
|
37
37
|
const audit_report_js_1 = require("./audit-report.js");
|
|
38
|
+
const rule_inventory_js_1 = require("./rule-inventory.js");
|
|
39
|
+
const rule_routing_js_1 = require("./rule-routing.js");
|
|
38
40
|
const adoptability_js_1 = require("./adoptability.js");
|
|
39
41
|
const compile_js_1 = require("./core/compile.js");
|
|
40
42
|
const proofs_js_1 = require("./core/proofs.js");
|
|
@@ -1402,6 +1404,132 @@ function discoverAdoptableForAudit(root, instructionFile) {
|
|
|
1402
1404
|
out.push(...discoverAdoptableSurfaces(root));
|
|
1403
1405
|
return out;
|
|
1404
1406
|
}
|
|
1407
|
+
/** Lint-config file BASENAMES whose CONTENTS (textual, NEVER executed) reveal
|
|
1408
|
+
* which rules a repo has configured — read best-effort for the rule-inventory
|
|
1409
|
+
* teaser. Deliberately not resolved/executed (that would be the RCE path); we
|
|
1410
|
+
* grep the raw text. Includes oxlint + biome (same rule names as ESLint) since
|
|
1411
|
+
* modern TS repos lint with them. */
|
|
1412
|
+
const RULE_INVENTORY_CONFIG_FILES = new Set([
|
|
1413
|
+
"eslint.config.js",
|
|
1414
|
+
"eslint.config.mjs",
|
|
1415
|
+
"eslint.config.cjs",
|
|
1416
|
+
"eslint.config.ts",
|
|
1417
|
+
".eslintrc",
|
|
1418
|
+
".eslintrc.json",
|
|
1419
|
+
".eslintrc.js",
|
|
1420
|
+
".eslintrc.cjs",
|
|
1421
|
+
".eslintrc.yml",
|
|
1422
|
+
".eslintrc.yaml",
|
|
1423
|
+
".oxlintrc.json",
|
|
1424
|
+
".oxlintrc.jsonc",
|
|
1425
|
+
"oxlint.json",
|
|
1426
|
+
"biome.json",
|
|
1427
|
+
"biome.jsonc",
|
|
1428
|
+
"ruff.toml",
|
|
1429
|
+
".ruff.toml",
|
|
1430
|
+
"pyproject.toml",
|
|
1431
|
+
".pylintrc",
|
|
1432
|
+
"clippy.toml",
|
|
1433
|
+
".clippy.toml",
|
|
1434
|
+
".rubocop.yml",
|
|
1435
|
+
".stylelintrc",
|
|
1436
|
+
".stylelintrc.json",
|
|
1437
|
+
]);
|
|
1438
|
+
/** Dirs never worth walking for a config file. */
|
|
1439
|
+
const RULE_INVENTORY_SKIP_DIRS = new Set([
|
|
1440
|
+
"node_modules",
|
|
1441
|
+
".git",
|
|
1442
|
+
"dist",
|
|
1443
|
+
"build",
|
|
1444
|
+
"out",
|
|
1445
|
+
"coverage",
|
|
1446
|
+
".next",
|
|
1447
|
+
".turbo",
|
|
1448
|
+
".cache",
|
|
1449
|
+
".yarn",
|
|
1450
|
+
]);
|
|
1451
|
+
/** readdir that returns [] instead of throwing (perms, races). */
|
|
1452
|
+
function safeReaddir(dir) {
|
|
1453
|
+
try {
|
|
1454
|
+
return (0, node_fs_1.readdirSync)(dir, { withFileTypes: true });
|
|
1455
|
+
}
|
|
1456
|
+
catch {
|
|
1457
|
+
return [];
|
|
1458
|
+
}
|
|
1459
|
+
}
|
|
1460
|
+
/** Collect lint-config CONTENTS from the repo root AND nested subdirs (depth ≤ 2)
|
|
1461
|
+
* — textual only, never executed. Nested because monorepos/webapps keep their
|
|
1462
|
+
* eslint/oxlint config under `web/`, `frontend/`, `packages/*`, etc. Bounded
|
|
1463
|
+
* (skips heavy dirs; caps files) so it stays cheap on huge repos. */
|
|
1464
|
+
function collectLintConfigText(root) {
|
|
1465
|
+
let text = "";
|
|
1466
|
+
let filesRead = 0;
|
|
1467
|
+
const visit = (dir, depth) => {
|
|
1468
|
+
if (filesRead >= 60)
|
|
1469
|
+
return;
|
|
1470
|
+
for (const e of safeReaddir(dir)) {
|
|
1471
|
+
if (e.isFile() && RULE_INVENTORY_CONFIG_FILES.has(e.name)) {
|
|
1472
|
+
try {
|
|
1473
|
+
text += (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(dir, e.name), "utf-8") + "\n";
|
|
1474
|
+
filesRead++;
|
|
1475
|
+
}
|
|
1476
|
+
catch {
|
|
1477
|
+
/* best-effort */
|
|
1478
|
+
}
|
|
1479
|
+
}
|
|
1480
|
+
else if (e.isDirectory() &&
|
|
1481
|
+
depth < 2 &&
|
|
1482
|
+
!e.name.startsWith(".") &&
|
|
1483
|
+
!RULE_INVENTORY_SKIP_DIRS.has(e.name)) {
|
|
1484
|
+
visit((0, node_path_1.resolve)(dir, e.name), depth + 1);
|
|
1485
|
+
}
|
|
1486
|
+
}
|
|
1487
|
+
};
|
|
1488
|
+
visit(root, 0);
|
|
1489
|
+
return text;
|
|
1490
|
+
}
|
|
1491
|
+
/** The deterministic rule-inventory teaser for `audit`: read the instruction
|
|
1492
|
+
* file(s) + lint-config TEXT (never executed) and map documented intents to
|
|
1493
|
+
* off-the-shelf rules + whether they're already configured. Best-effort, fs-only;
|
|
1494
|
+
* NO model, NO config execution — safe on any repo. Composition-root. */
|
|
1495
|
+
function computeRuleInventory(root, instructionFile) {
|
|
1496
|
+
try {
|
|
1497
|
+
const instructionText = readInstructionText(root, instructionFile);
|
|
1498
|
+
if (!instructionText.trim())
|
|
1499
|
+
return [];
|
|
1500
|
+
return (0, rule_inventory_js_1.buildRuleInventory)(instructionText, collectLintConfigText(root));
|
|
1501
|
+
}
|
|
1502
|
+
catch {
|
|
1503
|
+
return [];
|
|
1504
|
+
}
|
|
1505
|
+
}
|
|
1506
|
+
/** Read EVERY agent instruction file present (not just the harness-native one) —
|
|
1507
|
+
* rules are often documented in AGENTS.md even under a claude-code harness. */
|
|
1508
|
+
function readInstructionText(root, instructionFile) {
|
|
1509
|
+
let instructionText = "";
|
|
1510
|
+
for (const name of new Set([instructionFile, "CLAUDE.md", "AGENTS.md"])) {
|
|
1511
|
+
const p = (0, node_path_1.resolve)(root, name);
|
|
1512
|
+
if ((0, node_fs_1.existsSync)(p))
|
|
1513
|
+
instructionText += (0, node_fs_1.readFileSync)(p, "utf-8") + "\n";
|
|
1514
|
+
}
|
|
1515
|
+
return instructionText;
|
|
1516
|
+
}
|
|
1517
|
+
/** The deterministic State-B routing preview for `audit`: segment the instruction
|
|
1518
|
+
* file(s) into atomic rules and route each (reuse / hook / semantic / unrouted) —
|
|
1519
|
+
* NO model, fs-only. `undefined` when there's nothing to segment (kept off the
|
|
1520
|
+
* report). Best-effort; a routing failure never breaks the audit. */
|
|
1521
|
+
function computeRuleRouting(root, instructionFile) {
|
|
1522
|
+
try {
|
|
1523
|
+
const instructionText = readInstructionText(root, instructionFile);
|
|
1524
|
+
if (!instructionText.trim())
|
|
1525
|
+
return undefined;
|
|
1526
|
+
const routing = (0, rule_routing_js_1.routeRules)(instructionText, instructionFile);
|
|
1527
|
+
return routing.segmented > 0 ? routing : undefined;
|
|
1528
|
+
}
|
|
1529
|
+
catch {
|
|
1530
|
+
return undefined;
|
|
1531
|
+
}
|
|
1532
|
+
}
|
|
1405
1533
|
/** The terminal "adoptable surfaces" nudge — N un-spec'd surfaces + the create-all
|
|
1406
1534
|
* command and up to ~5 per-surface commands (then "+K more"). "" when nothing to
|
|
1407
1535
|
* adopt (a fully spec-managed repo says nothing). */
|
|
@@ -5109,6 +5237,8 @@ async function main() {
|
|
|
5109
5237
|
vigilesVersion: getVersion(),
|
|
5110
5238
|
adoptableSurfaces,
|
|
5111
5239
|
observations: (0, observe_js_1.summarizeObservations)(ledgerRecords),
|
|
5240
|
+
rulesInventory: computeRuleInventory(root, adapter.layout.instructionFile),
|
|
5241
|
+
ruleRouting: computeRuleRouting(root, adapter.layout.instructionFile),
|
|
5112
5242
|
});
|
|
5113
5243
|
const sc = auditReport.score;
|
|
5114
5244
|
const plan = (0, optimize_js_1.optimize)(report);
|
package/dist/eval-cost.d.ts
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* the `claude` CLI) + a running session tally. We deliberately do NOT show a
|
|
10
10
|
* "% of your subscription" — Anthropic does not expose a subscription's quota or
|
|
11
11
|
* limit programmatically (and the real limits are rolling rate windows, not a
|
|
12
|
-
* dollar bucket), so any percentage would be fiction. See
|
|
12
|
+
* dollar bucket), so any percentage would be fiction. See research/eval-architecture.md.
|
|
13
13
|
*
|
|
14
14
|
* Pure + injectable (env + an output sink), so the whole thing is unit-tested
|
|
15
15
|
* without a model or a real key.
|
package/dist/eval.d.ts
CHANGED
|
@@ -43,7 +43,7 @@ export interface EvalArm {
|
|
|
43
43
|
* opus: { model: "claude-opus-4-8" } }` — so model-as-an-arm answers "does my
|
|
44
44
|
* harness still hold on the cheaper tier / after a model upgrade?" through the
|
|
45
45
|
* same significance machinery, with no separate model-matrix runner. Omit to
|
|
46
|
-
* use the eval-level model. See `
|
|
46
|
+
* use the eval-level model. See `research/eval-architecture.md` (model strategy).
|
|
47
47
|
*/
|
|
48
48
|
readonly model?: string;
|
|
49
49
|
}
|
|
@@ -491,7 +491,7 @@ export declare function aggregateUsage(usages: readonly EvalUsage[]): ArmUsage;
|
|
|
491
491
|
* e.g. `claude-haiku-4-5-20251001`. A floating alias (`haiku`, `sonnet`, or even
|
|
492
492
|
* `claude-sonnet-4-6` with no date) can change underneath you — so a cached or
|
|
493
493
|
* baselined result pinned to it can silently hide model drift. See
|
|
494
|
-
* `
|
|
494
|
+
* `research/eval-architecture.md` (honest model pinning).
|
|
495
495
|
*/
|
|
496
496
|
export declare function isDatedModel(model: string): boolean;
|
|
497
497
|
/**
|
package/dist/eval.js
CHANGED
|
@@ -665,7 +665,7 @@ function isRecord(v) {
|
|
|
665
665
|
* e.g. `claude-haiku-4-5-20251001`. A floating alias (`haiku`, `sonnet`, or even
|
|
666
666
|
* `claude-sonnet-4-6` with no date) can change underneath you — so a cached or
|
|
667
667
|
* baselined result pinned to it can silently hide model drift. See
|
|
668
|
-
* `
|
|
668
|
+
* `research/eval-architecture.md` (honest model pinning).
|
|
669
669
|
*/
|
|
670
670
|
function isDatedModel(model) {
|
|
671
671
|
return /\d{8}$/.test(model);
|
package/dist/leaderboard.js
CHANGED
|
@@ -214,6 +214,12 @@ function computeIntegrityScore(deductions) {
|
|
|
214
214
|
}
|
|
215
215
|
return { score: Math.max(0, 100 - penalty), penalty };
|
|
216
216
|
}
|
|
217
|
+
/** Resolve the terse "thing(s)" plural placeholder against a count:
|
|
218
|
+
* n===1 drops the "(s)" ("1 tool"); otherwise it becomes "s" ("3 tools").
|
|
219
|
+
* (Kept local — audit-score.ts has its own copy to avoid a circular import.) */
|
|
220
|
+
function pluralizeLabel(n, label) {
|
|
221
|
+
return label.replace(/\(s\)/g, n === 1 ? "" : "s");
|
|
222
|
+
}
|
|
217
223
|
/** Deterministic structural-health score for one scanned plugin. */
|
|
218
224
|
function scoreReport(r) {
|
|
219
225
|
// An empty/unloadable machine isn't healthy — it's a non-plugin or a broken
|
|
@@ -230,7 +236,7 @@ function scoreReport(r) {
|
|
|
230
236
|
for (const d of deductions) {
|
|
231
237
|
if (d.n === 0)
|
|
232
238
|
continue;
|
|
233
|
-
issues.push(`${String(d.n)} ${d.label}`);
|
|
239
|
+
issues.push(`${String(d.n)} ${pluralizeLabel(d.n, d.label)}`);
|
|
234
240
|
}
|
|
235
241
|
// Sort issues by cost (worst first) so the report leads with what matters.
|
|
236
242
|
issues.sort((a, b) => Number(b.split(" ")[0]) - Number(a.split(" ")[0]));
|
|
@@ -241,10 +247,10 @@ function scoreReport(r) {
|
|
|
241
247
|
// - untested surfaces: a hardening gap, not breakage.
|
|
242
248
|
const noContract = r.agents.filter((a) => a.tools === null).length;
|
|
243
249
|
if (noContract > 0) {
|
|
244
|
-
issues.push(`${String(noContract)} agent(s) inherit all tools (no contract) (advisory)`);
|
|
250
|
+
issues.push(`${String(noContract)} ${pluralizeLabel(noContract, "agent(s) inherit all tools (no contract) (advisory)")}`);
|
|
245
251
|
}
|
|
246
252
|
if (r.untested > 0) {
|
|
247
|
-
issues.push(`${String(r.untested)} untested surface(s) (advisory)`);
|
|
253
|
+
issues.push(`${String(r.untested)} ${pluralizeLabel(r.untested, "untested surface(s) (advisory)")}`);
|
|
248
254
|
}
|
|
249
255
|
return { score, issues };
|
|
250
256
|
}
|