@gaunt-sloth/batch 2.0.0-beta.9 → 2.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin.d.ts +2 -2
- package/dist/bin.js +2 -2
- package/dist/classificationReport.d.ts +2 -2
- package/dist/classificationReport.js +2 -2
- package/dist/classificationTypes.d.ts +7 -7
- package/dist/classificationTypes.js +1 -1
- package/dist/evalCompare.d.ts +128 -1
- package/dist/evalCompare.js +253 -2
- package/dist/evalCompare.js.map +1 -1
- package/dist/evalOutput.d.ts +1 -1
- package/dist/evalOutput.js +1 -1
- package/dist/evalRunner.d.ts +15 -3
- package/dist/evalRunner.js +82 -5
- package/dist/evalRunner.js.map +1 -1
- package/dist/evalSuite.d.ts +5 -1
- package/dist/evalSuite.js +218 -7
- package/dist/evalSuite.js.map +1 -1
- package/dist/evalTypes.d.ts +79 -18
- package/dist/evalTypes.js +2 -2
- package/dist/evalTypes.js.map +1 -1
- package/dist/index.d.ts +8 -2
- package/dist/index.js +9 -1
- package/dist/index.js.map +1 -1
- package/dist/judge.d.ts +2 -2
- package/dist/judge.js +10 -14
- package/dist/judge.js.map +1 -1
- package/dist/output.d.ts +1 -1
- package/dist/output.js +1 -1
- package/dist/parseOver.d.ts +1 -1
- package/dist/parseOver.js +1 -1
- package/dist/pipelineCli.d.ts +3 -3
- package/dist/pipelineCli.js +5 -4
- package/dist/pipelineCli.js.map +1 -1
- package/dist/raterPromptArm.d.ts +140 -0
- package/dist/raterPromptArm.js +306 -0
- package/dist/raterPromptArm.js.map +1 -0
- package/dist/raterTarget.d.ts +34 -12
- package/dist/raterTarget.js +124 -33
- package/dist/raterTarget.js.map +1 -1
- package/dist/reporters/registry.d.ts +1 -1
- package/dist/reporters/registry.js +1 -1
- package/dist/reporters/registry.js.map +1 -1
- package/dist/reporters/reporterTypes.d.ts +3 -3
- package/dist/reporters/textReporter.js +16 -0
- package/dist/reporters/textReporter.js.map +1 -1
- package/dist/toolCoverage.d.ts +244 -0
- package/dist/toolCoverage.js +414 -0
- package/dist/toolCoverage.js.map +1 -0
- package/dist/toolCoverageRender.d.ts +31 -0
- package/dist/toolCoverageRender.js +71 -0
- package/dist/toolCoverageRender.js.map +1 -0
- package/dist/toolResultChecks.d.ts +8 -2
- package/dist/toolResultChecks.js +45 -3
- package/dist/toolResultChecks.js.map +1 -1
- package/dist/types.d.ts +38 -3
- package/dist/types.js +0 -9
- package/dist/types.js.map +1 -1
- package/dist/workflow/runWorkflow.d.ts +4 -4
- package/dist/workflow/runWorkflow.js +5 -4
- package/dist/workflow/runWorkflow.js.map +1 -1
- package/package.json +9 -8
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One advertised tool, as the batch layer carries it — a structural mirror of core's
|
|
3
|
+
* `GthAdvertisedTool`, kept local for the same reason `ToolResultRecord` (`#src/types.js`) is: this
|
|
4
|
+
* package's outcome shapes stay independent of the runner/LLM types.
|
|
5
|
+
*/
|
|
6
|
+
export interface AdvertisedToolRecord {
|
|
7
|
+
/** The registered tool name, exactly as the model would call it. */
|
|
8
|
+
name: string;
|
|
9
|
+
/**
|
|
10
|
+
* The `mcpServers` key that explains the name, resolved against the configured keys at capture.
|
|
11
|
+
* Absent for a tool outside the MCP namespace (a built-in or a user-configured one); the empty
|
|
12
|
+
* string for an MCP-namespaced name no configured key explains.
|
|
13
|
+
*/
|
|
14
|
+
server?: string;
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* One run's advertised-tool inventory, as captured by the agent at `init` — a structural mirror of
|
|
18
|
+
* core's `GthAdvertisedTools`.
|
|
19
|
+
*/
|
|
20
|
+
export interface AdvertisedToolInventory {
|
|
21
|
+
/** Every named tool the agent loaded, BEFORE any `allowedTools` narrowing. */
|
|
22
|
+
tools: AdvertisedToolRecord[];
|
|
23
|
+
/** The subset an `allowedTools` allow-list removed. Empty when no allow-list is configured. */
|
|
24
|
+
filteredOut: AdvertisedToolRecord[];
|
|
25
|
+
/** How many loaded tools carry no name — counted in neither half of the fraction. */
|
|
26
|
+
unnamed: number;
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* A suite's `tool_coverage:` declaration: which tools are deliberately not covered, and what would
|
|
30
|
+
* make the run fail.
|
|
31
|
+
*/
|
|
32
|
+
export interface ToolCoverageSpec {
|
|
33
|
+
/**
|
|
34
|
+
* Globs (the `must_call` matcher) whose matching advertised tools LEAVE the denominator.
|
|
35
|
+
*
|
|
36
|
+
* Load-bearing, not a nicety: a read-only suite must not be shamed for never calling the mutating
|
|
37
|
+
* tools, and without waivers any real server's coverage starts red and is then ignored — which is
|
|
38
|
+
* the same failure as a metric nobody reads. A deliberate decision not to cover something belongs
|
|
39
|
+
* in the suite, where it is reviewable, rather than in a comment.
|
|
40
|
+
*/
|
|
41
|
+
waive: string[];
|
|
42
|
+
/**
|
|
43
|
+
* Optional floor: the minimum percentage (0-100) of the post-waiver denominator that must have
|
|
44
|
+
* been exercised. Breaching it is a product signal, graded like a declared metric gate.
|
|
45
|
+
*/
|
|
46
|
+
min?: number;
|
|
47
|
+
/**
|
|
48
|
+
* Globs that must EACH match at least one exercised tool, whatever the percentage says. A floor
|
|
49
|
+
* answers "is enough of the surface covered"; this answers "was this specific tool reached", which
|
|
50
|
+
* a percentage can always satisfy by covering something else.
|
|
51
|
+
*/
|
|
52
|
+
require: string[];
|
|
53
|
+
}
|
|
54
|
+
/** One bucket of the per-server breakdown. */
|
|
55
|
+
export interface ToolCoverageServerReport {
|
|
56
|
+
/**
|
|
57
|
+
* Which kind of bucket this is — explicit rather than encoded in {@link server}, so a JSON reader
|
|
58
|
+
* never has to tell "no server" from "a server named empty", and so a configured key cannot
|
|
59
|
+
* collide with a label this file made up.
|
|
60
|
+
*/
|
|
61
|
+
kind: 'mcp' | 'builtin' | 'unresolved';
|
|
62
|
+
/** The `mcpServers` key, for a `mcp` bucket only. */
|
|
63
|
+
server?: string;
|
|
64
|
+
/** Exercised tool names in this bucket (post-waiver). */
|
|
65
|
+
covered: string[];
|
|
66
|
+
/** Advertised-but-never-called names in this bucket (post-waiver). */
|
|
67
|
+
uncovered: string[];
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* The coverage block written to `results.json` and rendered on the console.
|
|
71
|
+
*
|
|
72
|
+
* Name LISTS rather than counts, deliberately: a directory run's aggregate is a set union across
|
|
73
|
+
* suites (the same 41-tool server advertised to every suite must be counted once, not once per
|
|
74
|
+
* suite), and counts cannot be unioned. `n/N` is `covered.length` over `covered.length +
|
|
75
|
+
* uncovered.length`.
|
|
76
|
+
*/
|
|
77
|
+
export interface ToolCoverageReport {
|
|
78
|
+
/** Advertised AND exercised, after waivers — the numerator. */
|
|
79
|
+
covered: string[];
|
|
80
|
+
/** Advertised, not waived, never called — the names the report exists to print. */
|
|
81
|
+
uncovered: string[];
|
|
82
|
+
/**
|
|
83
|
+
* Advertised names a `waive:` glob removed from the denominator, reported BESIDE the fraction.
|
|
84
|
+
*
|
|
85
|
+
* A waived tool leaves the denominator, so a suite waiving 38 of 41 reports 100% while covering
|
|
86
|
+
* three. Printing the waived count next to the number is what stops that reading as full coverage
|
|
87
|
+
* — the blind-denominator shape a metric facility cannot afford, because its whole value is being
|
|
88
|
+
* trusted.
|
|
89
|
+
*/
|
|
90
|
+
waived: string[];
|
|
91
|
+
/** Advertised names an `allowedTools` allow-list removed before the agent bound them. */
|
|
92
|
+
filteredOut: string[];
|
|
93
|
+
/** Nameless provider-native tools, counted in neither half (there is no name to count). */
|
|
94
|
+
unnamed: number;
|
|
95
|
+
/** Per-server breakdown — a single percentage hides which server is uncovered. */
|
|
96
|
+
byServer: ToolCoverageServerReport[];
|
|
97
|
+
/** Everything that makes the figure mean less than it appears to. Never silently dropped. */
|
|
98
|
+
warnings: string[];
|
|
99
|
+
/**
|
|
100
|
+
* Breached `min` / unmet `require:` entries. Non-empty forces the run's exit code to 1 through the
|
|
101
|
+
* same contract a breached metric gate uses — a product signal, never a harness one.
|
|
102
|
+
*/
|
|
103
|
+
gateFailures: string[];
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* The waived share at which the fraction stops describing the surface and starts describing the
|
|
107
|
+
* exemptions. Half is the point where the tools NOT being measured outnumber the tools that are, so
|
|
108
|
+
* a reader who takes the percentage at face value is wrong more often than right.
|
|
109
|
+
*
|
|
110
|
+
* A threshold rather than a hard failure because a high waiver share is legitimate — a read-only
|
|
111
|
+
* suite against a mostly-mutating server is exactly that — and failing it would push authors to
|
|
112
|
+
* delete the waivers, which is how the denominator goes blind in the first place.
|
|
113
|
+
*/
|
|
114
|
+
export declare const WAIVER_SHARE_WARN_THRESHOLD = 0.5;
|
|
115
|
+
/** Inputs to {@link computeToolCoverage}. */
|
|
116
|
+
export interface ToolCoverageInput {
|
|
117
|
+
/**
|
|
118
|
+
* One entry per cell that reported an inventory. **Empty means no cell reported one**, which is
|
|
119
|
+
* not the same as a cell reporting an empty inventory: the first cannot supply a denominator at
|
|
120
|
+
* all (an external target, or a run whose SUT never initialised) and yields no report; the second
|
|
121
|
+
* is a real denominator of zero and does.
|
|
122
|
+
*/
|
|
123
|
+
inventories: readonly AdvertisedToolInventory[];
|
|
124
|
+
/** Every tool name any cell invoked, in any order, deduplicated or not. */
|
|
125
|
+
exercised: readonly string[];
|
|
126
|
+
/** The suite's `tool_coverage:` declaration, when it made one. */
|
|
127
|
+
spec?: ToolCoverageSpec;
|
|
128
|
+
}
|
|
129
|
+
/**
|
|
130
|
+
* Compute the coverage report, or `undefined` when no inventory was observed at all.
|
|
131
|
+
*
|
|
132
|
+
* `undefined` is the honest answer for a target that cannot supply a denominator, and is why this
|
|
133
|
+
* returns a value rather than a zeroed report: **`0/0` printed as a coverage figure is the "metric
|
|
134
|
+
* nobody can trust" failure in its purest form** — it looks like a measurement, it is green, and it
|
|
135
|
+
* measured nothing. A caller that declared gates on a target that cannot be measured is rejected at
|
|
136
|
+
* parse time instead (`evalSuite.ts`), so a silent pass is unreachable from both ends.
|
|
137
|
+
*/
|
|
138
|
+
export declare function computeToolCoverage(input: ToolCoverageInput): ToolCoverageReport | undefined;
|
|
139
|
+
/**
|
|
140
|
+
* Union several suites' reports into the run-level figure, or `undefined` when none was produced.
|
|
141
|
+
*
|
|
142
|
+
* **A set union, never a sum.** A directory run points every suite at the same agent, so the same 41
|
|
143
|
+
* advertised tools appear in each report; summing would report 123 tools and make one suite's
|
|
144
|
+
* coverage of a tool count three times. Coverage is a directory-level figure precisely because one
|
|
145
|
+
* suite covering 3 tools is fine if its sibling covers the other 38 — which only reads correctly
|
|
146
|
+
* when a tool covered anywhere is covered once.
|
|
147
|
+
*
|
|
148
|
+
* **Carries no warnings and no gate failures, deliberately.** A gate is declared in a suite and
|
|
149
|
+
* graded against that suite (the `metrics:` precedent), and the run's exit is the OR of those. If a
|
|
150
|
+
* declared floor were re-applied to the aggregate, the same suite would pass when run alone and fail
|
|
151
|
+
* when run as part of a directory, with nothing in either output saying the threshold had moved.
|
|
152
|
+
* This function stays reporting only.
|
|
153
|
+
*
|
|
154
|
+
* **A run-level floor is a different declaration, graded elsewhere.** BATCH-48 adds one —
|
|
155
|
+
* `evalToolCoverage` in gth config, resolved once per run from the base config and never from an
|
|
156
|
+
* identity's — and {@link gradeRunToolCoverage} is what grades it, against the report this function
|
|
157
|
+
* returns. Keeping the two apart is what preserves the argument above: a suite's `min:` is never
|
|
158
|
+
* promoted to the aggregate, and the run's floor is never read back as a suite's.
|
|
159
|
+
*/
|
|
160
|
+
export declare function aggregateToolCoverage(reports: readonly ToolCoverageReport[]): ToolCoverageReport | undefined;
|
|
161
|
+
/**
|
|
162
|
+
* BATCH-48 — the run-level coverage spec, as `gth eval` resolved it from the base config.
|
|
163
|
+
*
|
|
164
|
+
* A structural mirror of the config key rather than a dependency on core's config types: the batch
|
|
165
|
+
* package owns the grading, and the command owns where the value came from. `min` is the same
|
|
166
|
+
* percentage a suite's `tool_coverage.min` is; `waive` is the same glob vocabulary.
|
|
167
|
+
*/
|
|
168
|
+
export interface RunToolCoverageSpec {
|
|
169
|
+
/** Minimum percentage (0–100) of the run's post-waiver denominator. Absent means no floor. */
|
|
170
|
+
min?: number;
|
|
171
|
+
/** Tool-name globs removed from the run denominator, whatever any suite waived. */
|
|
172
|
+
waive: string[];
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* Where the run-level spec was read from, so the output can name it.
|
|
176
|
+
*
|
|
177
|
+
* Naming the source is what keeps a threshold from silently moving scope: a reader can see that the
|
|
178
|
+
* number was graded against the run's floor and where that floor was declared, rather than having to
|
|
179
|
+
* infer it from which suites happened to run.
|
|
180
|
+
*/
|
|
181
|
+
export interface RunToolCoverageSource {
|
|
182
|
+
/**
|
|
183
|
+
* `profile` — the base `-i` profile the run was started with. `project` — the plain project
|
|
184
|
+
* config found by discovery. `config-file` — an explicit `-c` file. `global` — the global config
|
|
185
|
+
* in `~/.gsloth`, when no project config exists or `-g` was given. The command decides which;
|
|
186
|
+
* this layer only prints what it is told.
|
|
187
|
+
*/
|
|
188
|
+
kind: 'profile' | 'project' | 'config-file' | 'global';
|
|
189
|
+
/** The profile name, when {@link kind} is `profile`. */
|
|
190
|
+
profile?: string;
|
|
191
|
+
/** The file path, when {@link kind} is `config-file`. */
|
|
192
|
+
path?: string;
|
|
193
|
+
}
|
|
194
|
+
/** One run graded against its run-level spec. */
|
|
195
|
+
export interface RunToolCoverageGrade {
|
|
196
|
+
/**
|
|
197
|
+
* The aggregate after the run-level `waive` has been applied — the figure the floor was graded
|
|
198
|
+
* against. `undefined` when no suite produced a report, which is a different outcome from an
|
|
199
|
+
* aggregate that exists and is below the floor: there is nothing to grade.
|
|
200
|
+
*/
|
|
201
|
+
report: ToolCoverageReport | undefined;
|
|
202
|
+
/**
|
|
203
|
+
* Breached `min`, or a floor that could not be graded because no suite produced a report. Empty
|
|
204
|
+
* when the floor holds, and when no floor was declared — a spec with only `waive` still narrows
|
|
205
|
+
* the denominator and warns, but it gates nothing.
|
|
206
|
+
*/
|
|
207
|
+
gateFailures: string[];
|
|
208
|
+
/**
|
|
209
|
+
* A run-level waiver that matched nothing advertised. The same stale-waiver warning a suite gets,
|
|
210
|
+
* because a renamed tool is back in the denominator under its new name while the author believes
|
|
211
|
+
* it is still waived — and at run scope that is the decay a duplicated per-suite list hides.
|
|
212
|
+
*/
|
|
213
|
+
warnings: string[];
|
|
214
|
+
/** The line that says which threshold this run was graded against and where it came from. */
|
|
215
|
+
sourceLine: string;
|
|
216
|
+
}
|
|
217
|
+
/**
|
|
218
|
+
* Whether a run-level spec declares anything to grade or apply.
|
|
219
|
+
*
|
|
220
|
+
* An absent config key and an empty object are the same thing to a caller: no floor, no exemption.
|
|
221
|
+
* Distinguishing them would make "the run has no floor" depend on which spelling the config used.
|
|
222
|
+
*/
|
|
223
|
+
export declare function hasRunToolCoverageSpec(spec: RunToolCoverageSpec | undefined): boolean;
|
|
224
|
+
/**
|
|
225
|
+
* Grade a run-level coverage spec against the run's aggregate.
|
|
226
|
+
*
|
|
227
|
+
* **Separate from {@link aggregateToolCoverage} on purpose.** That function reports the union and
|
|
228
|
+
* carries no gate, because re-applying a SUITE's declared floor to a different denominator would
|
|
229
|
+
* make the same suite pass alone and fail inside a directory. This function grades a floor that was
|
|
230
|
+
* declared for the run, against the aggregate it names, so the two thresholds stay independent: a
|
|
231
|
+
* suite's `min:` is never promoted here, and this floor is never written back onto a suite.
|
|
232
|
+
*
|
|
233
|
+
* **The denominator is the aggregate's reconciled one, minus this spec's `waive`.** The aggregate
|
|
234
|
+
* has already applied the reconciliation — a tool waived in one suite but counted in another stays
|
|
235
|
+
* counted, and a tool waived in every suite is already out. A suite's own waiver therefore does not
|
|
236
|
+
* shrink the run denominator further; only the run-level list does. A tool the run-level list waives
|
|
237
|
+
* leaves the denominator even when it was covered, the same way a suite-level waiver does inside
|
|
238
|
+
* {@link computeToolCoverage}.
|
|
239
|
+
*
|
|
240
|
+
* **A floor with nothing to grade fails.** `min` set and no report at all — every suite targeted
|
|
241
|
+
* something that cannot supply a denominator — must not pass: a floor quietly skipped reports green
|
|
242
|
+
* forever over a run it never measured.
|
|
243
|
+
*/
|
|
244
|
+
export declare function gradeRunToolCoverage(aggregate: ToolCoverageReport | undefined, spec: RunToolCoverageSpec, source: RunToolCoverageSource): RunToolCoverageGrade;
|
|
@@ -0,0 +1,414 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @packageDocumentation
|
|
3
|
+
* BATCH-32 — **tool coverage**: which of the tools the agent actually advertised to the model a
|
|
4
|
+
* suite exercised, and which it never touched.
|
|
5
|
+
*
|
|
6
|
+
* The numerator has existed since GS2-16 (`runStats.tools`, per cell, already in every
|
|
7
|
+
* `<case>.json`). What was missing is the DENOMINATOR: a run that calls 3 of 41 advertised tools
|
|
8
|
+
* prints the same all-green summary as one that calls all 41, so "every cell passed" reads as
|
|
9
|
+
* reassurance it has not earned, and coverage decays silently as the server grows a 42nd tool
|
|
10
|
+
* nobody wrote a case for.
|
|
11
|
+
*
|
|
12
|
+
* Pure and dependency-free on purpose (the `classificationReport.ts` precedent): every input is
|
|
13
|
+
* handed in, so the whole facility is unit-testable without an agent, an MCP server or a model.
|
|
14
|
+
*
|
|
15
|
+
* ## The denominator is the FULL advertised list, and that is the design decision here
|
|
16
|
+
*
|
|
17
|
+
* The list the agent binds is narrowed by `allowedTools` before it is bound. Taking the denominator
|
|
18
|
+
* from the narrowed list would let any suite reach 100% by narrowing the allow-list to the tools it
|
|
19
|
+
* already calls — the tools nobody exercises would leave the bottom of the fraction instead of being
|
|
20
|
+
* named. So the denominator is the pre-filter inventory, and the removed tools are reported as their
|
|
21
|
+
* own {@link ToolCoverageReport.filteredOut} category.
|
|
22
|
+
*
|
|
23
|
+
* ## What "covered" means, stated in the output rather than assumed
|
|
24
|
+
*
|
|
25
|
+
* A tool counts as covered when it was **called**. A call whose result came back an error still
|
|
26
|
+
* counts: the tool was reached, which is what a coverage figure measures. Grading what a tool
|
|
27
|
+
* RETURNED is what `must_error` / `tool_result_json_path` assertions are for, and they are a
|
|
28
|
+
* different question asked per case.
|
|
29
|
+
*/
|
|
30
|
+
import { toolNameMatchesPattern } from '@gaunt-sloth/core/utils/toolMatching.js';
|
|
31
|
+
/**
|
|
32
|
+
* The waived share at which the fraction stops describing the surface and starts describing the
|
|
33
|
+
* exemptions. Half is the point where the tools NOT being measured outnumber the tools that are, so
|
|
34
|
+
* a reader who takes the percentage at face value is wrong more often than right.
|
|
35
|
+
*
|
|
36
|
+
* A threshold rather than a hard failure because a high waiver share is legitimate — a read-only
|
|
37
|
+
* suite against a mostly-mutating server is exactly that — and failing it would push authors to
|
|
38
|
+
* delete the waivers, which is how the denominator goes blind in the first place.
|
|
39
|
+
*/
|
|
40
|
+
export const WAIVER_SHARE_WARN_THRESHOLD = 0.5;
|
|
41
|
+
/** Percentage of `total` that `part` represents, to one decimal place. `total` of 0 yields 0. */
|
|
42
|
+
function percent(part, total) {
|
|
43
|
+
if (total === 0)
|
|
44
|
+
return 0;
|
|
45
|
+
return Math.round((part / total) * 1000) / 10;
|
|
46
|
+
}
|
|
47
|
+
/** Bucket key for the per-server breakdown — one stable string per bucket identity. */
|
|
48
|
+
function bucketKeyFor(record) {
|
|
49
|
+
if (record.server === undefined)
|
|
50
|
+
return 'builtin';
|
|
51
|
+
if (record.server === '')
|
|
52
|
+
return 'unresolved';
|
|
53
|
+
return `mcp:${record.server}`;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Fold the per-cell inventories into one, and say so when they disagreed.
|
|
57
|
+
*
|
|
58
|
+
* **The union is the denominator, not the first cell's list.** Each cell resolves its own tools (a
|
|
59
|
+
* fresh MCP client per cell), so a cell whose server failed to connect advertises a shorter list. A
|
|
60
|
+
* denominator taken from that cell would quietly shrink, and coverage would IMPROVE because a server
|
|
61
|
+
* broke — the most misleading direction a coverage number can move. The union keeps the tool in the
|
|
62
|
+
* denominator, where it shows up as uncovered.
|
|
63
|
+
*
|
|
64
|
+
* Disagreement is still a warning, because a varying inventory means the cells did not all see the
|
|
65
|
+
* same surface and no single fraction describes the run exactly.
|
|
66
|
+
*/
|
|
67
|
+
function unionInventories(inventories) {
|
|
68
|
+
const tools = new Map();
|
|
69
|
+
const filteredOut = new Map();
|
|
70
|
+
let unnamed = 0;
|
|
71
|
+
let disagreed = false;
|
|
72
|
+
let firstSignature;
|
|
73
|
+
for (const inventory of inventories) {
|
|
74
|
+
for (const record of inventory.tools) {
|
|
75
|
+
if (!tools.has(record.name))
|
|
76
|
+
tools.set(record.name, record);
|
|
77
|
+
}
|
|
78
|
+
for (const record of inventory.filteredOut) {
|
|
79
|
+
if (!filteredOut.has(record.name))
|
|
80
|
+
filteredOut.set(record.name, record);
|
|
81
|
+
}
|
|
82
|
+
// The same session's inventory repeated per cell must not multiply, so take the largest count
|
|
83
|
+
// rather than the sum — `unnamed` describes one agent's surface, like every other field here.
|
|
84
|
+
unnamed = Math.max(unnamed, inventory.unnamed);
|
|
85
|
+
// Compared as JSON rather than as a joined string: any separator character that could itself
|
|
86
|
+
// occur in a tool name would make two different inventories compare equal.
|
|
87
|
+
const signature = JSON.stringify(inventory.tools.map((record) => record.name).sort());
|
|
88
|
+
if (firstSignature === undefined)
|
|
89
|
+
firstSignature = signature;
|
|
90
|
+
else if (signature !== firstSignature)
|
|
91
|
+
disagreed = true;
|
|
92
|
+
}
|
|
93
|
+
return {
|
|
94
|
+
tools: [...tools.values()],
|
|
95
|
+
filteredOut: [...filteredOut.values()],
|
|
96
|
+
unnamed,
|
|
97
|
+
disagreed,
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* Compute the coverage report, or `undefined` when no inventory was observed at all.
|
|
102
|
+
*
|
|
103
|
+
* `undefined` is the honest answer for a target that cannot supply a denominator, and is why this
|
|
104
|
+
* returns a value rather than a zeroed report: **`0/0` printed as a coverage figure is the "metric
|
|
105
|
+
* nobody can trust" failure in its purest form** — it looks like a measurement, it is green, and it
|
|
106
|
+
* measured nothing. A caller that declared gates on a target that cannot be measured is rejected at
|
|
107
|
+
* parse time instead (`evalSuite.ts`), so a silent pass is unreachable from both ends.
|
|
108
|
+
*/
|
|
109
|
+
export function computeToolCoverage(input) {
|
|
110
|
+
if (input.inventories.length === 0)
|
|
111
|
+
return undefined;
|
|
112
|
+
const { tools, filteredOut, unnamed, disagreed } = unionInventories(input.inventories);
|
|
113
|
+
const spec = input.spec;
|
|
114
|
+
const waivePatterns = spec?.waive ?? [];
|
|
115
|
+
const exercised = new Set(input.exercised);
|
|
116
|
+
const warnings = [];
|
|
117
|
+
const gateFailures = [];
|
|
118
|
+
const waived = [];
|
|
119
|
+
const counted = [];
|
|
120
|
+
for (const record of tools) {
|
|
121
|
+
if (waivePatterns.some((pattern) => toolNameMatchesPattern(record.name, pattern))) {
|
|
122
|
+
waived.push(record);
|
|
123
|
+
}
|
|
124
|
+
else {
|
|
125
|
+
counted.push(record);
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
const covered = counted.filter((record) => exercised.has(record.name));
|
|
129
|
+
const uncovered = counted.filter((record) => !exercised.has(record.name));
|
|
130
|
+
const total = counted.length;
|
|
131
|
+
// Per-server buckets over the counted (post-waiver) tools — the same population the headline
|
|
132
|
+
// fraction describes, so the buckets sum to it.
|
|
133
|
+
const buckets = new Map();
|
|
134
|
+
for (const record of counted) {
|
|
135
|
+
const key = bucketKeyFor(record);
|
|
136
|
+
let bucket = buckets.get(key);
|
|
137
|
+
if (!bucket) {
|
|
138
|
+
bucket =
|
|
139
|
+
record.server === undefined
|
|
140
|
+
? { kind: 'builtin', covered: [], uncovered: [] }
|
|
141
|
+
: record.server === ''
|
|
142
|
+
? { kind: 'unresolved', covered: [], uncovered: [] }
|
|
143
|
+
: { kind: 'mcp', server: record.server, covered: [], uncovered: [] };
|
|
144
|
+
buckets.set(key, bucket);
|
|
145
|
+
}
|
|
146
|
+
if (exercised.has(record.name))
|
|
147
|
+
bucket.covered.push(record.name);
|
|
148
|
+
else
|
|
149
|
+
bucket.uncovered.push(record.name);
|
|
150
|
+
}
|
|
151
|
+
if (disagreed) {
|
|
152
|
+
warnings.push('the cells did not all advertise the same tools — the denominator is their union, so a ' +
|
|
153
|
+
'tool missing from some cells still counts (check for an MCP server that failed to connect)');
|
|
154
|
+
}
|
|
155
|
+
// A waiver that matches nothing is dead config, and the likely cause is the dangerous one: the
|
|
156
|
+
// tool was renamed, so it is back in the denominator under its new name while the author believes
|
|
157
|
+
// it is still waived. Cheap to say, and it is the decay this whole feature exists to catch.
|
|
158
|
+
for (const pattern of waivePatterns) {
|
|
159
|
+
if (!tools.some((record) => toolNameMatchesPattern(record.name, pattern))) {
|
|
160
|
+
warnings.push(`waive "${pattern}" matched no advertised tool — stale waiver, or a typo`);
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
if (waived.length > 0 && waived.length / tools.length >= WAIVER_SHARE_WARN_THRESHOLD) {
|
|
164
|
+
warnings.push(`${waived.length} of ${tools.length} advertised tool(s) are waived ` +
|
|
165
|
+
`(${percent(waived.length, tools.length)}%) — the figure describes the exemptions more ` +
|
|
166
|
+
'than the surface');
|
|
167
|
+
}
|
|
168
|
+
if (filteredOut.length > 0) {
|
|
169
|
+
warnings.push(`${filteredOut.length} advertised tool(s) were removed by allowedTools and never reached ` +
|
|
170
|
+
'the model — they are in the denominator, so they can only ever read as uncovered');
|
|
171
|
+
}
|
|
172
|
+
// Exercised names the inventory never listed. Not folded into the numerator (that would let a
|
|
173
|
+
// fraction exceed 1 and would credit coverage of a tool nobody advertised), and not dropped
|
|
174
|
+
// either — a name that ran but was never advertised means the two halves are measuring different
|
|
175
|
+
// populations, which is exactly the kind of quiet mismatch this report must surface.
|
|
176
|
+
const advertisedNames = new Set(tools.map((record) => record.name));
|
|
177
|
+
const unexpected = [...exercised].filter((name) => !advertisedNames.has(name)).sort();
|
|
178
|
+
if (unexpected.length > 0) {
|
|
179
|
+
warnings.push(`${unexpected.length} exercised tool(s) were not in the advertised inventory ` +
|
|
180
|
+
`(${unexpected.join(', ')}) — they are not counted in the numerator`);
|
|
181
|
+
}
|
|
182
|
+
if (spec) {
|
|
183
|
+
// Deliberately asymmetric with `waive`, which warns when a pattern matches nothing. A `require`
|
|
184
|
+
// pattern that matches nothing — the mistyped or stale entry — lands in the first branch and
|
|
185
|
+
// fails the gate, which already stops the run and names the pattern, so a stale-config warning
|
|
186
|
+
// beside it would only be quieter duplication. The warning below is the narrower case the gate
|
|
187
|
+
// cannot state: the pattern WAS satisfied by an exercised tool, but no advertised tool matches
|
|
188
|
+
// it, so the run passes while the two halves disagree about what exists.
|
|
189
|
+
for (const pattern of spec.require) {
|
|
190
|
+
if (![...exercised].some((name) => toolNameMatchesPattern(name, pattern))) {
|
|
191
|
+
gateFailures.push(`require "${pattern}": no case exercised a tool matching it`);
|
|
192
|
+
}
|
|
193
|
+
else if (!tools.some((record) => toolNameMatchesPattern(record.name, pattern))) {
|
|
194
|
+
warnings.push(`require "${pattern}" matched no ADVERTISED tool — it was satisfied by a tool the ` +
|
|
195
|
+
'inventory does not list');
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
if (spec.min !== undefined) {
|
|
199
|
+
if (total === 0) {
|
|
200
|
+
// A floor over an empty denominator cannot be met, and must not be treated as met. This is
|
|
201
|
+
// the case where every tool went missing (an MCP server that never connected, or a waiver
|
|
202
|
+
// list that swallowed the whole surface) — the one run where a vacuous pass would be most
|
|
203
|
+
// misleading, because it is indistinguishable from perfect coverage.
|
|
204
|
+
gateFailures.push(`min ${spec.min}%: no tools remain in the denominator — ` +
|
|
205
|
+
(tools.length === 0
|
|
206
|
+
? 'the agent advertised none'
|
|
207
|
+
: `all ${tools.length} advertised tool(s) are waived`));
|
|
208
|
+
}
|
|
209
|
+
else if (percent(covered.length, total) < spec.min) {
|
|
210
|
+
gateFailures.push(`min ${spec.min}%: covered ${covered.length}/${total} ` +
|
|
211
|
+
`(${percent(covered.length, total)}%)`);
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
return {
|
|
216
|
+
covered: covered.map((record) => record.name),
|
|
217
|
+
uncovered: uncovered.map((record) => record.name),
|
|
218
|
+
waived: waived.map((record) => record.name),
|
|
219
|
+
filteredOut: filteredOut.map((record) => record.name),
|
|
220
|
+
unnamed,
|
|
221
|
+
byServer: [...buckets.values()],
|
|
222
|
+
warnings,
|
|
223
|
+
gateFailures,
|
|
224
|
+
};
|
|
225
|
+
}
|
|
226
|
+
/**
|
|
227
|
+
* Union several suites' reports into the run-level figure, or `undefined` when none was produced.
|
|
228
|
+
*
|
|
229
|
+
* **A set union, never a sum.** A directory run points every suite at the same agent, so the same 41
|
|
230
|
+
* advertised tools appear in each report; summing would report 123 tools and make one suite's
|
|
231
|
+
* coverage of a tool count three times. Coverage is a directory-level figure precisely because one
|
|
232
|
+
* suite covering 3 tools is fine if its sibling covers the other 38 — which only reads correctly
|
|
233
|
+
* when a tool covered anywhere is covered once.
|
|
234
|
+
*
|
|
235
|
+
* **Carries no warnings and no gate failures, deliberately.** A gate is declared in a suite and
|
|
236
|
+
* graded against that suite (the `metrics:` precedent), and the run's exit is the OR of those. If a
|
|
237
|
+
* declared floor were re-applied to the aggregate, the same suite would pass when run alone and fail
|
|
238
|
+
* when run as part of a directory, with nothing in either output saying the threshold had moved.
|
|
239
|
+
* This function stays reporting only.
|
|
240
|
+
*
|
|
241
|
+
* **A run-level floor is a different declaration, graded elsewhere.** BATCH-48 adds one —
|
|
242
|
+
* `evalToolCoverage` in gth config, resolved once per run from the base config and never from an
|
|
243
|
+
* identity's — and {@link gradeRunToolCoverage} is what grades it, against the report this function
|
|
244
|
+
* returns. Keeping the two apart is what preserves the argument above: a suite's `min:` is never
|
|
245
|
+
* promoted to the aggregate, and the run's floor is never read back as a suite's.
|
|
246
|
+
*/
|
|
247
|
+
export function aggregateToolCoverage(reports) {
|
|
248
|
+
if (reports.length === 0)
|
|
249
|
+
return undefined;
|
|
250
|
+
const covered = new Set();
|
|
251
|
+
const uncovered = new Set();
|
|
252
|
+
const waived = new Set();
|
|
253
|
+
const filteredOut = new Set();
|
|
254
|
+
const buckets = new Map();
|
|
255
|
+
let unnamed = 0;
|
|
256
|
+
for (const report of reports) {
|
|
257
|
+
for (const name of report.covered)
|
|
258
|
+
covered.add(name);
|
|
259
|
+
for (const name of report.uncovered)
|
|
260
|
+
uncovered.add(name);
|
|
261
|
+
for (const name of report.waived)
|
|
262
|
+
waived.add(name);
|
|
263
|
+
for (const name of report.filteredOut)
|
|
264
|
+
filteredOut.add(name);
|
|
265
|
+
unnamed = Math.max(unnamed, report.unnamed);
|
|
266
|
+
for (const bucket of report.byServer) {
|
|
267
|
+
const key = bucket.kind === 'mcp' ? `mcp:${bucket.server}` : bucket.kind;
|
|
268
|
+
let merged = buckets.get(key);
|
|
269
|
+
if (!merged) {
|
|
270
|
+
merged = {
|
|
271
|
+
kind: bucket.kind,
|
|
272
|
+
...(bucket.server !== undefined ? { server: bucket.server } : {}),
|
|
273
|
+
covered: [],
|
|
274
|
+
uncovered: [],
|
|
275
|
+
};
|
|
276
|
+
buckets.set(key, merged);
|
|
277
|
+
}
|
|
278
|
+
merged.covered.push(...bucket.covered);
|
|
279
|
+
merged.uncovered.push(...bucket.uncovered);
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
// A tool covered by ANY suite is covered: that is what makes the aggregate the interesting
|
|
283
|
+
// number. Without this, a tool one suite exercised would still be listed as uncovered because a
|
|
284
|
+
// sibling suite never touched it.
|
|
285
|
+
for (const name of covered)
|
|
286
|
+
uncovered.delete(name);
|
|
287
|
+
// Likewise a tool that is waived in one suite but counted in another stays counted — a waiver is
|
|
288
|
+
// one suite's statement about its own scope, and the run as a whole did measure that tool.
|
|
289
|
+
for (const name of [...covered, ...uncovered])
|
|
290
|
+
waived.delete(name);
|
|
291
|
+
return {
|
|
292
|
+
covered: [...covered],
|
|
293
|
+
uncovered: [...uncovered],
|
|
294
|
+
waived: [...waived],
|
|
295
|
+
filteredOut: [...filteredOut],
|
|
296
|
+
unnamed,
|
|
297
|
+
byServer: [...buckets.values()].map((bucket) => {
|
|
298
|
+
const bucketCovered = [...new Set(bucket.covered)];
|
|
299
|
+
const coveredSet = new Set(bucketCovered);
|
|
300
|
+
return {
|
|
301
|
+
...bucket,
|
|
302
|
+
covered: bucketCovered,
|
|
303
|
+
uncovered: [...new Set(bucket.uncovered)].filter((name) => !coveredSet.has(name)),
|
|
304
|
+
};
|
|
305
|
+
}),
|
|
306
|
+
warnings: [],
|
|
307
|
+
gateFailures: [],
|
|
308
|
+
};
|
|
309
|
+
}
|
|
310
|
+
/** The human name of a {@link RunToolCoverageSource}, shared by the source line and the gate text. */
|
|
311
|
+
function describeRunCoverageSource(source) {
|
|
312
|
+
switch (source.kind) {
|
|
313
|
+
case 'profile':
|
|
314
|
+
return `profile ${source.profile ?? '(unnamed)'}`;
|
|
315
|
+
case 'config-file':
|
|
316
|
+
return `config file ${source.path ?? '(unnamed)'}`;
|
|
317
|
+
case 'global':
|
|
318
|
+
return 'the global config';
|
|
319
|
+
default:
|
|
320
|
+
return 'the project config';
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
/**
|
|
324
|
+
* Whether a run-level spec declares anything to grade or apply.
|
|
325
|
+
*
|
|
326
|
+
* An absent config key and an empty object are the same thing to a caller: no floor, no exemption.
|
|
327
|
+
* Distinguishing them would make "the run has no floor" depend on which spelling the config used.
|
|
328
|
+
*/
|
|
329
|
+
export function hasRunToolCoverageSpec(spec) {
|
|
330
|
+
return spec !== undefined && (spec.min !== undefined || spec.waive.length > 0);
|
|
331
|
+
}
|
|
332
|
+
/**
|
|
333
|
+
* Grade a run-level coverage spec against the run's aggregate.
|
|
334
|
+
*
|
|
335
|
+
* **Separate from {@link aggregateToolCoverage} on purpose.** That function reports the union and
|
|
336
|
+
* carries no gate, because re-applying a SUITE's declared floor to a different denominator would
|
|
337
|
+
* make the same suite pass alone and fail inside a directory. This function grades a floor that was
|
|
338
|
+
* declared for the run, against the aggregate it names, so the two thresholds stay independent: a
|
|
339
|
+
* suite's `min:` is never promoted here, and this floor is never written back onto a suite.
|
|
340
|
+
*
|
|
341
|
+
* **The denominator is the aggregate's reconciled one, minus this spec's `waive`.** The aggregate
|
|
342
|
+
* has already applied the reconciliation — a tool waived in one suite but counted in another stays
|
|
343
|
+
* counted, and a tool waived in every suite is already out. A suite's own waiver therefore does not
|
|
344
|
+
* shrink the run denominator further; only the run-level list does. A tool the run-level list waives
|
|
345
|
+
* leaves the denominator even when it was covered, the same way a suite-level waiver does inside
|
|
346
|
+
* {@link computeToolCoverage}.
|
|
347
|
+
*
|
|
348
|
+
* **A floor with nothing to grade fails.** `min` set and no report at all — every suite targeted
|
|
349
|
+
* something that cannot supply a denominator — must not pass: a floor quietly skipped reports green
|
|
350
|
+
* forever over a run it never measured.
|
|
351
|
+
*/
|
|
352
|
+
export function gradeRunToolCoverage(aggregate, spec, source) {
|
|
353
|
+
const sourceText = describeRunCoverageSource(source);
|
|
354
|
+
const sourceLine = spec.min !== undefined
|
|
355
|
+
? `graded against evalToolCoverage.min ${spec.min}% from ${sourceText}`
|
|
356
|
+
: `evalToolCoverage applied from ${sourceText}`;
|
|
357
|
+
if (!aggregate) {
|
|
358
|
+
// Nothing was measured. A waiver list has nothing to match against either, so it neither warns
|
|
359
|
+
// nor silently passes a floor: the floor is the thing that was declared, and it could not be
|
|
360
|
+
// graded.
|
|
361
|
+
const gateFailures = spec.min !== undefined
|
|
362
|
+
? [
|
|
363
|
+
`evalToolCoverage.min ${spec.min}% from ${sourceText} could not be graded: no suite ` +
|
|
364
|
+
'produced a coverage report',
|
|
365
|
+
]
|
|
366
|
+
: [];
|
|
367
|
+
return { report: undefined, gateFailures, warnings: [], sourceLine };
|
|
368
|
+
}
|
|
369
|
+
const warnings = [];
|
|
370
|
+
const names = [...aggregate.covered, ...aggregate.uncovered, ...aggregate.waived];
|
|
371
|
+
const waivedByRun = new Set();
|
|
372
|
+
for (const pattern of spec.waive) {
|
|
373
|
+
const matched = names.filter((name) => toolNameMatchesPattern(name, pattern));
|
|
374
|
+
if (matched.length === 0) {
|
|
375
|
+
warnings.push(`evalToolCoverage.waive "${pattern}" matched no advertised tool — stale waiver, or a typo`);
|
|
376
|
+
}
|
|
377
|
+
for (const name of matched)
|
|
378
|
+
waivedByRun.add(name);
|
|
379
|
+
}
|
|
380
|
+
// A covered tool the run waives leaves the numerator, exactly as a suite-level waiver of a
|
|
381
|
+
// covered tool leaves it. Keeping it covered would let the run exempt a tool and still be
|
|
382
|
+
// credited for exercising it, which is the opposite of what a waiver means.
|
|
383
|
+
const covered = aggregate.covered.filter((name) => !waivedByRun.has(name));
|
|
384
|
+
const uncovered = aggregate.uncovered.filter((name) => !waivedByRun.has(name));
|
|
385
|
+
const waived = [...new Set([...aggregate.waived, ...waivedByRun])];
|
|
386
|
+
const gateFailures = [];
|
|
387
|
+
if (spec.min !== undefined) {
|
|
388
|
+
const total = covered.length + uncovered.length;
|
|
389
|
+
if (total === 0) {
|
|
390
|
+
gateFailures.push(`evalToolCoverage.min ${spec.min}% from ${sourceText}: no tools remain in the ` +
|
|
391
|
+
`denominator — all ${names.length} advertised tool(s) are waived`);
|
|
392
|
+
}
|
|
393
|
+
else if (percent(covered.length, total) < spec.min) {
|
|
394
|
+
gateFailures.push(`evalToolCoverage.min ${spec.min}% from ${sourceText}: covered ${covered.length}/${total} ` +
|
|
395
|
+
`(${percent(covered.length, total)}%)`);
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
// The per-server buckets describe the same post-waiver population as the headline, so they sum
|
|
399
|
+
// to it; a run-waived tool leaves its bucket too, and a bucket left empty is dropped.
|
|
400
|
+
const byServer = aggregate.byServer
|
|
401
|
+
.map((bucket) => ({
|
|
402
|
+
...bucket,
|
|
403
|
+
covered: bucket.covered.filter((name) => !waivedByRun.has(name)),
|
|
404
|
+
uncovered: bucket.uncovered.filter((name) => !waivedByRun.has(name)),
|
|
405
|
+
}))
|
|
406
|
+
.filter((bucket) => bucket.covered.length + bucket.uncovered.length > 0);
|
|
407
|
+
return {
|
|
408
|
+
report: { ...aggregate, covered, uncovered, waived, byServer, gateFailures: [], warnings: [] },
|
|
409
|
+
gateFailures,
|
|
410
|
+
warnings,
|
|
411
|
+
sourceLine,
|
|
412
|
+
};
|
|
413
|
+
}
|
|
414
|
+
//# sourceMappingURL=toolCoverage.js.map
|