webmcp-gauge 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +121 -0
- package/action.yml +162 -0
- package/bin/webmcp-gauge.mjs +544 -0
- package/bin/webmcp-gauge.test.mjs +354 -0
- package/browser/launch.mjs +188 -0
- package/browser/serve.mjs +78 -0
- package/browser/session.mjs +210 -0
- package/browser/webmcp.mjs +432 -0
- package/browser/webmcp.test.mjs +299 -0
- package/core/args.mjs +93 -0
- package/core/args.test.mjs +85 -0
- package/core/capture-seam.test.mjs +86 -0
- package/core/cohort.mjs +432 -0
- package/core/cohort.test.mjs +370 -0
- package/core/gallery.mjs +145 -0
- package/core/gallery.test.mjs +128 -0
- package/core/gate.mjs +164 -0
- package/core/gate.test.mjs +213 -0
- package/core/lint.mjs +381 -0
- package/core/lint.test.mjs +346 -0
- package/core/orchestrate.mjs +128 -0
- package/core/orchestrate.test.mjs +191 -0
- package/core/stats.mjs +172 -0
- package/core/stats.test.mjs +156 -0
- package/core/sweep.mjs +274 -0
- package/core/sweep.test.mjs +162 -0
- package/core/taxonomy.mjs +175 -0
- package/core/taxonomy.test.mjs +198 -0
- package/core/trial.mjs +248 -0
- package/core/visibility.mjs +163 -0
- package/core/visibility.test.mjs +164 -0
- package/docs/concept.md +468 -0
- package/docs/explainer.md +161 -0
- package/docs/getting-started.md +331 -0
- package/fixtures/README.md +42 -0
- package/fixtures/airlock.utterances.json +284 -0
- package/fixtures/broken/compose.mjs +52 -0
- package/fixtures/broken/compose.test.mjs +270 -0
- package/fixtures/broken/sample-expenses.csv +966 -0
- package/fixtures/broken/tools.json +1311 -0
- package/fixtures/broken/twin.html +482 -0
- package/fixtures/broken/widget.html +62 -0
- package/fixtures/gallery/gallery.html +56 -0
- package/judges/openai-compatible.mjs +145 -0
- package/package.json +53 -0
- package/report/badge.mjs +110 -0
- package/report/badge.test.mjs +97 -0
- package/report/emit.mjs +282 -0
- package/report/published-runs.test.mjs +77 -0
- package/report/scorecard.mjs +157 -0
- package/report/scorecard.test.mjs +130 -0
package/core/lint.mjs
ADDED
|
@@ -0,0 +1,381 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* L0 - the static linter. No browser needed to reason, no model, no API key.
|
|
3
|
+
*
|
|
4
|
+
* It reads one thing: the manifest a page exposes (names, descriptions, input
|
|
5
|
+
* schemas, annotations). Everything it reports is a property of that manifest, so
|
|
6
|
+
* the same function lints a live page's `getTools()` and a hand-written manifest in
|
|
7
|
+
* a JSON file, and its answers are deterministic - which is the point. The harness
|
|
8
|
+
* needs a judge, a browser and minutes; the linter needs a manifest and no
|
|
9
|
+
* network, so it is the free on-ramp and the fast half of the same question.
|
|
10
|
+
*
|
|
11
|
+
* Thresholds are calibrated against the one page in this project with a measured
|
|
12
|
+
* invocation rate. The reference page reads 100% [96.9%, 100.0%] on five tools and
|
|
13
|
+
* 99.2% on the sixth over 960 trials, its longest schema has exactly six
|
|
14
|
+
* properties, its thinnest description is 76 characters, and every property it
|
|
15
|
+
* declares carries a description. Defaults are therefore set so that page lints
|
|
16
|
+
* clean: a default that flags a manifest known to work is a broken default, not a
|
|
17
|
+
* strict one. Every threshold is an option, because they are heuristics with a
|
|
18
|
+
* measurement behind them rather than constants from a spec.
|
|
19
|
+
*
|
|
20
|
+
* What it deliberately does not do: guess. It never rewrites a description, never
|
|
21
|
+
* scores "quality" on a scale, and never claims a flagged manifest will invoke
|
|
22
|
+
* badly - that claim belongs to the harness, which measures it.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
export const DEFAULT_OPTIONS = Object.freeze({
|
|
26
|
+
/** Below this, a description is too thin to separate a tool from its neighbour. */
|
|
27
|
+
minDescriptionChars: 60,
|
|
28
|
+
/** The reference page's largest schema has six properties and invokes at 99.2%. */
|
|
29
|
+
maxProperties: 6,
|
|
30
|
+
/**
|
|
31
|
+
* There is no published per-page tool budget. The only field figure is a report of
|
|
32
|
+
* 296 registered tools silently disabling WebMCP for an entire page — and that
|
|
33
|
+
* figure **does not reproduce on Chrome 152.0.7977.65**: measured 2026-08-31, a
|
|
34
|
+
* page registering 507 tools had all 507 accepted, listed by `getTools()` and
|
|
35
|
+
* surfaced by the browser's own WebMCP domain. So this rule warns rather than
|
|
36
|
+
* asserts, and both levels are warnings: an error on something measured to work
|
|
37
|
+
* would be the false positive the calibration principle above exists to prevent.
|
|
38
|
+
*/
|
|
39
|
+
budgetWarnAt: 64,
|
|
40
|
+
budgetBreakAt: 296,
|
|
41
|
+
/** Token-set overlap above which two descriptions are hard to tell apart. */
|
|
42
|
+
nearDuplicateThreshold: 0.7,
|
|
43
|
+
/**
|
|
44
|
+
* Token-set overlap above which two tool *names* are close enough that a model
|
|
45
|
+
* with no useful description to go on may pick between them on the name.
|
|
46
|
+
* `sum_by_category` against `summarise_by_category` scores 0.500.
|
|
47
|
+
*/
|
|
48
|
+
nameSimilarityThreshold: 0.4,
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Names that survive registration are not the same set as names that are safe.
|
|
53
|
+
* Measured on Chrome 152.0.7977.65 (2026-08-30): `registerTool` throws
|
|
54
|
+
* "Invalid tool name" for a name containing a space - so spec issue #145's
|
|
55
|
+
* "silently does nothing" is not what this build does - while a dotted name like
|
|
56
|
+
* `top.expenses.v2` is accepted and appears in getTools(). A live manifest
|
|
57
|
+
* therefore cannot show the worst names at all; they show up as a tool that is
|
|
58
|
+
* missing, which the harness scores as not_registered. This rule earns its keep on
|
|
59
|
+
* the names a build does accept, and on static manifests read from source.
|
|
60
|
+
*/
|
|
61
|
+
const SAFE_NAME = /^[A-Za-z0-9_][A-Za-z0-9_-]*$/;
|
|
62
|
+
const MAX_NAME_CHARS = 64;
|
|
63
|
+
|
|
64
|
+
export const RULES = Object.freeze([
|
|
65
|
+
{ id: 'name/invalid-characters', family: 'names', severity: 'error' },
|
|
66
|
+
{ id: 'name/too-long', family: 'names', severity: 'warning' },
|
|
67
|
+
{ id: 'name/duplicate', family: 'names', severity: 'error' },
|
|
68
|
+
{ id: 'description/missing', family: 'descriptions', severity: 'error' },
|
|
69
|
+
{ id: 'description/thin', family: 'descriptions', severity: 'warning' },
|
|
70
|
+
{ id: 'description/duplicate', family: 'descriptions', severity: 'error' },
|
|
71
|
+
{ id: 'description/near-duplicate', family: 'descriptions', severity: 'warning' },
|
|
72
|
+
/**
|
|
73
|
+
* Adopted 2026-09-05 on measurement, and it is the only rule here aimed at a
|
|
74
|
+
* mechanism rather than at a property of the text.
|
|
75
|
+
*
|
|
76
|
+
* Five arms took `sum_by_category` from 95.0% to between 48.3% and 58.3% while the
|
|
77
|
+
* overlap between its description and its competitor's fell from 1.000 to 0.130 —
|
|
78
|
+
* so *similarity* is not what does the damage, and no threshold on
|
|
79
|
+
* `nearDuplicateThreshold` can catch it (`reports/ladder-2026-09-05.md`). What the
|
|
80
|
+
* judge actually did, in its own recorded words, was pick on the **name**:
|
|
81
|
+
* "summarise_by_category seems designed for summarizing by category". It does that
|
|
82
|
+
* when neither description tells it which tool is which.
|
|
83
|
+
*
|
|
84
|
+
* So this rule asks the two questions that mechanism needs, both model-free:
|
|
85
|
+
* are the names close, and does *neither* description say what its own tool is
|
|
86
|
+
* for. The second half is absolute rather than relative on purpose — the failed
|
|
87
|
+
* option was the relative one.
|
|
88
|
+
*
|
|
89
|
+
* **Warning, not error**, and the reason is this project's own precedent:
|
|
90
|
+
* `budget/headroom` was demoted because a linter that fails a build on a
|
|
91
|
+
* threshold nobody has reproduced is a linter people disable. This rule
|
|
92
|
+
* reproduces every arm across 13 manifests (`probes/name-proxy-rule.mjs`), but all
|
|
93
|
+
* 13 were written here. The cohort corpus settled the open question on
|
|
94
|
+
* 2026-09-26, and the maintainer decided to **keep the warning**: on 1,507
|
|
95
|
+
* captured manifests (`reports/census-2026-09-26.md`) the rule fires on 60 of
|
|
96
|
+
* them (4.0%), one a textbook indistinguishable pair, the rest prefix-heavy
|
|
97
|
+
* namespaces whose descriptions differentiate semantically — promoting would
|
|
98
|
+
* fail real builds on a naming pattern rather than a measured confusion.
|
|
99
|
+
*/
|
|
100
|
+
{ id: 'description/indistinguishable-pair', family: 'descriptions', severity: 'warning' },
|
|
101
|
+
{ id: 'schema/not-object', family: 'schemas', severity: 'error' },
|
|
102
|
+
{ id: 'schema/required-without-description', family: 'schemas', severity: 'error' },
|
|
103
|
+
{ id: 'schema/over-parameterised', family: 'schemas', severity: 'warning' },
|
|
104
|
+
{ id: 'schema/undocumented-property', family: 'schemas', severity: 'warning' },
|
|
105
|
+
{ id: 'schema/missing-type', family: 'schemas', severity: 'warning' },
|
|
106
|
+
{ id: 'budget/headroom', family: 'budget', severity: 'warning' },
|
|
107
|
+
]);
|
|
108
|
+
|
|
109
|
+
const severityOf = (ruleId) => RULES.find((rule) => rule.id === ruleId)?.severity ?? 'warning';
|
|
110
|
+
|
|
111
|
+
const isPlainObject = (value) =>
|
|
112
|
+
typeof value === 'object' && value !== null && !Array.isArray(value);
|
|
113
|
+
|
|
114
|
+
const normaliseDescription = (text) =>
|
|
115
|
+
String(text ?? '')
|
|
116
|
+
.toLowerCase()
|
|
117
|
+
.replace(/\s+/g, ' ')
|
|
118
|
+
.trim();
|
|
119
|
+
|
|
120
|
+
const tokenise = (text) =>
|
|
121
|
+
new Set(
|
|
122
|
+
normaliseDescription(text)
|
|
123
|
+
.replace(/[^a-z0-9 ]+/g, ' ')
|
|
124
|
+
.split(' ')
|
|
125
|
+
.filter((token) => token.length > 0)
|
|
126
|
+
);
|
|
127
|
+
|
|
128
|
+
/** Token-set Jaccard: shared vocabulary over total vocabulary, symmetric and cheap. */
|
|
129
|
+
export const similarity = (a, b) => {
|
|
130
|
+
const left = tokenise(a);
|
|
131
|
+
const right = tokenise(b);
|
|
132
|
+
if (left.size === 0 || right.size === 0) return 0;
|
|
133
|
+
let shared = 0;
|
|
134
|
+
for (const token of left) if (right.has(token)) shared += 1;
|
|
135
|
+
return shared / (left.size + right.size - shared);
|
|
136
|
+
};
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Does this description say which tool it belongs to? Answered without a model, by
|
|
140
|
+
* asking whether it mentions the substantial tokens of its own name — plus any short
|
|
141
|
+
* token that distinguishes it from the sibling it is being compared against, since
|
|
142
|
+
* `sum` against `summarise` is exactly the distinction that matters and `by` is not.
|
|
143
|
+
*/
|
|
144
|
+
const describesItsOwnName = (name, description, distinguishing = []) => {
|
|
145
|
+
const said = tokenise(description);
|
|
146
|
+
const wanted = [...tokenise(name)].filter((token) => token.length > 2).concat(distinguishing);
|
|
147
|
+
return wanted.some((token) => said.has(token));
|
|
148
|
+
};
|
|
149
|
+
|
|
150
|
+
export const lintManifest = ({ manifest, options = {} }) => {
|
|
151
|
+
const settings = { ...DEFAULT_OPTIONS, ...options };
|
|
152
|
+
const tools = Array.isArray(manifest?.tools) ? manifest.tools : [];
|
|
153
|
+
const findings = [];
|
|
154
|
+
|
|
155
|
+
const add = (rule, tool, detail, evidence = null) =>
|
|
156
|
+
findings.push({ rule, severity: severityOf(rule), tool, detail, evidence });
|
|
157
|
+
|
|
158
|
+
const seenNames = new Map();
|
|
159
|
+
const descriptions = [];
|
|
160
|
+
|
|
161
|
+
for (const tool of tools) {
|
|
162
|
+
const name = typeof tool?.name === 'string' ? tool.name : '';
|
|
163
|
+
|
|
164
|
+
if (!SAFE_NAME.test(name)) {
|
|
165
|
+
const offenders = [...name].filter((character) => !/[A-Za-z0-9_-]/.test(character));
|
|
166
|
+
add(
|
|
167
|
+
'name/invalid-characters',
|
|
168
|
+
name || '(unnamed)',
|
|
169
|
+
offenders.length > 0
|
|
170
|
+
? `contains ${offenders.map((character) => (character === ' ' ? 'a space' : `'${character}'`)).join(', ')}; a client may refuse to register it or a model may fail to reference it`
|
|
171
|
+
: 'does not start with a letter, digit or underscore',
|
|
172
|
+
{ name }
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
if (name.length > MAX_NAME_CHARS) {
|
|
176
|
+
add('name/too-long', name, `${name.length} characters; keep names under ${MAX_NAME_CHARS}`);
|
|
177
|
+
}
|
|
178
|
+
if (seenNames.has(name)) {
|
|
179
|
+
add('name/duplicate', name, 'two tools share this name, so only one of them is reachable');
|
|
180
|
+
}
|
|
181
|
+
seenNames.set(name, true);
|
|
182
|
+
|
|
183
|
+
const description = typeof tool?.description === 'string' ? tool.description : '';
|
|
184
|
+
const trimmed = description.trim();
|
|
185
|
+
if (trimmed.length === 0) {
|
|
186
|
+
add(
|
|
187
|
+
'description/missing',
|
|
188
|
+
name,
|
|
189
|
+
'no description at all: selection is the model reading this text, so an empty one is an unreachable tool'
|
|
190
|
+
);
|
|
191
|
+
} else {
|
|
192
|
+
if (trimmed.length < settings.minDescriptionChars) {
|
|
193
|
+
add(
|
|
194
|
+
'description/thin',
|
|
195
|
+
name,
|
|
196
|
+
`${trimmed.length} characters, below the ${settings.minDescriptionChars}-character floor calibrated on the reference page`,
|
|
197
|
+
{ description: trimmed }
|
|
198
|
+
);
|
|
199
|
+
}
|
|
200
|
+
descriptions.push({ name, description: trimmed, key: normaliseDescription(trimmed) });
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
const schema = tool?.inputSchema;
|
|
204
|
+
if (!isPlainObject(schema) || schema.type !== 'object' || (schema.properties !== undefined && !isPlainObject(schema.properties))) {
|
|
205
|
+
add(
|
|
206
|
+
'schema/not-object',
|
|
207
|
+
name,
|
|
208
|
+
`inputSchema must be an object schema with an object 'properties' map; got ${isPlainObject(schema) ? `type '${schema.type}'` : typeof schema}`,
|
|
209
|
+
{ inputSchema: schema ?? null }
|
|
210
|
+
);
|
|
211
|
+
continue;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
const properties = isPlainObject(schema.properties) ? schema.properties : {};
|
|
215
|
+
const propertyNames = Object.keys(properties);
|
|
216
|
+
const required = Array.isArray(schema.required) ? schema.required : [];
|
|
217
|
+
|
|
218
|
+
if (propertyNames.length > settings.maxProperties) {
|
|
219
|
+
add(
|
|
220
|
+
'schema/over-parameterised',
|
|
221
|
+
name,
|
|
222
|
+
`${propertyNames.length} properties against a ${settings.maxProperties}-property reference maximum; every extra argument is one more thing a model has to invent`
|
|
223
|
+
);
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
for (const [property, definition] of Object.entries(properties)) {
|
|
227
|
+
const documented =
|
|
228
|
+
isPlainObject(definition) && typeof definition.description === 'string' && definition.description.trim().length > 0;
|
|
229
|
+
const typed = isPlainObject(definition) && typeof definition.type === 'string' && definition.type.length > 0;
|
|
230
|
+
|
|
231
|
+
if (required.includes(property) && !documented) {
|
|
232
|
+
add(
|
|
233
|
+
'schema/required-without-description',
|
|
234
|
+
name,
|
|
235
|
+
`'${property}' is required and undocumented, so a model must guess a value it cannot derive from the request`
|
|
236
|
+
);
|
|
237
|
+
} else if (!documented) {
|
|
238
|
+
add('schema/undocumented-property', name, `'${property}' has no description`);
|
|
239
|
+
}
|
|
240
|
+
if (!typed) add('schema/missing-type', name, `'${property}' declares no type`);
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
for (const property of required) {
|
|
244
|
+
if (!propertyNames.includes(property)) {
|
|
245
|
+
add(
|
|
246
|
+
'schema/required-without-description',
|
|
247
|
+
name,
|
|
248
|
+
`'${property}' is listed in required but is not declared in properties`
|
|
249
|
+
);
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
for (let i = 0; i < descriptions.length; i += 1) {
|
|
255
|
+
for (let j = i + 1; j < descriptions.length; j += 1) {
|
|
256
|
+
const left = descriptions[i];
|
|
257
|
+
const right = descriptions[j];
|
|
258
|
+
if (left.key === right.key) {
|
|
259
|
+
add(
|
|
260
|
+
'description/duplicate',
|
|
261
|
+
`${left.name} + ${right.name}`,
|
|
262
|
+
'identical descriptions: nothing in the manifest can tell these two tools apart',
|
|
263
|
+
{ description: left.description }
|
|
264
|
+
);
|
|
265
|
+
continue;
|
|
266
|
+
}
|
|
267
|
+
const overlap = similarity(left.description, right.description);
|
|
268
|
+
if (overlap >= settings.nearDuplicateThreshold) {
|
|
269
|
+
add(
|
|
270
|
+
'description/near-duplicate',
|
|
271
|
+
`${left.name} + ${right.name}`,
|
|
272
|
+
`descriptions share ${(overlap * 100).toFixed(0)}% of their vocabulary, at or above the ${(settings.nearDuplicateThreshold * 100).toFixed(0)}% threshold`,
|
|
273
|
+
{ left: left.description, right: right.description }
|
|
274
|
+
);
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
// Close names plus two descriptions that never say which tool is which. Measured
|
|
280
|
+
// as the mechanism behind a 45-point loss that no similarity threshold catches;
|
|
281
|
+
// see the rule's entry in RULES for why it is a warning rather than an error.
|
|
282
|
+
for (let i = 0; i < descriptions.length; i += 1) {
|
|
283
|
+
for (let j = i + 1; j < descriptions.length; j += 1) {
|
|
284
|
+
const left = descriptions[i];
|
|
285
|
+
const right = descriptions[j];
|
|
286
|
+
const nameOverlap = similarity(left.name, right.name);
|
|
287
|
+
if (nameOverlap < settings.nameSimilarityThreshold) continue;
|
|
288
|
+
|
|
289
|
+
const leftTokens = tokenise(left.name);
|
|
290
|
+
const rightTokens = tokenise(right.name);
|
|
291
|
+
const leftOnly = [...leftTokens].filter((token) => !rightTokens.has(token));
|
|
292
|
+
const rightOnly = [...rightTokens].filter((token) => !leftTokens.has(token));
|
|
293
|
+
|
|
294
|
+
if (describesItsOwnName(left.name, left.description, leftOnly)) continue;
|
|
295
|
+
if (describesItsOwnName(right.name, right.description, rightOnly)) continue;
|
|
296
|
+
|
|
297
|
+
add(
|
|
298
|
+
'description/indistinguishable-pair',
|
|
299
|
+
`${left.name} + ${right.name}`,
|
|
300
|
+
`names share ${(nameOverlap * 100).toFixed(0)}% of their tokens and neither description says which tool it is, so a model with nothing to choose on will choose on the name`,
|
|
301
|
+
{ nameSimilarity: Number(nameOverlap.toFixed(3)), left: left.description, right: right.description }
|
|
302
|
+
);
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
if (tools.length >= settings.budgetBreakAt) {
|
|
307
|
+
add(
|
|
308
|
+
'budget/headroom',
|
|
309
|
+
null,
|
|
310
|
+
`${tools.length} tools, at or past the ${settings.budgetBreakAt} reported to disable WebMCP for a whole page with no error. That report does not reproduce on Chrome 152.0.7977.65, where 507 tools were all registered and surfaced — so this is a warning about an unknown, not a verified ceiling`,
|
|
311
|
+
{ toolCount: tools.length }
|
|
312
|
+
);
|
|
313
|
+
} else if (tools.length >= settings.budgetWarnAt) {
|
|
314
|
+
add(
|
|
315
|
+
'budget/headroom',
|
|
316
|
+
null,
|
|
317
|
+
`${tools.length} tools; no per-page budget is published, and ${settings.budgetBreakAt} has been reported to disable the feature silently, so headroom here is unknown rather than fine`,
|
|
318
|
+
{ toolCount: tools.length }
|
|
319
|
+
);
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
const counts = findings.reduce(
|
|
323
|
+
(totals, finding) => ({ ...totals, [finding.severity]: (totals[finding.severity] ?? 0) + 1 }),
|
|
324
|
+
{ error: 0, warning: 0 }
|
|
325
|
+
);
|
|
326
|
+
|
|
327
|
+
return {
|
|
328
|
+
schema: 'webmcp-gauge/lint/1',
|
|
329
|
+
generatedAt: new Date().toISOString(),
|
|
330
|
+
thresholds: settings,
|
|
331
|
+
manifest: {
|
|
332
|
+
present: manifest?.present ?? null,
|
|
333
|
+
settled: manifest?.settled ?? null,
|
|
334
|
+
settledAtMs: manifest?.settledAtMs ?? null,
|
|
335
|
+
toolCount: tools.length,
|
|
336
|
+
names: tools.map((tool) => tool?.name ?? null),
|
|
337
|
+
},
|
|
338
|
+
findings,
|
|
339
|
+
counts,
|
|
340
|
+
families: [...new Set(findings.map((finding) => finding.rule.split('/')[0]))],
|
|
341
|
+
};
|
|
342
|
+
};
|
|
343
|
+
|
|
344
|
+
export const lintToText = (result, { subject = null } = {}) => {
|
|
345
|
+
const lines = [];
|
|
346
|
+
const { manifest, counts, findings } = result;
|
|
347
|
+
|
|
348
|
+
lines.push(`webmcp-gauge lint — ${subject ?? '(manifest)'}`);
|
|
349
|
+
lines.push(
|
|
350
|
+
`${manifest.toolCount} tools · ${counts.error} error${counts.error === 1 ? '' : 's'} · ${counts.warning} warning${counts.warning === 1 ? '' : 's'}${manifest.settled === false ? ' · MANIFEST NEVER SETTLED' : ''}`
|
|
351
|
+
);
|
|
352
|
+
lines.push('');
|
|
353
|
+
|
|
354
|
+
if (findings.length === 0) {
|
|
355
|
+
lines.push('No findings. Every rule in this build passed against this manifest.');
|
|
356
|
+
lines.push('');
|
|
357
|
+
lines.push(
|
|
358
|
+
'A clean lint is not a measured invocation rate: L0 reads the manifest, and whether an agent picks these tools is what `run` measures.'
|
|
359
|
+
);
|
|
360
|
+
return `${lines.join('\n')}\n`;
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
const order = { error: 0, warning: 1 };
|
|
364
|
+
for (const finding of [...findings].sort(
|
|
365
|
+
(a, b) => order[a.severity] - order[b.severity] || a.rule.localeCompare(b.rule)
|
|
366
|
+
)) {
|
|
367
|
+
lines.push(
|
|
368
|
+
`${finding.severity === 'error' ? 'ERROR ' : 'WARN '} ${finding.rule}${finding.tool ? ` · ${finding.tool}` : ''}`
|
|
369
|
+
);
|
|
370
|
+
lines.push(` ${finding.detail}`);
|
|
371
|
+
}
|
|
372
|
+
lines.push('');
|
|
373
|
+
lines.push(
|
|
374
|
+
'Thresholds: ' +
|
|
375
|
+
`descriptions ≥ ${result.thresholds.minDescriptionChars} chars, ≤ ${result.thresholds.maxProperties} schema properties, ` +
|
|
376
|
+
`near-duplicate at ${(result.thresholds.nearDuplicateThreshold * 100).toFixed(0)}% vocabulary overlap, budget warning at ${result.thresholds.budgetWarnAt} tools. ` +
|
|
377
|
+
'Calibrated on the reference page, not taken from the spec.'
|
|
378
|
+
);
|
|
379
|
+
|
|
380
|
+
return `${lines.join('\n')}\n`;
|
|
381
|
+
};
|