mcp-context-cost 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +202 -43
- package/dist/audit/audit.d.ts +101 -0
- package/dist/audit/audit.js +492 -16
- package/dist/audit/config.d.ts +38 -0
- package/dist/audit/config.js +64 -0
- package/dist/audit/deferral.d.ts +346 -0
- package/dist/audit/deferral.js +376 -0
- package/dist/audit/diff.d.ts +124 -0
- package/dist/audit/diff.js +318 -0
- package/dist/audit/run.d.ts +34 -0
- package/dist/audit/run.js +45 -2
- package/dist/cli.d.ts +21 -0
- package/dist/cli.js +141 -7
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +18 -0
- package/dist/sweep/dashboard.js +74 -10
- package/dist/sweep/docker.d.ts +31 -0
- package/dist/sweep/docker.js +20 -11
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +18 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +43 -0
- package/dist/sweep/run.js +135 -37
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +57 -2
- package/package.json +3 -1
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
/** Parse and shape-check a stored report. A baseline that cannot be read is never "no change". */
|
|
2
|
+
export function parseBaselineReport(text) {
|
|
3
|
+
let doc;
|
|
4
|
+
try {
|
|
5
|
+
doc = JSON.parse(text);
|
|
6
|
+
}
|
|
7
|
+
catch (e) {
|
|
8
|
+
return { report: null, problem: `baseline is not JSON: ${e.message}` };
|
|
9
|
+
}
|
|
10
|
+
if (!doc || typeof doc !== 'object' || Array.isArray(doc)) {
|
|
11
|
+
return { report: null, problem: 'baseline is not an audit report object' };
|
|
12
|
+
}
|
|
13
|
+
const r = doc;
|
|
14
|
+
if (!Array.isArray(r.configs)) {
|
|
15
|
+
return { report: null, problem: "baseline has no 'configs' array — is it the output of `audit --json`?" };
|
|
16
|
+
}
|
|
17
|
+
if (typeof r.methodologyVersion !== 'string' || typeof r.encoding !== 'string') {
|
|
18
|
+
return { report: null, problem: 'baseline is missing methodologyVersion/encoding — is it the output of `audit --json`?' };
|
|
19
|
+
}
|
|
20
|
+
for (const c of r.configs) {
|
|
21
|
+
if (!c || typeof c !== 'object' || typeof c.source !== 'string') {
|
|
22
|
+
return { report: null, problem: 'baseline has a config entry without a source path' };
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
return { report: doc };
|
|
26
|
+
}
|
|
27
|
+
function statesOf(cfg) {
|
|
28
|
+
const out = new Map();
|
|
29
|
+
for (const s of cfg.servers ?? [])
|
|
30
|
+
out.set(s.name, { present: true, tokens: typeof s.tokens === 'number' ? s.tokens : null });
|
|
31
|
+
// A skipped server IS in the config; it just has no number. Keeping it distinct
|
|
32
|
+
// from absent is the whole reason `removed` and `unmeasured-now` are separate kinds.
|
|
33
|
+
for (const s of cfg.skipped ?? [])
|
|
34
|
+
if (!out.has(s.name))
|
|
35
|
+
out.set(s.name, { present: true, tokens: null });
|
|
36
|
+
return out;
|
|
37
|
+
}
|
|
38
|
+
export function diffConfig(before, after, matchedBy) {
|
|
39
|
+
const afterShare = after.contextShare;
|
|
40
|
+
if (!before) {
|
|
41
|
+
return {
|
|
42
|
+
client: after.client,
|
|
43
|
+
source: after.source,
|
|
44
|
+
matchedBy: 'unmatched',
|
|
45
|
+
beforeTotal: null,
|
|
46
|
+
afterTotal: after.totalTokens,
|
|
47
|
+
delta: null,
|
|
48
|
+
beforeShare: null,
|
|
49
|
+
afterShare,
|
|
50
|
+
exact: false,
|
|
51
|
+
understatedBy: 0,
|
|
52
|
+
overstatedBy: 0,
|
|
53
|
+
servers: [],
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
const b = statesOf(before);
|
|
57
|
+
const a = statesOf(after);
|
|
58
|
+
const servers = [];
|
|
59
|
+
let understatedBy = 0;
|
|
60
|
+
let overstatedBy = 0;
|
|
61
|
+
let exact = true;
|
|
62
|
+
for (const name of new Set([...b.keys(), ...a.keys()])) {
|
|
63
|
+
const bs = b.get(name);
|
|
64
|
+
const as = a.get(name);
|
|
65
|
+
if (bs && !as) {
|
|
66
|
+
servers.push(bs.tokens === null
|
|
67
|
+
? { name, kind: 'removed', before: null, after: null, delta: null, note: 'was in the config but never measured — removing it changed no measured cost' }
|
|
68
|
+
: { name, kind: 'removed', before: bs.tokens, after: null, delta: -bs.tokens });
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
if (!bs && as) {
|
|
72
|
+
servers.push(as.tokens === null
|
|
73
|
+
? { name, kind: 'added', before: null, after: null, delta: null, note: 'added but not measurable — its cost is unknown, not zero' }
|
|
74
|
+
: { name, kind: 'added', before: null, after: as.tokens, delta: as.tokens });
|
|
75
|
+
continue;
|
|
76
|
+
}
|
|
77
|
+
if (!bs || !as)
|
|
78
|
+
continue;
|
|
79
|
+
if (bs.tokens !== null && as.tokens !== null) {
|
|
80
|
+
const delta = as.tokens - bs.tokens;
|
|
81
|
+
servers.push({ name, kind: delta === 0 ? 'unchanged' : 'changed', before: bs.tokens, after: as.tokens, delta });
|
|
82
|
+
}
|
|
83
|
+
else if (bs.tokens !== null && as.tokens === null) {
|
|
84
|
+
exact = false;
|
|
85
|
+
understatedBy += bs.tokens;
|
|
86
|
+
servers.push({
|
|
87
|
+
name,
|
|
88
|
+
kind: 'unmeasured-now',
|
|
89
|
+
before: bs.tokens,
|
|
90
|
+
after: null,
|
|
91
|
+
delta: null,
|
|
92
|
+
note: `measured ${bs.tokens.toLocaleString('en-US')} in the baseline and could not be measured now — its cost is missing from the total, not gone from your config`,
|
|
93
|
+
});
|
|
94
|
+
}
|
|
95
|
+
else if (bs.tokens === null && as.tokens !== null) {
|
|
96
|
+
exact = false;
|
|
97
|
+
overstatedBy += as.tokens;
|
|
98
|
+
servers.push({
|
|
99
|
+
name,
|
|
100
|
+
kind: 'unmeasured-before',
|
|
101
|
+
before: null,
|
|
102
|
+
after: as.tokens,
|
|
103
|
+
delta: null,
|
|
104
|
+
note: `could not be measured in the baseline and measures ${as.tokens.toLocaleString('en-US')} now — this cost is newly visible, not necessarily new`,
|
|
105
|
+
});
|
|
106
|
+
}
|
|
107
|
+
else {
|
|
108
|
+
servers.push({
|
|
109
|
+
name,
|
|
110
|
+
kind: 'unmeasured-both',
|
|
111
|
+
before: null,
|
|
112
|
+
after: null,
|
|
113
|
+
delta: null,
|
|
114
|
+
note: 'not measurable in either run — contributes 0 to both totals and hides an unknown cost',
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
// Biggest movers first; ties and non-deltas fall to the bottom in name order.
|
|
119
|
+
servers.sort((x, y) => Math.abs(y.delta ?? 0) - Math.abs(x.delta ?? 0) || x.name.localeCompare(y.name));
|
|
120
|
+
return {
|
|
121
|
+
client: after.client,
|
|
122
|
+
source: after.source,
|
|
123
|
+
matchedBy,
|
|
124
|
+
beforeTotal: before.totalTokens,
|
|
125
|
+
afterTotal: after.totalTokens,
|
|
126
|
+
delta: after.totalTokens - before.totalTokens,
|
|
127
|
+
beforeShare: before.contextShare ?? null,
|
|
128
|
+
afterShare,
|
|
129
|
+
exact,
|
|
130
|
+
understatedBy,
|
|
131
|
+
overstatedBy,
|
|
132
|
+
servers,
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Pair current configs with baseline configs.
|
|
137
|
+
*
|
|
138
|
+
* Exact source path first. Then one deliberate fallback: if each side has
|
|
139
|
+
* exactly one config, they are the same config seen from two machines — the CI
|
|
140
|
+
* case, where a baseline recorded at /Users/… meets a checkout at /home/runner/….
|
|
141
|
+
* Anything looser would pair two unrelated clients and call the difference a
|
|
142
|
+
* change, so everything else stays unmatched and says so.
|
|
143
|
+
*/
|
|
144
|
+
export function pairConfigs(before, after) {
|
|
145
|
+
const unusedBefore = new Map(before.map((c) => [c.source, c]));
|
|
146
|
+
const pairs = [];
|
|
147
|
+
for (const cur of after) {
|
|
148
|
+
const hit = unusedBefore.get(cur.source);
|
|
149
|
+
if (hit) {
|
|
150
|
+
unusedBefore.delete(cur.source);
|
|
151
|
+
pairs.push({ before: hit, after: cur, matchedBy: 'source' });
|
|
152
|
+
}
|
|
153
|
+
else {
|
|
154
|
+
pairs.push({ before: null, after: cur, matchedBy: 'unmatched' });
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
if (before.length === 1 && after.length === 1 && pairs[0].before === null) {
|
|
158
|
+
pairs[0] = { before: before[0], after: after[0], matchedBy: 'sole-config' };
|
|
159
|
+
unusedBefore.delete(before[0].source);
|
|
160
|
+
}
|
|
161
|
+
return { pairs, dropped: [...unusedBefore.values()] };
|
|
162
|
+
}
|
|
163
|
+
export function buildDiff(baseline, current) {
|
|
164
|
+
const warnings = [];
|
|
165
|
+
let comparable = true;
|
|
166
|
+
if (baseline.methodologyVersion !== current.methodologyVersion) {
|
|
167
|
+
comparable = false;
|
|
168
|
+
warnings.push(`methodology changed (${baseline.methodologyVersion} → ${current.methodologyVersion}) — token counts from the two runs are not the same measurement`);
|
|
169
|
+
}
|
|
170
|
+
if (baseline.encoding !== current.encoding) {
|
|
171
|
+
comparable = false;
|
|
172
|
+
warnings.push(`encoding changed (${baseline.encoding} → ${current.encoding}) — the counts are in different units`);
|
|
173
|
+
}
|
|
174
|
+
if (baseline.contextWindow !== current.contextWindow) {
|
|
175
|
+
// Shares move, token counts do not. Worth saying, not worth invalidating.
|
|
176
|
+
warnings.push(`context window changed (${baseline.contextWindow.toLocaleString('en-US')} → ${current.contextWindow.toLocaleString('en-US')}) — shares are not comparable, token counts still are`);
|
|
177
|
+
}
|
|
178
|
+
const { pairs, dropped } = pairConfigs(baseline.configs, current.configs);
|
|
179
|
+
const configs = pairs.map((p) => diffConfig(p.before, p.after, p.matchedBy));
|
|
180
|
+
for (const c of configs) {
|
|
181
|
+
if (c.matchedBy === 'unmatched') {
|
|
182
|
+
warnings.push(`${c.source}: no matching config in the baseline — its ${c.afterTotal.toLocaleString('en-US')} tokens are shown as a total, not a change`);
|
|
183
|
+
}
|
|
184
|
+
if (c.matchedBy === 'sole-config' && c.source !== baseline.configs[0]?.source) {
|
|
185
|
+
warnings.push(`paired ${c.source} with the baseline's ${baseline.configs[0]?.source} — one config on each side, different paths`);
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
for (const d of dropped) {
|
|
189
|
+
warnings.push(`${d.source}: in the baseline (${d.totalTokens.toLocaleString('en-US')} tokens) and not found now — a config that disappeared is not a config that got cheaper`);
|
|
190
|
+
}
|
|
191
|
+
const increases = configs.filter((c) => typeof c.delta === 'number' && c.delta > 0);
|
|
192
|
+
increases.sort((a, b) => b.delta - a.delta);
|
|
193
|
+
return {
|
|
194
|
+
baselineGeneratedAt: baseline.generatedAt,
|
|
195
|
+
baselineMethodologyVersion: baseline.methodologyVersion,
|
|
196
|
+
comparable,
|
|
197
|
+
droppedConfigs: dropped.map((d) => ({ client: d.client, source: d.source, totalTokens: d.totalTokens })),
|
|
198
|
+
warnings,
|
|
199
|
+
configs,
|
|
200
|
+
worstIncrease: increases.length ? { source: increases[0].source, delta: increases[0].delta } : null,
|
|
201
|
+
};
|
|
202
|
+
}
|
|
203
|
+
const n = (x) => x.toLocaleString('en-US');
|
|
204
|
+
const signed = (x) => `${x >= 0 ? '+' : '−'}${n(Math.abs(x))}`;
|
|
205
|
+
const pct = (x) => `${(x * 100).toFixed(1)}%`;
|
|
206
|
+
/** `--config <path>` records the client as 'explicit', which is a parser detail, not a name. */
|
|
207
|
+
const clientLabel = (client) => (client === 'explicit' || !client ? 'this client' : client);
|
|
208
|
+
export function formatDiff(diff, contextWindow) {
|
|
209
|
+
const lines = [];
|
|
210
|
+
lines.push('');
|
|
211
|
+
lines.push(`diff vs baseline measured ${diff.baselineGeneratedAt} (methodology ${diff.baselineMethodologyVersion})`);
|
|
212
|
+
for (const c of diff.configs) {
|
|
213
|
+
lines.push('');
|
|
214
|
+
if (c.matchedBy === 'unmatched' || c.delta === null || c.beforeTotal === null) {
|
|
215
|
+
lines.push(` ${c.source} ${n(c.afterTotal)} tokens — no baseline for this config, so nothing to compare`);
|
|
216
|
+
continue;
|
|
217
|
+
}
|
|
218
|
+
const rows = c.servers.filter((s) => s.kind !== 'unchanged');
|
|
219
|
+
lines.push(` ${c.source}`);
|
|
220
|
+
lines.push(` ${n(c.beforeTotal)} → ${n(c.afterTotal)} ${signed(c.delta)}`);
|
|
221
|
+
if (rows.length) {
|
|
222
|
+
lines.push('');
|
|
223
|
+
const w = Math.max(...rows.map((r) => r.name.length), 6);
|
|
224
|
+
for (const r of rows) {
|
|
225
|
+
const from = r.before === null ? '—' : n(r.before);
|
|
226
|
+
const to = r.after === null ? '—' : n(r.after);
|
|
227
|
+
const d = r.delta === null ? '' : ` ${signed(r.delta)}`;
|
|
228
|
+
lines.push(` ${r.kind.padEnd(17)} ${r.name.padEnd(w)} ${from.padStart(9)} → ${to.padStart(9)}${d}`);
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
const unchanged = c.servers.length - rows.length;
|
|
232
|
+
if (unchanged)
|
|
233
|
+
lines.push(` (${unchanged} server${unchanged === 1 ? '' : 's'} unchanged)`);
|
|
234
|
+
lines.push('');
|
|
235
|
+
if (!c.exact) {
|
|
236
|
+
// The headline sentence is where a skimmer stops, so it must not assert a change
|
|
237
|
+
// this run could not establish. A server that died takes its tokens out of the
|
|
238
|
+
// total exactly like a server you uninstalled — printing "removes 2,378 tokens"
|
|
239
|
+
// and correcting it two lines down is the flattering reading getting read.
|
|
240
|
+
lines.push(` Not a clean comparison: a server changed measured-ness between the two runs.`);
|
|
241
|
+
lines.push(` The measured total moved ${signed(c.delta)}, but that is not what your config did.`);
|
|
242
|
+
lines.push('');
|
|
243
|
+
for (const r of c.servers) {
|
|
244
|
+
if (r.kind === 'unmeasured-now' || r.kind === 'unmeasured-before')
|
|
245
|
+
lines.push(` ${r.name}: ${r.note}`);
|
|
246
|
+
}
|
|
247
|
+
if (c.understatedBy)
|
|
248
|
+
lines.push(` → true cost is at least ${n(c.understatedBy)} higher than the ${n(c.afterTotal)} measured now.`);
|
|
249
|
+
if (c.overstatedBy)
|
|
250
|
+
lines.push(` → up to ${n(c.overstatedBy)} of that movement was already being paid, just unmeasured.`);
|
|
251
|
+
}
|
|
252
|
+
else if (c.delta === 0) {
|
|
253
|
+
lines.push(` No change: this config costs the same ${n(c.afterTotal)} tokens per request as the baseline.`);
|
|
254
|
+
}
|
|
255
|
+
else if (c.delta > 0) {
|
|
256
|
+
lines.push(` This change adds ${n(c.delta)} tokens to every request in ${clientLabel(c.client)} — ` +
|
|
257
|
+
`${pct(c.beforeShare ?? 0)} → ${pct(c.afterShare)} of a ${n(contextWindow)}-token context window.`);
|
|
258
|
+
}
|
|
259
|
+
else {
|
|
260
|
+
lines.push(` This change removes ${n(Math.abs(c.delta))} tokens from every request in ${clientLabel(c.client)} — ` +
|
|
261
|
+
`${pct(c.beforeShare ?? 0)} → ${pct(c.afterShare)} of a ${n(contextWindow)}-token context window.`);
|
|
262
|
+
}
|
|
263
|
+
const blind = c.servers.filter((r) => r.kind === 'unmeasured-both' || ((r.kind === 'added' || r.kind === 'removed') && r.delta === null));
|
|
264
|
+
if (blind.length) {
|
|
265
|
+
lines.push('');
|
|
266
|
+
for (const r of blind)
|
|
267
|
+
lines.push(` ${r.name}: ${r.note}`);
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
if (diff.warnings.length) {
|
|
271
|
+
lines.push('');
|
|
272
|
+
lines.push(' diff warnings');
|
|
273
|
+
for (const w of diff.warnings)
|
|
274
|
+
lines.push(` ${w}`);
|
|
275
|
+
}
|
|
276
|
+
if (!diff.comparable) {
|
|
277
|
+
lines.push('');
|
|
278
|
+
lines.push(' The two runs are not the same measurement, so the numbers above are not a change.');
|
|
279
|
+
lines.push(' Re-record the baseline with this version: mcp-context-cost audit --json > baseline.json');
|
|
280
|
+
}
|
|
281
|
+
return lines.join('\n');
|
|
282
|
+
}
|
|
283
|
+
/**
|
|
284
|
+
* `--max-increase N` — the CI gate. Fails on an increase over the limit, and
|
|
285
|
+
* equally on any reason the increase could not be established.
|
|
286
|
+
*
|
|
287
|
+
* That second half is the point. A gate that passes when a server failed to
|
|
288
|
+
* start, or when the baseline covered a config this run never found, is a green
|
|
289
|
+
* check on a question nobody asked. Everything this portfolio has learned says
|
|
290
|
+
* unchecked must not read as clean, so an inexact diff fails and names why.
|
|
291
|
+
*/
|
|
292
|
+
export function evaluateIncreaseGate(diff, limit) {
|
|
293
|
+
const reasons = [];
|
|
294
|
+
if (!diff.comparable)
|
|
295
|
+
reasons.push('the baseline is not the same measurement as this run — nothing was compared');
|
|
296
|
+
for (const c of diff.configs) {
|
|
297
|
+
if (c.matchedBy === 'unmatched') {
|
|
298
|
+
reasons.push(`${c.source}: no baseline to check its ${n(c.afterTotal)} tokens against`);
|
|
299
|
+
}
|
|
300
|
+
else if (!c.exact) {
|
|
301
|
+
reasons.push(`${c.source}: a server changed measured-ness, so the change could not be established exactly`);
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
for (const d of diff.droppedConfigs) {
|
|
305
|
+
reasons.push(`${d.source}: covered by the baseline and not found in this run`);
|
|
306
|
+
}
|
|
307
|
+
const increase = diff.worstIncrease?.delta ?? (diff.configs.some((c) => typeof c.delta === 'number') ? 0 : null);
|
|
308
|
+
if (reasons.length === 0 && increase !== null && increase > limit) {
|
|
309
|
+
reasons.push(`${diff.worstIncrease.source}: +${n(increase)} tokens per request, over the ${n(limit)} allowed`);
|
|
310
|
+
}
|
|
311
|
+
return { limit, pass: reasons.length === 0, increase, reasons };
|
|
312
|
+
}
|
|
313
|
+
export function formatGate(gate) {
|
|
314
|
+
if (gate.pass) {
|
|
315
|
+
return `increase ok: ${gate.increase === null ? 'no change to measure' : `${signed(gate.increase)} tokens`} ≤ ${n(gate.limit)} allowed`;
|
|
316
|
+
}
|
|
317
|
+
return ['INCREASE FAIL:', ...gate.reasons.map((r) => ` ${r}`)].join('\n');
|
|
318
|
+
}
|
package/dist/audit/run.d.ts
CHANGED
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
import type { Measurement } from '../core/types.js';
|
|
2
|
+
import { type DivergenceRun } from '../core/divergence.js';
|
|
2
3
|
import { type AuditReport } from './audit.js';
|
|
4
|
+
import { type ToolSearchEnv, type ToolSearchSource } from './deferral.js';
|
|
3
5
|
import { type LoadedConfig } from './config.js';
|
|
6
|
+
/** Where the published `tools-delta/v1` run lives when `--claude` doesn't override it. */
|
|
7
|
+
export declare const DEFAULT_DIVERGENCE_URL = "https://raw.githubusercontent.com/athakur3/mcp-context-cost/main/results/divergence.json";
|
|
4
8
|
export interface AuditOptions {
|
|
5
9
|
/** Explicit config path(s); when empty, every known client location is tried. */
|
|
6
10
|
configPaths?: string[];
|
|
@@ -11,9 +15,39 @@ export interface AuditOptions {
|
|
|
11
15
|
docker?: boolean;
|
|
12
16
|
contextWindow?: number;
|
|
13
17
|
budget?: number;
|
|
18
|
+
/** Join each measured server against the published Claude divergence run. */
|
|
19
|
+
claude?: boolean;
|
|
20
|
+
/** Override the divergence.json source — mainly for tests and self-hosted mirrors. */
|
|
21
|
+
divergenceUrl?: string;
|
|
22
|
+
/**
|
|
23
|
+
* The tool-search variables as this process's SHELL has them. Defaults to
|
|
24
|
+
* this process's environment. Overridable so a test can state a machine
|
|
25
|
+
* rather than inherit the one it runs on.
|
|
26
|
+
*/
|
|
27
|
+
env?: ToolSearchEnv;
|
|
28
|
+
/**
|
|
29
|
+
* The same variables as Claude Code's own settings files set them — the other
|
|
30
|
+
* half of the answer, and the half a shell cannot show. Defaults to reading
|
|
31
|
+
* those files off the machine being audited (`discoverSettings`).
|
|
32
|
+
*/
|
|
33
|
+
settings?: ToolSearchSource[];
|
|
14
34
|
onProgress?: (name: string, done: number, total: number) => void;
|
|
15
35
|
}
|
|
36
|
+
/** Fetch and parse the published divergence run. Never throws: a failure is a report problem, not a crash. */
|
|
37
|
+
export declare function fetchDivergence(url: string): Promise<{
|
|
38
|
+
run: DivergenceRun | null;
|
|
39
|
+
problem?: string;
|
|
40
|
+
}>;
|
|
16
41
|
export declare function discover(opts?: AuditOptions): LoadedConfig[];
|
|
42
|
+
/**
|
|
43
|
+
* Read every settings file Claude Code would take its deferral setting from.
|
|
44
|
+
*
|
|
45
|
+
* A sibling of `discover` above and deliberately separate from it: that one
|
|
46
|
+
* finds which servers exist, this one finds how the client treats them. The
|
|
47
|
+
* setting does not live in a client config, and reading only the shell that
|
|
48
|
+
* launched this audit answers for the wrong machine.
|
|
49
|
+
*/
|
|
50
|
+
export declare function discoverSettings(opts?: AuditOptions): ToolSearchSource[];
|
|
17
51
|
/** Measure every distinct stdio server across the given configs, once each. */
|
|
18
52
|
export declare function measureAll(configs: LoadedConfig[], opts?: AuditOptions): Promise<Map<string, Measurement>>;
|
|
19
53
|
export declare function runAudit(opts?: AuditOptions): Promise<AuditReport>;
|
package/dist/audit/run.js
CHANGED
|
@@ -6,8 +6,25 @@
|
|
|
6
6
|
*/
|
|
7
7
|
import { homedir } from 'node:os';
|
|
8
8
|
import { measureServer } from '../sweep/run.js';
|
|
9
|
+
import { parseDivergence } from '../core/divergence.js';
|
|
9
10
|
import { buildReport, serverKey } from './audit.js';
|
|
10
|
-
import {
|
|
11
|
+
import { toolSearchEnv } from './deferral.js';
|
|
12
|
+
import { configCandidates, loadConfigs, loadSettingsSources, settingsCandidates, } from './config.js';
|
|
13
|
+
/** Where the published `tools-delta/v1` run lives when `--claude` doesn't override it. */
|
|
14
|
+
export const DEFAULT_DIVERGENCE_URL = 'https://raw.githubusercontent.com/athakur3/mcp-context-cost/main/results/divergence.json';
|
|
15
|
+
/** Fetch and parse the published divergence run. Never throws: a failure is a report problem, not a crash. */
|
|
16
|
+
export async function fetchDivergence(url) {
|
|
17
|
+
try {
|
|
18
|
+
const res = await fetch(url, { signal: AbortSignal.timeout(15_000) });
|
|
19
|
+
if (!res.ok)
|
|
20
|
+
return { run: null, problem: `claude divergence: HTTP ${res.status} fetching ${url}` };
|
|
21
|
+
const run = parseDivergence(await res.text());
|
|
22
|
+
return run ? { run } : { run: null, problem: `claude divergence: malformed data at ${url}` };
|
|
23
|
+
}
|
|
24
|
+
catch (e) {
|
|
25
|
+
return { run: null, problem: `claude divergence: failed to fetch ${url}: ${e.message}` };
|
|
26
|
+
}
|
|
27
|
+
}
|
|
11
28
|
export function discover(opts = {}) {
|
|
12
29
|
const cwd = opts.cwd ?? process.cwd();
|
|
13
30
|
const home = opts.home ?? homedir();
|
|
@@ -16,6 +33,19 @@ export function discover(opts = {}) {
|
|
|
16
33
|
: configCandidates({ home, cwd, platform: process.platform, appData: process.env.APPDATA });
|
|
17
34
|
return loadConfigs(candidates, cwd);
|
|
18
35
|
}
|
|
36
|
+
/**
|
|
37
|
+
* Read every settings file Claude Code would take its deferral setting from.
|
|
38
|
+
*
|
|
39
|
+
* A sibling of `discover` above and deliberately separate from it: that one
|
|
40
|
+
* finds which servers exist, this one finds how the client treats them. The
|
|
41
|
+
* setting does not live in a client config, and reading only the shell that
|
|
42
|
+
* launched this audit answers for the wrong machine.
|
|
43
|
+
*/
|
|
44
|
+
export function discoverSettings(opts = {}) {
|
|
45
|
+
const cwd = opts.cwd ?? process.cwd();
|
|
46
|
+
const home = opts.home ?? homedir();
|
|
47
|
+
return loadSettingsSources(settingsCandidates({ home, cwd, platform: process.platform, programData: process.env.ProgramData }));
|
|
48
|
+
}
|
|
19
49
|
/** Measure every distinct stdio server across the given configs, once each. */
|
|
20
50
|
export async function measureAll(configs, opts = {}) {
|
|
21
51
|
const unique = new Map();
|
|
@@ -52,8 +82,21 @@ export async function measureAll(configs, opts = {}) {
|
|
|
52
82
|
export async function runAudit(opts = {}) {
|
|
53
83
|
const configs = discover(opts);
|
|
54
84
|
const measured = await measureAll(configs, opts);
|
|
55
|
-
|
|
85
|
+
let divergence = null;
|
|
86
|
+
let divergenceProblem;
|
|
87
|
+
if (opts.claude) {
|
|
88
|
+
const fetched = await fetchDivergence(opts.divergenceUrl ?? DEFAULT_DIVERGENCE_URL);
|
|
89
|
+
divergence = fetched.run;
|
|
90
|
+
divergenceProblem = fetched.problem;
|
|
91
|
+
}
|
|
92
|
+
const report = buildReport(configs, measured, {
|
|
56
93
|
contextWindow: opts.contextWindow,
|
|
57
94
|
budget: opts.budget,
|
|
95
|
+
divergence,
|
|
96
|
+
env: opts.env ?? toolSearchEnv(process.env),
|
|
97
|
+
settings: opts.settings ?? discoverSettings(opts),
|
|
58
98
|
});
|
|
99
|
+
if (divergenceProblem)
|
|
100
|
+
report.problems.push(divergenceProblem);
|
|
101
|
+
return report;
|
|
59
102
|
}
|
package/dist/cli.d.ts
CHANGED
|
@@ -6,3 +6,24 @@ export declare function verifyMeasurement(m: Measurement): {
|
|
|
6
6
|
rederivedSha: string | null;
|
|
7
7
|
problems: string[];
|
|
8
8
|
};
|
|
9
|
+
/** Derives a servers.yaml-style slug from a remote URL's hostname, e.g. mcp.deepwiki.com -> deepwiki. */
|
|
10
|
+
export declare function slugFromUrl(url: string): string;
|
|
11
|
+
/** Installed version, for error messages that need to say which one you are running. */
|
|
12
|
+
export declare function cliVersion(): string;
|
|
13
|
+
/**
|
|
14
|
+
* Reject flags this build does not know.
|
|
15
|
+
*
|
|
16
|
+
* An older CLI used to ignore an unrecognised flag and carry on. That is the exact failure
|
|
17
|
+
* this project exists to catch, in our own tool: `audit --baseline base.json
|
|
18
|
+
* --max-increase 2000` on a build without those flags ran a plain audit and **exited 0** —
|
|
19
|
+
* a green CI check on a gate that never ran. The README documents flags before they are
|
|
20
|
+
* published, so the version skew is not hypothetical; it is the normal case for anyone
|
|
21
|
+
* running `npx -y mcp-context-cost`.
|
|
22
|
+
*
|
|
23
|
+
* So an unknown flag is a usage error, and the message names the running version, because
|
|
24
|
+
* the likeliest cause is that the reader's command is newer than their install.
|
|
25
|
+
*/
|
|
26
|
+
export declare function unknownFlags(argv: string[], spec: {
|
|
27
|
+
value: string[];
|
|
28
|
+
boolean: string[];
|
|
29
|
+
}): string[];
|