mcp-context-cost 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +91 -25
- package/dist/audit/audit.d.ts +38 -0
- package/dist/audit/audit.js +372 -7
- package/dist/audit/config.d.ts +38 -0
- package/dist/audit/config.js +64 -0
- package/dist/audit/deferral.d.ts +346 -0
- package/dist/audit/deferral.js +376 -0
- package/dist/audit/run.d.ts +22 -0
- package/dist/audit/run.js +17 -1
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +10 -0
- package/dist/sweep/dashboard.js +32 -16
- package/dist/sweep/docker.d.ts +23 -0
- package/dist/sweep/docker.js +16 -12
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +16 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +41 -0
- package/dist/sweep/run.js +129 -36
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +56 -2
- package/package.json +3 -1
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
/** Share of the context window at which deferral activates under `auto`. */
|
|
2
|
+
export const TOOL_SEARCH_AUTO_SHARE = 0.1;
|
|
3
|
+
/** The variables that decide whether this machine's Claude Code defers. */
|
|
4
|
+
export const TOOL_SEARCH_VARS = [
|
|
5
|
+
'ENABLE_TOOL_SEARCH',
|
|
6
|
+
'CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS',
|
|
7
|
+
'ANTHROPIC_BASE_URL',
|
|
8
|
+
];
|
|
9
|
+
/** Pick the three variables that matter out of a process environment. */
|
|
10
|
+
export function toolSearchEnv(env) {
|
|
11
|
+
return {
|
|
12
|
+
ENABLE_TOOL_SEARCH: env.ENABLE_TOOL_SEARCH,
|
|
13
|
+
CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS: env.CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS,
|
|
14
|
+
ANTHROPIC_BASE_URL: env.ANTHROPIC_BASE_URL,
|
|
15
|
+
};
|
|
16
|
+
}
|
|
17
|
+
/** What `source` says for the process environment, which has no path. */
|
|
18
|
+
export const SHELL_SOURCE = '(shell environment)';
|
|
19
|
+
/** A source as it is published: what it sets, by name. */
|
|
20
|
+
export function toolSearchSourceRecord(s) {
|
|
21
|
+
return {
|
|
22
|
+
scope: s.scope,
|
|
23
|
+
source: s.source,
|
|
24
|
+
state: s.state,
|
|
25
|
+
sets: TOOL_SEARCH_VARS.filter((n) => (s.vars[n] ?? '').trim() !== ''),
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
/** The one host Claude Code treats as first-party for the tool-search fallback. */
|
|
29
|
+
const FIRST_PARTY_API_HOST = 'api.anthropic.com';
|
|
30
|
+
/** What is printed for a base URL we could not parse — a marker, not the value. */
|
|
31
|
+
const UNREADABLE_BASE_URL = '(unreadable URL)';
|
|
32
|
+
/**
|
|
33
|
+
* The hostname of a base URL, or null if it does not parse.
|
|
34
|
+
*
|
|
35
|
+
* The hostname is the whole of what the mode decision needs, and it is also the
|
|
36
|
+
* whole of what may leave this function: the rest of the value can carry a
|
|
37
|
+
* credential. A value that does not parse is not first-party, which is the
|
|
38
|
+
* reading that says tokens are paid — never the one that says they are free.
|
|
39
|
+
*/
|
|
40
|
+
function baseUrlHost(raw) {
|
|
41
|
+
try {
|
|
42
|
+
return new URL(raw).hostname.toLowerCase();
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
return null;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Read the tool-search setting out of ONE environment. Values are matched
|
|
50
|
+
* exactly as documented: an unrecognized value produces `setting-unrecognized`
|
|
51
|
+
* rather than a guess, because guessing here would print a definite verdict
|
|
52
|
+
* about tokens the reader may or may not be paying.
|
|
53
|
+
*
|
|
54
|
+
* `source` is left null here: this function is given one environment and has no
|
|
55
|
+
* way to say which of the machine's places it came from. `resolveToolSearchSources`,
|
|
56
|
+
* which does, fills it in.
|
|
57
|
+
*/
|
|
58
|
+
export function resolveToolSearch(env) {
|
|
59
|
+
const betas = env.CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS?.trim();
|
|
60
|
+
// Read first: documented as not overridable by ENABLE_TOOL_SEARCH.
|
|
61
|
+
if (betas) {
|
|
62
|
+
return {
|
|
63
|
+
mode: 'loads-upfront',
|
|
64
|
+
thresholdShare: null,
|
|
65
|
+
variable: 'CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS',
|
|
66
|
+
value: betas,
|
|
67
|
+
source: null,
|
|
68
|
+
readFromMachine: true,
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
const raw = env.ENABLE_TOOL_SEARCH?.trim();
|
|
72
|
+
const set = (mode, thresholdShare) => ({
|
|
73
|
+
mode,
|
|
74
|
+
thresholdShare,
|
|
75
|
+
variable: 'ENABLE_TOOL_SEARCH',
|
|
76
|
+
value: raw ?? null,
|
|
77
|
+
source: null,
|
|
78
|
+
readFromMachine: true,
|
|
79
|
+
});
|
|
80
|
+
if (raw === undefined || raw === '') {
|
|
81
|
+
const base = env.ANTHROPIC_BASE_URL?.trim();
|
|
82
|
+
if (base) {
|
|
83
|
+
const host = baseUrlHost(base);
|
|
84
|
+
if (host !== FIRST_PARTY_API_HOST) {
|
|
85
|
+
return {
|
|
86
|
+
mode: 'loads-upfront',
|
|
87
|
+
thresholdShare: null,
|
|
88
|
+
variable: 'ANTHROPIC_BASE_URL',
|
|
89
|
+
// The hostname alone. `base` itself is never carried out of here.
|
|
90
|
+
value: host ?? UNREADABLE_BASE_URL,
|
|
91
|
+
source: null,
|
|
92
|
+
readFromMachine: true,
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
return {
|
|
97
|
+
mode: 'defers-all',
|
|
98
|
+
thresholdShare: null,
|
|
99
|
+
variable: 'ENABLE_TOOL_SEARCH',
|
|
100
|
+
value: null,
|
|
101
|
+
source: null,
|
|
102
|
+
readFromMachine: false,
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
if (raw === 'true')
|
|
106
|
+
return set('defers-all', null);
|
|
107
|
+
if (raw === 'false')
|
|
108
|
+
return set('loads-upfront', null);
|
|
109
|
+
if (raw === 'auto')
|
|
110
|
+
return set('threshold', TOOL_SEARCH_AUTO_SHARE);
|
|
111
|
+
const custom = /^auto:(\d{1,3})$/.exec(raw);
|
|
112
|
+
if (custom) {
|
|
113
|
+
const pct = Number(custom[1]);
|
|
114
|
+
if (pct >= 0 && pct <= 100)
|
|
115
|
+
return set('threshold', pct / 100);
|
|
116
|
+
}
|
|
117
|
+
return set('setting-unrecognized', null);
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Read the posture from every place the audited machine can set it.
|
|
121
|
+
*
|
|
122
|
+
* Claude Code takes these variables from the shell it was started in AND from
|
|
123
|
+
* the `env` block of its own settings files, so an audit that reads only the
|
|
124
|
+
* shell answers the machine's question with someone else's environment. The
|
|
125
|
+
* case that made this necessary: `~/.claude/settings.json` sets
|
|
126
|
+
* `ENABLE_TOOL_SEARCH: "false"`, the shell running the audit sets nothing, and
|
|
127
|
+
* every request on that machine pays for every tool definition while the report
|
|
128
|
+
* calls it the documented default and says the tokens are not loaded at all.
|
|
129
|
+
*
|
|
130
|
+
* Among the settings files the order is Claude Code's documented precedence —
|
|
131
|
+
* enterprise managed policy, then project-local, then project, then user
|
|
132
|
+
* (Claude Code settings documentation, §"Settings files", read 2026-08-20) — so
|
|
133
|
+
* the first of them that sets a variable is the one that would win.
|
|
134
|
+
*
|
|
135
|
+
* Between the settings files and the shell there is NO order on record here, so
|
|
136
|
+
* a disagreement is refused rather than resolved: `setting-unresolved` names the
|
|
137
|
+
* variable and every place, and no verdict is given. A place that exists and
|
|
138
|
+
* could not be read is the same refusal for the same reason — what it sets is
|
|
139
|
+
* unknown, and an unknown that could flip the answer is not a default.
|
|
140
|
+
*
|
|
141
|
+
* Not visible from here at all, and so not claimed: a variable set on Claude
|
|
142
|
+
* Code's own command line.
|
|
143
|
+
*/
|
|
144
|
+
export function resolveToolSearchSources(sources) {
|
|
145
|
+
const unresolved = (reason, variable) => ({
|
|
146
|
+
mode: 'setting-unresolved',
|
|
147
|
+
thresholdShare: null,
|
|
148
|
+
variable,
|
|
149
|
+
value: null,
|
|
150
|
+
source: null,
|
|
151
|
+
readFromMachine: false,
|
|
152
|
+
unresolved: reason,
|
|
153
|
+
});
|
|
154
|
+
if (sources.some((s) => s.state === 'unreadable'))
|
|
155
|
+
return unresolved('source-unreadable', null);
|
|
156
|
+
const settings = sources.filter((s) => s.scope !== 'shell');
|
|
157
|
+
const shell = sources.find((s) => s.scope === 'shell');
|
|
158
|
+
/** The value that would win for one variable, or the fact that two places disagree. */
|
|
159
|
+
const read = (name) => {
|
|
160
|
+
const winner = settings
|
|
161
|
+
.map((s) => ({ value: (s.vars[name] ?? '').trim(), source: s.source }))
|
|
162
|
+
.find((v) => v.value !== '');
|
|
163
|
+
const shellValue = (shell?.vars[name] ?? '').trim();
|
|
164
|
+
if (winner && shellValue && winner.value !== shellValue)
|
|
165
|
+
return 'conflict';
|
|
166
|
+
if (winner)
|
|
167
|
+
return winner;
|
|
168
|
+
if (shellValue && shell)
|
|
169
|
+
return { value: shellValue, source: shell.source };
|
|
170
|
+
return null;
|
|
171
|
+
};
|
|
172
|
+
// Consulted in the same order `resolveToolSearch` consults them, so a
|
|
173
|
+
// disagreement over a variable that would not have decided anything —
|
|
174
|
+
// ANTHROPIC_BASE_URL behind an explicit ENABLE_TOOL_SEARCH — does not refuse
|
|
175
|
+
// an answer the machine actually gives.
|
|
176
|
+
const betas = read('CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS');
|
|
177
|
+
if (betas === 'conflict')
|
|
178
|
+
return unresolved('sources-disagree', 'CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS');
|
|
179
|
+
if (betas) {
|
|
180
|
+
return {
|
|
181
|
+
...resolveToolSearch({ CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS: betas.value }),
|
|
182
|
+
source: betas.source,
|
|
183
|
+
};
|
|
184
|
+
}
|
|
185
|
+
const enable = read('ENABLE_TOOL_SEARCH');
|
|
186
|
+
if (enable === 'conflict')
|
|
187
|
+
return unresolved('sources-disagree', 'ENABLE_TOOL_SEARCH');
|
|
188
|
+
if (enable) {
|
|
189
|
+
return { ...resolveToolSearch({ ENABLE_TOOL_SEARCH: enable.value }), source: enable.source };
|
|
190
|
+
}
|
|
191
|
+
const base = read('ANTHROPIC_BASE_URL');
|
|
192
|
+
if (base === 'conflict')
|
|
193
|
+
return unresolved('sources-disagree', 'ANTHROPIC_BASE_URL');
|
|
194
|
+
const resolved = resolveToolSearch(base ? { ANTHROPIC_BASE_URL: base.value } : {});
|
|
195
|
+
return { ...resolved, source: resolved.readFromMachine ? (base?.source ?? null) : null };
|
|
196
|
+
}
|
|
197
|
+
/**
|
|
198
|
+
* The band as published in this repository's own `results/divergence.json`
|
|
199
|
+
* (claude-opus-5, 2026-08-19, 20 servers). Used when no divergence run was
|
|
200
|
+
* supplied; `--claude` recomputes it from the run it fetched.
|
|
201
|
+
*/
|
|
202
|
+
export const PUBLISHED_WIRE_TO_CLIENT_RATIO = {
|
|
203
|
+
low: 0.2,
|
|
204
|
+
high: 1.92,
|
|
205
|
+
servers: 20,
|
|
206
|
+
source: 'the published claude-opus-5 divergence run',
|
|
207
|
+
};
|
|
208
|
+
/** Derive the band from a supplied divergence run, falling back to the published one. */
|
|
209
|
+
export function wireToClientRatio(run) {
|
|
210
|
+
if (!run)
|
|
211
|
+
return PUBLISHED_WIRE_TO_CLIENT_RATIO;
|
|
212
|
+
let low = Infinity;
|
|
213
|
+
let high = -Infinity;
|
|
214
|
+
let servers = 0;
|
|
215
|
+
for (const row of Object.values(run.servers)) {
|
|
216
|
+
if (!row || row.error || typeof row.claudeDelta !== 'number' || !(row.o200kFull > 0))
|
|
217
|
+
continue;
|
|
218
|
+
const ratio = row.claudeDelta / row.o200kFull;
|
|
219
|
+
low = Math.min(low, ratio);
|
|
220
|
+
high = Math.max(high, ratio);
|
|
221
|
+
servers++;
|
|
222
|
+
}
|
|
223
|
+
if (servers === 0)
|
|
224
|
+
return PUBLISHED_WIRE_TO_CLIENT_RATIO;
|
|
225
|
+
return { low, high, servers, source: `the ${run.measuredAt} ${run.model} divergence run` };
|
|
226
|
+
}
|
|
227
|
+
/** Clients this tool discovers that have no default deferral on record. */
|
|
228
|
+
const NO_DEFERRAL_ON_RECORD = new Set(['claude-desktop', 'cursor', 'vscode', 'windsurf']);
|
|
229
|
+
/**
|
|
230
|
+
* Where deferral does not apply even when the machine's setting says it should.
|
|
231
|
+
* None of these can be read from the config or the environment, so they are
|
|
232
|
+
* printed as conditions for the reader to check rather than folded into the
|
|
233
|
+
* verdict.
|
|
234
|
+
*/
|
|
235
|
+
const EXCEPTIONS = [
|
|
236
|
+
'a Microsoft Foundry deployment hosted on Azure, which rejects tool search server-side',
|
|
237
|
+
"Google Cloud's Agent Platform on a model earlier than the Claude 4.5 generation",
|
|
238
|
+
'a model without support for tool_reference blocks (before Sonnet 4.5 / Haiku 4.5 / Opus 4.5)',
|
|
239
|
+
'a server pinned with "alwaysLoad": true, whose tools load at session start regardless',
|
|
240
|
+
];
|
|
241
|
+
function estimate(servers, ratio) {
|
|
242
|
+
let low = 0;
|
|
243
|
+
let high = 0;
|
|
244
|
+
let exact = 0;
|
|
245
|
+
let estimated = 0;
|
|
246
|
+
for (const s of servers) {
|
|
247
|
+
if (typeof s.claudeTokens === 'number') {
|
|
248
|
+
// A published Anthropic count for this exact capture. It carries the tool
|
|
249
|
+
// framework overhead the API charges once per request rather than once
|
|
250
|
+
// per server, so a multi-server sum leans high by at most that overhead —
|
|
251
|
+
// far inside the band the converted servers already contribute.
|
|
252
|
+
low += s.claudeTokens;
|
|
253
|
+
high += s.claudeTokens;
|
|
254
|
+
exact++;
|
|
255
|
+
}
|
|
256
|
+
else {
|
|
257
|
+
low += s.tokens * ratio.low;
|
|
258
|
+
high += s.tokens * ratio.high;
|
|
259
|
+
estimated++;
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
return { low: Math.round(low), high: Math.round(high), exact, estimated };
|
|
263
|
+
}
|
|
264
|
+
/**
|
|
265
|
+
* Read one session's deferral position. Pure arithmetic over a built scope — no
|
|
266
|
+
* config file is re-read and no server is launched. The environment is passed
|
|
267
|
+
* in rather than read here, so the answer is reproducible from its inputs.
|
|
268
|
+
*/
|
|
269
|
+
export function evaluateDeferral(scope, opts) {
|
|
270
|
+
const wireTokens = scope.servers.reduce((a, s) => a + s.tokens, 0);
|
|
271
|
+
const isFloor = scope.skippedCount > 0;
|
|
272
|
+
const sharedMeasurements = scope.sharedMeasurements;
|
|
273
|
+
// Every field a verdict carries, at its "nothing to say" value. Each mode
|
|
274
|
+
// below overrides only what it can actually answer.
|
|
275
|
+
const base = {
|
|
276
|
+
client: scope.client,
|
|
277
|
+
sources: scope.sources,
|
|
278
|
+
wireTokens,
|
|
279
|
+
isFloor,
|
|
280
|
+
// Carried by every mode, not just the one that returns early on it below.
|
|
281
|
+
// That early return withholds `clientTokens` and `crosses`, which only
|
|
282
|
+
// threshold mode ever computes; the other modes derive nothing from the
|
|
283
|
+
// total and so have nothing to withhold. What they do all do is PRINT it,
|
|
284
|
+
// so the caveat belongs in every branch of the report — see
|
|
285
|
+
// `sharedMeasurementLines` in audit.ts, which every mode calls.
|
|
286
|
+
sharedMeasurements,
|
|
287
|
+
mechanism: null,
|
|
288
|
+
setting: null,
|
|
289
|
+
thresholdShare: null,
|
|
290
|
+
thresholdTokens: null,
|
|
291
|
+
clientTokens: null,
|
|
292
|
+
ratio: null,
|
|
293
|
+
distanceTokens: null,
|
|
294
|
+
crosses: null,
|
|
295
|
+
exceptions: [],
|
|
296
|
+
};
|
|
297
|
+
if (scope.client !== 'claude-code') {
|
|
298
|
+
return {
|
|
299
|
+
...base,
|
|
300
|
+
mode: NO_DEFERRAL_ON_RECORD.has(scope.client) ? 'no-deferral-on-record' : 'client-unknown',
|
|
301
|
+
};
|
|
302
|
+
}
|
|
303
|
+
// The shell is one place among several, not the machine. Everything Claude
|
|
304
|
+
// Code would read is resolved together, and a disagreement between them is
|
|
305
|
+
// refused rather than decided by whichever this happened to open.
|
|
306
|
+
const sources = [
|
|
307
|
+
{ scope: 'shell', source: SHELL_SOURCE, state: 'read', vars: opts.env ?? {} },
|
|
308
|
+
...(opts.settings ?? []),
|
|
309
|
+
];
|
|
310
|
+
const resolved = resolveToolSearchSources(sources);
|
|
311
|
+
const setting = {
|
|
312
|
+
variable: resolved.variable,
|
|
313
|
+
value: resolved.value,
|
|
314
|
+
source: resolved.source,
|
|
315
|
+
readFromMachine: resolved.readFromMachine,
|
|
316
|
+
sources: sources.map(toolSearchSourceRecord),
|
|
317
|
+
...(resolved.unresolved ? { unresolved: resolved.unresolved } : {}),
|
|
318
|
+
};
|
|
319
|
+
if (resolved.mode !== 'threshold') {
|
|
320
|
+
return {
|
|
321
|
+
...base,
|
|
322
|
+
mode: resolved.mode,
|
|
323
|
+
mechanism: 'tool search',
|
|
324
|
+
setting,
|
|
325
|
+
// Nothing is deferred in the other two modes, so the conditions under
|
|
326
|
+
// which deferral fails to apply are not worth printing there.
|
|
327
|
+
exceptions: resolved.mode === 'defers-all' ? EXCEPTIONS : [],
|
|
328
|
+
};
|
|
329
|
+
}
|
|
330
|
+
const thresholdShare = resolved.thresholdShare ?? TOOL_SEARCH_AUTO_SHARE;
|
|
331
|
+
const thresholdTokens = Math.round(opts.contextWindow * thresholdShare);
|
|
332
|
+
if (sharedMeasurements > 0) {
|
|
333
|
+
// There is a threshold, and no total to hold against it. A shared
|
|
334
|
+
// measurement is not a floor: a twin can serve more tools than the one that
|
|
335
|
+
// was launched or fewer, so the sum can be wrong in either direction and
|
|
336
|
+
// neither side can be ruled out. The same machine has already been seen to
|
|
337
|
+
// report 13,834 wire tokens or 392 for one stack depending on which twin
|
|
338
|
+
// the cache happened to hold, and to print a confident — opposite — side
|
|
339
|
+
// each time. This is the rule `evaluateIncreaseGate` states and `crosses`
|
|
340
|
+
// already follows: an answer that could not be established fails rather
|
|
341
|
+
// than resolves. The threshold itself is still reported, because where the
|
|
342
|
+
// line sits is known even when this stack's distance from it is not.
|
|
343
|
+
return {
|
|
344
|
+
...base,
|
|
345
|
+
mode: 'threshold',
|
|
346
|
+
mechanism: 'tool search',
|
|
347
|
+
setting,
|
|
348
|
+
thresholdShare,
|
|
349
|
+
thresholdTokens,
|
|
350
|
+
exceptions: EXCEPTIONS,
|
|
351
|
+
};
|
|
352
|
+
}
|
|
353
|
+
const ratio = wireToClientRatio(opts.divergence);
|
|
354
|
+
const clientTokens = estimate(scope.servers, ratio);
|
|
355
|
+
// At-or-above, on the documented "defers all of them once the definitions
|
|
356
|
+
// reach 10%". A range that is entirely over is over even if it is a floor:
|
|
357
|
+
// more unmeasured tokens cannot take it back under.
|
|
358
|
+
const crosses = clientTokens.low >= thresholdTokens
|
|
359
|
+
? true
|
|
360
|
+
: isFloor || clientTokens.high >= thresholdTokens
|
|
361
|
+
? null
|
|
362
|
+
: false;
|
|
363
|
+
return {
|
|
364
|
+
...base,
|
|
365
|
+
mode: 'threshold',
|
|
366
|
+
mechanism: 'tool search',
|
|
367
|
+
setting,
|
|
368
|
+
thresholdShare,
|
|
369
|
+
thresholdTokens,
|
|
370
|
+
clientTokens,
|
|
371
|
+
ratio,
|
|
372
|
+
distanceTokens: { low: clientTokens.low - thresholdTokens, high: clientTokens.high - thresholdTokens },
|
|
373
|
+
crosses,
|
|
374
|
+
exceptions: EXCEPTIONS,
|
|
375
|
+
};
|
|
376
|
+
}
|
package/dist/audit/run.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { Measurement } from '../core/types.js';
|
|
2
2
|
import { type DivergenceRun } from '../core/divergence.js';
|
|
3
3
|
import { type AuditReport } from './audit.js';
|
|
4
|
+
import { type ToolSearchEnv, type ToolSearchSource } from './deferral.js';
|
|
4
5
|
import { type LoadedConfig } from './config.js';
|
|
5
6
|
/** Where the published `tools-delta/v1` run lives when `--claude` doesn't override it. */
|
|
6
7
|
export declare const DEFAULT_DIVERGENCE_URL = "https://raw.githubusercontent.com/athakur3/mcp-context-cost/main/results/divergence.json";
|
|
@@ -18,6 +19,18 @@ export interface AuditOptions {
|
|
|
18
19
|
claude?: boolean;
|
|
19
20
|
/** Override the divergence.json source — mainly for tests and self-hosted mirrors. */
|
|
20
21
|
divergenceUrl?: string;
|
|
22
|
+
/**
|
|
23
|
+
* The tool-search variables as this process's SHELL has them. Defaults to
|
|
24
|
+
* this process's environment. Overridable so a test can state a machine
|
|
25
|
+
* rather than inherit the one it runs on.
|
|
26
|
+
*/
|
|
27
|
+
env?: ToolSearchEnv;
|
|
28
|
+
/**
|
|
29
|
+
* The same variables as Claude Code's own settings files set them — the other
|
|
30
|
+
* half of the answer, and the half a shell cannot show. Defaults to reading
|
|
31
|
+
* those files off the machine being audited (`discoverSettings`).
|
|
32
|
+
*/
|
|
33
|
+
settings?: ToolSearchSource[];
|
|
21
34
|
onProgress?: (name: string, done: number, total: number) => void;
|
|
22
35
|
}
|
|
23
36
|
/** Fetch and parse the published divergence run. Never throws: a failure is a report problem, not a crash. */
|
|
@@ -26,6 +39,15 @@ export declare function fetchDivergence(url: string): Promise<{
|
|
|
26
39
|
problem?: string;
|
|
27
40
|
}>;
|
|
28
41
|
export declare function discover(opts?: AuditOptions): LoadedConfig[];
|
|
42
|
+
/**
|
|
43
|
+
* Read every settings file Claude Code would take its deferral setting from.
|
|
44
|
+
*
|
|
45
|
+
* A sibling of `discover` above and deliberately separate from it: that one
|
|
46
|
+
* finds which servers exist, this one finds how the client treats them. The
|
|
47
|
+
* setting does not live in a client config, and reading only the shell that
|
|
48
|
+
* launched this audit answers for the wrong machine.
|
|
49
|
+
*/
|
|
50
|
+
export declare function discoverSettings(opts?: AuditOptions): ToolSearchSource[];
|
|
29
51
|
/** Measure every distinct stdio server across the given configs, once each. */
|
|
30
52
|
export declare function measureAll(configs: LoadedConfig[], opts?: AuditOptions): Promise<Map<string, Measurement>>;
|
|
31
53
|
export declare function runAudit(opts?: AuditOptions): Promise<AuditReport>;
|
package/dist/audit/run.js
CHANGED
|
@@ -8,7 +8,8 @@ import { homedir } from 'node:os';
|
|
|
8
8
|
import { measureServer } from '../sweep/run.js';
|
|
9
9
|
import { parseDivergence } from '../core/divergence.js';
|
|
10
10
|
import { buildReport, serverKey } from './audit.js';
|
|
11
|
-
import {
|
|
11
|
+
import { toolSearchEnv } from './deferral.js';
|
|
12
|
+
import { configCandidates, loadConfigs, loadSettingsSources, settingsCandidates, } from './config.js';
|
|
12
13
|
/** Where the published `tools-delta/v1` run lives when `--claude` doesn't override it. */
|
|
13
14
|
export const DEFAULT_DIVERGENCE_URL = 'https://raw.githubusercontent.com/athakur3/mcp-context-cost/main/results/divergence.json';
|
|
14
15
|
/** Fetch and parse the published divergence run. Never throws: a failure is a report problem, not a crash. */
|
|
@@ -32,6 +33,19 @@ export function discover(opts = {}) {
|
|
|
32
33
|
: configCandidates({ home, cwd, platform: process.platform, appData: process.env.APPDATA });
|
|
33
34
|
return loadConfigs(candidates, cwd);
|
|
34
35
|
}
|
|
36
|
+
/**
|
|
37
|
+
* Read every settings file Claude Code would take its deferral setting from.
|
|
38
|
+
*
|
|
39
|
+
* A sibling of `discover` above and deliberately separate from it: that one
|
|
40
|
+
* finds which servers exist, this one finds how the client treats them. The
|
|
41
|
+
* setting does not live in a client config, and reading only the shell that
|
|
42
|
+
* launched this audit answers for the wrong machine.
|
|
43
|
+
*/
|
|
44
|
+
export function discoverSettings(opts = {}) {
|
|
45
|
+
const cwd = opts.cwd ?? process.cwd();
|
|
46
|
+
const home = opts.home ?? homedir();
|
|
47
|
+
return loadSettingsSources(settingsCandidates({ home, cwd, platform: process.platform, programData: process.env.ProgramData }));
|
|
48
|
+
}
|
|
35
49
|
/** Measure every distinct stdio server across the given configs, once each. */
|
|
36
50
|
export async function measureAll(configs, opts = {}) {
|
|
37
51
|
const unique = new Map();
|
|
@@ -79,6 +93,8 @@ export async function runAudit(opts = {}) {
|
|
|
79
93
|
contextWindow: opts.contextWindow,
|
|
80
94
|
budget: opts.budget,
|
|
81
95
|
divergence,
|
|
96
|
+
env: opts.env ?? toolSearchEnv(process.env),
|
|
97
|
+
settings: opts.settings ?? discoverSettings(opts),
|
|
82
98
|
});
|
|
83
99
|
if (divergenceProblem)
|
|
84
100
|
report.problems.push(divergenceProblem);
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Badge adoption — how many projects outside this one actually display the
|
|
3
|
+
* badge, and on what day someone last looked.
|
|
4
|
+
*
|
|
5
|
+
* Every other number this project publishes is about MCP servers. This one is
|
|
6
|
+
* about the project itself, and it exists because the alternative is a launch
|
|
7
|
+
* that produces a number nobody can attribute. "Nobody is using the badge" and
|
|
8
|
+
* "nobody has checked whether anybody is using the badge" are the same sentence
|
|
9
|
+
* to a reader, and only one of them is a measurement. So the reading published
|
|
10
|
+
* here is a dated observation with its own working shown: the exact queries
|
|
11
|
+
* that were run, every file they turned up, and what each of those files was
|
|
12
|
+
* judged to be. A zero from this instrument means *these queries ran on this
|
|
13
|
+
* date and found none*, which is a fact. A missing reading says so in those
|
|
14
|
+
* words and publishes no number at all.
|
|
15
|
+
*
|
|
16
|
+
* ## What counts as displaying the badge
|
|
17
|
+
*
|
|
18
|
+
* A file, in a repository owned by someone else, carrying a shields.io endpoint
|
|
19
|
+
* badge that is auditable — and the project publishes two ways to make one, so
|
|
20
|
+
* the rule counts both. Either the badge's JSON is served from this
|
|
21
|
+
* repository's `badges/` directory, or the author hosts their own
|
|
22
|
+
* `badges/<name>.json` — which is what `npm run sweep` produces, and what every
|
|
23
|
+
* badge snippet this project ships tells a reader to point shields at — and
|
|
24
|
+
* links the badge back at the measurement here, which the same snippet
|
|
25
|
+
* requires.
|
|
26
|
+
*
|
|
27
|
+
* Counting only the first form would have published a zero about a spelling
|
|
28
|
+
* nobody was ever told to write: no snippet in README, in the dashboard or in
|
|
29
|
+
* the staged upstream patch produces a URL inside this repository's `badges/`,
|
|
30
|
+
* because each of them is addressed to an author measuring their own server.
|
|
31
|
+
*
|
|
32
|
+
* What is still outside the rule is a self-hosted badge that links back to
|
|
33
|
+
* nothing. Nothing in such a file names this project, so no query can nominate
|
|
34
|
+
* it — and this project's own README says a badge nobody can audit is
|
|
35
|
+
* decoration. That limit is published on the page rather than papered over; see
|
|
36
|
+
* `renderAdoptionPage`.
|
|
37
|
+
*
|
|
38
|
+
* The link is paired with the image it wraps rather than looked for anywhere in
|
|
39
|
+
* the file, because a README carrying an unrelated shields badge and, elsewhere,
|
|
40
|
+
* a sentence naming this project is not a project displaying this badge.
|
|
41
|
+
*
|
|
42
|
+
* ## Why the search is a net and not the judgement
|
|
43
|
+
*
|
|
44
|
+
* Code search matches text, and the same badge is written two ways: shields
|
|
45
|
+
* percent-encodes the `url` parameter, so the published snippet carries
|
|
46
|
+
* `raw.githubusercontent.com%2Fathakur3%2F…`, while a hand-written badge may
|
|
47
|
+
* carry the plain path. Measured against GitHub code search on 2026-08-20, the
|
|
48
|
+
* two forms do not find each other: a repository whose README carries only the
|
|
49
|
+
* encoded form of a raw URL returns 0 for the plain form of that same path,
|
|
50
|
+
* while the encoded literal returns matches. Neither query alone is the
|
|
51
|
+
* question being asked.
|
|
52
|
+
*
|
|
53
|
+
* So the queries only nominate candidates. What a candidate *is* gets decided
|
|
54
|
+
* by reading the file — `classifyFile` below, applied to content that has had
|
|
55
|
+
* its percent-encoding undone, so one rule covers both spellings. Files that
|
|
56
|
+
* name the project without displaying the badge are kept in the reading as
|
|
57
|
+
* rejections, because a zero is worth much more next to the list of things that
|
|
58
|
+
* were examined and turned down.
|
|
59
|
+
*/
|
|
60
|
+
/** Method identifier, versioned independently of the o200k methodology. */
|
|
61
|
+
export declare const ADOPTION_METHOD = "badge-sightings/v1";
|
|
62
|
+
/** Where this project's badge JSON is published — the thing a badge points at. */
|
|
63
|
+
export interface BadgeSource {
|
|
64
|
+
owner: string;
|
|
65
|
+
repo: string;
|
|
66
|
+
branch: string;
|
|
67
|
+
}
|
|
68
|
+
export declare const BADGE_SOURCE: BadgeSource;
|
|
69
|
+
/** A query as published: what was asked, and why it is part of the question. */
|
|
70
|
+
export interface QueryDef {
|
|
71
|
+
name: string;
|
|
72
|
+
q: string;
|
|
73
|
+
why: string;
|
|
74
|
+
}
|
|
75
|
+
/** A query as answered. `hits` is null whenever `state` is not `ok`. */
|
|
76
|
+
export interface QueryResult extends QueryDef {
|
|
77
|
+
state: 'ok' | 'failed';
|
|
78
|
+
hits: number | null;
|
|
79
|
+
/** Set when the query did not answer — a reading with one of these is refused. */
|
|
80
|
+
error?: string;
|
|
81
|
+
/** Set when more results existed than were collected. Also refuses the reading. */
|
|
82
|
+
truncated?: boolean;
|
|
83
|
+
}
|
|
84
|
+
/** `badge`: displays it. `mention`: names the project without displaying it. */
|
|
85
|
+
export type SightingKind = 'badge' | 'mention';
|
|
86
|
+
export interface Sighting {
|
|
87
|
+
/** `owner/repo` of the third-party repository. */
|
|
88
|
+
repo: string;
|
|
89
|
+
path: string;
|
|
90
|
+
url: string;
|
|
91
|
+
kind: SightingKind;
|
|
92
|
+
/** Which query nominated it — so a reader can re-run the one that found it. */
|
|
93
|
+
foundBy: string;
|
|
94
|
+
firstSeenAt: string;
|
|
95
|
+
lastSeenAt: string;
|
|
96
|
+
}
|
|
97
|
+
export interface AdoptionRun {
|
|
98
|
+
method: string;
|
|
99
|
+
/** UTC day the queries were run (YYYY-MM-DD). */
|
|
100
|
+
checkedAt: string;
|
|
101
|
+
source: BadgeSource;
|
|
102
|
+
queries: QueryResult[];
|
|
103
|
+
/** Distinct third-party files examined this run, whatever they turned out to be. */
|
|
104
|
+
candidates: number;
|
|
105
|
+
/** Every sighting ever recorded, including ones not seen this run. */
|
|
106
|
+
sightings: Sighting[];
|
|
107
|
+
/**
|
|
108
|
+
* Third-party repositories displaying the badge on `checkedAt`, or null when
|
|
109
|
+
* the queries did not establish it. Never 0 for want of looking.
|
|
110
|
+
*/
|
|
111
|
+
thirdPartyRepos: number | null;
|
|
112
|
+
/** Why no number is published. Null exactly when `thirdPartyRepos` is a number. */
|
|
113
|
+
unresolved: string | null;
|
|
114
|
+
/**
|
|
115
|
+
* The most recent reading that did establish a number, carried across runs.
|
|
116
|
+
* A search that falls over must not publish a count, but it also must not
|
|
117
|
+
* erase the last one that stood: "unknown today, 3 on the 20th" is a reading,
|
|
118
|
+
* and "unknown" alone throws away one that was already paid for.
|
|
119
|
+
*/
|
|
120
|
+
lastResolved: {
|
|
121
|
+
checkedAt: string;
|
|
122
|
+
thirdPartyRepos: number;
|
|
123
|
+
} | null;
|
|
124
|
+
}
|
|
125
|
+
/**
|
|
126
|
+
* The published query set. Both spellings of the badge URL are asked for
|
|
127
|
+
* separately (see the header), plus the click-through the badge is supposed to
|
|
128
|
+
* carry, plus the project's own name as the widest net — anything that names
|
|
129
|
+
* the project becomes a candidate and is then judged by its contents.
|
|
130
|
+
*/
|
|
131
|
+
export declare function adoptionQueries(src?: BadgeSource): QueryDef[];
|
|
132
|
+
/**
|
|
133
|
+
* Undo percent-encoding without throwing on the malformed sequences that turn
|
|
134
|
+
* up in real files. Decoded per-escape rather than over the whole string, so
|
|
135
|
+
* one bad `%zz` costs that escape and nothing around it.
|
|
136
|
+
*/
|
|
137
|
+
export declare function decodeLoose(text: string): string;
|
|
138
|
+
/**
|
|
139
|
+
* A shields endpoint badge as it stands in a file: the JSON it renders from,
|
|
140
|
+
* and the link a reader clicking it would follow. `linkTarget` is null when the
|
|
141
|
+
* image is not wrapped in a link at all.
|
|
142
|
+
*/
|
|
143
|
+
export interface EndpointBadge {
|
|
144
|
+
url: string;
|
|
145
|
+
linkTarget: string | null;
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
148
|
+
* Every shields endpoint badge in a file, decoded, each paired with its own
|
|
149
|
+
* link target. Both spellings a README uses are read: markdown
|
|
150
|
+
* `[](target)`, which is the snippet this project publishes, and an
|
|
151
|
+
* HTML anchor wrapping an `<img>`.
|
|
152
|
+
*/
|
|
153
|
+
export declare function endpointBadges(text: string): EndpointBadge[];
|
|
154
|
+
/**
|
|
155
|
+
* Whether a badge's JSON is served from this repository — auditable by
|
|
156
|
+
* construction, whatever the badge links to. Matched without regard to case:
|
|
157
|
+
* GitHub owner and repository names are case-insensitive, a badge written
|
|
158
|
+
* `MCP-Context-Cost` renders exactly the same one, and code search found the
|
|
159
|
+
* file that way too.
|
|
160
|
+
*/
|
|
161
|
+
export declare function hostedHere(url: string, src?: BadgeSource): boolean;
|
|
162
|
+
/**
|
|
163
|
+
* Whether a badge's link target leads back to this project — the repository,
|
|
164
|
+
* the published pages, or a raw file in either. This is exactly what the
|
|
165
|
+
* published snippet asks an author to do with a badge whose JSON they host
|
|
166
|
+
* themselves, and it is what makes that badge auditable rather than decoration.
|
|
167
|
+
*/
|
|
168
|
+
export declare function linksBackToProject(target: string, src?: BadgeSource): boolean;
|
|
169
|
+
/**
|
|
170
|
+
* Whether a file displays this project's badge, in either published form: the
|
|
171
|
+
* JSON served from here, or self-hosted JSON with the badge linked back at the
|
|
172
|
+
* measurement here. See the header for why both count and why the link is
|
|
173
|
+
* paired with the image rather than looked for anywhere in the file.
|
|
174
|
+
*/
|
|
175
|
+
export declare function displaysBadge(text: string, src?: BadgeSource): boolean;
|
|
176
|
+
/**
|
|
177
|
+
* Every shields endpoint `url` in a file whose JSON is served from this
|
|
178
|
+
* repository's `badges/` directory, decoded. The first of the two forms above,
|
|
179
|
+
* on its own.
|
|
180
|
+
*/
|
|
181
|
+
export declare function endpointUrls(text: string, src?: BadgeSource): string[];
|
|
182
|
+
/**
|
|
183
|
+
* What a candidate file is. `null` when it turns out to be neither — a search
|
|
184
|
+
* index can be older than the file it points at.
|
|
185
|
+
*
|
|
186
|
+
* Case is ignored here for the same reason it is ignored above, and the first
|
|
187
|
+
* real run is why it is stated rather than assumed: a file discussing
|
|
188
|
+
* "MCP-context-cost" was found by the search and would have been thrown out by
|
|
189
|
+
* an exact-case test, which is a rejection that looks identical to a file that
|
|
190
|
+
* genuinely stopped mentioning the project.
|
|
191
|
+
*/
|
|
192
|
+
export declare function classifyFile(text: string, src?: BadgeSource): SightingKind | null;
|
|
193
|
+
/** `owner/repo` → is that owner someone other than this project's? */
|
|
194
|
+
export declare function isThirdParty(repoFullName: string, src?: BadgeSource): boolean;
|
|
195
|
+
/** A sighting as observed this run, before it is dated against the previous one. */
|
|
196
|
+
export type FreshSighting = Omit<Sighting, 'firstSeenAt' | 'lastSeenAt'>;
|
|
197
|
+
/**
|
|
198
|
+
* Date this run's sightings against the last one. A file seen before keeps its
|
|
199
|
+
* `firstSeenAt`; a file no longer found is kept with the date it was last seen
|
|
200
|
+
* rather than deleted, so a badge that disappears is visible as a badge that
|
|
201
|
+
* disappeared instead of as one that never existed.
|
|
202
|
+
*/
|
|
203
|
+
export declare function mergeSightings(previous: Sighting[], fresh: FreshSighting[], checkedAt: string): Sighting[];
|
|
204
|
+
/** Repositories displaying the badge as of `checkedAt` — sorted, deduplicated. */
|
|
205
|
+
export declare function badgeRepos(sightings: Sighting[], checkedAt: string): string[];
|
|
206
|
+
/**
|
|
207
|
+
* Whether this run may publish a number, and if not, why not. A query that did
|
|
208
|
+
* not answer, or one whose results were cut short, means the set of files that
|
|
209
|
+
* carry the badge was never established — and a count taken from an incomplete
|
|
210
|
+
* search is a zero that means "we did not finish", which is the exact confusion
|
|
211
|
+
* this instrument exists to remove.
|
|
212
|
+
*/
|
|
213
|
+
export declare function resolveCount(queries: QueryResult[], sightings: Sighting[], checkedAt: string, unreadableCandidates?: number): Pick<AdoptionRun, 'thirdPartyRepos' | 'unresolved'>;
|
|
214
|
+
/**
|
|
215
|
+
* The last completed reading to carry into this run's record: this one if it
|
|
216
|
+
* completed, otherwise whatever the previous run was carrying.
|
|
217
|
+
*/
|
|
218
|
+
export declare function carryResolved(previous: AdoptionRun | null, current: Pick<AdoptionRun, 'checkedAt' | 'thirdPartyRepos'>): AdoptionRun['lastResolved'];
|
|
219
|
+
/** Parse results/badge-adoption.json; anything malformed yields null, never throws. */
|
|
220
|
+
export declare function parseAdoption(text: string): AdoptionRun | null;
|
|
221
|
+
/**
|
|
222
|
+
* The page a reader opens. `null` is the state that matters most: no run on
|
|
223
|
+
* record renders as "nobody has looked", in those words, with no number — which
|
|
224
|
+
* is the whole distinction this instrument exists to make readable.
|
|
225
|
+
*/
|
|
226
|
+
export declare function renderAdoptionPage(run: AdoptionRun | null, src?: BadgeSource): string;
|