mcp-context-cost 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +202 -43
- package/dist/audit/audit.d.ts +101 -0
- package/dist/audit/audit.js +492 -16
- package/dist/audit/config.d.ts +38 -0
- package/dist/audit/config.js +64 -0
- package/dist/audit/deferral.d.ts +346 -0
- package/dist/audit/deferral.js +376 -0
- package/dist/audit/diff.d.ts +124 -0
- package/dist/audit/diff.js +318 -0
- package/dist/audit/run.d.ts +34 -0
- package/dist/audit/run.js +45 -2
- package/dist/cli.d.ts +21 -0
- package/dist/cli.js +141 -7
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +18 -0
- package/dist/sweep/dashboard.js +74 -10
- package/dist/sweep/docker.d.ts +31 -0
- package/dist/sweep/docker.js +20 -11
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +18 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +43 -0
- package/dist/sweep/run.js +135 -37
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +57 -2
- package/package.json +3 -1
package/dist/audit/audit.js
CHANGED
|
@@ -13,14 +13,151 @@
|
|
|
13
13
|
* commands shared by two configs are still only measured once.
|
|
14
14
|
*/
|
|
15
15
|
import { METHODOLOGY_VERSION } from '../core/canonical.js';
|
|
16
|
+
import { isCurrent } from '../core/divergence.js';
|
|
17
|
+
import { evaluateDeferral, PUBLISHED_WIRE_TO_CLIENT_RATIO, SHELL_SOURCE, } from './deferral.js';
|
|
18
|
+
import { formatDiff, formatGate } from './diff.js';
|
|
16
19
|
export const DEFAULT_CONTEXT_WINDOW = 200_000;
|
|
20
|
+
const TRIM_TOOL_COUNT = 3;
|
|
21
|
+
function buildTrimAdvice(sortedTools, totalTokens) {
|
|
22
|
+
if (totalTokens <= 0 || sortedTools.length < 2)
|
|
23
|
+
return null;
|
|
24
|
+
const trimmed = sortedTools.slice(0, TRIM_TOOL_COUNT);
|
|
25
|
+
const recoverableTokens = trimmed.reduce((a, t) => a + t.tokens, 0);
|
|
26
|
+
return { tools: trimmed, recoverableTokens, recoverableShare: recoverableTokens / totalTokens };
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* The smallest heaviest-first set of servers that gets a config under its budget.
|
|
30
|
+
*
|
|
31
|
+
* `--budget` used to print "BUDGET FAIL: 84,455 > 20,000" and stop, which tells a reader
|
|
32
|
+
* they have a problem and nothing about the shape of it. The whole point of the audit
|
|
33
|
+
* surface is that the person running it is the person paying the tokens, and "you are over"
|
|
34
|
+
* is a measurement where "these two are why" is a decision.
|
|
35
|
+
*
|
|
36
|
+
* Heaviest-first is ONE ordering, not a recommendation: this cannot know which servers you
|
|
37
|
+
* need, and dropping by weight will sometimes name the one you cannot live without. That
|
|
38
|
+
* caveat is printed with the result rather than left implied.
|
|
39
|
+
*/
|
|
40
|
+
export function planBudgetFit(config, limit) {
|
|
41
|
+
const measured = config.servers
|
|
42
|
+
.filter((srv) => typeof srv.tokens === 'number' && srv.tokens > 0)
|
|
43
|
+
.sort((a, b) => b.tokens - a.tokens);
|
|
44
|
+
const overBy = config.totalTokens - limit;
|
|
45
|
+
const drop = [];
|
|
46
|
+
let remaining = config.totalTokens;
|
|
47
|
+
for (const srv of measured) {
|
|
48
|
+
if (remaining <= limit)
|
|
49
|
+
break;
|
|
50
|
+
remaining -= srv.tokens;
|
|
51
|
+
drop.push({ name: srv.name, tokens: srv.tokens, remaining });
|
|
52
|
+
}
|
|
53
|
+
return {
|
|
54
|
+
overBy,
|
|
55
|
+
drop,
|
|
56
|
+
keptCount: measured.length - drop.length,
|
|
57
|
+
keptTokens: remaining,
|
|
58
|
+
// Removing everything measured still leaves unmeasured/base cost behind, so the
|
|
59
|
+
// honest test is whether the remainder actually landed under the limit.
|
|
60
|
+
feasible: remaining <= limit,
|
|
61
|
+
};
|
|
62
|
+
}
|
|
17
63
|
/** Cache key for measurement reuse: the exact argv two configs would spawn. */
|
|
18
64
|
export function serverKey(s) {
|
|
19
65
|
return JSON.stringify(s.argv ?? [s.url ?? s.name]);
|
|
20
66
|
}
|
|
67
|
+
/**
|
|
68
|
+
* One entry's environment, as a value two entries can be compared by.
|
|
69
|
+
*
|
|
70
|
+
* Values are read here and compared here, and nothing derived from them leaves
|
|
71
|
+
* this function — the count that does is a count of entries. Same rule as
|
|
72
|
+
* `config.ts`: env values are read to spawn a server, never written to a report.
|
|
73
|
+
*/
|
|
74
|
+
function envSignature(s) {
|
|
75
|
+
const env = s.env ?? {};
|
|
76
|
+
return JSON.stringify(Object.keys(env).sort().map((k) => [k, env[k]]));
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* The measurement keys that stand for more than one distinct server.
|
|
80
|
+
*
|
|
81
|
+
* `serverKey` is the argv alone, so two entries running the same command under
|
|
82
|
+
* different environments are measured once and both are given that one number.
|
|
83
|
+
* Environment decides what a server serves — `GITHUB_TOOLSETS` on
|
|
84
|
+
* `github-mcp-server` selects which toolsets it lists — so for entries under one
|
|
85
|
+
* of these keys, the number reported is one entry's, not each one's.
|
|
86
|
+
*
|
|
87
|
+
* Same argv AND same environment is not collapsed: two clients pointing at an
|
|
88
|
+
* identical server are one measurement, which is the reuse this key is for.
|
|
89
|
+
*/
|
|
90
|
+
export function collapsedKeys(configs) {
|
|
91
|
+
const envs = new Map();
|
|
92
|
+
for (const cfg of configs) {
|
|
93
|
+
if (cfg.error)
|
|
94
|
+
continue;
|
|
95
|
+
for (const s of cfg.servers) {
|
|
96
|
+
if (s.transport !== 'stdio')
|
|
97
|
+
continue;
|
|
98
|
+
const key = serverKey(s);
|
|
99
|
+
const seen = envs.get(key);
|
|
100
|
+
if (seen)
|
|
101
|
+
seen.add(envSignature(s));
|
|
102
|
+
else
|
|
103
|
+
envs.set(key, new Set([envSignature(s)]));
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
return new Set([...envs].filter(([, sigs]) => sigs.size > 1).map(([key]) => key));
|
|
107
|
+
}
|
|
21
108
|
function measuredOk(m) {
|
|
22
109
|
return (m.status === 'measured' || m.status === 'dynamic') && typeof m.totalTokens === 'number';
|
|
23
110
|
}
|
|
111
|
+
/**
|
|
112
|
+
* Clients whose several config files are read into ONE session.
|
|
113
|
+
*
|
|
114
|
+
* Claude Code loads user-scope `~/.claude.json` and project-scope
|
|
115
|
+
* `<cwd>/.mcp.json` together, so a stack split across them faces the deferral
|
|
116
|
+
* threshold as a sum. Judging each file alone tells the standard setup it is
|
|
117
|
+
* under a line the session it actually runs is over.
|
|
118
|
+
*
|
|
119
|
+
* This does not merge the reported totals, which stay per file: a context
|
|
120
|
+
* window belongs to one session, and that is the argument for adding these two
|
|
121
|
+
* together — not for adding one client's servers to another's.
|
|
122
|
+
*/
|
|
123
|
+
const ONE_SESSION_PER_CLIENT = new Set(['claude-code']);
|
|
124
|
+
/** Which configs share a deferral verdict. */
|
|
125
|
+
function deferralScopeKey(client, source) {
|
|
126
|
+
return ONE_SESSION_PER_CLIENT.has(client) ? client : `${client}\0${source}`;
|
|
127
|
+
}
|
|
128
|
+
/**
|
|
129
|
+
* Give every config a deferral verdict, computed once per session scope and
|
|
130
|
+
* shared by identity across the configs that scope covers — so the report can
|
|
131
|
+
* print it once and a `--json` consumer can see which files it spans.
|
|
132
|
+
*/
|
|
133
|
+
function attachDeferral(configs, contextWindow, opts) {
|
|
134
|
+
const scopes = new Map();
|
|
135
|
+
for (const cfg of configs) {
|
|
136
|
+
const key = deferralScopeKey(cfg.client, cfg.source);
|
|
137
|
+
const group = scopes.get(key);
|
|
138
|
+
if (group)
|
|
139
|
+
group.push(cfg);
|
|
140
|
+
else
|
|
141
|
+
scopes.set(key, [cfg]);
|
|
142
|
+
}
|
|
143
|
+
const verdicts = new Map();
|
|
144
|
+
for (const [key, group] of scopes) {
|
|
145
|
+
verdicts.set(key,
|
|
146
|
+
// Computed against the same context window the share uses, so any
|
|
147
|
+
// threshold moves with `--context` instead of being pinned to 200,000.
|
|
148
|
+
evaluateDeferral({
|
|
149
|
+
client: group[0].client,
|
|
150
|
+
sources: group.map((c) => c.source),
|
|
151
|
+
servers: group.flatMap((c) => c.servers.map((s) => ({ tokens: s.tokens ?? 0, claudeTokens: s.claudeTokens }))),
|
|
152
|
+
skippedCount: group.reduce((a, c) => a + c.skipped.length, 0),
|
|
153
|
+
sharedMeasurements: group.reduce((a, c) => a + (opts.shared.get(c) ?? 0), 0),
|
|
154
|
+
}, { contextWindow, env: opts.env, settings: opts.settings, divergence: opts.divergence }));
|
|
155
|
+
}
|
|
156
|
+
return configs.map((cfg) => ({
|
|
157
|
+
...cfg,
|
|
158
|
+
deferral: verdicts.get(deferralScopeKey(cfg.client, cfg.source)),
|
|
159
|
+
}));
|
|
160
|
+
}
|
|
24
161
|
/**
|
|
25
162
|
* Assemble the report from configs + measurements. Pure: `runAudit` does the
|
|
26
163
|
* spawning, this does the arithmetic, so totals and shares are testable without
|
|
@@ -29,7 +166,11 @@ function measuredOk(m) {
|
|
|
29
166
|
export function buildReport(configs, measured, opts = {}) {
|
|
30
167
|
const contextWindow = opts.contextWindow ?? DEFAULT_CONTEXT_WINDOW;
|
|
31
168
|
const problems = [];
|
|
32
|
-
const
|
|
169
|
+
const built = [];
|
|
170
|
+
// Across every config at once: a twin in one client's file is measured for
|
|
171
|
+
// the other client's entry just the same.
|
|
172
|
+
const collapsed = collapsedKeys(configs);
|
|
173
|
+
const shared = new Map();
|
|
33
174
|
for (const cfg of configs) {
|
|
34
175
|
if (cfg.error) {
|
|
35
176
|
problems.push(`${cfg.source}: ${cfg.error}`);
|
|
@@ -38,6 +179,9 @@ export function buildReport(configs, measured, opts = {}) {
|
|
|
38
179
|
const ok = [];
|
|
39
180
|
const skipped = [];
|
|
40
181
|
const tools = [];
|
|
182
|
+
// Counted only for servers that put a number into the total: a twin that
|
|
183
|
+
// failed to launch is already a floor, and adds nothing to a sum.
|
|
184
|
+
let sharedHere = 0;
|
|
41
185
|
for (const s of cfg.servers) {
|
|
42
186
|
const base = {
|
|
43
187
|
name: s.name,
|
|
@@ -73,6 +217,9 @@ export function buildReport(configs, measured, opts = {}) {
|
|
|
73
217
|
});
|
|
74
218
|
continue;
|
|
75
219
|
}
|
|
220
|
+
if (collapsed.has(serverKey(s)))
|
|
221
|
+
sharedHere++;
|
|
222
|
+
const divRow = opts.divergence?.servers[s.name];
|
|
76
223
|
ok.push({
|
|
77
224
|
...base,
|
|
78
225
|
status: m.status,
|
|
@@ -80,6 +227,7 @@ export function buildReport(configs, measured, opts = {}) {
|
|
|
80
227
|
toolCount: m.toolCount,
|
|
81
228
|
share: null, // filled once the total is known
|
|
82
229
|
canonicalSha256: m.canonicalSha256,
|
|
230
|
+
claudeTokens: opts.divergence ? (isCurrent(divRow, m.canonicalSha256 ?? null) ? divRow.claudeDelta : null) : undefined,
|
|
83
231
|
notes: m.status === 'dynamic' ? m.notes : undefined,
|
|
84
232
|
});
|
|
85
233
|
for (const t of m.tools)
|
|
@@ -91,7 +239,7 @@ export function buildReport(configs, measured, opts = {}) {
|
|
|
91
239
|
for (const s of ok)
|
|
92
240
|
s.share = totalTokens > 0 ? (s.tokens ?? 0) / totalTokens : 0;
|
|
93
241
|
tools.sort((a, b) => b.tokens - a.tokens);
|
|
94
|
-
|
|
242
|
+
const result = {
|
|
95
243
|
client: cfg.client,
|
|
96
244
|
source: cfg.source,
|
|
97
245
|
totalTokens,
|
|
@@ -105,9 +253,13 @@ export function buildReport(configs, measured, opts = {}) {
|
|
|
105
253
|
servers: ok,
|
|
106
254
|
skipped,
|
|
107
255
|
heaviestTools: tools.slice(0, 5),
|
|
108
|
-
|
|
256
|
+
trimAdvice: buildTrimAdvice(tools, totalTokens),
|
|
257
|
+
};
|
|
258
|
+
built.push(result);
|
|
259
|
+
shared.set(result, sharedHere);
|
|
109
260
|
}
|
|
110
|
-
|
|
261
|
+
built.sort((a, b) => b.totalTokens - a.totalTokens);
|
|
262
|
+
const results = attachDeferral(built, contextWindow, { ...opts, shared });
|
|
111
263
|
const report = {
|
|
112
264
|
methodologyVersion: METHODOLOGY_VERSION,
|
|
113
265
|
encoding: 'o200k_base',
|
|
@@ -116,25 +268,284 @@ export function buildReport(configs, measured, opts = {}) {
|
|
|
116
268
|
configs: results,
|
|
117
269
|
problems,
|
|
118
270
|
};
|
|
271
|
+
if (opts.divergence) {
|
|
272
|
+
report.claudeDivergence = { model: opts.divergence.model, measuredAt: opts.divergence.measuredAt };
|
|
273
|
+
}
|
|
119
274
|
if (typeof opts.budget === 'number') {
|
|
120
275
|
// The worst config is the gate: passing because your *lightest* client fits
|
|
121
276
|
// would be a green check on a session you don't run.
|
|
122
277
|
const worst = results[0];
|
|
278
|
+
const over = (worst?.totalTokens ?? 0) > opts.budget;
|
|
123
279
|
report.budget = {
|
|
124
280
|
limit: opts.budget,
|
|
125
281
|
worstTotal: worst?.totalTokens ?? 0,
|
|
126
282
|
worstSource: worst?.source ?? '(none)',
|
|
127
|
-
over
|
|
283
|
+
over,
|
|
284
|
+
fit: over && worst ? planBudgetFit(worst, opts.budget) : undefined,
|
|
128
285
|
};
|
|
129
286
|
}
|
|
130
287
|
return report;
|
|
131
288
|
}
|
|
132
289
|
const n = (x) => x.toLocaleString('en-US');
|
|
133
290
|
const pct = (x) => `${(x * 100).toFixed(1)}%`;
|
|
291
|
+
/** The measurement itself — unconditional, and separate from any claim about who pays it. */
|
|
292
|
+
function measurementLine(cfg, contextWindow) {
|
|
293
|
+
return (` ${n(cfg.totalTokens)} tokens of tool schemas — ${pct(cfg.contextShare)} of a ` +
|
|
294
|
+
`${n(contextWindow)}-token context window.`);
|
|
295
|
+
}
|
|
296
|
+
/** How a place is named in a sentence; the shell has a path-shaped stand-in in JSON. */
|
|
297
|
+
function sourceName(source) {
|
|
298
|
+
return source === SHELL_SOURCE ? 'this shell' : source;
|
|
299
|
+
}
|
|
300
|
+
/** "ENABLE_TOOL_SEARCH=auto on this machine" / "unset here, the documented default". */
|
|
301
|
+
function settingPhrase(d) {
|
|
302
|
+
const s = d.setting;
|
|
303
|
+
if (!s || !s.variable)
|
|
304
|
+
return 'by default';
|
|
305
|
+
// Which place it was read in is not spelled here: a settings path is longer
|
|
306
|
+
// than this report's line, and both cases — the value and the silence — are
|
|
307
|
+
// answered by the list `postureSourceLines` prints directly underneath, which
|
|
308
|
+
// marks the place that decided it.
|
|
309
|
+
if (!s.readFromMachine)
|
|
310
|
+
return `${s.variable} is unset here, which is the documented default`;
|
|
311
|
+
// A base URL is reported by hostname only (see ToolSearchSetting.value), so it
|
|
312
|
+
// is phrased as where the variable points and never as what it equals.
|
|
313
|
+
if (s.variable === 'ANTHROPIC_BASE_URL')
|
|
314
|
+
return `${s.variable} points at ${s.value} on this machine`;
|
|
315
|
+
return `${s.variable}=${s.value} on this machine`;
|
|
316
|
+
}
|
|
317
|
+
/**
|
|
318
|
+
* Which places this posture was read from, and what each one held.
|
|
319
|
+
*
|
|
320
|
+
* The sentence above this list states a verdict about tokens on somebody's
|
|
321
|
+
* machine, and the commonest of those verdicts — the documented default — is
|
|
322
|
+
* an argument from silence. Silence is only evidence across the places that
|
|
323
|
+
* were opened, so they are named: a reader who sets `ENABLE_TOOL_SEARCH` in a
|
|
324
|
+
* file this audit did not read can see that from the report instead of
|
|
325
|
+
* believing a default that does not apply to them.
|
|
326
|
+
*/
|
|
327
|
+
function postureSourceLines(d) {
|
|
328
|
+
const recs = d.setting?.sources ?? [];
|
|
329
|
+
if (recs.length === 0)
|
|
330
|
+
return [];
|
|
331
|
+
const lines = [
|
|
332
|
+
' Where this was read — Claude Code takes these variables from the shell it',
|
|
333
|
+
' starts in and from the env block of its own settings files:',
|
|
334
|
+
];
|
|
335
|
+
let absent = 0;
|
|
336
|
+
for (const r of recs) {
|
|
337
|
+
if (r.state === 'absent') {
|
|
338
|
+
absent++;
|
|
339
|
+
continue;
|
|
340
|
+
}
|
|
341
|
+
const held = r.state === 'unreadable'
|
|
342
|
+
? 'could not be read — what it sets is unknown'
|
|
343
|
+
: r.sets.length
|
|
344
|
+
? `sets ${r.sets.join(', ')}`
|
|
345
|
+
: 'sets none of them';
|
|
346
|
+
// Which place the verdict came out of, said once rather than left to a
|
|
347
|
+
// reader to work out from two lists.
|
|
348
|
+
const decided = d.setting?.source === r.source ? ', which decided this' : '';
|
|
349
|
+
lines.push(` ${sourceName(r.source)} — ${held}${decided}`);
|
|
350
|
+
}
|
|
351
|
+
if (absent > 0) {
|
|
352
|
+
lines.push(` ${absent} other settings file(s) it reads are not on this machine`);
|
|
353
|
+
}
|
|
354
|
+
if (!recs.some((r) => r.scope !== 'shell')) {
|
|
355
|
+
lines.push(" its settings files were NOT read here, so what they set is unknown");
|
|
356
|
+
}
|
|
357
|
+
return lines;
|
|
358
|
+
}
|
|
359
|
+
/**
|
|
360
|
+
* The total the lines around this one are about was summed from a measurement
|
|
361
|
+
* that stood for more than one entry, so it is not a number to reason from.
|
|
362
|
+
*
|
|
363
|
+
* Printed in EVERY mode, not only the one that weighs the total against a
|
|
364
|
+
* threshold. The other modes state what these tokens cost just as plainly —
|
|
365
|
+
* `loads-upfront` says every request carries them, and there the number IS the
|
|
366
|
+
* whole cost claim — so a total this report will not stand behind must not be
|
|
367
|
+
* left uncaveated in any of them. `evaluateDeferral` additionally withholds
|
|
368
|
+
* `clientTokens` and `crosses`, which only threshold mode derives; the other
|
|
369
|
+
* modes derive nothing from the total, so this paragraph is the whole of the
|
|
370
|
+
* rule there.
|
|
371
|
+
*/
|
|
372
|
+
function sharedMeasurementLines(d, skippedNames, consequence) {
|
|
373
|
+
const lines = [
|
|
374
|
+
` How big this stack is cannot be said here: ${d.sharedMeasurements} of the servers above run`,
|
|
375
|
+
' the same command as another entry and differ only in the environment they',
|
|
376
|
+
' are given, so one launch was measured and its number counted for each of',
|
|
377
|
+
' them. Environment decides what a server serves, so that sum can be wrong in',
|
|
378
|
+
...consequence,
|
|
379
|
+
];
|
|
380
|
+
if (d.isFloor) {
|
|
381
|
+
lines.push(` ${skippedNames} server(s) here also produced no number — see "not measured" above.`);
|
|
382
|
+
}
|
|
383
|
+
return lines;
|
|
384
|
+
}
|
|
385
|
+
/** Where a threshold is in play, the unknown size is the whole verdict. */
|
|
386
|
+
const SIDE_UNKNOWN = [
|
|
387
|
+
' either direction — which side of the threshold this stack falls on cannot be',
|
|
388
|
+
' read off it. Measurements here are keyed by command line alone; nothing is',
|
|
389
|
+
' wrong with the config.',
|
|
390
|
+
];
|
|
391
|
+
/** Where none is, the size is simply not a figure to quote. */
|
|
392
|
+
const SIZE_UNKNOWN = [
|
|
393
|
+
' either direction — how many tokens this stack costs cannot be read off it.',
|
|
394
|
+
' Measurements here are keyed by command line alone; nothing is wrong with',
|
|
395
|
+
' the config.',
|
|
396
|
+
];
|
|
397
|
+
/**
|
|
398
|
+
* Whether the tokens above are paid up front, and — only where the client
|
|
399
|
+
* decides that by a threshold — which side of it this stack is on.
|
|
400
|
+
*
|
|
401
|
+
* Two things this must not do, both of which an earlier version did. It must
|
|
402
|
+
* not present the threshold as what decides the default case: Claude Code's
|
|
403
|
+
* default defers every MCP tool definition unconditionally, so a stack of any
|
|
404
|
+
* size is deferred and telling its owner they pay those tokens is wrong in the
|
|
405
|
+
* commonest case there is. And where a threshold does apply, it must not
|
|
406
|
+
* compare the audit's wire count against it as though they were the same unit;
|
|
407
|
+
* they differ by a factor this repository publishes and cannot narrow, so the
|
|
408
|
+
* answer is a range and sometimes an admission.
|
|
409
|
+
*/
|
|
410
|
+
function deferralLines(d, skippedNames) {
|
|
411
|
+
const lines = [];
|
|
412
|
+
if (d.sources.length > 1) {
|
|
413
|
+
lines.push(` These ${d.sources.length} config files are read into one ${d.client} session:`);
|
|
414
|
+
for (const s of d.sources)
|
|
415
|
+
lines.push(` ${s}`);
|
|
416
|
+
lines.push(' so they face the question below together, as their sum.');
|
|
417
|
+
}
|
|
418
|
+
if (d.mode === 'client-unknown') {
|
|
419
|
+
lines.push(' Which client reads this config is not known here, so whether it defers');
|
|
420
|
+
lines.push(' tool definitions by default is not known either. Read as loaded up front.');
|
|
421
|
+
if (d.sharedMeasurements > 0)
|
|
422
|
+
lines.push(...sharedMeasurementLines(d, skippedNames, SIZE_UNKNOWN));
|
|
423
|
+
return lines;
|
|
424
|
+
}
|
|
425
|
+
if (d.mode === 'no-deferral-on-record') {
|
|
426
|
+
lines.push(` No default deferral is on record for ${d.client}, so every request`);
|
|
427
|
+
lines.push(' carries these tokens before you type anything — an absence of a record');
|
|
428
|
+
lines.push(' about the client, not a measurement of it.');
|
|
429
|
+
if (d.sharedMeasurements > 0)
|
|
430
|
+
lines.push(...sharedMeasurementLines(d, skippedNames, SIZE_UNKNOWN));
|
|
431
|
+
return lines;
|
|
432
|
+
}
|
|
433
|
+
if (d.mode === 'setting-unrecognized') {
|
|
434
|
+
lines.push(` ${d.setting?.variable} is set to "${d.setting?.value}" on this machine, which is not`);
|
|
435
|
+
lines.push(' one of the values Claude Code documents (unset, true, false, auto, auto:N).');
|
|
436
|
+
lines.push(' Whether these tokens are deferred cannot be said from it.');
|
|
437
|
+
lines.push(...postureSourceLines(d));
|
|
438
|
+
if (d.sharedMeasurements > 0)
|
|
439
|
+
lines.push(...sharedMeasurementLines(d, skippedNames, SIZE_UNKNOWN));
|
|
440
|
+
return lines;
|
|
441
|
+
}
|
|
442
|
+
if (d.mode === 'setting-unresolved') {
|
|
443
|
+
if (d.setting?.unresolved === 'sources-disagree') {
|
|
444
|
+
lines.push(` ${d.setting.variable} is set to different values by more than one place this`);
|
|
445
|
+
lines.push(' machine reads it from, and which one Claude Code takes is not on record');
|
|
446
|
+
lines.push(' here. Whether these tokens are deferred cannot be said from them.');
|
|
447
|
+
}
|
|
448
|
+
else {
|
|
449
|
+
lines.push(' A settings file Claude Code reads exists here and could not be read, so');
|
|
450
|
+
lines.push(' what it sets is unknown — and it can set the variable that decides this.');
|
|
451
|
+
lines.push(' Whether these tokens are deferred cannot be said from them.');
|
|
452
|
+
}
|
|
453
|
+
lines.push(...postureSourceLines(d));
|
|
454
|
+
if (d.sharedMeasurements > 0)
|
|
455
|
+
lines.push(...sharedMeasurementLines(d, skippedNames, SIZE_UNKNOWN));
|
|
456
|
+
return lines;
|
|
457
|
+
}
|
|
458
|
+
if (d.mode === 'loads-upfront') {
|
|
459
|
+
lines.push(` ${d.client} loads every tool definition up front here: ${d.mechanism} is off`);
|
|
460
|
+
lines.push(` because ${settingPhrase(d)}. Every request carries these`);
|
|
461
|
+
lines.push(' tokens before you type anything.');
|
|
462
|
+
lines.push(...postureSourceLines(d));
|
|
463
|
+
if (d.sharedMeasurements > 0)
|
|
464
|
+
lines.push(...sharedMeasurementLines(d, skippedNames, SIZE_UNKNOWN));
|
|
465
|
+
return lines;
|
|
466
|
+
}
|
|
467
|
+
if (d.mode === 'defers-all') {
|
|
468
|
+
lines.push(` ${d.client} defers every MCP tool definition (${d.mechanism}), with no threshold —`);
|
|
469
|
+
lines.push(` ${settingPhrase(d)}. These tokens are NOT loaded`);
|
|
470
|
+
lines.push(' up front at any size; they load when the model reaches for a tool. Size');
|
|
471
|
+
lines.push(' decides nothing here, so none of the arithmetic above changes the answer.');
|
|
472
|
+
lines.push(...postureSourceLines(d));
|
|
473
|
+
if (d.sharedMeasurements > 0)
|
|
474
|
+
lines.push(...sharedMeasurementLines(d, skippedNames, SIZE_UNKNOWN));
|
|
475
|
+
lines.push(' The full number is paid where deferral does not apply:');
|
|
476
|
+
for (const e of d.exceptions)
|
|
477
|
+
lines.push(` ${e}`);
|
|
478
|
+
return lines;
|
|
479
|
+
}
|
|
480
|
+
// Threshold mode: the only mode where how big this stack is matters at all.
|
|
481
|
+
const t = d.thresholdTokens ?? 0;
|
|
482
|
+
lines.push(` ${d.client} defers tool definitions above a threshold here (${d.mechanism}):`);
|
|
483
|
+
lines.push(` ${settingPhrase(d)}, so deferral activates once the`);
|
|
484
|
+
lines.push(` definitions reach ${n(t)} tokens — ${pct(d.thresholdShare ?? 0)} of the context window.`);
|
|
485
|
+
lines.push(...postureSourceLines(d));
|
|
486
|
+
const c = d.clientTokens;
|
|
487
|
+
if (!c) {
|
|
488
|
+
// No total was established, so there is no side to be on and no number to
|
|
489
|
+
// say it with. Both are withheld together: printing the sum here and only
|
|
490
|
+
// withholding the verdict would leave a figure a reader would compare
|
|
491
|
+
// against the threshold themselves.
|
|
492
|
+
lines.push(...sharedMeasurementLines(d, skippedNames, SIDE_UNKNOWN));
|
|
493
|
+
return lines;
|
|
494
|
+
}
|
|
495
|
+
const total = d.isFloor ? `at least ${n(d.wireTokens)}` : n(d.wireTokens);
|
|
496
|
+
if (c.estimated === 0) {
|
|
497
|
+
lines.push(` This stack is ${total} tokens on the wire, and ${n(c.low)} by the published`);
|
|
498
|
+
lines.push(` Anthropic counts for all ${c.exact} measured server(s) — which is the side the`);
|
|
499
|
+
lines.push(' threshold is counted on.');
|
|
500
|
+
}
|
|
501
|
+
else {
|
|
502
|
+
lines.push(` This stack is ${total} tokens on the wire. The threshold is counted in what`);
|
|
503
|
+
lines.push(` the client sends to the API, which is a different number: across ${d.ratio.servers}`);
|
|
504
|
+
lines.push(` servers in ${d.ratio.source} the two differ by ${ratioBand(d)}, putting this stack`);
|
|
505
|
+
lines.push(` between ${n(c.low)} and ${n(c.high)} tokens on that side` +
|
|
506
|
+
(c.exact > 0 ? `, with ${c.exact} of them taken from published counts.` : '.'));
|
|
507
|
+
}
|
|
508
|
+
if (d.crosses === true) {
|
|
509
|
+
lines.push(` That is at or above the threshold — over by ${n(d.distanceTokens.low)} at the low end —`);
|
|
510
|
+
lines.push(' so these tokens are NOT loaded up front. The full number is still paid');
|
|
511
|
+
lines.push(' where deferral does not apply:');
|
|
512
|
+
for (const e of d.exceptions)
|
|
513
|
+
lines.push(` ${e}`);
|
|
514
|
+
}
|
|
515
|
+
else if (d.crosses === false) {
|
|
516
|
+
lines.push(` That is below the threshold — under by ${n(-d.distanceTokens.high)} at the high end —`);
|
|
517
|
+
lines.push(' so deferral does not activate and every request carries these tokens');
|
|
518
|
+
lines.push(' before you type anything.');
|
|
519
|
+
}
|
|
520
|
+
else {
|
|
521
|
+
lines.push(' Which side this stack falls on cannot be said:');
|
|
522
|
+
if (c.low < t && c.high >= t) {
|
|
523
|
+
lines.push(` that range straddles the ${n(t)}-token threshold`);
|
|
524
|
+
}
|
|
525
|
+
if (d.isFloor) {
|
|
526
|
+
lines.push(` ${skippedNames} server(s) here produced no number, and what they serve`);
|
|
527
|
+
lines.push(' counts toward it too — see "not measured" above');
|
|
528
|
+
}
|
|
529
|
+
if (c.estimated > 0) {
|
|
530
|
+
lines.push(' run with --claude to replace the estimate with published Anthropic');
|
|
531
|
+
lines.push(' counts wherever a capture still matches');
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
return lines;
|
|
535
|
+
}
|
|
536
|
+
/** "0.20×–1.92×", from whichever band the verdict was actually computed against. */
|
|
537
|
+
function ratioBand(d) {
|
|
538
|
+
const r = d.ratio ?? PUBLISHED_WIRE_TO_CLIENT_RATIO;
|
|
539
|
+
return `${r.low.toFixed(2)}×–${r.high.toFixed(2)}×`;
|
|
540
|
+
}
|
|
134
541
|
/** Human output. JSON output is the report object itself. */
|
|
135
542
|
export function formatReport(report) {
|
|
136
543
|
const lines = [];
|
|
137
544
|
lines.push(`mcp-context-cost audit · methodology ${report.methodologyVersion} · ${report.encoding} · context window ${n(report.contextWindow)}`);
|
|
545
|
+
const showClaude = !!report.claudeDivergence;
|
|
546
|
+
// Configs one session loads together share a verdict object; it answers for
|
|
547
|
+
// all of them at once, so it is printed under the first one and not repeated.
|
|
548
|
+
const verdictPrinted = new Set();
|
|
138
549
|
for (const cfg of report.configs) {
|
|
139
550
|
lines.push('');
|
|
140
551
|
lines.push(`${cfg.client} ${cfg.source}`);
|
|
@@ -143,21 +554,31 @@ export function formatReport(report) {
|
|
|
143
554
|
tools: s.toolCount === null ? '—' : String(s.toolCount),
|
|
144
555
|
tokens: s.tokens === null ? '—' : n(s.tokens),
|
|
145
556
|
share: s.share === null ? '—' : pct(s.share),
|
|
557
|
+
claude: s.claudeTokens == null ? '—' : n(s.claudeTokens),
|
|
146
558
|
}));
|
|
147
559
|
const w = {
|
|
148
560
|
name: Math.max(6, ...rows.map((r) => r.name.length), 'total'.length),
|
|
149
561
|
tools: Math.max(5, ...rows.map((r) => r.tools.length)),
|
|
150
562
|
tokens: Math.max(6, ...rows.map((r) => r.tokens.length), n(cfg.totalTokens).length),
|
|
563
|
+
claude: Math.max(6, ...rows.map((r) => r.claude.length)),
|
|
151
564
|
};
|
|
152
|
-
const line = (name, tools, tokens, share) => ` ${name.padEnd(w.name)} ${tools.padStart(w.tools)} ${tokens.padStart(w.tokens)} ${share.padStart(6)}
|
|
153
|
-
|
|
565
|
+
const line = (name, tools, tokens, share, claude) => ` ${name.padEnd(w.name)} ${tools.padStart(w.tools)} ${tokens.padStart(w.tokens)} ${share.padStart(6)}` +
|
|
566
|
+
(showClaude ? ` ${claude.padStart(w.claude)}` : '');
|
|
567
|
+
lines.push(line('server', 'tools', 'tokens', 'share', 'claude'));
|
|
154
568
|
for (const r of rows)
|
|
155
|
-
lines.push(line(r.name, r.tools, r.tokens, r.share));
|
|
156
|
-
lines.push(` ${'─'.repeat(w.name + w.tools + w.tokens + 14)}`);
|
|
157
|
-
lines.push(line('total', String(cfg.toolCount), n(cfg.totalTokens), ''));
|
|
569
|
+
lines.push(line(r.name, r.tools, r.tokens, r.share, r.claude));
|
|
570
|
+
lines.push(` ${'─'.repeat(w.name + w.tools + w.tokens + 14 + (showClaude ? w.claude + 2 : 0))}`);
|
|
571
|
+
lines.push(line('total', String(cfg.toolCount), n(cfg.totalTokens), '', ''));
|
|
158
572
|
lines.push('');
|
|
159
|
-
lines.push(
|
|
160
|
-
|
|
573
|
+
lines.push(measurementLine(cfg, report.contextWindow));
|
|
574
|
+
if (!verdictPrinted.has(cfg.deferral)) {
|
|
575
|
+
verdictPrinted.add(cfg.deferral);
|
|
576
|
+
const skippedInScope = report.configs
|
|
577
|
+
.filter((c) => c.deferral === cfg.deferral)
|
|
578
|
+
.reduce((a, c) => a + c.skipped.length, 0);
|
|
579
|
+
for (const line of deferralLines(cfg.deferral, skippedInScope))
|
|
580
|
+
lines.push(line);
|
|
581
|
+
}
|
|
161
582
|
if (cfg.heaviestTools.length) {
|
|
162
583
|
lines.push('');
|
|
163
584
|
lines.push(' heaviest tools');
|
|
@@ -166,6 +587,13 @@ export function formatReport(report) {
|
|
|
166
587
|
lines.push(` ${`${t.server} · ${t.tool}`.padEnd(tw)} ${n(t.tokens).padStart(7)}`);
|
|
167
588
|
}
|
|
168
589
|
}
|
|
590
|
+
if (cfg.trimAdvice) {
|
|
591
|
+
const names = cfg.trimAdvice.tools.map((t) => `${t.server}·${t.tool}`).join(', ');
|
|
592
|
+
lines.push('');
|
|
593
|
+
lines.push(` trim: disabling ${cfg.trimAdvice.tools.length} tool${cfg.trimAdvice.tools.length === 1 ? '' : 's'} ` +
|
|
594
|
+
`(${names}) would recover ${n(cfg.trimAdvice.recoverableTokens)} tokens ` +
|
|
595
|
+
`(${pct(cfg.trimAdvice.recoverableShare)} of this config) — if your client supports per-tool filtering.`);
|
|
596
|
+
}
|
|
169
597
|
if (cfg.skipped.length) {
|
|
170
598
|
lines.push('');
|
|
171
599
|
lines.push(' not measured');
|
|
@@ -175,6 +603,11 @@ export function formatReport(report) {
|
|
|
175
603
|
}
|
|
176
604
|
}
|
|
177
605
|
}
|
|
606
|
+
if (showClaude) {
|
|
607
|
+
lines.push('');
|
|
608
|
+
lines.push(` claude = Anthropic-request cost from the ${report.claudeDivergence.measuredAt} ${report.claudeDivergence.model} ` +
|
|
609
|
+
`divergence run, shown only where the published capture hash matches this install; '—' means no current match.`);
|
|
610
|
+
}
|
|
178
611
|
if (report.problems.length) {
|
|
179
612
|
lines.push('');
|
|
180
613
|
lines.push('problems');
|
|
@@ -182,14 +615,57 @@ export function formatReport(report) {
|
|
|
182
615
|
lines.push(` ${p}`);
|
|
183
616
|
}
|
|
184
617
|
if (report.budget) {
|
|
618
|
+
const b = report.budget;
|
|
619
|
+
lines.push('');
|
|
620
|
+
if (!b.over) {
|
|
621
|
+
const headroom = b.limit - b.worstTotal;
|
|
622
|
+
lines.push(`budget ok: ${n(b.worstTotal)} ≤ ${n(b.limit)} — ${n(headroom)} to spare`);
|
|
623
|
+
}
|
|
624
|
+
else {
|
|
625
|
+
lines.push(`BUDGET FAIL: ${n(b.worstTotal)} > ${n(b.limit)} (${b.worstSource})`);
|
|
626
|
+
const fit = b.fit;
|
|
627
|
+
if (fit && fit.drop.length) {
|
|
628
|
+
lines.push('');
|
|
629
|
+
lines.push(` over by ${n(fit.overBy)}. Heaviest-first, this is what gets you under:`);
|
|
630
|
+
for (const step of fit.drop) {
|
|
631
|
+
const verdict = step.remaining <= b.limit ? 'fits' : 'still over';
|
|
632
|
+
lines.push(` drop ${step.name.padEnd(22)} ${n(step.tokens).padStart(9)} → ${n(step.remaining).padStart(9)} ${verdict}`);
|
|
633
|
+
}
|
|
634
|
+
lines.push('');
|
|
635
|
+
if (fit.feasible && fit.keptCount === 0) {
|
|
636
|
+
// Arithmetically it fits, and the answer is useless: the only way under this
|
|
637
|
+
// limit is to run no servers at all. Saying "fits" here would be true and
|
|
638
|
+
// misleading, which is the pair this whole tool exists to keep apart.
|
|
639
|
+
lines.push(` no subset fits: every measured server would have to go. The limit is below`);
|
|
640
|
+
lines.push(` what any one of these servers costs.`);
|
|
641
|
+
}
|
|
642
|
+
else if (fit.feasible) {
|
|
643
|
+
const share = b.limit > 0 ? ` (${pct(fit.keptTokens / b.limit)} of budget)` : '';
|
|
644
|
+
lines.push(` keeps ${fit.keptCount} server(s) at ${n(fit.keptTokens)} tokens${share}`);
|
|
645
|
+
}
|
|
646
|
+
else {
|
|
647
|
+
lines.push(` even removing every measured server leaves ${n(fit.keptTokens)} — the limit is below this config's floor`);
|
|
648
|
+
}
|
|
649
|
+
lines.push('');
|
|
650
|
+
lines.push(' This is arithmetic, not advice: it cannot know which servers you need, and by');
|
|
651
|
+
lines.push(' weight alone it will sometimes name the one you cannot work without. Use --json');
|
|
652
|
+
lines.push(' for the full per-server list and pick your own order.');
|
|
653
|
+
}
|
|
654
|
+
else if (fit) {
|
|
655
|
+
lines.push(' nothing measured could be removed to get under the limit.');
|
|
656
|
+
}
|
|
657
|
+
}
|
|
658
|
+
}
|
|
659
|
+
if (report.diff)
|
|
660
|
+
lines.push(formatDiff(report.diff, report.contextWindow));
|
|
661
|
+
if (report.increaseGate) {
|
|
185
662
|
lines.push('');
|
|
186
|
-
lines.push(report.
|
|
187
|
-
? `BUDGET FAIL: ${n(report.budget.worstTotal)} > ${n(report.budget.limit)} (${report.budget.worstSource})`
|
|
188
|
-
: `budget ok: ${n(report.budget.worstTotal)} ≤ ${n(report.budget.limit)}`);
|
|
663
|
+
lines.push(formatGate(report.increaseGate));
|
|
189
664
|
}
|
|
190
665
|
lines.push('');
|
|
191
666
|
lines.push('These are wire tokens — what the server puts on the wire, counted with o200k_base. What your model is billed');
|
|
192
|
-
lines.push(
|
|
667
|
+
lines.push(`differs per provider: measured ratios run ${PUBLISHED_WIRE_TO_CLIENT_RATIO.low.toFixed(2)}×–` +
|
|
668
|
+
`${PUBLISHED_WIRE_TO_CLIENT_RATIO.high.toFixed(2)}× on Anthropic requests. See docs/METHODOLOGY.md §claude-divergence.`);
|
|
193
669
|
return lines.map((l) => l.replace(/\s+$/, '')).join('\n');
|
|
194
670
|
}
|
|
195
671
|
/** Top-level tool list across every config — used by nothing yet, handy for --json consumers. */
|
package/dist/audit/config.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type ToolSearchScope, type ToolSearchSource } from './deferral.js';
|
|
1
2
|
export interface ConfiguredServer {
|
|
2
3
|
name: string;
|
|
3
4
|
/** Which client's config this came from ('claude-desktop', 'cursor', ...). */
|
|
@@ -51,3 +52,40 @@ export interface LoadedConfig {
|
|
|
51
52
|
}
|
|
52
53
|
/** Read + parse the candidates that exist. Unreadable files are reported, not thrown. */
|
|
53
54
|
export declare function loadConfigs(candidates: ConfigCandidate[], cwd: string): LoadedConfig[];
|
|
55
|
+
/** One file Claude Code reads its `env` block from. */
|
|
56
|
+
export interface SettingsCandidate {
|
|
57
|
+
scope: ToolSearchScope;
|
|
58
|
+
path: string;
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Every settings file Claude Code takes environment variables from, highest
|
|
62
|
+
* precedence first.
|
|
63
|
+
*
|
|
64
|
+
* A client config says which servers there are; these say how the client
|
|
65
|
+
* behaves — including whether it defers their tool definitions. They are a
|
|
66
|
+
* different set of files from `configCandidates` above and are read for a
|
|
67
|
+
* different question, so they are listed separately rather than folded in.
|
|
68
|
+
*
|
|
69
|
+
* Order is Claude Code's documented settings precedence (enterprise managed
|
|
70
|
+
* policy, then project-local, then project, then user), read 2026-08-20. Paths
|
|
71
|
+
* in, candidates out — nothing here touches a disk.
|
|
72
|
+
*/
|
|
73
|
+
export declare function settingsCandidates(env: {
|
|
74
|
+
home: string;
|
|
75
|
+
cwd: string;
|
|
76
|
+
platform: NodeJS.Platform;
|
|
77
|
+
programData?: string;
|
|
78
|
+
}): SettingsCandidate[];
|
|
79
|
+
/**
|
|
80
|
+
* Read what each settings file sets, of the variables that decide deferral.
|
|
81
|
+
*
|
|
82
|
+
* Every candidate comes back, present or not, because "this file is not on the
|
|
83
|
+
* machine" and "this file was never opened" are different answers to the
|
|
84
|
+
* question the report asks, and only one of them means the default stands.
|
|
85
|
+
*
|
|
86
|
+
* Only the three tool-search variables are picked out; the rest of an `env`
|
|
87
|
+
* block — which is where a person keeps their API keys — is left where it is.
|
|
88
|
+
* A file that exists but cannot be parsed is `unreadable`, never silently
|
|
89
|
+
* treated as setting nothing.
|
|
90
|
+
*/
|
|
91
|
+
export declare function loadSettingsSources(candidates: SettingsCandidate[]): ToolSearchSource[];
|