mcp-context-cost 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +101 -33
  2. package/dist/audit/audit.d.ts +48 -0
  3. package/dist/audit/audit.js +392 -7
  4. package/dist/audit/config.d.ts +46 -0
  5. package/dist/audit/config.js +86 -2
  6. package/dist/audit/deferral.d.ts +366 -0
  7. package/dist/audit/deferral.js +403 -0
  8. package/dist/audit/run.d.ts +22 -0
  9. package/dist/audit/run.js +17 -1
  10. package/dist/cli.js +11 -0
  11. package/dist/core/adoption.d.ts +226 -0
  12. package/dist/core/adoption.js +432 -0
  13. package/dist/core/canonical.d.ts +6 -0
  14. package/dist/core/canonical.js +3 -0
  15. package/dist/core/index.d.ts +1 -0
  16. package/dist/core/index.js +1 -0
  17. package/dist/core/session-start.d.ts +102 -0
  18. package/dist/core/session-start.js +186 -0
  19. package/dist/core/types.d.ts +8 -0
  20. package/dist/sweep/client.d.ts +6 -1
  21. package/dist/sweep/client.js +1 -0
  22. package/dist/sweep/dashboard.d.ts +10 -0
  23. package/dist/sweep/dashboard.js +32 -16
  24. package/dist/sweep/docker.d.ts +23 -0
  25. package/dist/sweep/docker.js +16 -12
  26. package/dist/sweep/harness-guard.d.ts +57 -0
  27. package/dist/sweep/harness-guard.js +144 -0
  28. package/dist/sweep/history.d.ts +33 -1
  29. package/dist/sweep/history.js +60 -5
  30. package/dist/sweep/regen.js +6 -1
  31. package/dist/sweep/report.d.ts +16 -0
  32. package/dist/sweep/report.js +79 -5
  33. package/dist/sweep/run.d.ts +41 -0
  34. package/dist/sweep/run.js +129 -36
  35. package/dist/sweep/server-pages.js +31 -6
  36. package/dist/sweep/session-start.d.ts +3 -0
  37. package/dist/sweep/session-start.js +103 -0
  38. package/dist/sweep/shard.d.ts +41 -0
  39. package/dist/sweep/shard.js +58 -0
  40. package/dist/sweep/sweep-all.js +56 -2
  41. package/package.json +3 -1
@@ -0,0 +1,366 @@
1
+ /**
2
+ * Whether the client reading this config loads MCP tool definitions up front —
3
+ * or defers them until the model reaches for one.
4
+ *
5
+ * The headline audit number is what a session pays to put every tool definition
6
+ * in the context window. Whether it pays that is a property of the client and
7
+ * of the machine it runs on, not of the servers. So this module answers with
8
+ * three separate things, because collapsing them is how the first version of
9
+ * this got the common case wrong:
10
+ *
11
+ * 1. **What mode is in force.** Claude Code's default is to defer EVERY MCP
12
+ * tool definition, unconditionally — there is no threshold in the default
13
+ * case. A threshold exists only in the opt-in `auto` mode, and `auto:N`
14
+ * lets that percentage be anything from 0 to 100. Which mode is in force
15
+ * is decided by environment variables on the machine being audited, so
16
+ * they are read rather than assumed.
17
+ * 2. **Where the threshold sits**, when there is one at all.
18
+ * 3. **Which side of it this stack falls on** — as a range, not a point,
19
+ * because the audit's number and the threshold are counted in different
20
+ * units (see `wireToClientRatio` below).
21
+ *
22
+ * Sources, and their dates, because these are claims about someone else's
23
+ * product and they will rot:
24
+ *
25
+ * - Claude Code MCP documentation, §"Scale with MCP tool search", read
26
+ * 2026-08-20. "Tool search is enabled by default. MCP tools are deferred
27
+ * rather than loaded into context upfront." The `ENABLE_TOOL_SEARCH` table:
28
+ * unset → "All MCP tools deferred and loaded on demand"; `true` → all
29
+ * deferred; `auto` → "Threshold mode: Claude Code loads the tools it would
30
+ * otherwise defer upfront while their definitions total less than 10% of
31
+ * the context window, and defers all of them once the definitions reach
32
+ * 10%"; `auto:N` → "Threshold mode with a custom percentage, where `N` is
33
+ * 0-100"; `false` → "All MCP tools loaded upfront, no deferral". Deferral
34
+ * also falls back to upfront loading behind a non-first-party
35
+ * `ANTHROPIC_BASE_URL`, on a Microsoft Foundry deployment hosted on Azure,
36
+ * and on Google Cloud Agent Platform models earlier than the Claude 4.5
37
+ * generation; `CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS` "keeps tool search
38
+ * off. You can't override it by setting `ENABLE_TOOL_SEARCH` yourself."
39
+ * A server with `alwaysLoad: true` loads at session start regardless.
40
+ *
41
+ * No default deferral is on record here for the other four clients this tool
42
+ * discovers. That is an absence of a record, not a measurement of those
43
+ * clients, and it is printed as such — the same rule the rest of this project
44
+ * follows for a value it has not observed.
45
+ */
46
+ import type { DivergenceRun } from '../core/divergence.js';
47
+ /** Share of the context window at which deferral activates under `auto`. */
48
+ export declare const TOOL_SEARCH_AUTO_SHARE = 0.1;
49
+ /** The variables that decide whether this machine's Claude Code defers. */
50
+ export declare const TOOL_SEARCH_VARS: readonly ['ENABLE_TOOL_SEARCH', 'CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS', 'ANTHROPIC_BASE_URL'];
51
+ export type ToolSearchVar = (typeof TOOL_SEARCH_VARS)[number];
52
+ /** The env vars that decide whether this machine's Claude Code defers. */
53
+ export interface ToolSearchEnv {
54
+ ENABLE_TOOL_SEARCH?: string;
55
+ CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS?: string;
56
+ ANTHROPIC_BASE_URL?: string;
57
+ }
58
+ /** Pick the three variables that matter out of a process environment. */
59
+ export declare function toolSearchEnv(env: Record<string, string | undefined>): ToolSearchEnv;
60
+ /**
61
+ * A place the audited machine can set those variables.
62
+ *
63
+ * Two kinds, because Claude Code reads two kinds: the environment of the shell
64
+ * it was started in, and the `env` block of its own settings files. Reading
65
+ * only the first is how this reported the documented default — "these tokens
66
+ * are NOT loaded up front at any size" — at a machine that had switched
67
+ * deferral off in `~/.claude/settings.json`.
68
+ */
69
+ export type ToolSearchScope = 'shell' | 'managed-settings' | 'local-settings' | 'project-settings' | 'user-settings';
70
+ /** What `source` says for the process environment, which has no path. */
71
+ export declare const SHELL_SOURCE = "(shell environment)";
72
+ export interface ToolSearchSource {
73
+ scope: ToolSearchScope;
74
+ /** Path of the settings file, or `SHELL_SOURCE`. */
75
+ source: string;
76
+ /**
77
+ * `read` — consulted, and `vars` is what it sets.
78
+ * `absent` — not on this machine, so it sets nothing.
79
+ * `unreadable` — it exists and could not be read: what it sets is UNKNOWN,
80
+ * which is not the same as nothing and is never resolved as though it were.
81
+ */
82
+ state: 'read' | 'absent' | 'unreadable';
83
+ /** What this place sets, of the three. Values are read here, never reported. */
84
+ vars: ToolSearchEnv;
85
+ /**
86
+ * Variables this place sets to something that is not a value this audit can
87
+ * read — an env block holding a JSON boolean, a number, or null.
88
+ *
89
+ * The file parsed and the variable IS set in it; what it is set to is
90
+ * unknown. That is not the same as unset, and reading it as unset is how a
91
+ * machine whose `~/.claude/settings.json` held `"ENABLE_TOOL_SEARCH": false`
92
+ * — the boolean, not the string — was told the documented default stands and
93
+ * these tokens are never loaded up front.
94
+ */
95
+ unreadable?: ToolSearchVar[];
96
+ }
97
+ /**
98
+ * A source as it appears in a report: names of what it sets, never values.
99
+ *
100
+ * This is the record that lets a reader tell "nothing is set anywhere" from
101
+ * "that place was never opened" — the two states `readFromMachine: false`
102
+ * cannot distinguish on its own.
103
+ */
104
+ export interface ToolSearchSourceRecord {
105
+ scope: ToolSearchScope;
106
+ source: string;
107
+ state: ToolSearchSource['state'];
108
+ /** Variable NAMES set here. A value can be a base URL carrying a credential. */
109
+ sets: ToolSearchVar[];
110
+ /**
111
+ * Variable NAMES this place sets to something unreadable, when it does.
112
+ * Omitted otherwise, so the common source record keeps its shape — a reader
113
+ * meets this field only where there is an unknown to meet.
114
+ */
115
+ unreadable?: ToolSearchVar[];
116
+ }
117
+ /** A source as it is published: what it sets, by name. */
118
+ export declare function toolSearchSourceRecord(s: ToolSearchSource): ToolSearchSourceRecord;
119
+ export type DeferralMode =
120
+ /** Every MCP tool definition is deferred, at any size. No threshold applies. */
121
+ 'defers-all'
122
+ /** Deferral activates only once the definitions reach a share of the window. */
123
+ | 'threshold'
124
+ /** Deferral is off here: every definition is in context at session start. */
125
+ | 'loads-upfront'
126
+ /** ENABLE_TOOL_SEARCH holds a value Claude Code does not document. */
127
+ | 'setting-unrecognized'
128
+ /** More than one place sets the variable, or one of them could not be read. */
129
+ | 'setting-unresolved'
130
+ /** A client we know about, with no default deferral on record. */
131
+ | 'no-deferral-on-record'
132
+ /** `--config <path>`: the file was read, but which client reads it is unknown. */
133
+ | 'client-unknown';
134
+ /** How the mode was decided — printed, so a reader can check it against their own shell. */
135
+ export interface ToolSearchSetting {
136
+ /** The variable that decided it, or null when nothing was set and the default stands. */
137
+ variable: string | null;
138
+ /**
139
+ * What is printed for that variable. Null when the decision came from the
140
+ * documented default.
141
+ *
142
+ * This is the value as read for `ENABLE_TOOL_SEARCH` and
143
+ * `CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS`, whose values are settings and not
144
+ * secrets. For `ANTHROPIC_BASE_URL` it is the hostname alone — never the
145
+ * value — because a base URL routed through a proxy commonly carries a
146
+ * credential in its userinfo or query, and a report is a thing meant to be
147
+ * shared (`examples/github-actions.yml` runs it in CI, `--baseline` reads a
148
+ * committed one). Same rule as `config.ts`: values are read, never written to
149
+ * a report.
150
+ */
151
+ value: string | null;
152
+ /** True when a variable on the audited machine decided this, false for the default. */
153
+ readFromMachine: boolean;
154
+ /**
155
+ * Which place the deciding value was read from — a settings file's path, or
156
+ * `SHELL_SOURCE`. Null when nothing was set anywhere and the documented
157
+ * default stands.
158
+ */
159
+ source: string | null;
160
+ /**
161
+ * Every place that was consulted, in the order Claude Code would take them,
162
+ * and what each one sets — by name. Printed and serialized so a reader can
163
+ * see what was opened, and a `--json` consumer can tell a variable that is
164
+ * set nowhere from a place this audit never read.
165
+ */
166
+ sources: ToolSearchSourceRecord[];
167
+ /** Set only in `setting-unresolved`: why no mode could be read off them. */
168
+ unresolved?: 'sources-disagree' | 'source-unreadable' | 'value-unreadable';
169
+ }
170
+ interface ResolvedToolSearch extends Omit<ToolSearchSetting, 'sources'> {
171
+ mode: Extract<DeferralMode, 'defers-all' | 'threshold' | 'loads-upfront' | 'setting-unrecognized' | 'setting-unresolved'>;
172
+ thresholdShare: number | null;
173
+ }
174
+ /**
175
+ * Read the tool-search setting out of ONE environment. Values are matched
176
+ * exactly as documented: an unrecognized value produces `setting-unrecognized`
177
+ * rather than a guess, because guessing here would print a definite verdict
178
+ * about tokens the reader may or may not be paying.
179
+ *
180
+ * `source` is left null here: this function is given one environment and has no
181
+ * way to say which of the machine's places it came from. `resolveToolSearchSources`,
182
+ * which does, fills it in.
183
+ */
184
+ export declare function resolveToolSearch(env: ToolSearchEnv): ResolvedToolSearch;
185
+ /**
186
+ * Read the posture from every place the audited machine can set it.
187
+ *
188
+ * Claude Code takes these variables from the shell it was started in AND from
189
+ * the `env` block of its own settings files, so an audit that reads only the
190
+ * shell answers the machine's question with someone else's environment. The
191
+ * case that made this necessary: `~/.claude/settings.json` sets
192
+ * `ENABLE_TOOL_SEARCH: "false"`, the shell running the audit sets nothing, and
193
+ * every request on that machine pays for every tool definition while the report
194
+ * calls it the documented default and says the tokens are not loaded at all.
195
+ *
196
+ * Among the settings files the order is Claude Code's documented precedence —
197
+ * enterprise managed policy, then project-local, then project, then user
198
+ * (Claude Code settings documentation, §"Settings files", read 2026-08-20) — so
199
+ * the first of them that sets a variable is the one that would win.
200
+ *
201
+ * Between the settings files and the shell there is NO order on record here, so
202
+ * a disagreement is refused rather than resolved: `setting-unresolved` names the
203
+ * variable and every place, and no verdict is given. A place that exists and
204
+ * could not be read is the same refusal for the same reason — what it sets is
205
+ * unknown, and an unknown that could flip the answer is not a default. So is a
206
+ * place that parsed and sets the deciding variable to something that is not a
207
+ * readable value: the variable is set there, and dropping it leaves the report
208
+ * arguing from a silence that is not silent.
209
+ *
210
+ * Not visible from here at all, and so not claimed: a variable set on Claude
211
+ * Code's own command line.
212
+ */
213
+ export declare function resolveToolSearchSources(sources: ToolSearchSource[]): ResolvedToolSearch;
214
+ /**
215
+ * The factor between the number this audit counts and the number the threshold
216
+ * is counted in.
217
+ *
218
+ * The audit's total is o200k_base over the bytes a server puts on the wire. The
219
+ * threshold is a share of the context window measured in what the client
220
+ * actually sends to the API — the name/description/input_schema projection,
221
+ * counted by Anthropic's tokenizer, plus the tool framework overhead. Those are
222
+ * not the same number and the gap is not small: across the published
223
+ * divergence run it runs from 0.20× to 1.92×, so a single stack total maps to a
224
+ * range roughly ten times as wide as itself. Comparing the wire number directly
225
+ * against the threshold understates the deferrable side for schema-heavy
226
+ * servers and overstates it for metadata-heavy ones, in one direction each.
227
+ */
228
+ export interface WireToClientRatio {
229
+ low: number;
230
+ high: number;
231
+ /** How many servers the band was measured across, for the printed caveat. */
232
+ servers: number;
233
+ /** The run it came from, so a reader can date it. */
234
+ source: string;
235
+ }
236
+ /**
237
+ * The band as published in this repository's own `results/divergence.json`
238
+ * (claude-opus-5, 2026-08-19, 20 servers). Used when no divergence run was
239
+ * supplied; `--claude` recomputes it from the run it fetched.
240
+ */
241
+ export declare const PUBLISHED_WIRE_TO_CLIENT_RATIO: WireToClientRatio;
242
+ /** Derive the band from a supplied divergence run, falling back to the published one. */
243
+ export declare function wireToClientRatio(run?: DivergenceRun | null): WireToClientRatio;
244
+ /** One measured server, as the deferral arithmetic needs it. */
245
+ export interface DeferralServer {
246
+ /** o200k tokens over the wire capture — the audit's own unit. */
247
+ tokens: number;
248
+ /**
249
+ * Anthropic's own count for this server from a current divergence row, when
250
+ * `--claude` supplied one. `null` means no current match, `undefined` means
251
+ * the join was not requested — either way it is converted through the band.
252
+ */
253
+ claudeTokens?: number | null;
254
+ }
255
+ /**
256
+ * The configs one session of one client loads together.
257
+ *
258
+ * Claude Code reads both `~/.claude.json` and `<cwd>/.mcp.json` into a single
259
+ * session, so they get one verdict against their sum rather than two verdicts
260
+ * each judged alone. The report still totals each config file separately — a
261
+ * context window belongs to one session, which is the argument for adding these
262
+ * two together, not for adding one client's servers to another's.
263
+ */
264
+ export interface DeferralScope {
265
+ client: string;
266
+ /** Every config file this verdict covers. */
267
+ sources: string[];
268
+ servers: DeferralServer[];
269
+ /** Entries discovered across those configs that produced no number. */
270
+ skippedCount: number;
271
+ /**
272
+ * Entries here whose number came from a measurement shared with another entry
273
+ * that differs only in its environment.
274
+ *
275
+ * Measurements are cached per command line, so two entries running the same
276
+ * command under different environments are launched once and both carry the
277
+ * one number. Environment decides what a server serves — `GITHUB_TOOLSETS` on
278
+ * `github-mcp-server` selects which toolsets it lists — so that number belongs
279
+ * to at most one of them, and which one is not knowable from here.
280
+ *
281
+ * Counted rather than flagged so the report can say how much of the stack it
282
+ * covers. Every entry sharing such a measurement counts, including whichever
283
+ * one was really launched.
284
+ */
285
+ sharedMeasurements: number;
286
+ }
287
+ /** What the client would count for this stack, as a range. */
288
+ export interface ClientSideEstimate {
289
+ low: number;
290
+ high: number;
291
+ /** Servers taken from a published Anthropic count rather than converted. */
292
+ exact: number;
293
+ /** Servers converted through the ratio band. */
294
+ estimated: number;
295
+ }
296
+ export interface DeferralVerdict {
297
+ client: string;
298
+ mode: DeferralMode;
299
+ /** What the client calls the mechanism, for a reader who wants to look it up. */
300
+ mechanism: string | null;
301
+ /** Every config file this one verdict covers. */
302
+ sources: string[];
303
+ /** Which variable decided the mode, and whether it was read or defaulted. */
304
+ setting: ToolSearchSetting | null;
305
+ /** Null whenever no threshold applies — which includes the default case. */
306
+ thresholdShare: number | null;
307
+ thresholdTokens: number | null;
308
+ /**
309
+ * o200k tokens summed across the scope — what this audit measured.
310
+ *
311
+ * Read together with `sharedMeasurements`: where that is non-zero this sum
312
+ * counts one measurement for several entries, so it is not the stack's total
313
+ * and nothing here is derived from it.
314
+ */
315
+ wireTokens: number;
316
+ /** Null when there is no threshold to compare against, or no total to convert. */
317
+ clientTokens: ClientSideEstimate | null;
318
+ ratio: WireToClientRatio | null;
319
+ /**
320
+ * True when the stack total is a lower bound rather than a count — some
321
+ * server in this scope could not be measured, and a session would still load
322
+ * whatever it serves. Absent is unknown, never zero.
323
+ */
324
+ isFloor: boolean;
325
+ /** clientTokens − thresholdTokens, at each end of the range. Positive is over. */
326
+ distanceTokens: {
327
+ low: number;
328
+ high: number;
329
+ } | null;
330
+ /**
331
+ * true = deferral activates, false = it does not, null = cannot be said.
332
+ * Null has four causes, all of them real: there is no threshold rule to be
333
+ * on a side of, the unit conversion straddles the threshold, an unmeasured
334
+ * server could carry an under-threshold stack over, or the stack has no
335
+ * established total at all (`sharedMeasurements`).
336
+ */
337
+ crosses: boolean | null;
338
+ /**
339
+ * How many entries in this scope carry a number measured for a twin that
340
+ * differs only in environment. Non-zero means the sum above is not this
341
+ * stack's total, in either direction, so no side of the threshold is claimed.
342
+ */
343
+ sharedMeasurements: number;
344
+ /** Conditions this cannot read, under which a deferring client pays in full. */
345
+ exceptions: string[];
346
+ }
347
+ /**
348
+ * Read one session's deferral position. Pure arithmetic over a built scope — no
349
+ * config file is re-read and no server is launched. The environment is passed
350
+ * in rather than read here, so the answer is reproducible from its inputs.
351
+ */
352
+ export declare function evaluateDeferral(scope: DeferralScope, opts: {
353
+ contextWindow: number;
354
+ /** The audited machine's SHELL variables. Omitted means the shell set nothing. */
355
+ env?: ToolSearchEnv;
356
+ /**
357
+ * The Claude Code settings files read on that machine, highest precedence
358
+ * first — the other place these variables come from. Omitted means they
359
+ * were not read here, which is published as such rather than as an absence
360
+ * of settings: see `ToolSearchSetting.sources`.
361
+ */
362
+ settings?: ToolSearchSource[];
363
+ /** Supplied by `--claude`; sharpens the unit conversion where rows match. */
364
+ divergence?: DivergenceRun | null;
365
+ }): DeferralVerdict;
366
+ export {};