mcp-context-cost 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +202 -43
- package/dist/audit/audit.d.ts +101 -0
- package/dist/audit/audit.js +492 -16
- package/dist/audit/config.d.ts +38 -0
- package/dist/audit/config.js +64 -0
- package/dist/audit/deferral.d.ts +346 -0
- package/dist/audit/deferral.js +376 -0
- package/dist/audit/diff.d.ts +124 -0
- package/dist/audit/diff.js +318 -0
- package/dist/audit/run.d.ts +34 -0
- package/dist/audit/run.js +45 -2
- package/dist/cli.d.ts +21 -0
- package/dist/cli.js +141 -7
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +18 -0
- package/dist/sweep/dashboard.js +74 -10
- package/dist/sweep/docker.d.ts +31 -0
- package/dist/sweep/docker.js +20 -11
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +18 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +43 -0
- package/dist/sweep/run.js +135 -37
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +57 -2
- package/package.json +3 -1
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
/** Share of the context window at which deferral activates under `auto`. */
|
|
2
|
+
export const TOOL_SEARCH_AUTO_SHARE = 0.1;
|
|
3
|
+
/** The variables that decide whether this machine's Claude Code defers. */
|
|
4
|
+
export const TOOL_SEARCH_VARS = [
|
|
5
|
+
'ENABLE_TOOL_SEARCH',
|
|
6
|
+
'CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS',
|
|
7
|
+
'ANTHROPIC_BASE_URL',
|
|
8
|
+
];
|
|
9
|
+
/** Pick the three variables that matter out of a process environment. */
|
|
10
|
+
export function toolSearchEnv(env) {
|
|
11
|
+
return {
|
|
12
|
+
ENABLE_TOOL_SEARCH: env.ENABLE_TOOL_SEARCH,
|
|
13
|
+
CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS: env.CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS,
|
|
14
|
+
ANTHROPIC_BASE_URL: env.ANTHROPIC_BASE_URL,
|
|
15
|
+
};
|
|
16
|
+
}
|
|
17
|
+
/** What `source` says for the process environment, which has no path. */
|
|
18
|
+
export const SHELL_SOURCE = '(shell environment)';
|
|
19
|
+
/** A source as it is published: what it sets, by name. */
|
|
20
|
+
export function toolSearchSourceRecord(s) {
|
|
21
|
+
return {
|
|
22
|
+
scope: s.scope,
|
|
23
|
+
source: s.source,
|
|
24
|
+
state: s.state,
|
|
25
|
+
sets: TOOL_SEARCH_VARS.filter((n) => (s.vars[n] ?? '').trim() !== ''),
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
/** The one host Claude Code treats as first-party for the tool-search fallback. */
|
|
29
|
+
const FIRST_PARTY_API_HOST = 'api.anthropic.com';
|
|
30
|
+
/** What is printed for a base URL we could not parse — a marker, not the value. */
|
|
31
|
+
const UNREADABLE_BASE_URL = '(unreadable URL)';
|
|
32
|
+
/**
|
|
33
|
+
* The hostname of a base URL, or null if it does not parse.
|
|
34
|
+
*
|
|
35
|
+
* The hostname is the whole of what the mode decision needs, and it is also the
|
|
36
|
+
* whole of what may leave this function: the rest of the value can carry a
|
|
37
|
+
* credential. A value that does not parse is not first-party, which is the
|
|
38
|
+
* reading that says tokens are paid — never the one that says they are free.
|
|
39
|
+
*/
|
|
40
|
+
function baseUrlHost(raw) {
|
|
41
|
+
try {
|
|
42
|
+
return new URL(raw).hostname.toLowerCase();
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
return null;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Read the tool-search setting out of ONE environment. Values are matched
|
|
50
|
+
* exactly as documented: an unrecognized value produces `setting-unrecognized`
|
|
51
|
+
* rather than a guess, because guessing here would print a definite verdict
|
|
52
|
+
* about tokens the reader may or may not be paying.
|
|
53
|
+
*
|
|
54
|
+
* `source` is left null here: this function is given one environment and has no
|
|
55
|
+
* way to say which of the machine's places it came from. `resolveToolSearchSources`,
|
|
56
|
+
* which does, fills it in.
|
|
57
|
+
*/
|
|
58
|
+
export function resolveToolSearch(env) {
|
|
59
|
+
const betas = env.CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS?.trim();
|
|
60
|
+
// Read first: documented as not overridable by ENABLE_TOOL_SEARCH.
|
|
61
|
+
if (betas) {
|
|
62
|
+
return {
|
|
63
|
+
mode: 'loads-upfront',
|
|
64
|
+
thresholdShare: null,
|
|
65
|
+
variable: 'CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS',
|
|
66
|
+
value: betas,
|
|
67
|
+
source: null,
|
|
68
|
+
readFromMachine: true,
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
const raw = env.ENABLE_TOOL_SEARCH?.trim();
|
|
72
|
+
const set = (mode, thresholdShare) => ({
|
|
73
|
+
mode,
|
|
74
|
+
thresholdShare,
|
|
75
|
+
variable: 'ENABLE_TOOL_SEARCH',
|
|
76
|
+
value: raw ?? null,
|
|
77
|
+
source: null,
|
|
78
|
+
readFromMachine: true,
|
|
79
|
+
});
|
|
80
|
+
if (raw === undefined || raw === '') {
|
|
81
|
+
const base = env.ANTHROPIC_BASE_URL?.trim();
|
|
82
|
+
if (base) {
|
|
83
|
+
const host = baseUrlHost(base);
|
|
84
|
+
if (host !== FIRST_PARTY_API_HOST) {
|
|
85
|
+
return {
|
|
86
|
+
mode: 'loads-upfront',
|
|
87
|
+
thresholdShare: null,
|
|
88
|
+
variable: 'ANTHROPIC_BASE_URL',
|
|
89
|
+
// The hostname alone. `base` itself is never carried out of here.
|
|
90
|
+
value: host ?? UNREADABLE_BASE_URL,
|
|
91
|
+
source: null,
|
|
92
|
+
readFromMachine: true,
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
return {
|
|
97
|
+
mode: 'defers-all',
|
|
98
|
+
thresholdShare: null,
|
|
99
|
+
variable: 'ENABLE_TOOL_SEARCH',
|
|
100
|
+
value: null,
|
|
101
|
+
source: null,
|
|
102
|
+
readFromMachine: false,
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
if (raw === 'true')
|
|
106
|
+
return set('defers-all', null);
|
|
107
|
+
if (raw === 'false')
|
|
108
|
+
return set('loads-upfront', null);
|
|
109
|
+
if (raw === 'auto')
|
|
110
|
+
return set('threshold', TOOL_SEARCH_AUTO_SHARE);
|
|
111
|
+
const custom = /^auto:(\d{1,3})$/.exec(raw);
|
|
112
|
+
if (custom) {
|
|
113
|
+
const pct = Number(custom[1]);
|
|
114
|
+
if (pct >= 0 && pct <= 100)
|
|
115
|
+
return set('threshold', pct / 100);
|
|
116
|
+
}
|
|
117
|
+
return set('setting-unrecognized', null);
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Read the posture from every place the audited machine can set it.
|
|
121
|
+
*
|
|
122
|
+
* Claude Code takes these variables from the shell it was started in AND from
|
|
123
|
+
* the `env` block of its own settings files, so an audit that reads only the
|
|
124
|
+
* shell answers the machine's question with someone else's environment. The
|
|
125
|
+
* case that made this necessary: `~/.claude/settings.json` sets
|
|
126
|
+
* `ENABLE_TOOL_SEARCH: "false"`, the shell running the audit sets nothing, and
|
|
127
|
+
* every request on that machine pays for every tool definition while the report
|
|
128
|
+
* calls it the documented default and says the tokens are not loaded at all.
|
|
129
|
+
*
|
|
130
|
+
* Among the settings files the order is Claude Code's documented precedence —
|
|
131
|
+
* enterprise managed policy, then project-local, then project, then user
|
|
132
|
+
* (Claude Code settings documentation, §"Settings files", read 2026-08-20) — so
|
|
133
|
+
* the first of them that sets a variable is the one that would win.
|
|
134
|
+
*
|
|
135
|
+
* Between the settings files and the shell there is NO order on record here, so
|
|
136
|
+
* a disagreement is refused rather than resolved: `setting-unresolved` names the
|
|
137
|
+
* variable and every place, and no verdict is given. A place that exists and
|
|
138
|
+
* could not be read is the same refusal for the same reason — what it sets is
|
|
139
|
+
* unknown, and an unknown that could flip the answer is not a default.
|
|
140
|
+
*
|
|
141
|
+
* Not visible from here at all, and so not claimed: a variable set on Claude
|
|
142
|
+
* Code's own command line.
|
|
143
|
+
*/
|
|
144
|
+
export function resolveToolSearchSources(sources) {
|
|
145
|
+
const unresolved = (reason, variable) => ({
|
|
146
|
+
mode: 'setting-unresolved',
|
|
147
|
+
thresholdShare: null,
|
|
148
|
+
variable,
|
|
149
|
+
value: null,
|
|
150
|
+
source: null,
|
|
151
|
+
readFromMachine: false,
|
|
152
|
+
unresolved: reason,
|
|
153
|
+
});
|
|
154
|
+
if (sources.some((s) => s.state === 'unreadable'))
|
|
155
|
+
return unresolved('source-unreadable', null);
|
|
156
|
+
const settings = sources.filter((s) => s.scope !== 'shell');
|
|
157
|
+
const shell = sources.find((s) => s.scope === 'shell');
|
|
158
|
+
/** The value that would win for one variable, or the fact that two places disagree. */
|
|
159
|
+
const read = (name) => {
|
|
160
|
+
const winner = settings
|
|
161
|
+
.map((s) => ({ value: (s.vars[name] ?? '').trim(), source: s.source }))
|
|
162
|
+
.find((v) => v.value !== '');
|
|
163
|
+
const shellValue = (shell?.vars[name] ?? '').trim();
|
|
164
|
+
if (winner && shellValue && winner.value !== shellValue)
|
|
165
|
+
return 'conflict';
|
|
166
|
+
if (winner)
|
|
167
|
+
return winner;
|
|
168
|
+
if (shellValue && shell)
|
|
169
|
+
return { value: shellValue, source: shell.source };
|
|
170
|
+
return null;
|
|
171
|
+
};
|
|
172
|
+
// Consulted in the same order `resolveToolSearch` consults them, so a
|
|
173
|
+
// disagreement over a variable that would not have decided anything —
|
|
174
|
+
// ANTHROPIC_BASE_URL behind an explicit ENABLE_TOOL_SEARCH — does not refuse
|
|
175
|
+
// an answer the machine actually gives.
|
|
176
|
+
const betas = read('CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS');
|
|
177
|
+
if (betas === 'conflict')
|
|
178
|
+
return unresolved('sources-disagree', 'CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS');
|
|
179
|
+
if (betas) {
|
|
180
|
+
return {
|
|
181
|
+
...resolveToolSearch({ CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS: betas.value }),
|
|
182
|
+
source: betas.source,
|
|
183
|
+
};
|
|
184
|
+
}
|
|
185
|
+
const enable = read('ENABLE_TOOL_SEARCH');
|
|
186
|
+
if (enable === 'conflict')
|
|
187
|
+
return unresolved('sources-disagree', 'ENABLE_TOOL_SEARCH');
|
|
188
|
+
if (enable) {
|
|
189
|
+
return { ...resolveToolSearch({ ENABLE_TOOL_SEARCH: enable.value }), source: enable.source };
|
|
190
|
+
}
|
|
191
|
+
const base = read('ANTHROPIC_BASE_URL');
|
|
192
|
+
if (base === 'conflict')
|
|
193
|
+
return unresolved('sources-disagree', 'ANTHROPIC_BASE_URL');
|
|
194
|
+
const resolved = resolveToolSearch(base ? { ANTHROPIC_BASE_URL: base.value } : {});
|
|
195
|
+
return { ...resolved, source: resolved.readFromMachine ? (base?.source ?? null) : null };
|
|
196
|
+
}
|
|
197
|
+
/**
|
|
198
|
+
* The band as published in this repository's own `results/divergence.json`
|
|
199
|
+
* (claude-opus-5, 2026-08-19, 20 servers). Used when no divergence run was
|
|
200
|
+
* supplied; `--claude` recomputes it from the run it fetched.
|
|
201
|
+
*/
|
|
202
|
+
export const PUBLISHED_WIRE_TO_CLIENT_RATIO = {
|
|
203
|
+
low: 0.2,
|
|
204
|
+
high: 1.92,
|
|
205
|
+
servers: 20,
|
|
206
|
+
source: 'the published claude-opus-5 divergence run',
|
|
207
|
+
};
|
|
208
|
+
/** Derive the band from a supplied divergence run, falling back to the published one. */
|
|
209
|
+
export function wireToClientRatio(run) {
|
|
210
|
+
if (!run)
|
|
211
|
+
return PUBLISHED_WIRE_TO_CLIENT_RATIO;
|
|
212
|
+
let low = Infinity;
|
|
213
|
+
let high = -Infinity;
|
|
214
|
+
let servers = 0;
|
|
215
|
+
for (const row of Object.values(run.servers)) {
|
|
216
|
+
if (!row || row.error || typeof row.claudeDelta !== 'number' || !(row.o200kFull > 0))
|
|
217
|
+
continue;
|
|
218
|
+
const ratio = row.claudeDelta / row.o200kFull;
|
|
219
|
+
low = Math.min(low, ratio);
|
|
220
|
+
high = Math.max(high, ratio);
|
|
221
|
+
servers++;
|
|
222
|
+
}
|
|
223
|
+
if (servers === 0)
|
|
224
|
+
return PUBLISHED_WIRE_TO_CLIENT_RATIO;
|
|
225
|
+
return { low, high, servers, source: `the ${run.measuredAt} ${run.model} divergence run` };
|
|
226
|
+
}
|
|
227
|
+
/** Clients this tool discovers that have no default deferral on record. */
|
|
228
|
+
const NO_DEFERRAL_ON_RECORD = new Set(['claude-desktop', 'cursor', 'vscode', 'windsurf']);
|
|
229
|
+
/**
|
|
230
|
+
* Where deferral does not apply even when the machine's setting says it should.
|
|
231
|
+
* None of these can be read from the config or the environment, so they are
|
|
232
|
+
* printed as conditions for the reader to check rather than folded into the
|
|
233
|
+
* verdict.
|
|
234
|
+
*/
|
|
235
|
+
const EXCEPTIONS = [
|
|
236
|
+
'a Microsoft Foundry deployment hosted on Azure, which rejects tool search server-side',
|
|
237
|
+
"Google Cloud's Agent Platform on a model earlier than the Claude 4.5 generation",
|
|
238
|
+
'a model without support for tool_reference blocks (before Sonnet 4.5 / Haiku 4.5 / Opus 4.5)',
|
|
239
|
+
'a server pinned with "alwaysLoad": true, whose tools load at session start regardless',
|
|
240
|
+
];
|
|
241
|
+
function estimate(servers, ratio) {
|
|
242
|
+
let low = 0;
|
|
243
|
+
let high = 0;
|
|
244
|
+
let exact = 0;
|
|
245
|
+
let estimated = 0;
|
|
246
|
+
for (const s of servers) {
|
|
247
|
+
if (typeof s.claudeTokens === 'number') {
|
|
248
|
+
// A published Anthropic count for this exact capture. It carries the tool
|
|
249
|
+
// framework overhead the API charges once per request rather than once
|
|
250
|
+
// per server, so a multi-server sum leans high by at most that overhead —
|
|
251
|
+
// far inside the band the converted servers already contribute.
|
|
252
|
+
low += s.claudeTokens;
|
|
253
|
+
high += s.claudeTokens;
|
|
254
|
+
exact++;
|
|
255
|
+
}
|
|
256
|
+
else {
|
|
257
|
+
low += s.tokens * ratio.low;
|
|
258
|
+
high += s.tokens * ratio.high;
|
|
259
|
+
estimated++;
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
return { low: Math.round(low), high: Math.round(high), exact, estimated };
|
|
263
|
+
}
|
|
264
|
+
/**
|
|
265
|
+
* Read one session's deferral position. Pure arithmetic over a built scope — no
|
|
266
|
+
* config file is re-read and no server is launched. The environment is passed
|
|
267
|
+
* in rather than read here, so the answer is reproducible from its inputs.
|
|
268
|
+
*/
|
|
269
|
+
export function evaluateDeferral(scope, opts) {
|
|
270
|
+
const wireTokens = scope.servers.reduce((a, s) => a + s.tokens, 0);
|
|
271
|
+
const isFloor = scope.skippedCount > 0;
|
|
272
|
+
const sharedMeasurements = scope.sharedMeasurements;
|
|
273
|
+
// Every field a verdict carries, at its "nothing to say" value. Each mode
|
|
274
|
+
// below overrides only what it can actually answer.
|
|
275
|
+
const base = {
|
|
276
|
+
client: scope.client,
|
|
277
|
+
sources: scope.sources,
|
|
278
|
+
wireTokens,
|
|
279
|
+
isFloor,
|
|
280
|
+
// Carried by every mode, not just the one that returns early on it below.
|
|
281
|
+
// That early return withholds `clientTokens` and `crosses`, which only
|
|
282
|
+
// threshold mode ever computes; the other modes derive nothing from the
|
|
283
|
+
// total and so have nothing to withhold. What they do all do is PRINT it,
|
|
284
|
+
// so the caveat belongs in every branch of the report — see
|
|
285
|
+
// `sharedMeasurementLines` in audit.ts, which every mode calls.
|
|
286
|
+
sharedMeasurements,
|
|
287
|
+
mechanism: null,
|
|
288
|
+
setting: null,
|
|
289
|
+
thresholdShare: null,
|
|
290
|
+
thresholdTokens: null,
|
|
291
|
+
clientTokens: null,
|
|
292
|
+
ratio: null,
|
|
293
|
+
distanceTokens: null,
|
|
294
|
+
crosses: null,
|
|
295
|
+
exceptions: [],
|
|
296
|
+
};
|
|
297
|
+
if (scope.client !== 'claude-code') {
|
|
298
|
+
return {
|
|
299
|
+
...base,
|
|
300
|
+
mode: NO_DEFERRAL_ON_RECORD.has(scope.client) ? 'no-deferral-on-record' : 'client-unknown',
|
|
301
|
+
};
|
|
302
|
+
}
|
|
303
|
+
// The shell is one place among several, not the machine. Everything Claude
|
|
304
|
+
// Code would read is resolved together, and a disagreement between them is
|
|
305
|
+
// refused rather than decided by whichever this happened to open.
|
|
306
|
+
const sources = [
|
|
307
|
+
{ scope: 'shell', source: SHELL_SOURCE, state: 'read', vars: opts.env ?? {} },
|
|
308
|
+
...(opts.settings ?? []),
|
|
309
|
+
];
|
|
310
|
+
const resolved = resolveToolSearchSources(sources);
|
|
311
|
+
const setting = {
|
|
312
|
+
variable: resolved.variable,
|
|
313
|
+
value: resolved.value,
|
|
314
|
+
source: resolved.source,
|
|
315
|
+
readFromMachine: resolved.readFromMachine,
|
|
316
|
+
sources: sources.map(toolSearchSourceRecord),
|
|
317
|
+
...(resolved.unresolved ? { unresolved: resolved.unresolved } : {}),
|
|
318
|
+
};
|
|
319
|
+
if (resolved.mode !== 'threshold') {
|
|
320
|
+
return {
|
|
321
|
+
...base,
|
|
322
|
+
mode: resolved.mode,
|
|
323
|
+
mechanism: 'tool search',
|
|
324
|
+
setting,
|
|
325
|
+
// Nothing is deferred in the other two modes, so the conditions under
|
|
326
|
+
// which deferral fails to apply are not worth printing there.
|
|
327
|
+
exceptions: resolved.mode === 'defers-all' ? EXCEPTIONS : [],
|
|
328
|
+
};
|
|
329
|
+
}
|
|
330
|
+
const thresholdShare = resolved.thresholdShare ?? TOOL_SEARCH_AUTO_SHARE;
|
|
331
|
+
const thresholdTokens = Math.round(opts.contextWindow * thresholdShare);
|
|
332
|
+
if (sharedMeasurements > 0) {
|
|
333
|
+
// There is a threshold, and no total to hold against it. A shared
|
|
334
|
+
// measurement is not a floor: a twin can serve more tools than the one that
|
|
335
|
+
// was launched or fewer, so the sum can be wrong in either direction and
|
|
336
|
+
// neither side can be ruled out. The same machine has already been seen to
|
|
337
|
+
// report 13,834 wire tokens or 392 for one stack depending on which twin
|
|
338
|
+
// the cache happened to hold, and to print a confident — opposite — side
|
|
339
|
+
// each time. This is the rule `evaluateIncreaseGate` states and `crosses`
|
|
340
|
+
// already follows: an answer that could not be established fails rather
|
|
341
|
+
// than resolves. The threshold itself is still reported, because where the
|
|
342
|
+
// line sits is known even when this stack's distance from it is not.
|
|
343
|
+
return {
|
|
344
|
+
...base,
|
|
345
|
+
mode: 'threshold',
|
|
346
|
+
mechanism: 'tool search',
|
|
347
|
+
setting,
|
|
348
|
+
thresholdShare,
|
|
349
|
+
thresholdTokens,
|
|
350
|
+
exceptions: EXCEPTIONS,
|
|
351
|
+
};
|
|
352
|
+
}
|
|
353
|
+
const ratio = wireToClientRatio(opts.divergence);
|
|
354
|
+
const clientTokens = estimate(scope.servers, ratio);
|
|
355
|
+
// At-or-above, on the documented "defers all of them once the definitions
|
|
356
|
+
// reach 10%". A range that is entirely over is over even if it is a floor:
|
|
357
|
+
// more unmeasured tokens cannot take it back under.
|
|
358
|
+
const crosses = clientTokens.low >= thresholdTokens
|
|
359
|
+
? true
|
|
360
|
+
: isFloor || clientTokens.high >= thresholdTokens
|
|
361
|
+
? null
|
|
362
|
+
: false;
|
|
363
|
+
return {
|
|
364
|
+
...base,
|
|
365
|
+
mode: 'threshold',
|
|
366
|
+
mechanism: 'tool search',
|
|
367
|
+
setting,
|
|
368
|
+
thresholdShare,
|
|
369
|
+
thresholdTokens,
|
|
370
|
+
clientTokens,
|
|
371
|
+
ratio,
|
|
372
|
+
distanceTokens: { low: clientTokens.low - thresholdTokens, high: clientTokens.high - thresholdTokens },
|
|
373
|
+
crosses,
|
|
374
|
+
exceptions: EXCEPTIONS,
|
|
375
|
+
};
|
|
376
|
+
}
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `audit --baseline <report.json>` — what a config change costs every future session.
|
|
3
|
+
*
|
|
4
|
+
* `audit` answers "what do my servers cost right now". That is a number a reader
|
|
5
|
+
* has to have an opinion about. A diff against a stored earlier report answers
|
|
6
|
+
* the question that needs no opinion at all: *this change adds 17,000 tokens to
|
|
7
|
+
* every request you will ever send from this client.* Same measurement path,
|
|
8
|
+
* same per-config discipline — a baseline is just an earlier `audit --json`.
|
|
9
|
+
*
|
|
10
|
+
* The trap this file exists to avoid: a server that measured fine before and
|
|
11
|
+
* fails to start now makes the total go DOWN. Subtracting two totals would
|
|
12
|
+
* report that as an improvement, which is the flattering reading and the true
|
|
13
|
+
* one having the same shape. So a server that changed measured-ness is never
|
|
14
|
+
* given a delta — it is named, its known side is printed, and the direction of
|
|
15
|
+
* the resulting error is stated ("understates by at least 9,246").
|
|
16
|
+
*/
|
|
17
|
+
import type { AuditConfigResult, AuditReport } from './audit.js';
|
|
18
|
+
export type ServerDeltaKind = 'added' | 'removed' | 'changed' | 'unchanged'
|
|
19
|
+
/** Measured in the baseline, not measurable now — the total understates. */
|
|
20
|
+
| 'unmeasured-now'
|
|
21
|
+
/** Not measurable in the baseline, measured now — the increase overstates. */
|
|
22
|
+
| 'unmeasured-before'
|
|
23
|
+
/** Present and unmeasured in both runs — contributes 0 to both totals, but hides cost. */
|
|
24
|
+
| 'unmeasured-both';
|
|
25
|
+
export interface ServerDelta {
|
|
26
|
+
name: string;
|
|
27
|
+
kind: ServerDeltaKind;
|
|
28
|
+
/** Baseline tokens; `null` when absent from the baseline or unmeasured in it. */
|
|
29
|
+
before: number | null;
|
|
30
|
+
/** Current tokens; `null` when gone from the config or unmeasured now. */
|
|
31
|
+
after: number | null;
|
|
32
|
+
/** Signed change. `null` whenever the two sides are not the same kind of number. */
|
|
33
|
+
delta: number | null;
|
|
34
|
+
/** Why a delta is missing, in a sentence a reader can act on. */
|
|
35
|
+
note?: string;
|
|
36
|
+
}
|
|
37
|
+
export interface ConfigDiff {
|
|
38
|
+
client: string;
|
|
39
|
+
source: string;
|
|
40
|
+
/** How this config was paired with a baseline config. */
|
|
41
|
+
matchedBy: 'source' | 'sole-config' | 'unmatched';
|
|
42
|
+
beforeTotal: number | null;
|
|
43
|
+
afterTotal: number;
|
|
44
|
+
/** afterTotal - beforeTotal, or `null` when there is no baseline to subtract. */
|
|
45
|
+
delta: number | null;
|
|
46
|
+
beforeShare: number | null;
|
|
47
|
+
afterShare: number;
|
|
48
|
+
/**
|
|
49
|
+
* True when `delta` is the exact change in measured cost. False when a server
|
|
50
|
+
* crossed the measured/unmeasured line, which moves the total for a reason
|
|
51
|
+
* that is not a config change.
|
|
52
|
+
*/
|
|
53
|
+
exact: boolean;
|
|
54
|
+
/** Tokens the diff is known to be missing, and which way it leans. */
|
|
55
|
+
understatedBy: number;
|
|
56
|
+
overstatedBy: number;
|
|
57
|
+
servers: ServerDelta[];
|
|
58
|
+
}
|
|
59
|
+
export interface AuditDiff {
|
|
60
|
+
baselineGeneratedAt: string;
|
|
61
|
+
baselineMethodologyVersion: string;
|
|
62
|
+
/** False when something makes the two reports incommensurable at all (methodology bump). */
|
|
63
|
+
comparable: boolean;
|
|
64
|
+
/** Baseline configs that no current config matched — never silently dropped. */
|
|
65
|
+
droppedConfigs: {
|
|
66
|
+
client: string;
|
|
67
|
+
source: string;
|
|
68
|
+
totalTokens: number;
|
|
69
|
+
}[];
|
|
70
|
+
warnings: string[];
|
|
71
|
+
configs: ConfigDiff[];
|
|
72
|
+
/**
|
|
73
|
+
* The largest per-config increase. Per config, never merged: a context window
|
|
74
|
+
* belongs to one client session, so a portfolio-wide "total delta" would
|
|
75
|
+
* describe a session nobody runs. `null` when nothing could be compared.
|
|
76
|
+
*/
|
|
77
|
+
worstIncrease: {
|
|
78
|
+
source: string;
|
|
79
|
+
delta: number;
|
|
80
|
+
} | null;
|
|
81
|
+
}
|
|
82
|
+
/** Parse and shape-check a stored report. A baseline that cannot be read is never "no change". */
|
|
83
|
+
export declare function parseBaselineReport(text: string): {
|
|
84
|
+
report: AuditReport | null;
|
|
85
|
+
problem?: string;
|
|
86
|
+
};
|
|
87
|
+
export declare function diffConfig(before: AuditConfigResult | null, after: AuditConfigResult, matchedBy: ConfigDiff['matchedBy']): ConfigDiff;
|
|
88
|
+
/**
|
|
89
|
+
* Pair current configs with baseline configs.
|
|
90
|
+
*
|
|
91
|
+
* Exact source path first. Then one deliberate fallback: if each side has
|
|
92
|
+
* exactly one config, they are the same config seen from two machines — the CI
|
|
93
|
+
* case, where a baseline recorded at /Users/… meets a checkout at /home/runner/….
|
|
94
|
+
* Anything looser would pair two unrelated clients and call the difference a
|
|
95
|
+
* change, so everything else stays unmatched and says so.
|
|
96
|
+
*/
|
|
97
|
+
export declare function pairConfigs(before: AuditConfigResult[], after: AuditConfigResult[]): {
|
|
98
|
+
pairs: {
|
|
99
|
+
before: AuditConfigResult | null;
|
|
100
|
+
after: AuditConfigResult;
|
|
101
|
+
matchedBy: ConfigDiff['matchedBy'];
|
|
102
|
+
}[];
|
|
103
|
+
dropped: AuditConfigResult[];
|
|
104
|
+
};
|
|
105
|
+
export declare function buildDiff(baseline: AuditReport, current: AuditReport): AuditDiff;
|
|
106
|
+
export declare function formatDiff(diff: AuditDiff, contextWindow: number): string;
|
|
107
|
+
export interface IncreaseGate {
|
|
108
|
+
limit: number;
|
|
109
|
+
pass: boolean;
|
|
110
|
+
/** The increase the gate measured, when it got far enough to measure one. */
|
|
111
|
+
increase: number | null;
|
|
112
|
+
reasons: string[];
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* `--max-increase N` — the CI gate. Fails on an increase over the limit, and
|
|
116
|
+
* equally on any reason the increase could not be established.
|
|
117
|
+
*
|
|
118
|
+
* That second half is the point. A gate that passes when a server failed to
|
|
119
|
+
* start, or when the baseline covered a config this run never found, is a green
|
|
120
|
+
* check on a question nobody asked. Everything this portfolio has learned says
|
|
121
|
+
* unchecked must not read as clean, so an inexact diff fails and names why.
|
|
122
|
+
*/
|
|
123
|
+
export declare function evaluateIncreaseGate(diff: AuditDiff, limit: number): IncreaseGate;
|
|
124
|
+
export declare function formatGate(gate: IncreaseGate): string;
|