mcp-context-cost 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,383 @@
1
+ /**
2
+ * The front pages' numbers, written by the same regeneration that writes the
3
+ * leaderboard — and checked against the same data in the suite.
4
+ *
5
+ * README and docs/index.md state numbers as prose: how many candidates, how
6
+ * many measured, the span, the sample tables, the Claude pair, the verify
7
+ * transcript. Those sentences were written by hand, so they were true on the
8
+ * day they were written and drifted with every scheduled re-sweep — by
9
+ * 2026-09-03 the leaderboard said 68 measured while both pages said 69, the
10
+ * exact front-page-contradicts-the-data failure repaired by hand once before
11
+ * (2026-08-20) and re-created by the first sweep after it.
12
+ *
13
+ * The deferral tables got the durable fix first: a test reads the page's own
14
+ * words against the resolver, so either side moving alone is a red check. That
15
+ * works there because the tables change only when code changes. These numbers
16
+ * change when *data* changes, on a schedule, with no human in the loop — a
17
+ * check alone would schedule its own red main. So the numbers get the
18
+ * leaderboard's treatment instead: regen patches them from results/, the
19
+ * scheduled jobs commit the pages beside the data, and the suite asserts the
20
+ * committed pages already agree — which fires only when someone changes data
21
+ * without running regen, or rewords a sentence regen maintains.
22
+ *
23
+ * Each claim is a template: fixed words with slots. The words have to be on
24
+ * the page as written (a missing anchor is a loud failure naming the claim,
25
+ * never a silent skip), and only the slots are ever rewritten — spliced in
26
+ * place, so the page's own line wrapping survives.
27
+ */
28
+ import { readFileSync, writeFileSync } from 'node:fs';
29
+ import { join } from 'node:path';
30
+ import { DEFAULT_CONTEXT_WINDOW } from '../audit/audit.js';
31
+ import { fieldSelectionShare, isCurrent } from '../core/divergence.js';
32
+ import { sessionStartLoad } from '../core/session-start.js';
33
+ import { isGood } from './harness-guard.js';
34
+ import { loadDivergence, loadRows, loadSessionStartRun } from './report.js';
35
+ /** Named in README's sample table — the choice is editorial, the numbers are not. */
36
+ export const SAMPLE_SERVERS = [
37
+ 'github',
38
+ 'xcodebuildmcp',
39
+ 'brave-search',
40
+ 'notion',
41
+ 'playwright',
42
+ 'filesystem',
43
+ 'markitdown',
44
+ ];
45
+ export function floorToTwoSignificant(n) {
46
+ const whole = Math.floor(n);
47
+ if (whole < 100)
48
+ return whole;
49
+ const magnitude = 10 ** (Math.floor(Math.log10(whole)) - 1);
50
+ return Math.floor(whole / magnitude) * magnitude;
51
+ }
52
+ export function computePublishedStats(entries, root = process.cwd()) {
53
+ const rows = loadRows(entries, root);
54
+ const div = loadDivergence(root);
55
+ const ss = loadSessionStartRun(root);
56
+ const measured = rows
57
+ .filter((r) => r.m !== null && isGood(r.m.status))
58
+ .filter((r) => typeof r.m.totalTokens === 'number')
59
+ .sort((a, b) => b.m.totalTokens - a.m.totalTokens);
60
+ if (measured.length < 2)
61
+ throw new Error('fewer than two measured servers on disk — published stats cannot be computed');
62
+ const asPair = (r) => ({ name: r.entry.name, tokens: r.m.totalTokens });
63
+ const max = asPair(measured[0]);
64
+ const min = asPair(measured[measured.length - 1]);
65
+ if (min.tokens <= 0)
66
+ throw new Error(`cheapest measured server (${min.name}) has no positive token count`);
67
+ const sample = {};
68
+ for (const name of SAMPLE_SERVERS) {
69
+ const r = measured.find((x) => x.entry.name === name);
70
+ if (!r || typeof r.m.toolCount !== 'number') {
71
+ throw new Error(`README's sample table names ${name}, which has no current measurement`);
72
+ }
73
+ sample[name] = { tokens: r.m.totalTokens, tools: r.m.toolCount };
74
+ }
75
+ if (!div)
76
+ throw new Error('results/divergence.json is missing — README states its numbers');
77
+ const divRow = (name) => {
78
+ const d = div.servers[name];
79
+ if (!d)
80
+ throw new Error(`README's Claude table names ${name}, which is not in the divergence run`);
81
+ return d;
82
+ };
83
+ const github = divRow('github');
84
+ const notion = divRow('notion');
85
+ const githubShare = fieldSelectionShare(github);
86
+ if (githubShare === null)
87
+ throw new Error('divergence run carries no o200k counts for github');
88
+ const runRows = Object.values(div.servers);
89
+ const shares = runRows.map((r) => fieldSelectionShare(r)).filter((s) => s !== null);
90
+ const ratios = runRows
91
+ .filter((r) => typeof r.claudeDelta === 'number' && r.claudeDelta > 0 && r.o200kFull > 0)
92
+ .map((r) => r.claudeDelta / r.o200kFull);
93
+ if (shares.length === 0 || ratios.length === 0) {
94
+ throw new Error('divergence run carries no usable rows — METHODOLOGY states its ranges');
95
+ }
96
+ const withClaude = measured.filter((r) => isCurrent(div.servers[r.entry.name], r.m.canonicalSha256));
97
+ const heaviest = [...withClaude].sort((a, b) => div.servers[b.entry.name].claudeDelta - div.servers[a.entry.name].claudeDelta)[0];
98
+ const costlier = measured.filter((r) => {
99
+ const load = sessionStartLoad(r.m, ss?.servers[r.entry.name]);
100
+ return load !== null && load.totalTokens >= r.m.totalTokens;
101
+ });
102
+ const githubRow = rows.find((r) => r.entry.name === 'github');
103
+ if (!githubRow?.m || typeof githubRow.m.totalTokens !== 'number') {
104
+ throw new Error('README quotes `verify` on results/github, which has no current measurement');
105
+ }
106
+ return {
107
+ candidateTotal: rows.length,
108
+ measuredCount: measured.length,
109
+ max,
110
+ second: asPair(measured[1]),
111
+ min,
112
+ spanTimes: floorToTwoSignificant(max.tokens / min.tokens),
113
+ maxContextSharePct: Math.round((max.tokens / DEFAULT_CONTEXT_WINDOW) * 100),
114
+ sample,
115
+ claude: {
116
+ runSize: Object.keys(div.servers).length,
117
+ currentCount: withClaude.length,
118
+ heaviestClaudeName: heaviest?.entry.name ?? null,
119
+ github: {
120
+ badgeTokens: github.o200kFull,
121
+ mappedTokens: github.o200kMapped,
122
+ claudeTokens: github.claudeDelta,
123
+ droppedPct: Math.round(githubShare * 100),
124
+ },
125
+ notion: { badgeTokens: notion.o200kFull, claudeTokens: notion.claudeDelta },
126
+ shareMin: Math.min(...shares),
127
+ shareMax: Math.max(...shares),
128
+ ratioMin: Math.min(...ratios),
129
+ ratioMax: Math.max(...ratios),
130
+ },
131
+ deferralCostlierCount: costlier.length,
132
+ verify: {
133
+ serverName: githubRow.m.serverName ?? 'github',
134
+ tokens: githubRow.m.totalTokens,
135
+ },
136
+ };
137
+ }
138
+ export const PAGE_FILES = ['README.md', 'docs/index.md', 'docs/METHODOLOGY.md'];
139
+ const fmt = (n) => n.toLocaleString('en-US');
140
+ export const PAGE_CLAIMS = [
141
+ {
142
+ file: 'README.md',
143
+ id: 'span',
144
+ template: 'across the {n} servers measured, cost spans **{n}×**, from `{w}` at {n} tokens to `{w}` at {n}.',
145
+ values: (s) => [fmt(s.measuredCount), fmt(s.spanTimes), s.min.name, fmt(s.min.tokens), s.max.name, fmt(s.max.tokens)],
146
+ },
147
+ {
148
+ file: 'README.md',
149
+ id: 'sample:github',
150
+ template: '| github (official) | **{n} tokens** | {n} |',
151
+ values: (s) => [fmt(s.sample.github.tokens), fmt(s.sample.github.tools)],
152
+ },
153
+ {
154
+ file: 'README.md',
155
+ id: 'sample:xcodebuildmcp',
156
+ template: '| xcodebuildmcp | {n} | {n} |',
157
+ values: (s) => [fmt(s.sample.xcodebuildmcp.tokens), fmt(s.sample.xcodebuildmcp.tools)],
158
+ },
159
+ {
160
+ file: 'README.md',
161
+ id: 'sample:brave-search',
162
+ template: '| brave-search | {n} | {n} |',
163
+ values: (s) => [fmt(s.sample['brave-search'].tokens), fmt(s.sample['brave-search'].tools)],
164
+ },
165
+ {
166
+ file: 'README.md',
167
+ id: 'sample:notion',
168
+ template: '| notion | {n} | {n} |',
169
+ values: (s) => [fmt(s.sample.notion.tokens), fmt(s.sample.notion.tools)],
170
+ },
171
+ {
172
+ file: 'README.md',
173
+ id: 'sample:playwright',
174
+ template: '| playwright *(4.8M installs/week)* | {n} | {n} |',
175
+ values: (s) => [fmt(s.sample.playwright.tokens), fmt(s.sample.playwright.tools)],
176
+ },
177
+ {
178
+ file: 'README.md',
179
+ id: 'sample:filesystem',
180
+ template: '| filesystem (reference) | {n} | {n} |',
181
+ values: (s) => [fmt(s.sample.filesystem.tokens), fmt(s.sample.filesystem.tools)],
182
+ },
183
+ {
184
+ file: 'README.md',
185
+ id: 'sample:markitdown',
186
+ template: '| markitdown | {n} | {n} |',
187
+ values: (s) => [fmt(s.sample.markitdown.tokens), fmt(s.sample.markitdown.tools)],
188
+ },
189
+ {
190
+ file: 'README.md',
191
+ id: 'measured-of-candidates',
192
+ template: '*({n} of {n} popular servers measured, each row dated by its own most recent sweep — full table in',
193
+ values: (s) => [fmt(s.measuredCount), fmt(s.candidateTotal)],
194
+ },
195
+ {
196
+ file: 'README.md',
197
+ id: 'divergence-run-size',
198
+ template: 'The run holds {n} rows — the top {n} measured servers by tokens when it ran',
199
+ values: (s) => [fmt(s.claude.runSize), fmt(s.claude.runSize)],
200
+ },
201
+ {
202
+ file: 'README.md',
203
+ id: 'divergence-current-count',
204
+ template: 'prints a claude number for the {n} that still match today and silence for the rest',
205
+ values: (s) => [fmt(s.claude.currentCount)],
206
+ },
207
+ {
208
+ file: 'README.md',
209
+ id: 'claude-table:github',
210
+ template: '| github | {n} | **{n}** | {d}% of the capture is `annotations`/`outputSchema` metadata Claude never sees |',
211
+ values: (s) => [fmt(s.claude.github.badgeTokens), fmt(s.claude.github.claudeTokens), String(s.claude.github.droppedPct)],
212
+ },
213
+ {
214
+ file: 'README.md',
215
+ id: 'claude-table:notion',
216
+ template: '| notion | {n} | **{n}** | almost no metadata to drop, so the tokenizer difference dominates |',
217
+ values: (s) => [fmt(s.claude.notion.badgeTokens), fmt(s.claude.notion.claudeTokens)],
218
+ },
219
+ {
220
+ file: 'README.md',
221
+ id: 'verify-transcript',
222
+ template: '# OK {w}: {d} tokens (o200k_base, methodology 1.0) — capture, hash, and count all agree',
223
+ values: (s) => [s.verify.serverName, String(s.verify.tokens)],
224
+ },
225
+ {
226
+ file: 'docs/index.md',
227
+ id: 'index:counts',
228
+ template: 'We measure {n} popular MCP servers; {n} have a number today, and every failure is listed with its reason.',
229
+ values: (s) => [fmt(s.candidateTotal), fmt(s.measuredCount)],
230
+ },
231
+ {
232
+ file: 'docs/index.md',
233
+ id: 'index:span',
234
+ template: 'The spread is {n}×: from `{w}` at {n} tokens to `{w}` at **{n} tokens** — {d}% of a 200K context window, before the agent takes a single action.',
235
+ values: (s) => [fmt(s.spanTimes), s.min.name, fmt(s.min.tokens), s.max.name, fmt(s.max.tokens), String(s.maxContextSharePct)],
236
+ },
237
+ {
238
+ file: 'docs/index.md',
239
+ id: 'index:second-heaviest',
240
+ template: 'Second-heaviest is `{w}` at {n}.',
241
+ values: (s) => [s.second.name, fmt(s.second.tokens)],
242
+ },
243
+ {
244
+ file: 'docs/METHODOLOGY.md',
245
+ id: 'divergence:share-range',
246
+ template: 'this removes between {f}% and **{f}%** of the payload (github: {n} → {n} tokens).',
247
+ values: (s) => [
248
+ (s.claude.shareMin * 100).toFixed(1),
249
+ (s.claude.shareMax * 100).toFixed(1),
250
+ fmt(s.claude.github.badgeTokens),
251
+ fmt(s.claude.github.mappedTokens),
252
+ ],
253
+ },
254
+ {
255
+ file: 'docs/METHODOLOGY.md',
256
+ id: 'divergence:ratio-range',
257
+ template: 'it ranged from {f}× to {f}× across the top {n},',
258
+ values: (s) => [s.claude.ratioMin.toFixed(2), s.claude.ratioMax.toFixed(2), fmt(s.claude.runSize)],
259
+ },
260
+ {
261
+ file: 'docs/METHODOLOGY.md',
262
+ id: 'divergence:heaviest-pair',
263
+ template: '{w} is the heaviest server on o200k and {w} is the heaviest on Claude.',
264
+ values: (s) => {
265
+ if (!s.claude.heaviestClaudeName) {
266
+ throw new Error('no row has a current claude number — the heaviest-on-Claude sentence cannot be maintained');
267
+ }
268
+ return [s.max.name, s.claude.heaviestClaudeName];
269
+ },
270
+ },
271
+ ];
272
+ export const CHECK_CLAIMS = [
273
+ {
274
+ file: 'README.md',
275
+ id: 'heaviest-differs-on-claude',
276
+ words: 'the heaviest server on the badge is not the heaviest server on Claude',
277
+ holds: (s) => s.claude.heaviestClaudeName === null
278
+ ? 'no row has a current claude number, so there is no heaviest server on Claude'
279
+ : s.claude.heaviestClaudeName === s.max.name
280
+ ? `the heaviest server on the badge (${s.max.name}) IS the heaviest on Claude now`
281
+ : null,
282
+ },
283
+ {
284
+ file: 'README.md',
285
+ id: 'deferring-costlier-somewhere',
286
+ words: 'for at least one server in the published set it costs **more** than loading the definitions would',
287
+ holds: (s) => s.deferralCostlierCount >= 1
288
+ ? null
289
+ : 'no measured server currently costs more at session start than its definitions',
290
+ },
291
+ ];
292
+ const escapeLiteral = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&').replace(/\s+/g, '\\s+');
293
+ /** Fixed words with `\s+` for every gap (prose wraps; a claim is its words, not its layout). */
294
+ export function compileTemplate(template) {
295
+ let source = '';
296
+ let last = 0;
297
+ for (const slot of template.matchAll(/\{[ndwf]\}/g)) {
298
+ source += escapeLiteral(template.slice(last, slot.index));
299
+ source +=
300
+ slot[0] === '{n}'
301
+ ? '([\\d,]+)'
302
+ : slot[0] === '{d}'
303
+ ? '(\\d+)'
304
+ : slot[0] === '{f}'
305
+ ? '(\\d+\\.\\d+)'
306
+ : '([A-Za-z0-9._-]+)';
307
+ last = slot.index + slot[0].length;
308
+ }
309
+ source += escapeLiteral(template.slice(last));
310
+ return new RegExp(source, 'dg');
311
+ }
312
+ /** Apply one claim: find its anchor exactly once, splice the slots to `want`, touch nothing else. */
313
+ export function applyClaim(text, claim, want) {
314
+ const matches = [...text.matchAll(compileTemplate(claim.template))];
315
+ if (matches.length !== 1) {
316
+ return {
317
+ text,
318
+ changed: false,
319
+ problem: matches.length === 0
320
+ ? `${claim.file}: claim '${claim.id}' not found — the sentence regen maintains is gone or reworded`
321
+ : `${claim.file}: claim '${claim.id}' matches ${matches.length} places — the anchor is ambiguous`,
322
+ };
323
+ }
324
+ const match = matches[0];
325
+ const got = match.slice(1);
326
+ if (got.length !== want.length) {
327
+ return {
328
+ text,
329
+ changed: false,
330
+ problem: `${claim.file}: claim '${claim.id}' has ${got.length} slots but ${want.length} values — the claim itself is broken`,
331
+ };
332
+ }
333
+ if (got.every((g, i) => g === want[i]))
334
+ return { text, problem: null, changed: false };
335
+ // Right-to-left so earlier slot offsets stay valid; the `d` flag supplies indices.
336
+ const indices = match.indices;
337
+ for (let g = want.length; g >= 1; g--) {
338
+ const [start, end] = indices[g];
339
+ text = text.slice(0, start) + want[g - 1] + text.slice(end);
340
+ }
341
+ return { text, problem: null, changed: true };
342
+ }
343
+ /** Apply every claim for one page to its text. Slots are spliced in place; anchors are never rewritten. */
344
+ export function patchPageText(file, text, stats) {
345
+ const problems = [];
346
+ const updated = [];
347
+ for (const claim of PAGE_CLAIMS.filter((c) => c.file === file)) {
348
+ const applied = applyClaim(text, claim, claim.values(stats));
349
+ text = applied.text;
350
+ if (applied.problem)
351
+ problems.push(applied.problem);
352
+ else if (applied.changed)
353
+ updated.push(claim.id);
354
+ }
355
+ return { text, problems, updated };
356
+ }
357
+ /** Compute stats and report what regen would rewrite, without writing anything. */
358
+ export function verifyPublishedPages(entries, root = process.cwd()) {
359
+ return applyTo(entries, root, false);
360
+ }
361
+ /** Compute stats and rewrite the pages in place. Returns what changed and any refusals. */
362
+ export function applyPublishedStats(entries, root = process.cwd()) {
363
+ return applyTo(entries, root, true);
364
+ }
365
+ function applyTo(entries, root, write) {
366
+ const stats = computePublishedStats(entries, root);
367
+ const problems = [];
368
+ const updated = [];
369
+ const changedFiles = [];
370
+ for (const file of PAGE_FILES) {
371
+ const path = join(root, file);
372
+ const before = readFileSync(path, 'utf8');
373
+ const patch = patchPageText(file, before, stats);
374
+ problems.push(...patch.problems);
375
+ updated.push(...patch.updated);
376
+ if (patch.text !== before) {
377
+ changedFiles.push(file);
378
+ if (write)
379
+ writeFileSync(path, patch.text);
380
+ }
381
+ }
382
+ return { problems, updated, changedFiles };
383
+ }
@@ -5,6 +5,7 @@ import { writeLeaderboard, percentiles } from './report.js';
5
5
  import { appendHistory } from './history.js';
6
6
  import { writeServerPages } from './server-pages.js';
7
7
  import { writeDashboard } from './dashboard.js';
8
+ import { applyPublishedStats } from './published-stats.js';
8
9
  const doc = parse(readFileSync('servers.yaml', 'utf8'));
9
10
  writeLeaderboard(doc.servers);
10
11
  // History first: the server pages read history.csv for their over-time table.
@@ -13,7 +14,19 @@ const p = writeServerPages(doc.servers);
13
14
  // The dashboard reads the same results/ and history.csv as the pages do, so it
14
15
  // belongs in the same refresh — see writeDashboard's note on why it wasn't.
15
16
  const d = writeDashboard();
17
+ // The front pages state numbers the sweep just changed; they are patched from
18
+ // the same results/ the leaderboard was. A missing anchor is a page regen can
19
+ // no longer maintain — refuse loudly rather than leave one number stale.
20
+ const stats = applyPublishedStats(doc.servers);
21
+ if (stats.problems.length > 0) {
22
+ for (const p of stats.problems)
23
+ console.error(`published stats: ${p}`);
24
+ process.exit(1);
25
+ }
16
26
  console.log('leaderboard:', JSON.stringify(percentiles(doc.servers)));
17
27
  console.log(`history: ${h.rows} rows (${h.added >= 0 ? '+' : ''}${h.added})`);
18
28
  console.log(`server pages: ${p.pages}`);
19
29
  console.log(`dashboard: ${d.out} (${(d.bytes / 1024).toFixed(0)}KB)`);
30
+ console.log(`published stats: ${stats.changedFiles.length > 0
31
+ ? `updated ${stats.updated.join(', ')} in ${stats.changedFiles.join(', ')}`
32
+ : 'pages already agree with the data'}`);
@@ -1,4 +1,6 @@
1
+ import type { Measurement } from '../core/types.js';
1
2
  import { type DivergenceRun } from '../core/divergence.js';
3
+ import { type CrossCheckRun } from '../core/cross-check.js';
2
4
  import { type SessionStartLoad, type SessionStartRun } from '../core/session-start.js';
3
5
  export interface ServerEntry {
4
6
  name: string;
@@ -21,12 +23,19 @@ export interface ServerEntry {
21
23
  */
22
24
  envValues?: Record<string, string>;
23
25
  }
26
+ export interface Row {
27
+ entry: ServerEntry;
28
+ m: Measurement | null;
29
+ }
24
30
  /** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
25
31
  export declare function mdCell(s: unknown): string;
32
+ export declare function loadRows(entries: ServerEntry[], root?: string): Row[];
26
33
  /** results/divergence.json if a divergence run has been recorded, else null. */
27
34
  export declare function loadDivergence(root?: string): DivergenceRun | null;
28
35
  /** results/session-start.json — the instructions backfill, if one exists. */
29
36
  export declare function loadSessionStartRun(root?: string): SessionStartRun | null;
37
+ /** results/cross-check.json if a CLI cross-check run has been recorded, else null. */
38
+ export declare function loadCrossCheckRun(root?: string): CrossCheckRun | null;
30
39
  /**
31
40
  * A session-start figure that is a floor reads `>= N`, never a bare `N`. The
32
41
  * marker is half the point of publishing the number: the names half is measured,
@@ -5,6 +5,7 @@
5
5
  import { existsSync, readFileSync, writeFileSync } from 'node:fs';
6
6
  import { join } from 'node:path';
7
7
  import { isCurrent, parseDivergence } from '../core/divergence.js';
8
+ import { divergencePct, isComparable, parseCrossCheck } from '../core/cross-check.js';
8
9
  import { SESSION_START_METHOD, parseSessionStart, sessionStartLoad, } from '../core/session-start.js';
9
10
  /** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
10
11
  export function mdCell(s) {
@@ -17,7 +18,7 @@ function csvCell(s) {
17
18
  const v = String(s ?? '');
18
19
  return /[",\n]/.test(v) ? `"${v.replace(/"/g, '""')}"` : v;
19
20
  }
20
- function loadRows(entries, root = process.cwd()) {
21
+ export function loadRows(entries, root = process.cwd()) {
21
22
  return entries.map((entry) => {
22
23
  const p = join(root, 'results', entry.name, 'measurement.json');
23
24
  return { entry, m: existsSync(p) ? JSON.parse(readFileSync(p, 'utf8')) : null };
@@ -33,6 +34,11 @@ export function loadSessionStartRun(root = process.cwd()) {
33
34
  const p = join(root, 'results', 'session-start.json');
34
35
  return existsSync(p) ? parseSessionStart(readFileSync(p, 'utf8')) : null;
35
36
  }
37
+ /** results/cross-check.json if a CLI cross-check run has been recorded, else null. */
38
+ export function loadCrossCheckRun(root = process.cwd()) {
39
+ const p = join(root, 'results', 'cross-check.json');
40
+ return existsSync(p) ? parseCrossCheck(readFileSync(p, 'utf8')) : null;
41
+ }
36
42
  /**
37
43
  * A session-start figure that is a floor reads `>= N`, never a bare `N`. The
38
44
  * marker is half the point of publishing the number: the names half is measured,
@@ -48,6 +54,7 @@ export function writeLeaderboard(entries, root = process.cwd()) {
48
54
  const rows = loadRows(entries, root);
49
55
  const div = loadDivergence(root);
50
56
  const ss = loadSessionStartRun(root);
57
+ const xc = loadCrossCheckRun(root);
51
58
  /** Session-start load for a row, or null when there is no capture to read. */
52
59
  const session = (r) => r.m ? sessionStartLoad(r.m, ss?.servers[r.entry.name]) : null;
53
60
  /** Claude tokens for a row, or null when not measured / stale / errored. */
@@ -57,6 +64,14 @@ export function writeLeaderboard(entries, root = process.cwd()) {
57
64
  const d = div.servers[r.entry.name];
58
65
  return isCurrent(d, r.m.canonicalSha256) ? d.claudeDelta : null;
59
66
  };
67
+ /** The other CLI's row, or null when the comparison is not like-with-like. */
68
+ const crossCheck = (r) => {
69
+ if (!xc || !r.m)
70
+ return null;
71
+ const row = xc.servers[r.entry.name];
72
+ return isComparable(row, r.m.canonicalSha256) ? row : null;
73
+ };
74
+ const signedPct = (p) => `${p >= 0 ? '+' : '−'}${Math.abs(p).toFixed(1)}%`;
60
75
  const measured = rows
61
76
  .filter((r) => r.m && (r.m.status === 'measured' || r.m.status === 'dynamic'))
62
77
  .sort((a, b) => (b.m.totalTokens ?? 0) - (a.m.totalTokens ?? 0));
@@ -77,6 +92,30 @@ export function writeLeaderboard(entries, root = process.cwd()) {
77
92
  `See [Claude divergence](../docs/METHODOLOGY.md#claude-divergence).`);
78
93
  md.push('');
79
94
  }
95
+ if (xc) {
96
+ const pcts = measured
97
+ .map((r) => {
98
+ const row = crossCheck(r);
99
+ return row ? divergencePct(row) : null;
100
+ })
101
+ .filter((p) => p !== null);
102
+ md.push(`The **mcp-tokens** column is the other CLI's count of the same server — ` +
103
+ `\`${mdCell(xc.cli)}\` \`${mdCell(xc.cliVersion)}\` (${mdCell(xc.measuredAt)}, method \`${mdCell(xc.method)}\`), ` +
104
+ `invoked with \`--model gpt-4o\` so both columns count o200k tokens. Its structs model the three ` +
105
+ `request fields (name/description/input\\_schema), so its number sits below the tokens column ` +
106
+ `wherever a server ships metadata those fields do not carry — that gap is each server's ` +
107
+ `field-selection share, published on its page, not a disagreement of counters. The parenthesized ` +
108
+ `percentage is the disagreement of counters: the CLI's count against ours of the same three-field ` +
109
+ `projection` +
110
+ (pcts.length > 0
111
+ ? `, ${signedPct(Math.min(...pcts))} to ${signedPct(Math.max(...pcts))} across the ` +
112
+ `${pcts.length} row${pcts.length === 1 ? '' : 's'} where both tools saw the same tool set.`
113
+ : `. No row currently compares like with like.`) +
114
+ ` A row prints only while the comparison is between like and like: the same tool names on both ` +
115
+ `sides, and our capture unchanged since the run. ` +
116
+ `See [CLI cross-check](../docs/METHODOLOGY.md#cli-cross-check).`);
117
+ md.push('');
118
+ }
80
119
  const floors = measured.filter((r) => session(r)?.isFloor).length;
81
120
  md.push(`The **session start** column is what a client puts in context when it *defers* tool definitions until they ` +
82
121
  `are used: the server's tool names plus the \`instructions\` string it returns from \`initialize\` ` +
@@ -119,16 +158,19 @@ export function writeLeaderboard(entries, root = process.cwd()) {
119
158
  `A client that defers definitions is better off on every other measured row and worse off on ${costlier.length === 1 ? 'this one' : 'these'}.`);
120
159
  md.push('');
121
160
  }
122
- md.push(`| # | server | tokens | session start |${div ? ' claude |' : ''} tools | largest tool | status | category |`);
123
- md.push(`|---:|---|---:|---:|${div ? '---:|' : ''}---:|---|---|---|`);
161
+ md.push(`| # | server | tokens | session start |${div ? ' claude |' : ''}${xc ? ' mcp-tokens |' : ''} tools | largest tool | status | category |`);
162
+ md.push(`|---:|---|---:|---:|${div ? '---:|' : ''}${xc ? '---:|' : ''}---:|---|---|---|`);
124
163
  measured.forEach((r, i) => {
125
164
  const m = r.m;
126
165
  const largest = [...m.tools].sort((a, b) => b.tokens - a.tokens)[0];
127
166
  const link = `[${mdCell(r.entry.name)}](../docs/servers/${encodeURIComponent(r.entry.name)}.md)`;
128
167
  const c = claude(r);
168
+ const x = crossCheck(r);
169
+ const xCell = x === null ? '—' : `${x.cliTokens.toLocaleString('en-US')} (${signedPct(divergencePct(x))})`;
129
170
  md.push(`| ${i + 1} | ${link} | ${m.totalTokens.toLocaleString('en-US')} |` +
130
171
  ` ${sessionStartCell(session(r))} |` +
131
172
  (div ? ` ${c === null ? '—' : c.toLocaleString('en-US')} |` : '') +
173
+ (xc ? ` ${xCell} |` : '') +
132
174
  ` ${m.toolCount} | ` +
133
175
  `${largest ? `${mdCell(largest.name)} (${largest.tokens.toLocaleString('en-US')})` : '—'} | ${m.status} | ${mdCell(r.entry.category)} |`);
134
176
  });
@@ -150,12 +192,14 @@ export function writeLeaderboard(entries, root = process.cwd()) {
150
192
  // every existing parser working.
151
193
  const csv = [
152
194
  'name,tokens,toolCount,status,category,metric,metricSource,claudeTokens,claudeModel,' +
153
- 'sessionStartTokens,sessionStartIsFloor,toolNameTokens,instructionsTokens',
195
+ 'sessionStartTokens,sessionStartIsFloor,toolNameTokens,instructionsTokens,' +
196
+ 'crossCheckTokens,crossCheckCliVersion',
154
197
  ];
155
198
  for (const r of rows) {
156
199
  const m = r.m;
157
200
  const c = claude(r);
158
201
  const ssl = session(r);
202
+ const x = crossCheck(r);
159
203
  csv.push([
160
204
  csvCell(r.entry.name),
161
205
  m?.totalTokens ?? '',
@@ -172,6 +216,8 @@ export function writeLeaderboard(entries, root = process.cwd()) {
172
216
  ssl ? String(ssl.isFloor) : '',
173
217
  ssl?.toolNameTokens ?? '',
174
218
  ssl?.instructionsTokens ?? '',
219
+ x?.cliTokens ?? '',
220
+ x === null ? '' : csvCell(xc.cliVersion),
175
221
  ].join(','));
176
222
  }
177
223
  writeFileSync(join(root, 'results', 'leaderboard.csv'), csv.join('\n') + '\n');
package/dist/sweep/run.js CHANGED
@@ -8,7 +8,7 @@ import { mkdirSync, writeFileSync } from 'node:fs';
8
8
  import { join, resolve } from 'node:path';
9
9
  import { fileURLToPath } from 'node:url';
10
10
  import { captureTools } from './client.js';
11
- import { dockerize } from './docker.js';
11
+ import { DockerHarnessFault, defaultImageFor, dockerize, ensureImage, isDockerRunFailure } from './docker.js';
12
12
  import { measureTools, failedMeasurement, canonicalString } from '../core/canonical.js';
13
13
  import { toBadge } from '../core/badge.js';
14
14
  function arg(name) {
@@ -66,6 +66,9 @@ export async function measureServer(name, command, opts = {}) {
66
66
  throw new Error(`invalid server name '${name}' — letters/digits/dot/dash/underscore only`);
67
67
  }
68
68
  const root = opts.root ?? process.cwd();
69
+ // True only when THIS code wraps the command in `docker run` — a command that
70
+ // is already its own `docker run` owns its exit codes and its image.
71
+ const dockerWrapped = opts.docker === true && !command.trimStart().startsWith('docker ');
69
72
  const hostSpec = opts.argv && opts.argv.length ? { command: opts.argv[0], argv: opts.argv.slice(1) } : command;
70
73
  let isolation = { docker: false };
71
74
  const containerNames = [];
@@ -123,7 +126,15 @@ export async function measureServer(name, command, opts = {}) {
123
126
  }
124
127
  }
125
128
  catch (err) {
129
+ if (err instanceof DockerHarnessFault)
130
+ throw err;
126
131
  const msg = err instanceof Error ? err.message : String(err);
132
+ // `docker run` failing as docker (exit 125, docker's own stderr) never
133
+ // launched the server — classifying it would publish a fact about this
134
+ // machine as a fact about the server, so it is thrown instead of returned.
135
+ if (dockerWrapped && isDockerRunFailure(msg)) {
136
+ throw new DockerHarnessFault(`docker could not run the container for ${name}: ${msg.slice(0, 400)}`);
137
+ }
127
138
  const status = msg.includes('timeout')
128
139
  ? 'timeout'
129
140
  : /auth|unauthorized|401|forbidden|credential|api.?key|token/i.test(msg)
@@ -135,6 +146,12 @@ export async function measureServer(name, command, opts = {}) {
135
146
  r.timeoutMs = attemptOpts.timeoutMs ?? 60_000;
136
147
  return r;
137
148
  }
149
+ // The base image is fetched before anything is measured: `--pull=missing`
150
+ // pulls lazily, so a registry hiccup would otherwise land mid-measurement and
151
+ // read as the server refusing to start. Throws DockerHarnessFault when the
152
+ // machine cannot produce the image at all — before any record is written.
153
+ if (dockerWrapped)
154
+ await ensureImage(opts.dockerImage ?? defaultImageFor(command));
138
155
  let m;
139
156
  try {
140
157
  m = await attempt(false);
@@ -189,11 +206,23 @@ if (isMain) {
189
206
  console.error('usage: npm run sweep -- --name <slug> --command "<launch command>"');
190
207
  process.exit(2);
191
208
  }
192
- const m = await measureServer(name, command, {
193
- timeoutMs: Number(arg('timeout') ?? 60_000),
194
- docker: process.argv.includes('--docker'),
195
- dockerImage: arg('docker-image'),
196
- });
209
+ let m;
210
+ try {
211
+ m = await measureServer(name, command, {
212
+ timeoutMs: Number(arg('timeout') ?? 60_000),
213
+ docker: process.argv.includes('--docker'),
214
+ dockerImage: arg('docker-image'),
215
+ });
216
+ }
217
+ catch (err) {
218
+ if (err instanceof DockerHarnessFault) {
219
+ // Nothing was measured and nothing was written — exiting non-zero here is
220
+ // what keeps a scheduled job from reaching its commit step.
221
+ console.error(`HARNESS FAULT: ${err.message}`);
222
+ process.exit(1);
223
+ }
224
+ throw err;
225
+ }
197
226
  // CLI path only — measureServer itself stays history-free so concurrent
198
227
  // sweep-all workers never race on the same file.
199
228
  const { appendHistory } = await import('./history.js');