mcp-context-cost 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +101 -33
  2. package/dist/audit/audit.d.ts +48 -0
  3. package/dist/audit/audit.js +392 -7
  4. package/dist/audit/config.d.ts +46 -0
  5. package/dist/audit/config.js +86 -2
  6. package/dist/audit/deferral.d.ts +366 -0
  7. package/dist/audit/deferral.js +403 -0
  8. package/dist/audit/run.d.ts +22 -0
  9. package/dist/audit/run.js +17 -1
  10. package/dist/cli.js +11 -0
  11. package/dist/core/adoption.d.ts +226 -0
  12. package/dist/core/adoption.js +432 -0
  13. package/dist/core/canonical.d.ts +6 -0
  14. package/dist/core/canonical.js +3 -0
  15. package/dist/core/index.d.ts +1 -0
  16. package/dist/core/index.js +1 -0
  17. package/dist/core/session-start.d.ts +102 -0
  18. package/dist/core/session-start.js +186 -0
  19. package/dist/core/types.d.ts +8 -0
  20. package/dist/sweep/client.d.ts +6 -1
  21. package/dist/sweep/client.js +1 -0
  22. package/dist/sweep/dashboard.d.ts +10 -0
  23. package/dist/sweep/dashboard.js +32 -16
  24. package/dist/sweep/docker.d.ts +23 -0
  25. package/dist/sweep/docker.js +16 -12
  26. package/dist/sweep/harness-guard.d.ts +57 -0
  27. package/dist/sweep/harness-guard.js +144 -0
  28. package/dist/sweep/history.d.ts +33 -1
  29. package/dist/sweep/history.js +60 -5
  30. package/dist/sweep/regen.js +6 -1
  31. package/dist/sweep/report.d.ts +16 -0
  32. package/dist/sweep/report.js +79 -5
  33. package/dist/sweep/run.d.ts +41 -0
  34. package/dist/sweep/run.js +129 -36
  35. package/dist/sweep/server-pages.js +31 -6
  36. package/dist/sweep/session-start.d.ts +3 -0
  37. package/dist/sweep/session-start.js +103 -0
  38. package/dist/sweep/shard.d.ts +41 -0
  39. package/dist/sweep/shard.js +58 -0
  40. package/dist/sweep/sweep-all.js +56 -2
  41. package/package.json +3 -1
@@ -0,0 +1,144 @@
1
+ /**
2
+ * A broken harness must not republish the whole set as failures.
3
+ *
4
+ * `measureServer` already adjudicates a *single* suspicious result — a
5
+ * startup-failure gets re-attempted off a cold cache, a timeout on double the
6
+ * budget (see run.ts). Neither retry can see the one failure mode that isn't
7
+ * about any individual server: the machine doing the measuring is broken.
8
+ * That is not hypothetical here — an orphaned Docker backend with a wedged
9
+ * network stack once returned 0/79 uniform timeouts, and every one of those
10
+ * results was a lie about a working server. The per-server retries make that
11
+ * case *worse*, not better: each one re-runs through the same broken harness
12
+ * and comes back failing again, which reads as confirmation.
13
+ *
14
+ * The signal a single measurement can't carry is population-level. A wave of
15
+ * servers that were measuring fine yesterday and all fail at once is a
16
+ * statement about the harness, not about the servers — upstream breakages
17
+ * arrive a package at a time, not by the dozen. So: snapshot what was on
18
+ * record before the sweep, count how many good measurements this sweep turned
19
+ * into failures, and if that count is both large in absolute terms and a
20
+ * majority of what it could have broken, refuse to publish, put the previous
21
+ * results back, and exit non-zero.
22
+ *
23
+ * Restoring is what makes this safe to run unattended: `measureServer`
24
+ * persists each result the moment it has one, so by the time a sweep-level
25
+ * verdict is possible the damage is already on disk. Holding the prior bytes
26
+ * in memory turns an irreversible overwrite into a reversible one.
27
+ */
28
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
29
+ import { join } from 'node:path';
30
+ /**
31
+ * Below this many regressions the population signal doesn't exist yet, and a
32
+ * deliberately narrow sweep (`--only redis,serena`, two servers known to be
33
+ * shaky) must never be able to trip a fault by failing completely.
34
+ */
35
+ export const MIN_REGRESSIONS = 5;
36
+ /**
37
+ * Share of the previously-good servers in this sweep that must regress before
38
+ * the harness is the likelier explanation. The largest genuine simultaneous
39
+ * upstream breakage on record here is a handful of PyPI servers that shared
40
+ * one unbounded dependency — single digits against ~65 measured, well under
41
+ * 10%. A harness fault, by contrast, takes everything it touches: the Docker
42
+ * outage above scored 100%. A majority sits far from the former and catches
43
+ * every instance of the latter this project has actually seen.
44
+ */
45
+ export const FAULT_RATIO = 0.5;
46
+ /** Statuses that represent a real number on record — the thing worth protecting. */
47
+ export function isGood(status) {
48
+ return status === 'measured' || status === 'dynamic';
49
+ }
50
+ /**
51
+ * Read what is currently published for each server. Called before the sweep
52
+ * starts; a server with nothing on record snapshots as `null` status, which no
53
+ * verdict counts either way.
54
+ */
55
+ export function snapshot(names, root = process.cwd()) {
56
+ return names.map((name) => {
57
+ const mPath = join(root, 'results', name, 'measurement.json');
58
+ const bPath = join(root, 'badges', `${name}.json`);
59
+ const measurementJson = existsSync(mPath) ? readFileSync(mPath, 'utf8') : null;
60
+ const badgeJson = existsSync(bPath) ? readFileSync(bPath, 'utf8') : null;
61
+ let status = null;
62
+ if (measurementJson) {
63
+ try {
64
+ status = JSON.parse(measurementJson).status;
65
+ }
66
+ catch {
67
+ // An unreadable prior record is not evidence of anything; treat it as
68
+ // no record rather than as a good one that just broke.
69
+ status = null;
70
+ }
71
+ }
72
+ return { name, status, measurementJson, badgeJson };
73
+ });
74
+ }
75
+ /**
76
+ * Compare pre-sweep snapshots against this sweep's outcomes.
77
+ *
78
+ * `current` maps server name to the status it just measured at. Servers absent
79
+ * from it were not swept and are ignored.
80
+ */
81
+ export function verdict(prior, current) {
82
+ const comparableNames = prior
83
+ .filter((s) => s.status !== null && isGood(s.status) && current.has(s.name))
84
+ .map((s) => s.name);
85
+ const regressed = comparableNames.filter((n) => !isGood(current.get(n)));
86
+ const comparable = comparableNames.length;
87
+ if (comparable === 0) {
88
+ return {
89
+ fault: false,
90
+ regressed,
91
+ comparable,
92
+ // Not a pass — an unavailable reading. Nothing on record measured well
93
+ // before this sweep, so there is no baseline to judge it against.
94
+ reason: 'no prior measurement to compare against — harness check not performed',
95
+ };
96
+ }
97
+ const ratio = regressed.length / comparable;
98
+ const pct = (ratio * 100).toFixed(0);
99
+ if (regressed.length >= MIN_REGRESSIONS && ratio >= FAULT_RATIO) {
100
+ return {
101
+ fault: true,
102
+ regressed,
103
+ comparable,
104
+ reason: `${regressed.length} of ${comparable} previously-measured servers (${pct}%) failed in ` +
105
+ `this sweep — at or above the ${MIN_REGRESSIONS}-server, ` +
106
+ `${(FAULT_RATIO * 100).toFixed(0)}% threshold that reads as a broken harness ` +
107
+ `rather than broken servers`,
108
+ };
109
+ }
110
+ return {
111
+ fault: false,
112
+ regressed,
113
+ comparable,
114
+ reason: `${regressed.length} of ${comparable} previously-measured servers (${pct}%) failed in ` +
115
+ `this sweep — below the harness-fault threshold, publishing normally`,
116
+ };
117
+ }
118
+ /**
119
+ * Put the snapshotted artifacts back, byte for byte. Only servers named in
120
+ * `names` are touched, and only where a prior file existed — a server whose
121
+ * first-ever measurement failed has nothing to restore and keeps its new
122
+ * (honest) failure record.
123
+ */
124
+ export function restore(prior, names, root = process.cwd()) {
125
+ const wanted = new Set(names);
126
+ const restored = [];
127
+ for (const s of prior) {
128
+ if (!wanted.has(s.name))
129
+ continue;
130
+ if (s.measurementJson === null && s.badgeJson === null)
131
+ continue;
132
+ if (s.measurementJson !== null) {
133
+ const dir = join(root, 'results', s.name);
134
+ mkdirSync(dir, { recursive: true });
135
+ writeFileSync(join(dir, 'measurement.json'), s.measurementJson);
136
+ }
137
+ if (s.badgeJson !== null) {
138
+ mkdirSync(join(root, 'badges'), { recursive: true });
139
+ writeFileSync(join(root, 'badges', `${s.name}.json`), s.badgeJson);
140
+ }
141
+ restored.push(s.name);
142
+ }
143
+ return restored;
144
+ }
@@ -7,8 +7,18 @@ export interface HistoryRow {
7
7
  toolCount: number;
8
8
  /** 'measured' or 'dynamic' — dynamic means the tool set moved between captures. */
9
9
  status: string;
10
+ /**
11
+ * How the measurement was taken: `docker` (isolated container), `host` (bare
12
+ * machine), or `''` when the row predates this column and the conditions are
13
+ * not on record. Two numbers taken under different isolation are not
14
+ * comparable — same server, different node, different resolution of an
15
+ * `@latest` tag, different ambient env — so a step between them is a property
16
+ * of the harness, not of the server. Recording it is what lets the trend line
17
+ * refuse to draw such a step; see `plottableSeries`.
18
+ */
19
+ isolation: string;
10
20
  }
11
- export declare const HISTORY_HEADER = "date,server,tokens,toolCount,status";
21
+ export declare const HISTORY_HEADER = "date,server,tokens,toolCount,status,isolation";
12
22
  /** Parse history.csv text; malformed or non-numeric rows are dropped, not thrown. */
13
23
  export declare function parseHistory(text: string): HistoryRow[];
14
24
  export declare function formatHistory(rows: HistoryRow[]): string;
@@ -19,6 +29,7 @@ export declare function upsert(rows: HistoryRow[], row: HistoryRow): HistoryRow[
19
29
  * auth walls are recorded in the leaderboard's "not measured" section, and
20
30
  * writing them here as zeros would fabricate a drop to zero in the series.
21
31
  */
32
+ export declare function isolationOf(m: Measurement): string;
22
33
  export declare function rowFor(server: string, m: Measurement): HistoryRow | null;
23
34
  /**
24
35
  * Fold every results/<server>/measurement.json into results/history.csv.
@@ -28,3 +39,24 @@ export declare function appendHistory(root?: string): {
28
39
  rows: number;
29
40
  added: number;
30
41
  };
42
+ export interface PlottableSeries {
43
+ /** The rows a trend may be drawn across, oldest first. */
44
+ rows: HistoryRow[];
45
+ /** Older rows excluded because they were measured under a different isolation. */
46
+ dropped: number;
47
+ /** True when a plotted row's conditions are not on record (a pre-`isolation` write). */
48
+ conditionsUnknown: boolean;
49
+ }
50
+ /**
51
+ * The longest run of a server's history, ending at its newest row, that a trend
52
+ * line may honestly be drawn across.
53
+ *
54
+ * Walking back from the newest row, a row stops the run when its isolation is
55
+ * known, the newest row's isolation is known, and the two differ — a step across
56
+ * that boundary would say the server changed when what changed is how it was
57
+ * measured. An *unknown* isolation is not evidence either way, so it stays in
58
+ * the run and is reported through `conditionsUnknown` instead: the alternative,
59
+ * treating unknown as its own incompatible value, would silently blank every
60
+ * series recorded before this column existed.
61
+ */
62
+ export declare function plottableSeries(rows: HistoryRow[]): PlottableSeries;
@@ -8,7 +8,7 @@
8
8
  */
9
9
  import { existsSync, readFileSync, readdirSync, writeFileSync } from 'node:fs';
10
10
  import { join } from 'node:path';
11
- export const HISTORY_HEADER = 'date,server,tokens,toolCount,status';
11
+ export const HISTORY_HEADER = 'date,server,tokens,toolCount,status,isolation';
12
12
  function csvCell(s) {
13
13
  const v = String(s ?? '');
14
14
  return /[",\n]/.test(v) ? `"${v.replace(/"/g, '""')}"` : v;
@@ -48,18 +48,27 @@ export function parseHistory(text) {
48
48
  for (const line of text.split('\n')) {
49
49
  if (!line.trim() || line.startsWith('date,'))
50
50
  continue;
51
- const [date, server, tokens, toolCount, status] = splitCsvLine(line);
51
+ const [date, server, tokens, toolCount, status, isolation] = splitCsvLine(line);
52
52
  if (!/^\d{4}-\d{2}-\d{2}$/.test(date ?? '') || !server)
53
53
  continue;
54
54
  if (!/^\d+$/.test(tokens ?? '') || !/^\d+$/.test(toolCount ?? ''))
55
55
  continue;
56
- rows.push({ date, server, tokens: Number(tokens), toolCount: Number(toolCount), status: status ?? '' });
56
+ // A 5-field row is a pre-`isolation` write: its conditions are unknown, and
57
+ // unknown is recorded as unknown rather than back-filled with a guess.
58
+ rows.push({
59
+ date,
60
+ server,
61
+ tokens: Number(tokens),
62
+ toolCount: Number(toolCount),
63
+ status: status ?? '',
64
+ isolation: isolation ?? '',
65
+ });
57
66
  }
58
67
  return rows;
59
68
  }
60
69
  export function formatHistory(rows) {
61
70
  const sorted = [...rows].sort((a, b) => a.date.localeCompare(b.date) || a.server.localeCompare(b.server));
62
- const lines = sorted.map((r) => [csvCell(r.date), csvCell(r.server), r.tokens, r.toolCount, csvCell(r.status)].join(','));
71
+ const lines = sorted.map((r) => [csvCell(r.date), csvCell(r.server), r.tokens, r.toolCount, csvCell(r.status), csvCell(r.isolation)].join(','));
63
72
  return [HISTORY_HEADER, ...lines].join('\n') + '\n';
64
73
  }
65
74
  /** Upsert one row by (date, server) — the newest write for a day wins. */
@@ -76,6 +85,11 @@ export function upsert(rows, row) {
76
85
  * auth walls are recorded in the leaderboard's "not measured" section, and
77
86
  * writing them here as zeros would fabricate a drop to zero in the series.
78
87
  */
88
+ export function isolationOf(m) {
89
+ if (!m.isolation)
90
+ return '';
91
+ return m.isolation.docker ? 'docker' : 'host';
92
+ }
79
93
  export function rowFor(server, m) {
80
94
  if (m.status !== 'measured' && m.status !== 'dynamic')
81
95
  return null;
@@ -84,7 +98,14 @@ export function rowFor(server, m) {
84
98
  const date = String(m.measuredAt ?? '').slice(0, 10);
85
99
  if (!/^\d{4}-\d{2}-\d{2}$/.test(date))
86
100
  return null;
87
- return { date, server, tokens: m.totalTokens, toolCount: m.toolCount, status: m.status };
101
+ return {
102
+ date,
103
+ server,
104
+ tokens: m.totalTokens,
105
+ toolCount: m.toolCount,
106
+ status: m.status,
107
+ isolation: isolationOf(m),
108
+ };
88
109
  }
89
110
  /**
90
111
  * Fold every results/<server>/measurement.json into results/history.csv.
@@ -118,3 +139,37 @@ export function appendHistory(root = process.cwd()) {
118
139
  writeFileSync(path, formatHistory(rows));
119
140
  return { rows: rows.length, added: rows.length - existing.length };
120
141
  }
142
+ /**
143
+ * The longest run of a server's history, ending at its newest row, that a trend
144
+ * line may honestly be drawn across.
145
+ *
146
+ * Walking back from the newest row, a row stops the run when its isolation is
147
+ * known, the newest row's isolation is known, and the two differ — a step across
148
+ * that boundary would say the server changed when what changed is how it was
149
+ * measured. An *unknown* isolation is not evidence either way, so it stays in
150
+ * the run and is reported through `conditionsUnknown` instead: the alternative,
151
+ * treating unknown as its own incompatible value, would silently blank every
152
+ * series recorded before this column existed.
153
+ */
154
+ export function plottableSeries(rows) {
155
+ const sorted = [...rows].sort((a, b) => a.date.localeCompare(b.date));
156
+ if (!sorted.length)
157
+ return { rows: [], dropped: 0, conditionsUnknown: false };
158
+ const current = sorted[sorted.length - 1].isolation;
159
+ let start = 0;
160
+ if (current) {
161
+ for (let i = sorted.length - 1; i >= 0; i--) {
162
+ const iso = sorted[i].isolation;
163
+ if (iso && iso !== current) {
164
+ start = i + 1;
165
+ break;
166
+ }
167
+ }
168
+ }
169
+ const kept = sorted.slice(start);
170
+ return {
171
+ rows: kept,
172
+ dropped: start,
173
+ conditionsUnknown: kept.some((r) => !r.isolation),
174
+ };
175
+ }
@@ -1,14 +1,19 @@
1
- /** Regenerate leaderboard + history + server pages from results/: npx tsx src/sweep/regen.ts */
1
+ /** Regenerate leaderboard + history + server pages + dashboard from results/: npx tsx src/sweep/regen.ts */
2
2
  import { readFileSync } from 'node:fs';
3
3
  import { parse } from 'yaml';
4
4
  import { writeLeaderboard, percentiles } from './report.js';
5
5
  import { appendHistory } from './history.js';
6
6
  import { writeServerPages } from './server-pages.js';
7
+ import { writeDashboard } from './dashboard.js';
7
8
  const doc = parse(readFileSync('servers.yaml', 'utf8'));
8
9
  writeLeaderboard(doc.servers);
9
10
  // History first: the server pages read history.csv for their over-time table.
10
11
  const h = appendHistory();
11
12
  const p = writeServerPages(doc.servers);
13
+ // The dashboard reads the same results/ and history.csv as the pages do, so it
14
+ // belongs in the same refresh — see writeDashboard's note on why it wasn't.
15
+ const d = writeDashboard();
12
16
  console.log('leaderboard:', JSON.stringify(percentiles(doc.servers)));
13
17
  console.log(`history: ${h.rows} rows (${h.added >= 0 ? '+' : ''}${h.added})`);
14
18
  console.log(`server pages: ${p.pages}`);
19
+ console.log(`dashboard: ${d.out} (${(d.bytes / 1024).toFixed(0)}KB)`);
@@ -1,4 +1,5 @@
1
1
  import { type DivergenceRun } from '../core/divergence.js';
2
+ import { type SessionStartLoad, type SessionStartRun } from '../core/session-start.js';
2
3
  export interface ServerEntry {
3
4
  name: string;
4
5
  command: string;
@@ -13,11 +14,26 @@ export interface ServerEntry {
13
14
  timeoutSeconds?: number;
14
15
  /** Launch needs the `git` binary (e.g. `uvx --from git+...`) — absent from the slim isolation images. */
15
16
  needsGit?: boolean;
17
+ /**
18
+ * Per-var overrides for the literal `dummy` placeholder docker mode injects
19
+ * for `env` names — for servers that parse a var's shape (URI scheme, URL)
20
+ * before ever reaching tools/list. See docker.ts `dummyEnvValues`.
21
+ */
22
+ envValues?: Record<string, string>;
16
23
  }
17
24
  /** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
18
25
  export declare function mdCell(s: unknown): string;
19
26
  /** results/divergence.json if a divergence run has been recorded, else null. */
20
27
  export declare function loadDivergence(root?: string): DivergenceRun | null;
28
+ /** results/session-start.json — the instructions backfill, if one exists. */
29
+ export declare function loadSessionStartRun(root?: string): SessionStartRun | null;
30
+ /**
31
+ * A session-start figure that is a floor reads `>= N`, never a bare `N`. The
32
+ * marker is half the point of publishing the number: the names half is measured,
33
+ * the instructions half has not been captured for this server, and a reader has
34
+ * to be able to tell that from a row where both halves are known.
35
+ */
36
+ export declare function sessionStartCell(load: SessionStartLoad | null): string;
21
37
  export declare function writeLeaderboard(entries: ServerEntry[], root?: string): void;
22
38
  /** Percentile helper for freezing color bands against the observed distribution. */
23
39
  export declare function percentiles(entries: ServerEntry[], root?: string): Record<string, number>;
@@ -5,6 +5,7 @@
5
5
  import { existsSync, readFileSync, writeFileSync } from 'node:fs';
6
6
  import { join } from 'node:path';
7
7
  import { isCurrent, parseDivergence } from '../core/divergence.js';
8
+ import { SESSION_START_METHOD, parseSessionStart, sessionStartLoad, } from '../core/session-start.js';
8
9
  /** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
9
10
  export function mdCell(s) {
10
11
  return String(s ?? '')
@@ -27,9 +28,28 @@ export function loadDivergence(root = process.cwd()) {
27
28
  const p = join(root, 'results', 'divergence.json');
28
29
  return existsSync(p) ? parseDivergence(readFileSync(p, 'utf8')) : null;
29
30
  }
31
+ /** results/session-start.json — the instructions backfill, if one exists. */
32
+ export function loadSessionStartRun(root = process.cwd()) {
33
+ const p = join(root, 'results', 'session-start.json');
34
+ return existsSync(p) ? parseSessionStart(readFileSync(p, 'utf8')) : null;
35
+ }
36
+ /**
37
+ * A session-start figure that is a floor reads `>= N`, never a bare `N`. The
38
+ * marker is half the point of publishing the number: the names half is measured,
39
+ * the instructions half has not been captured for this server, and a reader has
40
+ * to be able to tell that from a row where both halves are known.
41
+ */
42
+ export function sessionStartCell(load) {
43
+ if (!load)
44
+ return '—';
45
+ return `${load.isFloor ? '≥' : ''}${load.totalTokens.toLocaleString('en-US')}`;
46
+ }
30
47
  export function writeLeaderboard(entries, root = process.cwd()) {
31
48
  const rows = loadRows(entries, root);
32
49
  const div = loadDivergence(root);
50
+ const ss = loadSessionStartRun(root);
51
+ /** Session-start load for a row, or null when there is no capture to read. */
52
+ const session = (r) => r.m ? sessionStartLoad(r.m, ss?.servers[r.entry.name]) : null;
33
53
  /** Claude tokens for a row, or null when not measured / stale / errored. */
34
54
  const claude = (r) => {
35
55
  if (!div || !r.m)
@@ -57,14 +77,57 @@ export function writeLeaderboard(entries, root = process.cwd()) {
57
77
  `See [Claude divergence](../docs/METHODOLOGY.md#claude-divergence).`);
58
78
  md.push('');
59
79
  }
60
- md.push(`| # | server | tokens |${div ? ' claude |' : ''} tools | largest tool | status | category |`);
61
- md.push(`|---:|---|---:|${div ? '---:|' : ''}---:|---|---|---|`);
80
+ const floors = measured.filter((r) => session(r)?.isFloor).length;
81
+ md.push(`The **session start** column is what a client puts in context when it *defers* tool definitions until they ` +
82
+ `are used: the server's tool names plus the \`instructions\` string it returns from \`initialize\` ` +
83
+ `(method \`${SESSION_START_METHOD}\`). The tokens column is what a client that loads every definition up ` +
84
+ `front pays; this one is what the same server costs a client that does not. ` +
85
+ `See [session-start load](../docs/METHODOLOGY.md#session-start-load).`);
86
+ md.push('');
87
+ if (floors > 0) {
88
+ md.push(`**\`≥\` marks a floor, on ${floors} of ${measured.length} rows.** Tool names are counted exactly from ` +
89
+ `the published capture, but \`instructions\` is not part of \`tools/list\` and has not been captured for ` +
90
+ `these servers — so the figure is the names half alone and the true number is that or higher. A row stops ` +
91
+ `being a floor the first time the server is measured with its instructions.`);
92
+ md.push('');
93
+ }
94
+ // Deferring usually saves almost everything, but it is not guaranteed to save
95
+ // anything: `instructions` are bytes the headline never counted, and a server
96
+ // that re-lists its tools in prose can charge a deferring client more than an
97
+ // eager one. Those rows are the most useful thing this column finds, so they
98
+ // are named here rather than left for a reader to spot by comparing columns.
99
+ // Derived on every write — no row is listed by hand, and the paragraph
100
+ // disappears if the set ever empties, rather than asserting a stale count.
101
+ const costlier = measured.filter((r) => {
102
+ const load = session(r);
103
+ return load !== null && r.m.totalTokens !== null && load.totalTokens >= r.m.totalTokens;
104
+ });
105
+ if (costlier.length > 0) {
106
+ // Server names are backticked, not bolded: the lead sentence is already
107
+ // bold and a nested `**` would close it early, silently un-bolding the
108
+ // half of the sentence that carries the finding.
109
+ const named = costlier
110
+ .map((r) => {
111
+ const load = session(r);
112
+ return `\`${mdCell(r.entry.name)}\` pays ${load.isFloor ? '≥' : ''}${load.totalTokens.toLocaleString('en-US')} at session start against ${r.m.totalTokens.toLocaleString('en-US')} of definitions`;
113
+ })
114
+ .join('; ');
115
+ md.push(`**Deferring costs more than it saves on ${costlier.length} of ${measured.length} rows.** ${named}. ` +
116
+ `The names half is always a fraction of the headline, but \`instructions\` are bytes the tokens column ` +
117
+ `never counted and their length is independent of the tool set — so a server that re-lists its tools in ` +
118
+ `its instructions makes a deferring client pay for a prose copy of the schemas it just skipped. ` +
119
+ `A client that defers definitions is better off on every other measured row and worse off on ${costlier.length === 1 ? 'this one' : 'these'}.`);
120
+ md.push('');
121
+ }
122
+ md.push(`| # | server | tokens | session start |${div ? ' claude |' : ''} tools | largest tool | status | category |`);
123
+ md.push(`|---:|---|---:|---:|${div ? '---:|' : ''}---:|---|---|---|`);
62
124
  measured.forEach((r, i) => {
63
125
  const m = r.m;
64
126
  const largest = [...m.tools].sort((a, b) => b.tokens - a.tokens)[0];
65
127
  const link = `[${mdCell(r.entry.name)}](../docs/servers/${encodeURIComponent(r.entry.name)}.md)`;
66
128
  const c = claude(r);
67
129
  md.push(`| ${i + 1} | ${link} | ${m.totalTokens.toLocaleString('en-US')} |` +
130
+ ` ${sessionStartCell(session(r))} |` +
68
131
  (div ? ` ${c === null ? '—' : c.toLocaleString('en-US')} |` : '') +
69
132
  ` ${m.toolCount} | ` +
70
133
  `${largest ? `${mdCell(largest.name)} (${largest.tokens.toLocaleString('en-US')})` : '—'} | ${m.status} | ${mdCell(r.entry.category)} |`);
@@ -82,12 +145,17 @@ export function writeLeaderboard(entries, root = process.cwd()) {
82
145
  md.push('');
83
146
  }
84
147
  writeFileSync(join(root, 'results', 'leaderboard.md'), md.join('\n') + '\n');
85
- // Columns are append-only: consumers key off the header, so adding the Claude
86
- // pair at the end leaves every existing parser working.
87
- const csv = ['name,tokens,toolCount,status,category,metric,metricSource,claudeTokens,claudeModel'];
148
+ // Columns are append-only: consumers key off the header, so each new group
149
+ // (the Claude pair, then the session-start four) goes on the end and leaves
150
+ // every existing parser working.
151
+ const csv = [
152
+ 'name,tokens,toolCount,status,category,metric,metricSource,claudeTokens,claudeModel,' +
153
+ 'sessionStartTokens,sessionStartIsFloor,toolNameTokens,instructionsTokens',
154
+ ];
88
155
  for (const r of rows) {
89
156
  const m = r.m;
90
157
  const c = claude(r);
158
+ const ssl = session(r);
91
159
  csv.push([
92
160
  csvCell(r.entry.name),
93
161
  m?.totalTokens ?? '',
@@ -98,6 +166,12 @@ export function writeLeaderboard(entries, root = process.cwd()) {
98
166
  csvCell(r.entry.metricSource),
99
167
  c ?? '',
100
168
  c === null ? '' : csvCell(div.model),
169
+ ssl?.totalTokens ?? '',
170
+ // Spelled out rather than left implicit: a consumer that ignores this
171
+ // column and sums the previous one is understating every floor row.
172
+ ssl ? String(ssl.isFloor) : '',
173
+ ssl?.toolNameTokens ?? '',
174
+ ssl?.instructionsTokens ?? '',
101
175
  ].join(','));
102
176
  }
103
177
  writeFileSync(join(root, 'results', 'leaderboard.csv'), csv.join('\n') + '\n');
@@ -7,6 +7,8 @@ export interface MeasureOptions {
7
7
  dockerImage?: string;
8
8
  /** env var NAMES to provide as dummy values (docker mode). */
9
9
  dummyEnv?: string[];
10
+ /** Override the literal `dummy` value for specific `dummyEnv` names — see docker.ts. */
11
+ dummyEnvValues?: Record<string, string>;
10
12
  /** Install `git` in the container before launch (docker mode) — see docker.ts. */
11
13
  needsGit?: boolean;
12
14
  /**
@@ -21,4 +23,43 @@ export interface MeasureOptions {
21
23
  */
22
24
  persist?: boolean;
23
25
  }
26
+ /**
27
+ * Marks a startup-failure that was re-attempted from a cold package cache and
28
+ * failed the same way. Its absence on a startup-failure means the retry never
29
+ * ran (host mode, or a self-containerised command), not that it passed.
30
+ */
31
+ export declare const RETRY_CONFIRMED_PREFIX = "reproduced with the shared package cache bypassed; ";
32
+ /**
33
+ * Whether a failed measurement gets a second attempt with the shared package
34
+ * caches bypassed (see `DockerOptions.noSharedCache`).
35
+ *
36
+ * A poisoned cache entry and a genuinely broken server produce the same exit
37
+ * code, so a `startup-failure` is not published until it reproduces from a cold
38
+ * cache. Deliberately only `startup-failure`: a `timeout` is usually the slow
39
+ * install a cold retry would only make slower (it has its own retry — see
40
+ * `retriesWithLongerTimeout`), `auth-required` is a real answer about the
41
+ * server, and a command that is already its own `docker run` has no cache mount
42
+ * to bypass.
43
+ */
44
+ export declare function retriesWithoutSharedCache(status: Measurement['status'], docker: boolean, command: string): boolean;
45
+ /**
46
+ * Marks a `timeout` that was re-attempted on a larger budget and timed out
47
+ * again. Its absence on a timeout means the retry never ran, not that it passed.
48
+ */
49
+ export declare const TIMEOUT_CONFIRMED_PREFIX = "reproduced on double the timeout budget; ";
50
+ /** A timed-out measurement is re-attempted on this multiple of its budget. */
51
+ export declare const TIMEOUT_RETRY_FACTOR = 2;
52
+ /**
53
+ * Whether a timed-out measurement gets a second attempt on a larger budget.
54
+ *
55
+ * Sweeps and `audit` both run a worker pool, and a server that starts slowly
56
+ * under that contention is indistinguishable from one that never starts: both
57
+ * come back `timeout`. Two servers were published as failures for exactly this
58
+ * reason (`puppeteer` and `kubernetes`, 2026-08-19) and measured normally when
59
+ * re-run alone, so a timeout is not published until it survives a second, wider
60
+ * budget. Unlike the cold-cache retry this is isolation-independent — contention
61
+ * is not a property of the package cache, so host runs and commands that are
62
+ * their own `docker run` are retried too.
63
+ */
64
+ export declare function retriesWithLongerTimeout(status: Measurement['status']): boolean;
24
65
  export declare function measureServer(name: string, command: string, opts?: MeasureOptions): Promise<Measurement>;