mcp-context-cost 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +91 -25
- package/dist/audit/audit.d.ts +38 -0
- package/dist/audit/audit.js +372 -7
- package/dist/audit/config.d.ts +38 -0
- package/dist/audit/config.js +64 -0
- package/dist/audit/deferral.d.ts +346 -0
- package/dist/audit/deferral.js +376 -0
- package/dist/audit/run.d.ts +22 -0
- package/dist/audit/run.js +17 -1
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +10 -0
- package/dist/sweep/dashboard.js +32 -16
- package/dist/sweep/docker.d.ts +23 -0
- package/dist/sweep/docker.js +16 -12
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +16 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +41 -0
- package/dist/sweep/run.js +129 -36
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +56 -2
- package/package.json +3 -1
package/dist/sweep/history.d.ts
CHANGED
|
@@ -7,8 +7,18 @@ export interface HistoryRow {
|
|
|
7
7
|
toolCount: number;
|
|
8
8
|
/** 'measured' or 'dynamic' — dynamic means the tool set moved between captures. */
|
|
9
9
|
status: string;
|
|
10
|
+
/**
|
|
11
|
+
* How the measurement was taken: `docker` (isolated container), `host` (bare
|
|
12
|
+
* machine), or `''` when the row predates this column and the conditions are
|
|
13
|
+
* not on record. Two numbers taken under different isolation are not
|
|
14
|
+
* comparable — same server, different node, different resolution of an
|
|
15
|
+
* `@latest` tag, different ambient env — so a step between them is a property
|
|
16
|
+
* of the harness, not of the server. Recording it is what lets the trend line
|
|
17
|
+
* refuse to draw such a step; see `plottableSeries`.
|
|
18
|
+
*/
|
|
19
|
+
isolation: string;
|
|
10
20
|
}
|
|
11
|
-
export declare const HISTORY_HEADER = "date,server,tokens,toolCount,status";
|
|
21
|
+
export declare const HISTORY_HEADER = "date,server,tokens,toolCount,status,isolation";
|
|
12
22
|
/** Parse history.csv text; malformed or non-numeric rows are dropped, not thrown. */
|
|
13
23
|
export declare function parseHistory(text: string): HistoryRow[];
|
|
14
24
|
export declare function formatHistory(rows: HistoryRow[]): string;
|
|
@@ -19,6 +29,7 @@ export declare function upsert(rows: HistoryRow[], row: HistoryRow): HistoryRow[
|
|
|
19
29
|
* auth walls are recorded in the leaderboard's "not measured" section, and
|
|
20
30
|
* writing them here as zeros would fabricate a drop to zero in the series.
|
|
21
31
|
*/
|
|
32
|
+
export declare function isolationOf(m: Measurement): string;
|
|
22
33
|
export declare function rowFor(server: string, m: Measurement): HistoryRow | null;
|
|
23
34
|
/**
|
|
24
35
|
* Fold every results/<server>/measurement.json into results/history.csv.
|
|
@@ -28,3 +39,24 @@ export declare function appendHistory(root?: string): {
|
|
|
28
39
|
rows: number;
|
|
29
40
|
added: number;
|
|
30
41
|
};
|
|
42
|
+
export interface PlottableSeries {
|
|
43
|
+
/** The rows a trend may be drawn across, oldest first. */
|
|
44
|
+
rows: HistoryRow[];
|
|
45
|
+
/** Older rows excluded because they were measured under a different isolation. */
|
|
46
|
+
dropped: number;
|
|
47
|
+
/** True when a plotted row's conditions are not on record (a pre-`isolation` write). */
|
|
48
|
+
conditionsUnknown: boolean;
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* The longest run of a server's history, ending at its newest row, that a trend
|
|
52
|
+
* line may honestly be drawn across.
|
|
53
|
+
*
|
|
54
|
+
* Walking back from the newest row, a row stops the run when its isolation is
|
|
55
|
+
* known, the newest row's isolation is known, and the two differ — a step across
|
|
56
|
+
* that boundary would say the server changed when what changed is how it was
|
|
57
|
+
* measured. An *unknown* isolation is not evidence either way, so it stays in
|
|
58
|
+
* the run and is reported through `conditionsUnknown` instead: the alternative,
|
|
59
|
+
* treating unknown as its own incompatible value, would silently blank every
|
|
60
|
+
* series recorded before this column existed.
|
|
61
|
+
*/
|
|
62
|
+
export declare function plottableSeries(rows: HistoryRow[]): PlottableSeries;
|
package/dist/sweep/history.js
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
*/
|
|
9
9
|
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'node:fs';
|
|
10
10
|
import { join } from 'node:path';
|
|
11
|
-
export const HISTORY_HEADER = 'date,server,tokens,toolCount,status';
|
|
11
|
+
export const HISTORY_HEADER = 'date,server,tokens,toolCount,status,isolation';
|
|
12
12
|
function csvCell(s) {
|
|
13
13
|
const v = String(s ?? '');
|
|
14
14
|
return /[",\n]/.test(v) ? `"${v.replace(/"/g, '""')}"` : v;
|
|
@@ -48,18 +48,27 @@ export function parseHistory(text) {
|
|
|
48
48
|
for (const line of text.split('\n')) {
|
|
49
49
|
if (!line.trim() || line.startsWith('date,'))
|
|
50
50
|
continue;
|
|
51
|
-
const [date, server, tokens, toolCount, status] = splitCsvLine(line);
|
|
51
|
+
const [date, server, tokens, toolCount, status, isolation] = splitCsvLine(line);
|
|
52
52
|
if (!/^\d{4}-\d{2}-\d{2}$/.test(date ?? '') || !server)
|
|
53
53
|
continue;
|
|
54
54
|
if (!/^\d+$/.test(tokens ?? '') || !/^\d+$/.test(toolCount ?? ''))
|
|
55
55
|
continue;
|
|
56
|
-
|
|
56
|
+
// A 5-field row is a pre-`isolation` write: its conditions are unknown, and
|
|
57
|
+
// unknown is recorded as unknown rather than back-filled with a guess.
|
|
58
|
+
rows.push({
|
|
59
|
+
date,
|
|
60
|
+
server,
|
|
61
|
+
tokens: Number(tokens),
|
|
62
|
+
toolCount: Number(toolCount),
|
|
63
|
+
status: status ?? '',
|
|
64
|
+
isolation: isolation ?? '',
|
|
65
|
+
});
|
|
57
66
|
}
|
|
58
67
|
return rows;
|
|
59
68
|
}
|
|
60
69
|
export function formatHistory(rows) {
|
|
61
70
|
const sorted = [...rows].sort((a, b) => a.date.localeCompare(b.date) || a.server.localeCompare(b.server));
|
|
62
|
-
const lines = sorted.map((r) => [csvCell(r.date), csvCell(r.server), r.tokens, r.toolCount, csvCell(r.status)].join(','));
|
|
71
|
+
const lines = sorted.map((r) => [csvCell(r.date), csvCell(r.server), r.tokens, r.toolCount, csvCell(r.status), csvCell(r.isolation)].join(','));
|
|
63
72
|
return [HISTORY_HEADER, ...lines].join('\n') + '\n';
|
|
64
73
|
}
|
|
65
74
|
/** Upsert one row by (date, server) — the newest write for a day wins. */
|
|
@@ -76,6 +85,11 @@ export function upsert(rows, row) {
|
|
|
76
85
|
* auth walls are recorded in the leaderboard's "not measured" section, and
|
|
77
86
|
* writing them here as zeros would fabricate a drop to zero in the series.
|
|
78
87
|
*/
|
|
88
|
+
export function isolationOf(m) {
|
|
89
|
+
if (!m.isolation)
|
|
90
|
+
return '';
|
|
91
|
+
return m.isolation.docker ? 'docker' : 'host';
|
|
92
|
+
}
|
|
79
93
|
export function rowFor(server, m) {
|
|
80
94
|
if (m.status !== 'measured' && m.status !== 'dynamic')
|
|
81
95
|
return null;
|
|
@@ -84,7 +98,14 @@ export function rowFor(server, m) {
|
|
|
84
98
|
const date = String(m.measuredAt ?? '').slice(0, 10);
|
|
85
99
|
if (!/^\d{4}-\d{2}-\d{2}$/.test(date))
|
|
86
100
|
return null;
|
|
87
|
-
return {
|
|
101
|
+
return {
|
|
102
|
+
date,
|
|
103
|
+
server,
|
|
104
|
+
tokens: m.totalTokens,
|
|
105
|
+
toolCount: m.toolCount,
|
|
106
|
+
status: m.status,
|
|
107
|
+
isolation: isolationOf(m),
|
|
108
|
+
};
|
|
88
109
|
}
|
|
89
110
|
/**
|
|
90
111
|
* Fold every results/<server>/measurement.json into results/history.csv.
|
|
@@ -118,3 +139,37 @@ export function appendHistory(root = process.cwd()) {
|
|
|
118
139
|
writeFileSync(path, formatHistory(rows));
|
|
119
140
|
return { rows: rows.length, added: rows.length - existing.length };
|
|
120
141
|
}
|
|
142
|
+
/**
|
|
143
|
+
* The longest run of a server's history, ending at its newest row, that a trend
|
|
144
|
+
* line may honestly be drawn across.
|
|
145
|
+
*
|
|
146
|
+
* Walking back from the newest row, a row stops the run when its isolation is
|
|
147
|
+
* known, the newest row's isolation is known, and the two differ — a step across
|
|
148
|
+
* that boundary would say the server changed when what changed is how it was
|
|
149
|
+
* measured. An *unknown* isolation is not evidence either way, so it stays in
|
|
150
|
+
* the run and is reported through `conditionsUnknown` instead: the alternative,
|
|
151
|
+
* treating unknown as its own incompatible value, would silently blank every
|
|
152
|
+
* series recorded before this column existed.
|
|
153
|
+
*/
|
|
154
|
+
export function plottableSeries(rows) {
|
|
155
|
+
const sorted = [...rows].sort((a, b) => a.date.localeCompare(b.date));
|
|
156
|
+
if (!sorted.length)
|
|
157
|
+
return { rows: [], dropped: 0, conditionsUnknown: false };
|
|
158
|
+
const current = sorted[sorted.length - 1].isolation;
|
|
159
|
+
let start = 0;
|
|
160
|
+
if (current) {
|
|
161
|
+
for (let i = sorted.length - 1; i >= 0; i--) {
|
|
162
|
+
const iso = sorted[i].isolation;
|
|
163
|
+
if (iso && iso !== current) {
|
|
164
|
+
start = i + 1;
|
|
165
|
+
break;
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
const kept = sorted.slice(start);
|
|
170
|
+
return {
|
|
171
|
+
rows: kept,
|
|
172
|
+
dropped: start,
|
|
173
|
+
conditionsUnknown: kept.some((r) => !r.isolation),
|
|
174
|
+
};
|
|
175
|
+
}
|
package/dist/sweep/regen.js
CHANGED
|
@@ -1,14 +1,19 @@
|
|
|
1
|
-
/** Regenerate leaderboard + history + server pages from results/: npx tsx src/sweep/regen.ts */
|
|
1
|
+
/** Regenerate leaderboard + history + server pages + dashboard from results/: npx tsx src/sweep/regen.ts */
|
|
2
2
|
import { readFileSync } from 'node:fs';
|
|
3
3
|
import { parse } from 'yaml';
|
|
4
4
|
import { writeLeaderboard, percentiles } from './report.js';
|
|
5
5
|
import { appendHistory } from './history.js';
|
|
6
6
|
import { writeServerPages } from './server-pages.js';
|
|
7
|
+
import { writeDashboard } from './dashboard.js';
|
|
7
8
|
const doc = parse(readFileSync('servers.yaml', 'utf8'));
|
|
8
9
|
writeLeaderboard(doc.servers);
|
|
9
10
|
// History first: the server pages read history.csv for their over-time table.
|
|
10
11
|
const h = appendHistory();
|
|
11
12
|
const p = writeServerPages(doc.servers);
|
|
13
|
+
// The dashboard reads the same results/ and history.csv as the pages do, so it
|
|
14
|
+
// belongs in the same refresh — see writeDashboard's note on why it wasn't.
|
|
15
|
+
const d = writeDashboard();
|
|
12
16
|
console.log('leaderboard:', JSON.stringify(percentiles(doc.servers)));
|
|
13
17
|
console.log(`history: ${h.rows} rows (${h.added >= 0 ? '+' : ''}${h.added})`);
|
|
14
18
|
console.log(`server pages: ${p.pages}`);
|
|
19
|
+
console.log(`dashboard: ${d.out} (${(d.bytes / 1024).toFixed(0)}KB)`);
|
package/dist/sweep/report.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type DivergenceRun } from '../core/divergence.js';
|
|
2
|
+
import { type SessionStartLoad, type SessionStartRun } from '../core/session-start.js';
|
|
2
3
|
export interface ServerEntry {
|
|
3
4
|
name: string;
|
|
4
5
|
command: string;
|
|
@@ -13,11 +14,26 @@ export interface ServerEntry {
|
|
|
13
14
|
timeoutSeconds?: number;
|
|
14
15
|
/** Launch needs the `git` binary (e.g. `uvx --from git+...`) — absent from the slim isolation images. */
|
|
15
16
|
needsGit?: boolean;
|
|
17
|
+
/**
|
|
18
|
+
* Per-var overrides for the literal `dummy` placeholder docker mode injects
|
|
19
|
+
* for `env` names — for servers that parse a var's shape (URI scheme, URL)
|
|
20
|
+
* before ever reaching tools/list. See docker.ts `dummyEnvValues`.
|
|
21
|
+
*/
|
|
22
|
+
envValues?: Record<string, string>;
|
|
16
23
|
}
|
|
17
24
|
/** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
|
|
18
25
|
export declare function mdCell(s: unknown): string;
|
|
19
26
|
/** results/divergence.json if a divergence run has been recorded, else null. */
|
|
20
27
|
export declare function loadDivergence(root?: string): DivergenceRun | null;
|
|
28
|
+
/** results/session-start.json — the instructions backfill, if one exists. */
|
|
29
|
+
export declare function loadSessionStartRun(root?: string): SessionStartRun | null;
|
|
30
|
+
/**
|
|
31
|
+
* A session-start figure that is a floor reads `>= N`, never a bare `N`. The
|
|
32
|
+
* marker is half the point of publishing the number: the names half is measured,
|
|
33
|
+
* the instructions half has not been captured for this server, and a reader has
|
|
34
|
+
* to be able to tell that from a row where both halves are known.
|
|
35
|
+
*/
|
|
36
|
+
export declare function sessionStartCell(load: SessionStartLoad | null): string;
|
|
21
37
|
export declare function writeLeaderboard(entries: ServerEntry[], root?: string): void;
|
|
22
38
|
/** Percentile helper for freezing color bands against the observed distribution. */
|
|
23
39
|
export declare function percentiles(entries: ServerEntry[], root?: string): Record<string, number>;
|
package/dist/sweep/report.js
CHANGED
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
import { existsSync, readFileSync, writeFileSync } from 'node:fs';
|
|
6
6
|
import { join } from 'node:path';
|
|
7
7
|
import { isCurrent, parseDivergence } from '../core/divergence.js';
|
|
8
|
+
import { SESSION_START_METHOD, parseSessionStart, sessionStartLoad, } from '../core/session-start.js';
|
|
8
9
|
/** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
|
|
9
10
|
export function mdCell(s) {
|
|
10
11
|
return String(s ?? '')
|
|
@@ -27,9 +28,28 @@ export function loadDivergence(root = process.cwd()) {
|
|
|
27
28
|
const p = join(root, 'results', 'divergence.json');
|
|
28
29
|
return existsSync(p) ? parseDivergence(readFileSync(p, 'utf8')) : null;
|
|
29
30
|
}
|
|
31
|
+
/** results/session-start.json — the instructions backfill, if one exists. */
|
|
32
|
+
export function loadSessionStartRun(root = process.cwd()) {
|
|
33
|
+
const p = join(root, 'results', 'session-start.json');
|
|
34
|
+
return existsSync(p) ? parseSessionStart(readFileSync(p, 'utf8')) : null;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* A session-start figure that is a floor reads `>= N`, never a bare `N`. The
|
|
38
|
+
* marker is half the point of publishing the number: the names half is measured,
|
|
39
|
+
* the instructions half has not been captured for this server, and a reader has
|
|
40
|
+
* to be able to tell that from a row where both halves are known.
|
|
41
|
+
*/
|
|
42
|
+
export function sessionStartCell(load) {
|
|
43
|
+
if (!load)
|
|
44
|
+
return '—';
|
|
45
|
+
return `${load.isFloor ? '≥' : ''}${load.totalTokens.toLocaleString('en-US')}`;
|
|
46
|
+
}
|
|
30
47
|
export function writeLeaderboard(entries, root = process.cwd()) {
|
|
31
48
|
const rows = loadRows(entries, root);
|
|
32
49
|
const div = loadDivergence(root);
|
|
50
|
+
const ss = loadSessionStartRun(root);
|
|
51
|
+
/** Session-start load for a row, or null when there is no capture to read. */
|
|
52
|
+
const session = (r) => r.m ? sessionStartLoad(r.m, ss?.servers[r.entry.name]) : null;
|
|
33
53
|
/** Claude tokens for a row, or null when not measured / stale / errored. */
|
|
34
54
|
const claude = (r) => {
|
|
35
55
|
if (!div || !r.m)
|
|
@@ -57,14 +77,57 @@ export function writeLeaderboard(entries, root = process.cwd()) {
|
|
|
57
77
|
`See [Claude divergence](../docs/METHODOLOGY.md#claude-divergence).`);
|
|
58
78
|
md.push('');
|
|
59
79
|
}
|
|
60
|
-
|
|
61
|
-
md.push(
|
|
80
|
+
const floors = measured.filter((r) => session(r)?.isFloor).length;
|
|
81
|
+
md.push(`The **session start** column is what a client puts in context when it *defers* tool definitions until they ` +
|
|
82
|
+
`are used: the server's tool names plus the \`instructions\` string it returns from \`initialize\` ` +
|
|
83
|
+
`(method \`${SESSION_START_METHOD}\`). The tokens column is what a client that loads every definition up ` +
|
|
84
|
+
`front pays; this one is what the same server costs a client that does not. ` +
|
|
85
|
+
`See [session-start load](../docs/METHODOLOGY.md#session-start-load).`);
|
|
86
|
+
md.push('');
|
|
87
|
+
if (floors > 0) {
|
|
88
|
+
md.push(`**\`≥\` marks a floor, on ${floors} of ${measured.length} rows.** Tool names are counted exactly from ` +
|
|
89
|
+
`the published capture, but \`instructions\` is not part of \`tools/list\` and has not been captured for ` +
|
|
90
|
+
`these servers — so the figure is the names half alone and the true number is that or higher. A row stops ` +
|
|
91
|
+
`being a floor the first time the server is measured with its instructions.`);
|
|
92
|
+
md.push('');
|
|
93
|
+
}
|
|
94
|
+
// Deferring usually saves almost everything, but it is not guaranteed to save
|
|
95
|
+
// anything: `instructions` are bytes the headline never counted, and a server
|
|
96
|
+
// that re-lists its tools in prose can charge a deferring client more than an
|
|
97
|
+
// eager one. Those rows are the most useful thing this column finds, so they
|
|
98
|
+
// are named here rather than left for a reader to spot by comparing columns.
|
|
99
|
+
// Derived on every write — no row is listed by hand, and the paragraph
|
|
100
|
+
// disappears if the set ever empties, rather than asserting a stale count.
|
|
101
|
+
const costlier = measured.filter((r) => {
|
|
102
|
+
const load = session(r);
|
|
103
|
+
return load !== null && r.m.totalTokens !== null && load.totalTokens >= r.m.totalTokens;
|
|
104
|
+
});
|
|
105
|
+
if (costlier.length > 0) {
|
|
106
|
+
// Server names are backticked, not bolded: the lead sentence is already
|
|
107
|
+
// bold and a nested `**` would close it early, silently un-bolding the
|
|
108
|
+
// half of the sentence that carries the finding.
|
|
109
|
+
const named = costlier
|
|
110
|
+
.map((r) => {
|
|
111
|
+
const load = session(r);
|
|
112
|
+
return `\`${mdCell(r.entry.name)}\` pays ${load.isFloor ? '≥' : ''}${load.totalTokens.toLocaleString('en-US')} at session start against ${r.m.totalTokens.toLocaleString('en-US')} of definitions`;
|
|
113
|
+
})
|
|
114
|
+
.join('; ');
|
|
115
|
+
md.push(`**Deferring costs more than it saves on ${costlier.length} of ${measured.length} rows.** ${named}. ` +
|
|
116
|
+
`The names half is always a fraction of the headline, but \`instructions\` are bytes the tokens column ` +
|
|
117
|
+
`never counted and their length is independent of the tool set — so a server that re-lists its tools in ` +
|
|
118
|
+
`its instructions makes a deferring client pay for a prose copy of the schemas it just skipped. ` +
|
|
119
|
+
`A client that defers definitions is better off on every other measured row and worse off on ${costlier.length === 1 ? 'this one' : 'these'}.`);
|
|
120
|
+
md.push('');
|
|
121
|
+
}
|
|
122
|
+
md.push(`| # | server | tokens | session start |${div ? ' claude |' : ''} tools | largest tool | status | category |`);
|
|
123
|
+
md.push(`|---:|---|---:|---:|${div ? '---:|' : ''}---:|---|---|---|`);
|
|
62
124
|
measured.forEach((r, i) => {
|
|
63
125
|
const m = r.m;
|
|
64
126
|
const largest = [...m.tools].sort((a, b) => b.tokens - a.tokens)[0];
|
|
65
127
|
const link = `[${mdCell(r.entry.name)}](../docs/servers/${encodeURIComponent(r.entry.name)}.md)`;
|
|
66
128
|
const c = claude(r);
|
|
67
129
|
md.push(`| ${i + 1} | ${link} | ${m.totalTokens.toLocaleString('en-US')} |` +
|
|
130
|
+
` ${sessionStartCell(session(r))} |` +
|
|
68
131
|
(div ? ` ${c === null ? '—' : c.toLocaleString('en-US')} |` : '') +
|
|
69
132
|
` ${m.toolCount} | ` +
|
|
70
133
|
`${largest ? `${mdCell(largest.name)} (${largest.tokens.toLocaleString('en-US')})` : '—'} | ${m.status} | ${mdCell(r.entry.category)} |`);
|
|
@@ -82,12 +145,17 @@ export function writeLeaderboard(entries, root = process.cwd()) {
|
|
|
82
145
|
md.push('');
|
|
83
146
|
}
|
|
84
147
|
writeFileSync(join(root, 'results', 'leaderboard.md'), md.join('\n') + '\n');
|
|
85
|
-
// Columns are append-only: consumers key off the header, so
|
|
86
|
-
// pair
|
|
87
|
-
|
|
148
|
+
// Columns are append-only: consumers key off the header, so each new group
|
|
149
|
+
// (the Claude pair, then the session-start four) goes on the end and leaves
|
|
150
|
+
// every existing parser working.
|
|
151
|
+
const csv = [
|
|
152
|
+
'name,tokens,toolCount,status,category,metric,metricSource,claudeTokens,claudeModel,' +
|
|
153
|
+
'sessionStartTokens,sessionStartIsFloor,toolNameTokens,instructionsTokens',
|
|
154
|
+
];
|
|
88
155
|
for (const r of rows) {
|
|
89
156
|
const m = r.m;
|
|
90
157
|
const c = claude(r);
|
|
158
|
+
const ssl = session(r);
|
|
91
159
|
csv.push([
|
|
92
160
|
csvCell(r.entry.name),
|
|
93
161
|
m?.totalTokens ?? '',
|
|
@@ -98,6 +166,12 @@ export function writeLeaderboard(entries, root = process.cwd()) {
|
|
|
98
166
|
csvCell(r.entry.metricSource),
|
|
99
167
|
c ?? '',
|
|
100
168
|
c === null ? '' : csvCell(div.model),
|
|
169
|
+
ssl?.totalTokens ?? '',
|
|
170
|
+
// Spelled out rather than left implicit: a consumer that ignores this
|
|
171
|
+
// column and sums the previous one is understating every floor row.
|
|
172
|
+
ssl ? String(ssl.isFloor) : '',
|
|
173
|
+
ssl?.toolNameTokens ?? '',
|
|
174
|
+
ssl?.instructionsTokens ?? '',
|
|
101
175
|
].join(','));
|
|
102
176
|
}
|
|
103
177
|
writeFileSync(join(root, 'results', 'leaderboard.csv'), csv.join('\n') + '\n');
|
package/dist/sweep/run.d.ts
CHANGED
|
@@ -7,6 +7,8 @@ export interface MeasureOptions {
|
|
|
7
7
|
dockerImage?: string;
|
|
8
8
|
/** env var NAMES to provide as dummy values (docker mode). */
|
|
9
9
|
dummyEnv?: string[];
|
|
10
|
+
/** Override the literal `dummy` value for specific `dummyEnv` names — see docker.ts. */
|
|
11
|
+
dummyEnvValues?: Record<string, string>;
|
|
10
12
|
/** Install `git` in the container before launch (docker mode) — see docker.ts. */
|
|
11
13
|
needsGit?: boolean;
|
|
12
14
|
/**
|
|
@@ -21,4 +23,43 @@ export interface MeasureOptions {
|
|
|
21
23
|
*/
|
|
22
24
|
persist?: boolean;
|
|
23
25
|
}
|
|
26
|
+
/**
|
|
27
|
+
* Marks a startup-failure that was re-attempted from a cold package cache and
|
|
28
|
+
* failed the same way. Its absence on a startup-failure means the retry never
|
|
29
|
+
* ran (host mode, or a self-containerised command), not that it passed.
|
|
30
|
+
*/
|
|
31
|
+
export declare const RETRY_CONFIRMED_PREFIX = "reproduced with the shared package cache bypassed; ";
|
|
32
|
+
/**
|
|
33
|
+
* Whether a failed measurement gets a second attempt with the shared package
|
|
34
|
+
* caches bypassed (see `DockerOptions.noSharedCache`).
|
|
35
|
+
*
|
|
36
|
+
* A poisoned cache entry and a genuinely broken server produce the same exit
|
|
37
|
+
* code, so a `startup-failure` is not published until it reproduces from a cold
|
|
38
|
+
* cache. Deliberately only `startup-failure`: a `timeout` is usually the slow
|
|
39
|
+
* install a cold retry would only make slower (it has its own retry — see
|
|
40
|
+
* `retriesWithLongerTimeout`), `auth-required` is a real answer about the
|
|
41
|
+
* server, and a command that is already its own `docker run` has no cache mount
|
|
42
|
+
* to bypass.
|
|
43
|
+
*/
|
|
44
|
+
export declare function retriesWithoutSharedCache(status: Measurement['status'], docker: boolean, command: string): boolean;
|
|
45
|
+
/**
|
|
46
|
+
* Marks a `timeout` that was re-attempted on a larger budget and timed out
|
|
47
|
+
* again. Its absence on a timeout means the retry never ran, not that it passed.
|
|
48
|
+
*/
|
|
49
|
+
export declare const TIMEOUT_CONFIRMED_PREFIX = "reproduced on double the timeout budget; ";
|
|
50
|
+
/** A timed-out measurement is re-attempted on this multiple of its budget. */
|
|
51
|
+
export declare const TIMEOUT_RETRY_FACTOR = 2;
|
|
52
|
+
/**
|
|
53
|
+
* Whether a timed-out measurement gets a second attempt on a larger budget.
|
|
54
|
+
*
|
|
55
|
+
* Sweeps and `audit` both run a worker pool, and a server that starts slowly
|
|
56
|
+
* under that contention is indistinguishable from one that never starts: both
|
|
57
|
+
* come back `timeout`. Two servers were published as failures for exactly this
|
|
58
|
+
* reason (`puppeteer` and `kubernetes`, 2026-08-19) and measured normally when
|
|
59
|
+
* re-run alone, so a timeout is not published until it survives a second, wider
|
|
60
|
+
* budget. Unlike the cold-cache retry this is isolation-independent — contention
|
|
61
|
+
* is not a property of the package cache, so host runs and commands that are
|
|
62
|
+
* their own `docker run` are retried too.
|
|
63
|
+
*/
|
|
64
|
+
export declare function retriesWithLongerTimeout(status: Measurement['status']): boolean;
|
|
24
65
|
export declare function measureServer(name: string, command: string, opts?: MeasureOptions): Promise<Measurement>;
|
package/dist/sweep/run.js
CHANGED
|
@@ -15,6 +15,49 @@ function arg(name) {
|
|
|
15
15
|
const i = process.argv.indexOf(`--${name}`);
|
|
16
16
|
return i >= 0 ? process.argv[i + 1] : undefined;
|
|
17
17
|
}
|
|
18
|
+
/**
|
|
19
|
+
* Marks a startup-failure that was re-attempted from a cold package cache and
|
|
20
|
+
* failed the same way. Its absence on a startup-failure means the retry never
|
|
21
|
+
* ran (host mode, or a self-containerised command), not that it passed.
|
|
22
|
+
*/
|
|
23
|
+
export const RETRY_CONFIRMED_PREFIX = 'reproduced with the shared package cache bypassed; ';
|
|
24
|
+
/**
|
|
25
|
+
* Whether a failed measurement gets a second attempt with the shared package
|
|
26
|
+
* caches bypassed (see `DockerOptions.noSharedCache`).
|
|
27
|
+
*
|
|
28
|
+
* A poisoned cache entry and a genuinely broken server produce the same exit
|
|
29
|
+
* code, so a `startup-failure` is not published until it reproduces from a cold
|
|
30
|
+
* cache. Deliberately only `startup-failure`: a `timeout` is usually the slow
|
|
31
|
+
* install a cold retry would only make slower (it has its own retry — see
|
|
32
|
+
* `retriesWithLongerTimeout`), `auth-required` is a real answer about the
|
|
33
|
+
* server, and a command that is already its own `docker run` has no cache mount
|
|
34
|
+
* to bypass.
|
|
35
|
+
*/
|
|
36
|
+
export function retriesWithoutSharedCache(status, docker, command) {
|
|
37
|
+
return status === 'startup-failure' && docker && !command.trimStart().startsWith('docker ');
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Marks a `timeout` that was re-attempted on a larger budget and timed out
|
|
41
|
+
* again. Its absence on a timeout means the retry never ran, not that it passed.
|
|
42
|
+
*/
|
|
43
|
+
export const TIMEOUT_CONFIRMED_PREFIX = 'reproduced on double the timeout budget; ';
|
|
44
|
+
/** A timed-out measurement is re-attempted on this multiple of its budget. */
|
|
45
|
+
export const TIMEOUT_RETRY_FACTOR = 2;
|
|
46
|
+
/**
|
|
47
|
+
* Whether a timed-out measurement gets a second attempt on a larger budget.
|
|
48
|
+
*
|
|
49
|
+
* Sweeps and `audit` both run a worker pool, and a server that starts slowly
|
|
50
|
+
* under that contention is indistinguishable from one that never starts: both
|
|
51
|
+
* come back `timeout`. Two servers were published as failures for exactly this
|
|
52
|
+
* reason (`puppeteer` and `kubernetes`, 2026-08-19) and measured normally when
|
|
53
|
+
* re-run alone, so a timeout is not published until it survives a second, wider
|
|
54
|
+
* budget. Unlike the cold-cache retry this is isolation-independent — contention
|
|
55
|
+
* is not a property of the package cache, so host runs and commands that are
|
|
56
|
+
* their own `docker run` are retried too.
|
|
57
|
+
*/
|
|
58
|
+
export function retriesWithLongerTimeout(status) {
|
|
59
|
+
return status === 'timeout';
|
|
60
|
+
}
|
|
18
61
|
export async function measureServer(name, command, opts = {}) {
|
|
19
62
|
const persist = opts.persist !== false;
|
|
20
63
|
// The name becomes a directory when persisting; that's the only reason it's
|
|
@@ -23,57 +66,107 @@ export async function measureServer(name, command, opts = {}) {
|
|
|
23
66
|
throw new Error(`invalid server name '${name}' — letters/digits/dot/dash/underscore only`);
|
|
24
67
|
}
|
|
25
68
|
const root = opts.root ?? process.cwd();
|
|
26
|
-
|
|
69
|
+
const hostSpec = opts.argv && opts.argv.length ? { command: opts.argv[0], argv: opts.argv.slice(1) } : command;
|
|
27
70
|
let isolation = { docker: false };
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
71
|
+
const containerNames = [];
|
|
72
|
+
// A fresh container name per capture: some servers don't exit on stdin close
|
|
73
|
+
// (background timers keep the event loop alive), so `close()`'s SIGKILL only
|
|
74
|
+
// detaches the host-side `docker run` CLI — the container itself can outlive
|
|
75
|
+
// it. Reusing one name across both tools/list captures then races `--rm`'s
|
|
76
|
+
// async cleanup: the second `docker run --name X` collides with the first
|
|
77
|
+
// container mid-removal. Distinct names sidestep the race entirely; the
|
|
78
|
+
// `finally` below force-removes every name this call created either way.
|
|
79
|
+
function buildSpec(noSharedCache = false) {
|
|
80
|
+
if (opts.docker && command.trimStart().startsWith('docker ')) {
|
|
81
|
+
// Command is already a container — wrapping it again would need docker-in-docker.
|
|
82
|
+
isolation = { docker: true, note: 'command is itself a docker run (host-spawned container)' };
|
|
83
|
+
return hostSpec;
|
|
84
|
+
}
|
|
85
|
+
if (!opts.docker)
|
|
86
|
+
return hostSpec;
|
|
87
|
+
const containerName = `mcp-ctx-${name}-${process.pid}-${Math.floor(Math.random() * 1e6)}-${containerNames.length}`;
|
|
88
|
+
containerNames.push(containerName);
|
|
35
89
|
const d = dockerize(command, {
|
|
36
90
|
image: opts.dockerImage,
|
|
37
91
|
dummyEnv: opts.dummyEnv,
|
|
92
|
+
dummyEnvValues: opts.dummyEnvValues,
|
|
38
93
|
containerName,
|
|
39
94
|
needsGit: opts.needsGit,
|
|
95
|
+
noSharedCache,
|
|
40
96
|
});
|
|
41
|
-
spec = { command: d.command, argv: d.argv };
|
|
42
97
|
isolation = d.isolation;
|
|
98
|
+
return { command: d.command, argv: d.argv };
|
|
99
|
+
}
|
|
100
|
+
async function attempt(noSharedCache, timeoutMsOverride) {
|
|
101
|
+
// A cold install has to fetch everything again, so the retry gets its own
|
|
102
|
+
// floor — otherwise the bypass would trade a cache-poisoned startup-failure
|
|
103
|
+
// for a timeout and report just as wrongly.
|
|
104
|
+
const attemptOpts = timeoutMsOverride !== undefined
|
|
105
|
+
? { ...opts, timeoutMs: timeoutMsOverride }
|
|
106
|
+
: noSharedCache
|
|
107
|
+
? { ...opts, timeoutMs: Math.max(opts.timeoutMs ?? 60_000, 240_000) }
|
|
108
|
+
: opts;
|
|
109
|
+
let r;
|
|
110
|
+
try {
|
|
111
|
+
const first = await captureTools(buildSpec(noSharedCache), attemptOpts);
|
|
112
|
+
const second = await captureTools(buildSpec(noSharedCache), attemptOpts);
|
|
113
|
+
r = measureTools(first.tools, {
|
|
114
|
+
serverName: first.serverInfo?.name ?? name,
|
|
115
|
+
serverVersion: first.serverInfo?.version,
|
|
116
|
+
launchCommand: command,
|
|
117
|
+
envVarNames: [...Object.keys(opts.env ?? {}), ...(opts.dummyEnv ?? [])],
|
|
118
|
+
instructions: first.instructions,
|
|
119
|
+
});
|
|
120
|
+
if (canonicalString(first.tools) !== canonicalString(second.tools)) {
|
|
121
|
+
r.status = 'dynamic';
|
|
122
|
+
r.notes = 'tools/list differed between two runs; value is for the first capture';
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
catch (err) {
|
|
126
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
127
|
+
const status = msg.includes('timeout')
|
|
128
|
+
? 'timeout'
|
|
129
|
+
: /auth|unauthorized|401|forbidden|credential|api.?key|token/i.test(msg)
|
|
130
|
+
? 'auth-required'
|
|
131
|
+
: 'startup-failure';
|
|
132
|
+
r = failedMeasurement(status, { serverName: name, launchCommand: command, notes: msg.slice(0, 700) });
|
|
133
|
+
}
|
|
134
|
+
r.isolation = isolation;
|
|
135
|
+
r.timeoutMs = attemptOpts.timeoutMs ?? 60_000;
|
|
136
|
+
return r;
|
|
43
137
|
}
|
|
44
138
|
let m;
|
|
45
139
|
try {
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
140
|
+
m = await attempt(false);
|
|
141
|
+
if (retriesWithoutSharedCache(m.status, opts.docker === true, command)) {
|
|
142
|
+
const retry = await attempt(true);
|
|
143
|
+
if (retry.status !== 'startup-failure') {
|
|
144
|
+
m = retry;
|
|
145
|
+
}
|
|
146
|
+
else {
|
|
147
|
+
// Keep the warm attempt — its stderr is the one worth reading — but say
|
|
148
|
+
// that the failure survived a cold cache, so a reader can tell a real
|
|
149
|
+
// breakage from a cache artifact without re-running the sweep.
|
|
150
|
+
m.notes = `${RETRY_CONFIRMED_PREFIX}${m.notes ?? ''}`.slice(0, 700);
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
else if (retriesWithLongerTimeout(m.status)) {
|
|
154
|
+
const retry = await attempt(false, (opts.timeoutMs ?? 60_000) * TIMEOUT_RETRY_FACTOR);
|
|
155
|
+
// The retry replaces the first attempt either way: it is the wider budget,
|
|
156
|
+
// so its `timeoutMs` is the number a reader needs to judge the verdict.
|
|
157
|
+
m = retry;
|
|
158
|
+
if (retry.status === 'timeout') {
|
|
159
|
+
m.notes = `${TIMEOUT_CONFIRMED_PREFIX}${m.notes ?? ''}`.slice(0, 700);
|
|
160
|
+
}
|
|
59
161
|
}
|
|
60
|
-
}
|
|
61
|
-
catch (err) {
|
|
62
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
63
|
-
const status = msg.includes('timeout')
|
|
64
|
-
? 'timeout'
|
|
65
|
-
: /auth|unauthorized|401|forbidden|credential|api.?key|token/i.test(msg)
|
|
66
|
-
? 'auth-required'
|
|
67
|
-
: 'startup-failure';
|
|
68
|
-
m = failedMeasurement(status, { serverName: name, launchCommand: command, notes: msg.slice(0, 700) });
|
|
69
|
-
m.isolation = isolation;
|
|
70
|
-
m.timeoutMs = opts.timeoutMs ?? 60_000;
|
|
71
162
|
}
|
|
72
163
|
finally {
|
|
73
|
-
if (
|
|
74
|
-
// Killing the docker CLI on timeout orphans the container — remove it.
|
|
164
|
+
if (containerNames.length) {
|
|
165
|
+
// Killing the docker CLI on timeout/non-exit orphans the container — remove it.
|
|
75
166
|
const { spawn } = await import('node:child_process');
|
|
76
|
-
|
|
167
|
+
for (const n of containerNames) {
|
|
168
|
+
spawn('docker', ['rm', '-f', n], { stdio: 'ignore' }).on('error', () => { });
|
|
169
|
+
}
|
|
77
170
|
}
|
|
78
171
|
}
|
|
79
172
|
if (!persist)
|
|
@@ -11,7 +11,7 @@ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
|
|
|
11
11
|
import { join } from 'node:path';
|
|
12
12
|
import { bandColor, BAND_META } from '../core/bands.js';
|
|
13
13
|
import { loadDivergence, mdCell } from './report.js';
|
|
14
|
-
import { parseHistory } from './history.js';
|
|
14
|
+
import { parseHistory, plottableSeries } from './history.js';
|
|
15
15
|
import { claudeRatio, fieldSelectionShare, isCurrent } from '../core/divergence.js';
|
|
16
16
|
/**
|
|
17
17
|
* Pages are served from GitHub Pages (docs/), but results/ and badges/ are not
|
|
@@ -123,17 +123,42 @@ export function renderServerPage(entry, m, history = [], divergence = null) {
|
|
|
123
123
|
md.push(...divergenceSection(divRow, divergence));
|
|
124
124
|
}
|
|
125
125
|
if (history.length > 1) {
|
|
126
|
+
// A change is only a change if both numbers were taken the same way, so the
|
|
127
|
+
// delta column is blank across an isolation boundary rather than reporting a
|
|
128
|
+
// difference the harness produced.
|
|
129
|
+
const plottable = plottableSeries(history);
|
|
130
|
+
const comparableFrom = plottable.rows[0]?.date;
|
|
126
131
|
md.push('## Over time');
|
|
127
132
|
md.push('');
|
|
128
|
-
md.push('| date | tokens | tools | change |');
|
|
129
|
-
md.push('
|
|
133
|
+
md.push('| date | tokens | tools | measured in | change |');
|
|
134
|
+
md.push('|---|---:|---:|---|---:|');
|
|
130
135
|
history.forEach((h, i) => {
|
|
131
136
|
const prev = history[i - 1];
|
|
132
|
-
const
|
|
133
|
-
const
|
|
134
|
-
|
|
137
|
+
const comparable = prev && (!prev.isolation || !h.isolation || prev.isolation === h.isolation);
|
|
138
|
+
const delta = prev && comparable ? h.tokens - prev.tokens : null;
|
|
139
|
+
const change = !prev
|
|
140
|
+
? '—'
|
|
141
|
+
: !comparable
|
|
142
|
+
? 'not comparable'
|
|
143
|
+
: delta === 0
|
|
144
|
+
? 'no change'
|
|
145
|
+
: `${delta > 0 ? '+' : ''}${fmt(delta)}`;
|
|
146
|
+
md.push(`| ${mdCell(h.date)} | ${fmt(h.tokens)} | ${h.toolCount} | ${mdCell(h.isolation || 'not recorded')} | ${change} |`);
|
|
135
147
|
});
|
|
136
148
|
md.push('');
|
|
149
|
+
if (plottable.dropped) {
|
|
150
|
+
md.push(`> The first ${plottable.dropped} row${plottable.dropped === 1 ? ' was' : 's were'} measured under a ` +
|
|
151
|
+
`different isolation than the current measurement, so the published trend starts at ` +
|
|
152
|
+
`${mdCell(comparableFrom)}. Numbers taken on the host and inside a container are not ` +
|
|
153
|
+
`interchangeable — the package a \`@latest\` tag resolves to, the runtime version and the ` +
|
|
154
|
+
`ambient environment can all differ.`);
|
|
155
|
+
md.push('');
|
|
156
|
+
}
|
|
157
|
+
else if (plottable.conditionsUnknown) {
|
|
158
|
+
md.push('> Some of these sweeps predate the `isolation` column, so the conditions they were ' +
|
|
159
|
+
'measured under are not on record.');
|
|
160
|
+
md.push('');
|
|
161
|
+
}
|
|
137
162
|
md.push(`Full series: [results/history.csv](${BLOB}/results/history.csv).`);
|
|
138
163
|
md.push('');
|
|
139
164
|
}
|