mcp-context-cost 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +202 -43
- package/dist/audit/audit.d.ts +101 -0
- package/dist/audit/audit.js +492 -16
- package/dist/audit/config.d.ts +38 -0
- package/dist/audit/config.js +64 -0
- package/dist/audit/deferral.d.ts +346 -0
- package/dist/audit/deferral.js +376 -0
- package/dist/audit/diff.d.ts +124 -0
- package/dist/audit/diff.js +318 -0
- package/dist/audit/run.d.ts +34 -0
- package/dist/audit/run.js +45 -2
- package/dist/cli.d.ts +21 -0
- package/dist/cli.js +141 -7
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +18 -0
- package/dist/sweep/dashboard.js +74 -10
- package/dist/sweep/docker.d.ts +31 -0
- package/dist/sweep/docker.js +20 -11
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +18 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +43 -0
- package/dist/sweep/run.js +135 -37
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +57 -2
- package/package.json +3 -1
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Session-start load — what a client actually puts in context when it defers
|
|
3
|
+
* tool definitions until they are used.
|
|
4
|
+
*
|
|
5
|
+
* The headline number counts every byte of `tools/list`, because that is what a
|
|
6
|
+
* client loading definitions eagerly pays. Clients increasingly do not: they put
|
|
7
|
+
* a *list of names* in context and fetch the definition only when the model
|
|
8
|
+
* reaches for the tool. That client pays a different bill at session start, and
|
|
9
|
+
* the headline number says nothing about it.
|
|
10
|
+
*
|
|
11
|
+
* Different, and usually far smaller — but *not* smaller by construction, and
|
|
12
|
+
* nothing here may claim it is. Only one of the two halves below is bounded by
|
|
13
|
+
* the definitions being deferred. Instructions are separate bytes that no part
|
|
14
|
+
* of the headline counts, and their length has nothing to do with the size of
|
|
15
|
+
* the tool set, so a server that re-lists its tools in its `instructions`
|
|
16
|
+
* charges a deferring client *more* than an eager one. That is not
|
|
17
|
+
* hypothetical: it is true of a server on the published leaderboard, which is
|
|
18
|
+
* the most useful thing this measurement has found, and every renderer of these
|
|
19
|
+
* numbers is expected to let the reader see it rather than smooth it away.
|
|
20
|
+
*
|
|
21
|
+
* Two quantities make up that bill, and only one of them is in the published
|
|
22
|
+
* capture:
|
|
23
|
+
*
|
|
24
|
+
* 1. **Tool names.** Derivable from `rawToolsCapture` — no re-measurement
|
|
25
|
+
* needed, and re-derivable by anyone from the same published bytes.
|
|
26
|
+
* 2. **Server instructions.** The `instructions` string a server returns from
|
|
27
|
+
* `initialize`. It is *not* part of `tools/list` and so is absent from every
|
|
28
|
+
* measurement recorded before this field existed.
|
|
29
|
+
*
|
|
30
|
+
* Which means a row can be in one of two honest states, and they are not the
|
|
31
|
+
* same claim: an exact figure (both halves known), or a **floor** (names
|
|
32
|
+
* measured, instructions never captured). A floor published as a figure is the
|
|
33
|
+
* failure mode this module exists to prevent — see `isFloor`, which every
|
|
34
|
+
* renderer must carry through to the reader.
|
|
35
|
+
*
|
|
36
|
+
* Versioned independently of the o200k methodology, on the same reasoning as
|
|
37
|
+
* `tools-delta/v1`: this adds a second published number and does not touch the
|
|
38
|
+
* definition of the first. No `totalTokens` and no canonical hash moves.
|
|
39
|
+
*/
|
|
40
|
+
import { countTokens, sha256Hex } from './canonical.js';
|
|
41
|
+
/** Method identifier, versioned independently of METHODOLOGY_VERSION. */
|
|
42
|
+
export const SESSION_START_METHOD = 'deferred-load/v1';
|
|
43
|
+
/**
|
|
44
|
+
* Tool names in server-returned order. A tool without a usable name is dropped
|
|
45
|
+
* rather than given a placeholder — the same rule `toAnthropicTools` follows,
|
|
46
|
+
* for the same reason: an invented name would change the count being published.
|
|
47
|
+
*/
|
|
48
|
+
export function toolNames(raw) {
|
|
49
|
+
const out = [];
|
|
50
|
+
for (const t of raw) {
|
|
51
|
+
const name = (t ?? {}).name;
|
|
52
|
+
if (typeof name === 'string' && name !== '')
|
|
53
|
+
out.push(name);
|
|
54
|
+
}
|
|
55
|
+
return out;
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* o200k count of the canonical tool-name list: `JSON.stringify` over the name
|
|
59
|
+
* array, the same serialization discipline as the headline number, so both
|
|
60
|
+
* re-derive from the same capture with the same five lines.
|
|
61
|
+
*/
|
|
62
|
+
export function toolNameTokens(raw) {
|
|
63
|
+
return countTokens(JSON.stringify(toolNames(raw)));
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* The two halves are counted separately and added, rather than counted over one
|
|
67
|
+
* concatenated string. Concatenation would let boundary tokens merge across the
|
|
68
|
+
* seam, so the published total would not equal the sum of the parts printed
|
|
69
|
+
* beside it — a discrepancy of a token or two that no reader could account for.
|
|
70
|
+
*/
|
|
71
|
+
export function sessionStartTokens(raw, instructions) {
|
|
72
|
+
return toolNameTokens(raw) + (instructions ? countTokens(instructions) : 0);
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* The instructions recorded *inside* a measurement, distinguishing three states
|
|
76
|
+
* that JSON round-trips faithfully:
|
|
77
|
+
* - a string → captured, the server sent this
|
|
78
|
+
* - null → captured, the server sent none
|
|
79
|
+
* - undefined → never captured (every measurement predating the field)
|
|
80
|
+
*
|
|
81
|
+
* Absent-means-unknown rather than absent-means-zero, on the precedent set by
|
|
82
|
+
* history.csv's `isolation` column: an old row reads as unknown and is never
|
|
83
|
+
* back-filled from a value it did not record.
|
|
84
|
+
*/
|
|
85
|
+
export function measuredInstructions(m) {
|
|
86
|
+
if (typeof m.serverInstructions === 'string')
|
|
87
|
+
return m.serverInstructions;
|
|
88
|
+
if (m.serverInstructions === null)
|
|
89
|
+
return '';
|
|
90
|
+
return undefined;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Resolve one server's session-start load from its measurement and, if it has
|
|
94
|
+
* one, the instructions captured beside it.
|
|
95
|
+
*
|
|
96
|
+
* Precedence is measurement over capture and never the other way round: the
|
|
97
|
+
* measurement's instructions came off the same server process as its tools, in
|
|
98
|
+
* the same run, so it cannot be stale relative to itself. The side capture is
|
|
99
|
+
* the backfill for measurements taken before the field existed, and it is used
|
|
100
|
+
* only while it still points at the measurement on disk.
|
|
101
|
+
*
|
|
102
|
+
* Returns null for a measurement with no capture to read names from — a
|
|
103
|
+
* `startup-failure` has no session-start load because it has no session.
|
|
104
|
+
*/
|
|
105
|
+
export function sessionStartLoad(m, row) {
|
|
106
|
+
if (!Array.isArray(m.rawToolsCapture))
|
|
107
|
+
return null;
|
|
108
|
+
const names = toolNameTokens(m.rawToolsCapture);
|
|
109
|
+
const toolCount = toolNames(m.rawToolsCapture).length;
|
|
110
|
+
const inMeasurement = measuredInstructions(m);
|
|
111
|
+
if (inMeasurement !== undefined) {
|
|
112
|
+
const t = inMeasurement ? countTokens(inMeasurement) : 0;
|
|
113
|
+
return {
|
|
114
|
+
toolCount,
|
|
115
|
+
toolNameTokens: names,
|
|
116
|
+
instructionsTokens: t,
|
|
117
|
+
totalTokens: names + t,
|
|
118
|
+
isFloor: false,
|
|
119
|
+
instructionsSource: 'measurement',
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
if (isCurrentInstructions(row, m.canonicalSha256)) {
|
|
123
|
+
return {
|
|
124
|
+
toolCount,
|
|
125
|
+
toolNameTokens: names,
|
|
126
|
+
instructionsTokens: row.instructionsTokens,
|
|
127
|
+
totalTokens: names + row.instructionsTokens,
|
|
128
|
+
isFloor: false,
|
|
129
|
+
instructionsSource: 'capture',
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
return {
|
|
133
|
+
toolCount,
|
|
134
|
+
toolNameTokens: names,
|
|
135
|
+
instructionsTokens: null,
|
|
136
|
+
totalTokens: names,
|
|
137
|
+
isFloor: true,
|
|
138
|
+
instructionsSource: 'not-captured',
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
/**
|
|
142
|
+
* A side capture is usable only if it carries a number and still points at the
|
|
143
|
+
* measurement on disk. Unlike a stale divergence row — which is hidden, because
|
|
144
|
+
* there is nothing else to print — a stale row here degrades the figure to its
|
|
145
|
+
* names-only floor. The column never blanks; it only ever stops claiming to
|
|
146
|
+
* know the half it no longer knows.
|
|
147
|
+
*/
|
|
148
|
+
export function isCurrentInstructions(row, canonicalSha256) {
|
|
149
|
+
if (!row || row.error)
|
|
150
|
+
return false;
|
|
151
|
+
if (typeof row.instructionsTokens !== 'number')
|
|
152
|
+
return false;
|
|
153
|
+
return !!canonicalSha256 && row.capturedSha256 === canonicalSha256;
|
|
154
|
+
}
|
|
155
|
+
/** Build a row from a freshly captured instructions string. */
|
|
156
|
+
export function toSessionStartRow(instructions, meta) {
|
|
157
|
+
const text = instructions ?? '';
|
|
158
|
+
return {
|
|
159
|
+
instructions: text,
|
|
160
|
+
instructionsTokens: text ? countTokens(text) : 0,
|
|
161
|
+
instructionsSha256: sha256Hex(text),
|
|
162
|
+
capturedSha256: meta.capturedSha256,
|
|
163
|
+
serverVersion: meta.serverVersion,
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
/** Parse results/session-start.json; anything malformed yields null, never throws. */
|
|
167
|
+
export function parseSessionStart(text) {
|
|
168
|
+
let run;
|
|
169
|
+
try {
|
|
170
|
+
run = JSON.parse(text);
|
|
171
|
+
}
|
|
172
|
+
catch {
|
|
173
|
+
return null;
|
|
174
|
+
}
|
|
175
|
+
const r = run;
|
|
176
|
+
if (!r || typeof r.measuredAt !== 'string')
|
|
177
|
+
return null;
|
|
178
|
+
if (!r.servers || typeof r.servers !== 'object')
|
|
179
|
+
return null;
|
|
180
|
+
return {
|
|
181
|
+
method: typeof r.method === 'string' ? r.method : SESSION_START_METHOD,
|
|
182
|
+
measuredAt: r.measuredAt,
|
|
183
|
+
isolation: typeof r.isolation === 'string' ? r.isolation : undefined,
|
|
184
|
+
servers: r.servers,
|
|
185
|
+
};
|
|
186
|
+
}
|
package/dist/core/types.d.ts
CHANGED
|
@@ -23,6 +23,14 @@ export interface Measurement {
|
|
|
23
23
|
canonicalSha256: string | null;
|
|
24
24
|
/** The tools array exactly as returned by tools/list, all pages concatenated. */
|
|
25
25
|
rawToolsCapture: unknown[] | null;
|
|
26
|
+
/**
|
|
27
|
+
* The `instructions` string returned by `initialize` — half of the
|
|
28
|
+
* session-start load (see core/session-start.ts). Three states, and they are
|
|
29
|
+
* different claims: a string (the server sent this), `null` (the server sent
|
|
30
|
+
* none), absent (never captured — every measurement predating this field).
|
|
31
|
+
* Absent is never read as zero.
|
|
32
|
+
*/
|
|
33
|
+
serverInstructions?: string | null;
|
|
26
34
|
measuredAt: string;
|
|
27
35
|
serverName: string;
|
|
28
36
|
serverVersion?: string;
|
package/dist/sweep/client.d.ts
CHANGED
|
@@ -5,7 +5,12 @@ export interface WireCapture {
|
|
|
5
5
|
};
|
|
6
6
|
protocolVersion?: string;
|
|
7
7
|
tools: unknown[];
|
|
8
|
-
/**
|
|
8
|
+
/**
|
|
9
|
+
* The `instructions` string from the initialize result — half of the
|
|
10
|
+
* session-start load. `null` means the server returned none, which is a
|
|
11
|
+
* different fact from never having asked, so the two are never conflated.
|
|
12
|
+
*/
|
|
13
|
+
instructions: string | null;
|
|
9
14
|
stderrTail: string;
|
|
10
15
|
}
|
|
11
16
|
export declare class McpStdioClient {
|
package/dist/sweep/client.js
CHANGED
|
@@ -164,6 +164,7 @@ export async function captureTools(spec, opts = {}) {
|
|
|
164
164
|
serverInfo: init?.serverInfo,
|
|
165
165
|
protocolVersion: init?.protocolVersion,
|
|
166
166
|
tools,
|
|
167
|
+
instructions: typeof init?.instructions === 'string' ? init.instructions : null,
|
|
167
168
|
stderrTail: client.stderrTail,
|
|
168
169
|
};
|
|
169
170
|
}
|
|
@@ -1 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A 12-point-max inline trend line, oldest to newest. Muted stroke (this is
|
|
3
|
+
* texture, not a headline number) with the current value picked out as an
|
|
4
|
+
* accent dot, per the sparkline spec: de-emphasis hue for the line, accent
|
|
5
|
+
* for "now". Flat series still draw a level line rather than faking a zero
|
|
6
|
+
* baseline. Returns '' when there's nothing to trend (0-1 points).
|
|
7
|
+
*/
|
|
8
|
+
export declare function renderSparkline(tokens: number[]): string;
|
|
1
9
|
export declare function generateDashboard(root?: string): string;
|
|
10
|
+
/**
|
|
11
|
+
* Render and write the dashboard. Exported so `regen.ts` refreshes it alongside
|
|
12
|
+
* the leaderboard and the server pages: this is the published Pages surface, and
|
|
13
|
+
* when only the CLI wrote it every sweep updated `results/` and `docs/servers/`
|
|
14
|
+
* while the page people actually open kept the previous run's numbers.
|
|
15
|
+
*/
|
|
16
|
+
export declare function writeDashboard(root?: string, out?: string): {
|
|
17
|
+
out: string;
|
|
18
|
+
bytes: number;
|
|
19
|
+
};
|
package/dist/sweep/dashboard.js
CHANGED
|
@@ -7,6 +7,38 @@ import { existsSync, readFileSync, writeFileSync, mkdirSync } from 'node:fs';
|
|
|
7
7
|
import { dirname, join } from 'node:path';
|
|
8
8
|
import { parse } from 'yaml';
|
|
9
9
|
import { bandColor, BAND_META } from '../core/bands.js';
|
|
10
|
+
import { parseHistory, plottableSeries } from './history.js';
|
|
11
|
+
/** Longest series a sparkline plots — a stat-tile trend, not a full chart. */
|
|
12
|
+
const SPARK_MAX_POINTS = 12;
|
|
13
|
+
/**
|
|
14
|
+
* A 12-point-max inline trend line, oldest to newest. Muted stroke (this is
|
|
15
|
+
* texture, not a headline number) with the current value picked out as an
|
|
16
|
+
* accent dot, per the sparkline spec: de-emphasis hue for the line, accent
|
|
17
|
+
* for "now". Flat series still draw a level line rather than faking a zero
|
|
18
|
+
* baseline. Returns '' when there's nothing to trend (0-1 points).
|
|
19
|
+
*/
|
|
20
|
+
export function renderSparkline(tokens) {
|
|
21
|
+
const points = tokens.slice(-SPARK_MAX_POINTS);
|
|
22
|
+
if (points.length < 2)
|
|
23
|
+
return '';
|
|
24
|
+
const w = 56;
|
|
25
|
+
const h = 18;
|
|
26
|
+
const pad = 3;
|
|
27
|
+
const min = Math.min(...points);
|
|
28
|
+
const max = Math.max(...points);
|
|
29
|
+
const range = max - min;
|
|
30
|
+
const stepX = (w - pad * 2) / (points.length - 1);
|
|
31
|
+
const coords = points.map((v, i) => {
|
|
32
|
+
const x = pad + i * stepX;
|
|
33
|
+
const y = range === 0 ? h / 2 : pad + (h - pad * 2) * (1 - (v - min) / range);
|
|
34
|
+
return `${x.toFixed(1)},${y.toFixed(1)}`;
|
|
35
|
+
});
|
|
36
|
+
const [lastX, lastY] = coords[coords.length - 1].split(',');
|
|
37
|
+
return `<svg class="spark" width="${w}" height="${h}" viewBox="0 0 ${w} ${h}" aria-hidden="true" focusable="false">
|
|
38
|
+
<polyline points="${coords.join(' ')}" fill="none" stroke="var(--muted)" stroke-width="1.6" stroke-linecap="round" stroke-linejoin="round"/>
|
|
39
|
+
<circle cx="${lastX}" cy="${lastY}" r="2.2" fill="var(--accent)" stroke="var(--surface)" stroke-width="1"/>
|
|
40
|
+
</svg>`;
|
|
41
|
+
}
|
|
10
42
|
const esc = (s) => String(s ?? '').replace(/&/g, '&').replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"');
|
|
11
43
|
export function generateDashboard(root = process.cwd()) {
|
|
12
44
|
const doc = parse(readFileSync(join(root, 'servers.yaml'), 'utf8'));
|
|
@@ -15,6 +47,11 @@ export function generateDashboard(root = process.cwd()) {
|
|
|
15
47
|
? JSON.parse(readFileSync(divergencePath, 'utf8'))
|
|
16
48
|
: {};
|
|
17
49
|
const dSrv = divergence.servers ?? {};
|
|
50
|
+
const historyPath = join(root, 'results', 'history.csv');
|
|
51
|
+
const history = existsSync(historyPath) ? parseHistory(readFileSync(historyPath, 'utf8')) : [];
|
|
52
|
+
// Only the run of sweeps taken under the same isolation is plotted: a step
|
|
53
|
+
// across an isolation change is the harness moving, not the server.
|
|
54
|
+
const seriesFor = (name) => plottableSeries(history.filter((h) => h.server === name));
|
|
18
55
|
const rows = doc.servers.map((entry) => {
|
|
19
56
|
const p = join(root, 'results', entry.name, 'measurement.json');
|
|
20
57
|
return { entry, m: existsSync(p) ? JSON.parse(readFileSync(p, 'utf8')) : null };
|
|
@@ -39,11 +76,23 @@ export function generateDashboard(root = process.cwd()) {
|
|
|
39
76
|
const pct = Math.max(1.2, (t / max) * 100);
|
|
40
77
|
const div = dSrv[r.entry.name];
|
|
41
78
|
const claudeTip = div ? ` · in a Claude request: ${fmt(div.claudeDelta)} tok` : '';
|
|
79
|
+
const series = seriesFor(r.entry.name);
|
|
80
|
+
const tokens = series.rows.map((h) => h.tokens);
|
|
81
|
+
const spark = renderSparkline(tokens);
|
|
82
|
+
const trendTip = tokens.length > 1
|
|
83
|
+
? ` · ${tokens.length}-sweep trend: ${tokens[0].toLocaleString('en-US')} → ${tokens[tokens.length - 1].toLocaleString('en-US')}` +
|
|
84
|
+
(series.dropped
|
|
85
|
+
? ` (${series.dropped} earlier sweep${series.dropped === 1 ? '' : 's'} measured under different isolation, not plotted)`
|
|
86
|
+
: '')
|
|
87
|
+
: series.dropped
|
|
88
|
+
? ` · earlier sweeps were measured under different isolation, so no trend is plotted`
|
|
89
|
+
: '';
|
|
42
90
|
// The whole row is the link to the server's detail page (docs/servers/).
|
|
43
|
-
return `<a class="row" href="servers/${encodeURIComponent(r.entry.name)}.html" data-tip="${esc(m.toolCount)} tools · largest: ${esc(largest?.name)} (${fmt(largest?.tokens ?? 0)} tok)${claudeTip} · ${esc(r.entry.category)} · ${esc(m.status)}${m.serverVersion ? ' · v' + esc(String(m.serverVersion).replace(/^v/, '')) : ''}">
|
|
91
|
+
return `<a class="row" href="servers/${encodeURIComponent(r.entry.name)}.html" data-tip="${esc(m.toolCount)} tools · largest: ${esc(largest?.name)} (${fmt(largest?.tokens ?? 0)} tok)${claudeTip}${trendTip} · ${esc(r.entry.category)} · ${esc(m.status)}${m.serverVersion ? ' · v' + esc(String(m.serverVersion).replace(/^v/, '')) : ''}">
|
|
44
92
|
<span class="rank">${i + 1}</span>
|
|
45
93
|
<span class="name">${esc(r.entry.name)}</span>
|
|
46
94
|
<span class="track"><span class="bar" style="width:${pct.toFixed(1)}%"></span></span>
|
|
95
|
+
<span class="spark-cell">${spark}</span>
|
|
47
96
|
<span class="val"><span class="dot dot-${band}" aria-hidden="true"></span>${fmt(t)}<span class="bandname">${meta.label}</span></span>
|
|
48
97
|
</a>`;
|
|
49
98
|
})
|
|
@@ -59,7 +108,11 @@ export function generateDashboard(root = process.cwd()) {
|
|
|
59
108
|
.map((r, i) => {
|
|
60
109
|
const m = r.m;
|
|
61
110
|
const div = dSrv[r.entry.name];
|
|
62
|
-
|
|
111
|
+
const tokens = seriesFor(r.entry.name).rows.map((h) => h.tokens);
|
|
112
|
+
const trend = tokens.length > 1
|
|
113
|
+
? `${tokens[tokens.length - 1] - tokens[0] >= 0 ? '+' : ''}${fmt(tokens[tokens.length - 1] - tokens[0])} / ${tokens.length}d`
|
|
114
|
+
: '—';
|
|
115
|
+
return `<tr><td>${i + 1}</td><td>${esc(r.entry.name)}</td><td class="num">${fmt(m.totalTokens)}</td><td class="num">${div ? fmt(div.claudeDelta) : '—'}</td><td class="num">${esc(m.toolCount)}</td><td>${esc(BAND_META[bandColor(m.totalTokens)].label)}</td><td>${esc(r.entry.category)}</td><td class="num">${trend}</td></tr>`;
|
|
63
116
|
})
|
|
64
117
|
.join('\n');
|
|
65
118
|
const specimens = [measured[measured.length - 1], measured[0]]
|
|
@@ -117,13 +170,15 @@ export function generateDashboard(root = process.cwd()) {
|
|
|
117
170
|
.stat .l { font-size: 11px; letter-spacing: 0.08em; text-transform: uppercase; color: var(--muted); }
|
|
118
171
|
|
|
119
172
|
.board { background: var(--surface); border: 1px solid var(--line); border-radius: 8px; padding: 14px 16px; }
|
|
120
|
-
.row { display: grid; grid-template-columns: 2ch minmax(120px, 190px) 1fr max-content; gap: 10px; align-items: center; padding: 3px 4px; border-radius: 4px; outline: none; color: inherit; text-decoration: none; }
|
|
173
|
+
.row { display: grid; grid-template-columns: 2ch minmax(120px, 190px) 1fr 56px max-content; gap: 10px; align-items: center; padding: 3px 4px; border-radius: 4px; outline: none; color: inherit; text-decoration: none; }
|
|
121
174
|
.row:hover, .row:focus-visible { background: var(--accent-soft); }
|
|
122
175
|
.row:hover .name, .row:focus-visible .name { text-decoration: underline; }
|
|
123
176
|
.rank { font-family: ui-monospace, Menlo, monospace; font-size: 11px; color: var(--muted); text-align: right; font-variant-numeric: tabular-nums; }
|
|
124
177
|
.name { font-size: 0.86rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
|
125
178
|
.track { background: var(--track); border-radius: 3px; height: 12px; overflow: hidden; }
|
|
126
179
|
.bar { display: block; height: 100%; background: var(--accent); border-radius: 3px 3px 3px 3px; min-width: 3px; }
|
|
180
|
+
.spark-cell { display: flex; align-items: center; justify-content: center; height: 18px; }
|
|
181
|
+
.spark-cell:empty { visibility: hidden; }
|
|
127
182
|
.val { font-family: ui-monospace, Menlo, monospace; font-variant-numeric: tabular-nums; font-size: 0.82rem; display: flex; align-items: center; gap: 6px; }
|
|
128
183
|
.dot { width: 8px; height: 8px; border-radius: 50%; display: inline-block; box-shadow: 0 0 0 2px var(--surface); }
|
|
129
184
|
.dot-brightgreen { background: var(--b-brightgreen); } .dot-green { background: var(--b-green); }
|
|
@@ -172,7 +227,7 @@ export function generateDashboard(root = process.cwd()) {
|
|
|
172
227
|
</div>
|
|
173
228
|
|
|
174
229
|
<h2>Leaderboard</h2>
|
|
175
|
-
<p class="h2sub">Tokens = o200k_base count of the canonical <code>tools/list</code> bytes — the wire payload. What a <em>Claude request</em> actually carries can differ sharply (github: 54,422 on the wire, ${dSrv['github'] ? fmt(dSrv['github'].claudeDelta) : '…'} in a request — 80% of its schema bytes are fields no Anthropic request sends). Hover a row for
|
|
230
|
+
<p class="h2sub">Tokens = o200k_base count of the canonical <code>tools/list</code> bytes — the wire payload. What a <em>Claude request</em> actually carries can differ sharply (github: 54,422 on the wire, ${dSrv['github'] ? fmt(dSrv['github'].claudeDelta) : '…'} in a request — 80% of its schema bytes are fields no Anthropic request sends). The trend line plots tokens across every sweep date on record (oldest→newest, dot = current); servers with one sweep so far show no line yet, and a sweep taken under different isolation than the current one is left out rather than drawn as a change in the server. Hover a row for exact numbers; <a href="METHODOLOGY.html#claude-divergence">method</a>.</p>
|
|
176
231
|
<div class="board">
|
|
177
232
|
${barRows || '<p class="h2sub">Sweep in progress — first results land shortly.</p>'}
|
|
178
233
|
</div>
|
|
@@ -198,7 +253,7 @@ ${barRows || '<p class="h2sub">Sweep in progress — first results land shortly.
|
|
|
198
253
|
|
|
199
254
|
<details><summary>Full data table</summary>
|
|
200
255
|
<div class="tablewrap" style="margin-top:10px"><table>
|
|
201
|
-
<thead><tr><th>#</th><th>server</th><th>tokens (o200k)</th><th>claude req</th><th>tools</th><th>band</th><th>category</th></tr></thead>
|
|
256
|
+
<thead><tr><th>#</th><th>server</th><th>tokens (o200k)</th><th>claude req</th><th>tools</th><th>band</th><th>category</th><th>trend</th></tr></thead>
|
|
202
257
|
<tbody>${tableRows}</tbody>
|
|
203
258
|
</table></div>
|
|
204
259
|
</details>
|
|
@@ -234,12 +289,21 @@ ${barRows || '<p class="h2sub">Sweep in progress — first results land shortly.
|
|
|
234
289
|
</script>
|
|
235
290
|
`;
|
|
236
291
|
}
|
|
292
|
+
/**
|
|
293
|
+
* Render and write the dashboard. Exported so `regen.ts` refreshes it alongside
|
|
294
|
+
* the leaderboard and the server pages: this is the published Pages surface, and
|
|
295
|
+
* when only the CLI wrote it every sweep updated `results/` and `docs/servers/`
|
|
296
|
+
* while the page people actually open kept the previous run's numbers.
|
|
297
|
+
*/
|
|
298
|
+
export function writeDashboard(root = process.cwd(), out = 'docs/dashboard.html') {
|
|
299
|
+
const html = generateDashboard(root);
|
|
300
|
+
mkdirSync(dirname(out), { recursive: true });
|
|
301
|
+
writeFileSync(out, html);
|
|
302
|
+
return { out, bytes: html.length };
|
|
303
|
+
}
|
|
237
304
|
const isMain = process.argv[1]?.endsWith('dashboard.ts') || process.argv[1]?.endsWith('dashboard.js');
|
|
238
305
|
if (isMain) {
|
|
239
306
|
const i = process.argv.indexOf('--out');
|
|
240
|
-
const
|
|
241
|
-
|
|
242
|
-
mkdirSync(dirname(out), { recursive: true });
|
|
243
|
-
writeFileSync(out, html);
|
|
244
|
-
console.log(`wrote ${out} (${(html.length / 1024).toFixed(0)}KB)`);
|
|
307
|
+
const w = writeDashboard(process.cwd(), i >= 0 ? process.argv[i + 1] : 'docs/dashboard.html');
|
|
308
|
+
console.log(`wrote ${w.out} (${(w.bytes / 1024).toFixed(0)}KB)`);
|
|
245
309
|
}
|
package/dist/sweep/docker.d.ts
CHANGED
|
@@ -10,8 +10,39 @@ export interface DockerOptions {
|
|
|
10
10
|
image?: string;
|
|
11
11
|
/** Extra env var NAMES to pass through with dummy values. */
|
|
12
12
|
dummyEnv?: string[];
|
|
13
|
+
/**
|
|
14
|
+
* Override the literal `dummy` value for specific `dummyEnv` names. Some
|
|
15
|
+
* servers parse an env var's shape before ever reaching tools/list (a URI
|
|
16
|
+
* scheme, a URL) and crash on a value that isn't real-looking rather than
|
|
17
|
+
* a value that isn't a real credential — e.g. `NEO4J_URI=dummy` fails
|
|
18
|
+
* `neo4j.exceptions.ConfigurationError` before the driver is even asked to
|
|
19
|
+
* connect. A locally-scoped placeholder (`bolt://localhost:7687`) clears
|
|
20
|
+
* that parse step without providing a working credential.
|
|
21
|
+
*/
|
|
22
|
+
dummyEnvValues?: Record<string, string>;
|
|
13
23
|
/** Container name, so a timed-out container can be force-removed. */
|
|
14
24
|
containerName?: string;
|
|
25
|
+
/**
|
|
26
|
+
* Install `git` inside the container before launch. The slim base images
|
|
27
|
+
* carry no VCS, so `uvx --from git+...` installs fail with "Git executable
|
|
28
|
+
* not found" — this only prefixes the containerized invocation, never the
|
|
29
|
+
* recorded `launchCommand`, so the published command stays what a user with
|
|
30
|
+
* git already on PATH would actually run.
|
|
31
|
+
*/
|
|
32
|
+
needsGit?: boolean;
|
|
33
|
+
/**
|
|
34
|
+
* Skip the shared npm/uv cache volumes, paying a cold install for a clean one.
|
|
35
|
+
*
|
|
36
|
+
* The caches are what make a re-sweep minutes instead of hours, but a cache
|
|
37
|
+
* entry can go bad and stay bad: npm remembers a failed postinstall, and every
|
|
38
|
+
* later install that resolves through it fails the same way. Observed
|
|
39
|
+
* 2026-08-19 — a cached `esbuild@0.25.12` whose postinstall exits 1 made
|
|
40
|
+
* `npx -y @shopify/dev-mcp@latest` exit 1 with no output, against a server that
|
|
41
|
+
* installs and answers fine from an empty cache. Nothing in the exit code tells
|
|
42
|
+
* that apart from a genuinely broken server, so the sweep retries here rather
|
|
43
|
+
* than publishing the downgrade.
|
|
44
|
+
*/
|
|
45
|
+
noSharedCache?: boolean;
|
|
15
46
|
}
|
|
16
47
|
export declare const DEFAULT_NODE_IMAGE = "public.ecr.aws/docker/library/node:22-slim";
|
|
17
48
|
export declare const DEFAULT_PYTHON_IMAGE = "ghcr.io/astral-sh/uv:python3.12-bookworm-slim";
|
package/dist/sweep/docker.js
CHANGED
|
@@ -23,19 +23,26 @@ export function dockerize(commandLine, opts = {}) {
|
|
|
23
23
|
'-w',
|
|
24
24
|
'/tmp',
|
|
25
25
|
// Shared package caches across containers — packages only, no credentials.
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
26
|
+
...(opts.noSharedCache
|
|
27
|
+
? []
|
|
28
|
+
: [
|
|
29
|
+
'-v',
|
|
30
|
+
'mcp-ctx-npm-cache:/tmp/.npm-cache',
|
|
31
|
+
'-e',
|
|
32
|
+
'npm_config_cache=/tmp/.npm-cache',
|
|
33
|
+
'-v',
|
|
34
|
+
'mcp-ctx-uv-cache:/tmp/.uv-cache',
|
|
35
|
+
'-e',
|
|
36
|
+
'UV_CACHE_DIR=/tmp/.uv-cache',
|
|
37
|
+
]),
|
|
34
38
|
];
|
|
35
39
|
for (const name of opts.dummyEnv ?? []) {
|
|
36
|
-
argv.push('-e', `${name}
|
|
40
|
+
argv.push('-e', `${name}=${opts.dummyEnvValues?.[name] ?? 'dummy'}`);
|
|
37
41
|
}
|
|
38
|
-
|
|
42
|
+
const gitPrefix = opts.needsGit
|
|
43
|
+
? 'command -v git >/dev/null 2>&1 || (apt-get update -qq && apt-get install -y -qq --no-install-recommends git >/dev/null 2>&1); '
|
|
44
|
+
: '';
|
|
45
|
+
argv.push(image, 'sh', '-lc', gitPrefix + commandLine);
|
|
39
46
|
return {
|
|
40
47
|
command: 'docker',
|
|
41
48
|
argv,
|
|
@@ -43,7 +50,9 @@ export function dockerize(commandLine, opts = {}) {
|
|
|
43
50
|
docker: true,
|
|
44
51
|
image,
|
|
45
52
|
network: 'bridge',
|
|
46
|
-
note: 'network enabled for package fetch; clean FS, no host credentials'
|
|
53
|
+
note: 'network enabled for package fetch; clean FS, no host credentials' +
|
|
54
|
+
(opts.needsGit ? ', git installed' : '') +
|
|
55
|
+
(opts.noSharedCache ? ', shared package cache bypassed' : ''),
|
|
47
56
|
},
|
|
48
57
|
};
|
|
49
58
|
}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import type { MeasurementStatus } from '../core/types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Below this many regressions the population signal doesn't exist yet, and a
|
|
4
|
+
* deliberately narrow sweep (`--only redis,serena`, two servers known to be
|
|
5
|
+
* shaky) must never be able to trip a fault by failing completely.
|
|
6
|
+
*/
|
|
7
|
+
export declare const MIN_REGRESSIONS = 5;
|
|
8
|
+
/**
|
|
9
|
+
* Share of the previously-good servers in this sweep that must regress before
|
|
10
|
+
* the harness is the likelier explanation. The largest genuine simultaneous
|
|
11
|
+
* upstream breakage on record here is a handful of PyPI servers that shared
|
|
12
|
+
* one unbounded dependency — single digits against ~65 measured, well under
|
|
13
|
+
* 10%. A harness fault, by contrast, takes everything it touches: the Docker
|
|
14
|
+
* outage above scored 100%. A majority sits far from the former and catches
|
|
15
|
+
* every instance of the latter this project has actually seen.
|
|
16
|
+
*/
|
|
17
|
+
export declare const FAULT_RATIO = 0.5;
|
|
18
|
+
/** Statuses that represent a real number on record — the thing worth protecting. */
|
|
19
|
+
export declare function isGood(status: MeasurementStatus): boolean;
|
|
20
|
+
/** The bytes of one server's published artifacts, exactly as they were pre-sweep. */
|
|
21
|
+
export interface Snapshot {
|
|
22
|
+
name: string;
|
|
23
|
+
status: MeasurementStatus | null;
|
|
24
|
+
/** Raw file contents, so a restore is byte-identical rather than re-serialized. */
|
|
25
|
+
measurementJson: string | null;
|
|
26
|
+
badgeJson: string | null;
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* Read what is currently published for each server. Called before the sweep
|
|
30
|
+
* starts; a server with nothing on record snapshots as `null` status, which no
|
|
31
|
+
* verdict counts either way.
|
|
32
|
+
*/
|
|
33
|
+
export declare function snapshot(names: string[], root?: string): Snapshot[];
|
|
34
|
+
export interface Verdict {
|
|
35
|
+
/** True when the sweep looks like a broken harness rather than broken servers. */
|
|
36
|
+
fault: boolean;
|
|
37
|
+
/** Servers that went from a real number to a failure in this sweep. */
|
|
38
|
+
regressed: string[];
|
|
39
|
+
/** How many servers had a real number on record and could therefore regress. */
|
|
40
|
+
comparable: number;
|
|
41
|
+
/** Human-readable reason, printed either way — a pass says why it passed. */
|
|
42
|
+
reason: string;
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Compare pre-sweep snapshots against this sweep's outcomes.
|
|
46
|
+
*
|
|
47
|
+
* `current` maps server name to the status it just measured at. Servers absent
|
|
48
|
+
* from it were not swept and are ignored.
|
|
49
|
+
*/
|
|
50
|
+
export declare function verdict(prior: Snapshot[], current: Map<string, MeasurementStatus>): Verdict;
|
|
51
|
+
/**
|
|
52
|
+
* Put the snapshotted artifacts back, byte for byte. Only servers named in
|
|
53
|
+
* `names` are touched, and only where a prior file existed — a server whose
|
|
54
|
+
* first-ever measurement failed has nothing to restore and keeps its new
|
|
55
|
+
* (honest) failure record.
|
|
56
|
+
*/
|
|
57
|
+
export declare function restore(prior: Snapshot[], names: string[], root?: string): string[];
|