mcp-context-cost 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +101 -33
- package/dist/audit/audit.d.ts +48 -0
- package/dist/audit/audit.js +392 -7
- package/dist/audit/config.d.ts +46 -0
- package/dist/audit/config.js +86 -2
- package/dist/audit/deferral.d.ts +366 -0
- package/dist/audit/deferral.js +403 -0
- package/dist/audit/run.d.ts +22 -0
- package/dist/audit/run.js +17 -1
- package/dist/cli.js +11 -0
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +10 -0
- package/dist/sweep/dashboard.js +32 -16
- package/dist/sweep/docker.d.ts +23 -0
- package/dist/sweep/docker.js +16 -12
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +16 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +41 -0
- package/dist/sweep/run.js +129 -36
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +56 -2
- package/package.json +3 -1
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A broken harness must not republish the whole set as failures.
|
|
3
|
+
*
|
|
4
|
+
* `measureServer` already adjudicates a *single* suspicious result — a
|
|
5
|
+
* startup-failure gets re-attempted off a cold cache, a timeout on double the
|
|
6
|
+
* budget (see run.ts). Neither retry can see the one failure mode that isn't
|
|
7
|
+
* about any individual server: the machine doing the measuring is broken.
|
|
8
|
+
* That is not hypothetical here — an orphaned Docker backend with a wedged
|
|
9
|
+
* network stack once returned 0/79 uniform timeouts, and every one of those
|
|
10
|
+
* results was a lie about a working server. The per-server retries make that
|
|
11
|
+
* case *worse*, not better: each one re-runs through the same broken harness
|
|
12
|
+
* and comes back failing again, which reads as confirmation.
|
|
13
|
+
*
|
|
14
|
+
* The signal a single measurement can't carry is population-level. A wave of
|
|
15
|
+
* servers that were measuring fine yesterday and all fail at once is a
|
|
16
|
+
* statement about the harness, not about the servers — upstream breakages
|
|
17
|
+
* arrive a package at a time, not by the dozen. So: snapshot what was on
|
|
18
|
+
* record before the sweep, count how many good measurements this sweep turned
|
|
19
|
+
* into failures, and if that count is both large in absolute terms and a
|
|
20
|
+
* majority of what it could have broken, refuse to publish, put the previous
|
|
21
|
+
* results back, and exit non-zero.
|
|
22
|
+
*
|
|
23
|
+
* Restoring is what makes this safe to run unattended: `measureServer`
|
|
24
|
+
* persists each result the moment it has one, so by the time a sweep-level
|
|
25
|
+
* verdict is possible the damage is already on disk. Holding the prior bytes
|
|
26
|
+
* in memory turns an irreversible overwrite into a reversible one.
|
|
27
|
+
*/
|
|
28
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
|
|
29
|
+
import { join } from 'node:path';
|
|
30
|
+
/**
|
|
31
|
+
* Below this many regressions the population signal doesn't exist yet, and a
|
|
32
|
+
* deliberately narrow sweep (`--only redis,serena`, two servers known to be
|
|
33
|
+
* shaky) must never be able to trip a fault by failing completely.
|
|
34
|
+
*/
|
|
35
|
+
export const MIN_REGRESSIONS = 5;
|
|
36
|
+
/**
|
|
37
|
+
* Share of the previously-good servers in this sweep that must regress before
|
|
38
|
+
* the harness is the likelier explanation. The largest genuine simultaneous
|
|
39
|
+
* upstream breakage on record here is a handful of PyPI servers that shared
|
|
40
|
+
* one unbounded dependency — single digits against ~65 measured, well under
|
|
41
|
+
* 10%. A harness fault, by contrast, takes everything it touches: the Docker
|
|
42
|
+
* outage above scored 100%. A majority sits far from the former and catches
|
|
43
|
+
* every instance of the latter this project has actually seen.
|
|
44
|
+
*/
|
|
45
|
+
export const FAULT_RATIO = 0.5;
|
|
46
|
+
/** Statuses that represent a real number on record — the thing worth protecting. */
|
|
47
|
+
export function isGood(status) {
|
|
48
|
+
return status === 'measured' || status === 'dynamic';
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* Read what is currently published for each server. Called before the sweep
|
|
52
|
+
* starts; a server with nothing on record snapshots as `null` status, which no
|
|
53
|
+
* verdict counts either way.
|
|
54
|
+
*/
|
|
55
|
+
export function snapshot(names, root = process.cwd()) {
|
|
56
|
+
return names.map((name) => {
|
|
57
|
+
const mPath = join(root, 'results', name, 'measurement.json');
|
|
58
|
+
const bPath = join(root, 'badges', `${name}.json`);
|
|
59
|
+
const measurementJson = existsSync(mPath) ? readFileSync(mPath, 'utf8') : null;
|
|
60
|
+
const badgeJson = existsSync(bPath) ? readFileSync(bPath, 'utf8') : null;
|
|
61
|
+
let status = null;
|
|
62
|
+
if (measurementJson) {
|
|
63
|
+
try {
|
|
64
|
+
status = JSON.parse(measurementJson).status;
|
|
65
|
+
}
|
|
66
|
+
catch {
|
|
67
|
+
// An unreadable prior record is not evidence of anything; treat it as
|
|
68
|
+
// no record rather than as a good one that just broke.
|
|
69
|
+
status = null;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
return { name, status, measurementJson, badgeJson };
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Compare pre-sweep snapshots against this sweep's outcomes.
|
|
77
|
+
*
|
|
78
|
+
* `current` maps server name to the status it just measured at. Servers absent
|
|
79
|
+
* from it were not swept and are ignored.
|
|
80
|
+
*/
|
|
81
|
+
export function verdict(prior, current) {
|
|
82
|
+
const comparableNames = prior
|
|
83
|
+
.filter((s) => s.status !== null && isGood(s.status) && current.has(s.name))
|
|
84
|
+
.map((s) => s.name);
|
|
85
|
+
const regressed = comparableNames.filter((n) => !isGood(current.get(n)));
|
|
86
|
+
const comparable = comparableNames.length;
|
|
87
|
+
if (comparable === 0) {
|
|
88
|
+
return {
|
|
89
|
+
fault: false,
|
|
90
|
+
regressed,
|
|
91
|
+
comparable,
|
|
92
|
+
// Not a pass — an unavailable reading. Nothing on record measured well
|
|
93
|
+
// before this sweep, so there is no baseline to judge it against.
|
|
94
|
+
reason: 'no prior measurement to compare against — harness check not performed',
|
|
95
|
+
};
|
|
96
|
+
}
|
|
97
|
+
const ratio = regressed.length / comparable;
|
|
98
|
+
const pct = (ratio * 100).toFixed(0);
|
|
99
|
+
if (regressed.length >= MIN_REGRESSIONS && ratio >= FAULT_RATIO) {
|
|
100
|
+
return {
|
|
101
|
+
fault: true,
|
|
102
|
+
regressed,
|
|
103
|
+
comparable,
|
|
104
|
+
reason: `${regressed.length} of ${comparable} previously-measured servers (${pct}%) failed in ` +
|
|
105
|
+
`this sweep — at or above the ${MIN_REGRESSIONS}-server, ` +
|
|
106
|
+
`${(FAULT_RATIO * 100).toFixed(0)}% threshold that reads as a broken harness ` +
|
|
107
|
+
`rather than broken servers`,
|
|
108
|
+
};
|
|
109
|
+
}
|
|
110
|
+
return {
|
|
111
|
+
fault: false,
|
|
112
|
+
regressed,
|
|
113
|
+
comparable,
|
|
114
|
+
reason: `${regressed.length} of ${comparable} previously-measured servers (${pct}%) failed in ` +
|
|
115
|
+
`this sweep — below the harness-fault threshold, publishing normally`,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* Put the snapshotted artifacts back, byte for byte. Only servers named in
|
|
120
|
+
* `names` are touched, and only where a prior file existed — a server whose
|
|
121
|
+
* first-ever measurement failed has nothing to restore and keeps its new
|
|
122
|
+
* (honest) failure record.
|
|
123
|
+
*/
|
|
124
|
+
export function restore(prior, names, root = process.cwd()) {
|
|
125
|
+
const wanted = new Set(names);
|
|
126
|
+
const restored = [];
|
|
127
|
+
for (const s of prior) {
|
|
128
|
+
if (!wanted.has(s.name))
|
|
129
|
+
continue;
|
|
130
|
+
if (s.measurementJson === null && s.badgeJson === null)
|
|
131
|
+
continue;
|
|
132
|
+
if (s.measurementJson !== null) {
|
|
133
|
+
const dir = join(root, 'results', s.name);
|
|
134
|
+
mkdirSync(dir, { recursive: true });
|
|
135
|
+
writeFileSync(join(dir, 'measurement.json'), s.measurementJson);
|
|
136
|
+
}
|
|
137
|
+
if (s.badgeJson !== null) {
|
|
138
|
+
mkdirSync(join(root, 'badges'), { recursive: true });
|
|
139
|
+
writeFileSync(join(root, 'badges', `${s.name}.json`), s.badgeJson);
|
|
140
|
+
}
|
|
141
|
+
restored.push(s.name);
|
|
142
|
+
}
|
|
143
|
+
return restored;
|
|
144
|
+
}
|
package/dist/sweep/history.d.ts
CHANGED
|
@@ -7,8 +7,18 @@ export interface HistoryRow {
|
|
|
7
7
|
toolCount: number;
|
|
8
8
|
/** 'measured' or 'dynamic' — dynamic means the tool set moved between captures. */
|
|
9
9
|
status: string;
|
|
10
|
+
/**
|
|
11
|
+
* How the measurement was taken: `docker` (isolated container), `host` (bare
|
|
12
|
+
* machine), or `''` when the row predates this column and the conditions are
|
|
13
|
+
* not on record. Two numbers taken under different isolation are not
|
|
14
|
+
* comparable — same server, different node, different resolution of an
|
|
15
|
+
* `@latest` tag, different ambient env — so a step between them is a property
|
|
16
|
+
* of the harness, not of the server. Recording it is what lets the trend line
|
|
17
|
+
* refuse to draw such a step; see `plottableSeries`.
|
|
18
|
+
*/
|
|
19
|
+
isolation: string;
|
|
10
20
|
}
|
|
11
|
-
export declare const HISTORY_HEADER = "date,server,tokens,toolCount,status";
|
|
21
|
+
export declare const HISTORY_HEADER = "date,server,tokens,toolCount,status,isolation";
|
|
12
22
|
/** Parse history.csv text; malformed or non-numeric rows are dropped, not thrown. */
|
|
13
23
|
export declare function parseHistory(text: string): HistoryRow[];
|
|
14
24
|
export declare function formatHistory(rows: HistoryRow[]): string;
|
|
@@ -19,6 +29,7 @@ export declare function upsert(rows: HistoryRow[], row: HistoryRow): HistoryRow[
|
|
|
19
29
|
* auth walls are recorded in the leaderboard's "not measured" section, and
|
|
20
30
|
* writing them here as zeros would fabricate a drop to zero in the series.
|
|
21
31
|
*/
|
|
32
|
+
export declare function isolationOf(m: Measurement): string;
|
|
22
33
|
export declare function rowFor(server: string, m: Measurement): HistoryRow | null;
|
|
23
34
|
/**
|
|
24
35
|
* Fold every results/<server>/measurement.json into results/history.csv.
|
|
@@ -28,3 +39,24 @@ export declare function appendHistory(root?: string): {
|
|
|
28
39
|
rows: number;
|
|
29
40
|
added: number;
|
|
30
41
|
};
|
|
42
|
+
export interface PlottableSeries {
|
|
43
|
+
/** The rows a trend may be drawn across, oldest first. */
|
|
44
|
+
rows: HistoryRow[];
|
|
45
|
+
/** Older rows excluded because they were measured under a different isolation. */
|
|
46
|
+
dropped: number;
|
|
47
|
+
/** True when a plotted row's conditions are not on record (a pre-`isolation` write). */
|
|
48
|
+
conditionsUnknown: boolean;
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* The longest run of a server's history, ending at its newest row, that a trend
|
|
52
|
+
* line may honestly be drawn across.
|
|
53
|
+
*
|
|
54
|
+
* Walking back from the newest row, a row stops the run when its isolation is
|
|
55
|
+
* known, the newest row's isolation is known, and the two differ — a step across
|
|
56
|
+
* that boundary would say the server changed when what changed is how it was
|
|
57
|
+
* measured. An *unknown* isolation is not evidence either way, so it stays in
|
|
58
|
+
* the run and is reported through `conditionsUnknown` instead: the alternative,
|
|
59
|
+
* treating unknown as its own incompatible value, would silently blank every
|
|
60
|
+
* series recorded before this column existed.
|
|
61
|
+
*/
|
|
62
|
+
export declare function plottableSeries(rows: HistoryRow[]): PlottableSeries;
|
package/dist/sweep/history.js
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
*/
|
|
9
9
|
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'node:fs';
|
|
10
10
|
import { join } from 'node:path';
|
|
11
|
-
export const HISTORY_HEADER = 'date,server,tokens,toolCount,status';
|
|
11
|
+
export const HISTORY_HEADER = 'date,server,tokens,toolCount,status,isolation';
|
|
12
12
|
function csvCell(s) {
|
|
13
13
|
const v = String(s ?? '');
|
|
14
14
|
return /[",\n]/.test(v) ? `"${v.replace(/"/g, '""')}"` : v;
|
|
@@ -48,18 +48,27 @@ export function parseHistory(text) {
|
|
|
48
48
|
for (const line of text.split('\n')) {
|
|
49
49
|
if (!line.trim() || line.startsWith('date,'))
|
|
50
50
|
continue;
|
|
51
|
-
const [date, server, tokens, toolCount, status] = splitCsvLine(line);
|
|
51
|
+
const [date, server, tokens, toolCount, status, isolation] = splitCsvLine(line);
|
|
52
52
|
if (!/^\d{4}-\d{2}-\d{2}$/.test(date ?? '') || !server)
|
|
53
53
|
continue;
|
|
54
54
|
if (!/^\d+$/.test(tokens ?? '') || !/^\d+$/.test(toolCount ?? ''))
|
|
55
55
|
continue;
|
|
56
|
-
|
|
56
|
+
// A 5-field row is a pre-`isolation` write: its conditions are unknown, and
|
|
57
|
+
// unknown is recorded as unknown rather than back-filled with a guess.
|
|
58
|
+
rows.push({
|
|
59
|
+
date,
|
|
60
|
+
server,
|
|
61
|
+
tokens: Number(tokens),
|
|
62
|
+
toolCount: Number(toolCount),
|
|
63
|
+
status: status ?? '',
|
|
64
|
+
isolation: isolation ?? '',
|
|
65
|
+
});
|
|
57
66
|
}
|
|
58
67
|
return rows;
|
|
59
68
|
}
|
|
60
69
|
export function formatHistory(rows) {
|
|
61
70
|
const sorted = [...rows].sort((a, b) => a.date.localeCompare(b.date) || a.server.localeCompare(b.server));
|
|
62
|
-
const lines = sorted.map((r) => [csvCell(r.date), csvCell(r.server), r.tokens, r.toolCount, csvCell(r.status)].join(','));
|
|
71
|
+
const lines = sorted.map((r) => [csvCell(r.date), csvCell(r.server), r.tokens, r.toolCount, csvCell(r.status), csvCell(r.isolation)].join(','));
|
|
63
72
|
return [HISTORY_HEADER, ...lines].join('\n') + '\n';
|
|
64
73
|
}
|
|
65
74
|
/** Upsert one row by (date, server) — the newest write for a day wins. */
|
|
@@ -76,6 +85,11 @@ export function upsert(rows, row) {
|
|
|
76
85
|
* auth walls are recorded in the leaderboard's "not measured" section, and
|
|
77
86
|
* writing them here as zeros would fabricate a drop to zero in the series.
|
|
78
87
|
*/
|
|
88
|
+
export function isolationOf(m) {
|
|
89
|
+
if (!m.isolation)
|
|
90
|
+
return '';
|
|
91
|
+
return m.isolation.docker ? 'docker' : 'host';
|
|
92
|
+
}
|
|
79
93
|
export function rowFor(server, m) {
|
|
80
94
|
if (m.status !== 'measured' && m.status !== 'dynamic')
|
|
81
95
|
return null;
|
|
@@ -84,7 +98,14 @@ export function rowFor(server, m) {
|
|
|
84
98
|
const date = String(m.measuredAt ?? '').slice(0, 10);
|
|
85
99
|
if (!/^\d{4}-\d{2}-\d{2}$/.test(date))
|
|
86
100
|
return null;
|
|
87
|
-
return {
|
|
101
|
+
return {
|
|
102
|
+
date,
|
|
103
|
+
server,
|
|
104
|
+
tokens: m.totalTokens,
|
|
105
|
+
toolCount: m.toolCount,
|
|
106
|
+
status: m.status,
|
|
107
|
+
isolation: isolationOf(m),
|
|
108
|
+
};
|
|
88
109
|
}
|
|
89
110
|
/**
|
|
90
111
|
* Fold every results/<server>/measurement.json into results/history.csv.
|
|
@@ -118,3 +139,37 @@ export function appendHistory(root = process.cwd()) {
|
|
|
118
139
|
writeFileSync(path, formatHistory(rows));
|
|
119
140
|
return { rows: rows.length, added: rows.length - existing.length };
|
|
120
141
|
}
|
|
142
|
+
/**
|
|
143
|
+
* The longest run of a server's history, ending at its newest row, that a trend
|
|
144
|
+
* line may honestly be drawn across.
|
|
145
|
+
*
|
|
146
|
+
* Walking back from the newest row, a row stops the run when its isolation is
|
|
147
|
+
* known, the newest row's isolation is known, and the two differ — a step across
|
|
148
|
+
* that boundary would say the server changed when what changed is how it was
|
|
149
|
+
* measured. An *unknown* isolation is not evidence either way, so it stays in
|
|
150
|
+
* the run and is reported through `conditionsUnknown` instead: the alternative,
|
|
151
|
+
* treating unknown as its own incompatible value, would silently blank every
|
|
152
|
+
* series recorded before this column existed.
|
|
153
|
+
*/
|
|
154
|
+
export function plottableSeries(rows) {
|
|
155
|
+
const sorted = [...rows].sort((a, b) => a.date.localeCompare(b.date));
|
|
156
|
+
if (!sorted.length)
|
|
157
|
+
return { rows: [], dropped: 0, conditionsUnknown: false };
|
|
158
|
+
const current = sorted[sorted.length - 1].isolation;
|
|
159
|
+
let start = 0;
|
|
160
|
+
if (current) {
|
|
161
|
+
for (let i = sorted.length - 1; i >= 0; i--) {
|
|
162
|
+
const iso = sorted[i].isolation;
|
|
163
|
+
if (iso && iso !== current) {
|
|
164
|
+
start = i + 1;
|
|
165
|
+
break;
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
const kept = sorted.slice(start);
|
|
170
|
+
return {
|
|
171
|
+
rows: kept,
|
|
172
|
+
dropped: start,
|
|
173
|
+
conditionsUnknown: kept.some((r) => !r.isolation),
|
|
174
|
+
};
|
|
175
|
+
}
|
package/dist/sweep/regen.js
CHANGED
|
@@ -1,14 +1,19 @@
|
|
|
1
|
-
/** Regenerate leaderboard + history + server pages from results/: npx tsx src/sweep/regen.ts */
|
|
1
|
+
/** Regenerate leaderboard + history + server pages + dashboard from results/: npx tsx src/sweep/regen.ts */
|
|
2
2
|
import { readFileSync } from 'node:fs';
|
|
3
3
|
import { parse } from 'yaml';
|
|
4
4
|
import { writeLeaderboard, percentiles } from './report.js';
|
|
5
5
|
import { appendHistory } from './history.js';
|
|
6
6
|
import { writeServerPages } from './server-pages.js';
|
|
7
|
+
import { writeDashboard } from './dashboard.js';
|
|
7
8
|
const doc = parse(readFileSync('servers.yaml', 'utf8'));
|
|
8
9
|
writeLeaderboard(doc.servers);
|
|
9
10
|
// History first: the server pages read history.csv for their over-time table.
|
|
10
11
|
const h = appendHistory();
|
|
11
12
|
const p = writeServerPages(doc.servers);
|
|
13
|
+
// The dashboard reads the same results/ and history.csv as the pages do, so it
|
|
14
|
+
// belongs in the same refresh — see writeDashboard's note on why it wasn't.
|
|
15
|
+
const d = writeDashboard();
|
|
12
16
|
console.log('leaderboard:', JSON.stringify(percentiles(doc.servers)));
|
|
13
17
|
console.log(`history: ${h.rows} rows (${h.added >= 0 ? '+' : ''}${h.added})`);
|
|
14
18
|
console.log(`server pages: ${p.pages}`);
|
|
19
|
+
console.log(`dashboard: ${d.out} (${(d.bytes / 1024).toFixed(0)}KB)`);
|
package/dist/sweep/report.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type DivergenceRun } from '../core/divergence.js';
|
|
2
|
+
import { type SessionStartLoad, type SessionStartRun } from '../core/session-start.js';
|
|
2
3
|
export interface ServerEntry {
|
|
3
4
|
name: string;
|
|
4
5
|
command: string;
|
|
@@ -13,11 +14,26 @@ export interface ServerEntry {
|
|
|
13
14
|
timeoutSeconds?: number;
|
|
14
15
|
/** Launch needs the `git` binary (e.g. `uvx --from git+...`) — absent from the slim isolation images. */
|
|
15
16
|
needsGit?: boolean;
|
|
17
|
+
/**
|
|
18
|
+
* Per-var overrides for the literal `dummy` placeholder docker mode injects
|
|
19
|
+
* for `env` names — for servers that parse a var's shape (URI scheme, URL)
|
|
20
|
+
* before ever reaching tools/list. See docker.ts `dummyEnvValues`.
|
|
21
|
+
*/
|
|
22
|
+
envValues?: Record<string, string>;
|
|
16
23
|
}
|
|
17
24
|
/** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
|
|
18
25
|
export declare function mdCell(s: unknown): string;
|
|
19
26
|
/** results/divergence.json if a divergence run has been recorded, else null. */
|
|
20
27
|
export declare function loadDivergence(root?: string): DivergenceRun | null;
|
|
28
|
+
/** results/session-start.json — the instructions backfill, if one exists. */
|
|
29
|
+
export declare function loadSessionStartRun(root?: string): SessionStartRun | null;
|
|
30
|
+
/**
|
|
31
|
+
* A session-start figure that is a floor reads `>= N`, never a bare `N`. The
|
|
32
|
+
* marker is half the point of publishing the number: the names half is measured,
|
|
33
|
+
* the instructions half has not been captured for this server, and a reader has
|
|
34
|
+
* to be able to tell that from a row where both halves are known.
|
|
35
|
+
*/
|
|
36
|
+
export declare function sessionStartCell(load: SessionStartLoad | null): string;
|
|
21
37
|
export declare function writeLeaderboard(entries: ServerEntry[], root?: string): void;
|
|
22
38
|
/** Percentile helper for freezing color bands against the observed distribution. */
|
|
23
39
|
export declare function percentiles(entries: ServerEntry[], root?: string): Record<string, number>;
|
package/dist/sweep/report.js
CHANGED
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
import { existsSync, readFileSync, writeFileSync } from 'node:fs';
|
|
6
6
|
import { join } from 'node:path';
|
|
7
7
|
import { isCurrent, parseDivergence } from '../core/divergence.js';
|
|
8
|
+
import { SESSION_START_METHOD, parseSessionStart, sessionStartLoad, } from '../core/session-start.js';
|
|
8
9
|
/** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
|
|
9
10
|
export function mdCell(s) {
|
|
10
11
|
return String(s ?? '')
|
|
@@ -27,9 +28,28 @@ export function loadDivergence(root = process.cwd()) {
|
|
|
27
28
|
const p = join(root, 'results', 'divergence.json');
|
|
28
29
|
return existsSync(p) ? parseDivergence(readFileSync(p, 'utf8')) : null;
|
|
29
30
|
}
|
|
31
|
+
/** results/session-start.json — the instructions backfill, if one exists. */
|
|
32
|
+
export function loadSessionStartRun(root = process.cwd()) {
|
|
33
|
+
const p = join(root, 'results', 'session-start.json');
|
|
34
|
+
return existsSync(p) ? parseSessionStart(readFileSync(p, 'utf8')) : null;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* A session-start figure that is a floor reads `>= N`, never a bare `N`. The
|
|
38
|
+
* marker is half the point of publishing the number: the names half is measured,
|
|
39
|
+
* the instructions half has not been captured for this server, and a reader has
|
|
40
|
+
* to be able to tell that from a row where both halves are known.
|
|
41
|
+
*/
|
|
42
|
+
export function sessionStartCell(load) {
|
|
43
|
+
if (!load)
|
|
44
|
+
return '—';
|
|
45
|
+
return `${load.isFloor ? '≥' : ''}${load.totalTokens.toLocaleString('en-US')}`;
|
|
46
|
+
}
|
|
30
47
|
export function writeLeaderboard(entries, root = process.cwd()) {
|
|
31
48
|
const rows = loadRows(entries, root);
|
|
32
49
|
const div = loadDivergence(root);
|
|
50
|
+
const ss = loadSessionStartRun(root);
|
|
51
|
+
/** Session-start load for a row, or null when there is no capture to read. */
|
|
52
|
+
const session = (r) => r.m ? sessionStartLoad(r.m, ss?.servers[r.entry.name]) : null;
|
|
33
53
|
/** Claude tokens for a row, or null when not measured / stale / errored. */
|
|
34
54
|
const claude = (r) => {
|
|
35
55
|
if (!div || !r.m)
|
|
@@ -57,14 +77,57 @@ export function writeLeaderboard(entries, root = process.cwd()) {
|
|
|
57
77
|
`See [Claude divergence](../docs/METHODOLOGY.md#claude-divergence).`);
|
|
58
78
|
md.push('');
|
|
59
79
|
}
|
|
60
|
-
|
|
61
|
-
md.push(
|
|
80
|
+
const floors = measured.filter((r) => session(r)?.isFloor).length;
|
|
81
|
+
md.push(`The **session start** column is what a client puts in context when it *defers* tool definitions until they ` +
|
|
82
|
+
`are used: the server's tool names plus the \`instructions\` string it returns from \`initialize\` ` +
|
|
83
|
+
`(method \`${SESSION_START_METHOD}\`). The tokens column is what a client that loads every definition up ` +
|
|
84
|
+
`front pays; this one is what the same server costs a client that does not. ` +
|
|
85
|
+
`See [session-start load](../docs/METHODOLOGY.md#session-start-load).`);
|
|
86
|
+
md.push('');
|
|
87
|
+
if (floors > 0) {
|
|
88
|
+
md.push(`**\`≥\` marks a floor, on ${floors} of ${measured.length} rows.** Tool names are counted exactly from ` +
|
|
89
|
+
`the published capture, but \`instructions\` is not part of \`tools/list\` and has not been captured for ` +
|
|
90
|
+
`these servers — so the figure is the names half alone and the true number is that or higher. A row stops ` +
|
|
91
|
+
`being a floor the first time the server is measured with its instructions.`);
|
|
92
|
+
md.push('');
|
|
93
|
+
}
|
|
94
|
+
// Deferring usually saves almost everything, but it is not guaranteed to save
|
|
95
|
+
// anything: `instructions` are bytes the headline never counted, and a server
|
|
96
|
+
// that re-lists its tools in prose can charge a deferring client more than an
|
|
97
|
+
// eager one. Those rows are the most useful thing this column finds, so they
|
|
98
|
+
// are named here rather than left for a reader to spot by comparing columns.
|
|
99
|
+
// Derived on every write — no row is listed by hand, and the paragraph
|
|
100
|
+
// disappears if the set ever empties, rather than asserting a stale count.
|
|
101
|
+
const costlier = measured.filter((r) => {
|
|
102
|
+
const load = session(r);
|
|
103
|
+
return load !== null && r.m.totalTokens !== null && load.totalTokens >= r.m.totalTokens;
|
|
104
|
+
});
|
|
105
|
+
if (costlier.length > 0) {
|
|
106
|
+
// Server names are backticked, not bolded: the lead sentence is already
|
|
107
|
+
// bold and a nested `**` would close it early, silently un-bolding the
|
|
108
|
+
// half of the sentence that carries the finding.
|
|
109
|
+
const named = costlier
|
|
110
|
+
.map((r) => {
|
|
111
|
+
const load = session(r);
|
|
112
|
+
return `\`${mdCell(r.entry.name)}\` pays ${load.isFloor ? '≥' : ''}${load.totalTokens.toLocaleString('en-US')} at session start against ${r.m.totalTokens.toLocaleString('en-US')} of definitions`;
|
|
113
|
+
})
|
|
114
|
+
.join('; ');
|
|
115
|
+
md.push(`**Deferring costs more than it saves on ${costlier.length} of ${measured.length} rows.** ${named}. ` +
|
|
116
|
+
`The names half is always a fraction of the headline, but \`instructions\` are bytes the tokens column ` +
|
|
117
|
+
`never counted and their length is independent of the tool set — so a server that re-lists its tools in ` +
|
|
118
|
+
`its instructions makes a deferring client pay for a prose copy of the schemas it just skipped. ` +
|
|
119
|
+
`A client that defers definitions is better off on every other measured row and worse off on ${costlier.length === 1 ? 'this one' : 'these'}.`);
|
|
120
|
+
md.push('');
|
|
121
|
+
}
|
|
122
|
+
md.push(`| # | server | tokens | session start |${div ? ' claude |' : ''} tools | largest tool | status | category |`);
|
|
123
|
+
md.push(`|---:|---|---:|---:|${div ? '---:|' : ''}---:|---|---|---|`);
|
|
62
124
|
measured.forEach((r, i) => {
|
|
63
125
|
const m = r.m;
|
|
64
126
|
const largest = [...m.tools].sort((a, b) => b.tokens - a.tokens)[0];
|
|
65
127
|
const link = `[${mdCell(r.entry.name)}](../docs/servers/${encodeURIComponent(r.entry.name)}.md)`;
|
|
66
128
|
const c = claude(r);
|
|
67
129
|
md.push(`| ${i + 1} | ${link} | ${m.totalTokens.toLocaleString('en-US')} |` +
|
|
130
|
+
` ${sessionStartCell(session(r))} |` +
|
|
68
131
|
(div ? ` ${c === null ? '—' : c.toLocaleString('en-US')} |` : '') +
|
|
69
132
|
` ${m.toolCount} | ` +
|
|
70
133
|
`${largest ? `${mdCell(largest.name)} (${largest.tokens.toLocaleString('en-US')})` : '—'} | ${m.status} | ${mdCell(r.entry.category)} |`);
|
|
@@ -82,12 +145,17 @@ export function writeLeaderboard(entries, root = process.cwd()) {
|
|
|
82
145
|
md.push('');
|
|
83
146
|
}
|
|
84
147
|
writeFileSync(join(root, 'results', 'leaderboard.md'), md.join('\n') + '\n');
|
|
85
|
-
// Columns are append-only: consumers key off the header, so
|
|
86
|
-
// pair
|
|
87
|
-
|
|
148
|
+
// Columns are append-only: consumers key off the header, so each new group
|
|
149
|
+
// (the Claude pair, then the session-start four) goes on the end and leaves
|
|
150
|
+
// every existing parser working.
|
|
151
|
+
const csv = [
|
|
152
|
+
'name,tokens,toolCount,status,category,metric,metricSource,claudeTokens,claudeModel,' +
|
|
153
|
+
'sessionStartTokens,sessionStartIsFloor,toolNameTokens,instructionsTokens',
|
|
154
|
+
];
|
|
88
155
|
for (const r of rows) {
|
|
89
156
|
const m = r.m;
|
|
90
157
|
const c = claude(r);
|
|
158
|
+
const ssl = session(r);
|
|
91
159
|
csv.push([
|
|
92
160
|
csvCell(r.entry.name),
|
|
93
161
|
m?.totalTokens ?? '',
|
|
@@ -98,6 +166,12 @@ export function writeLeaderboard(entries, root = process.cwd()) {
|
|
|
98
166
|
csvCell(r.entry.metricSource),
|
|
99
167
|
c ?? '',
|
|
100
168
|
c === null ? '' : csvCell(div.model),
|
|
169
|
+
ssl?.totalTokens ?? '',
|
|
170
|
+
// Spelled out rather than left implicit: a consumer that ignores this
|
|
171
|
+
// column and sums the previous one is understating every floor row.
|
|
172
|
+
ssl ? String(ssl.isFloor) : '',
|
|
173
|
+
ssl?.toolNameTokens ?? '',
|
|
174
|
+
ssl?.instructionsTokens ?? '',
|
|
101
175
|
].join(','));
|
|
102
176
|
}
|
|
103
177
|
writeFileSync(join(root, 'results', 'leaderboard.csv'), csv.join('\n') + '\n');
|
package/dist/sweep/run.d.ts
CHANGED
|
@@ -7,6 +7,8 @@ export interface MeasureOptions {
|
|
|
7
7
|
dockerImage?: string;
|
|
8
8
|
/** env var NAMES to provide as dummy values (docker mode). */
|
|
9
9
|
dummyEnv?: string[];
|
|
10
|
+
/** Override the literal `dummy` value for specific `dummyEnv` names — see docker.ts. */
|
|
11
|
+
dummyEnvValues?: Record<string, string>;
|
|
10
12
|
/** Install `git` in the container before launch (docker mode) — see docker.ts. */
|
|
11
13
|
needsGit?: boolean;
|
|
12
14
|
/**
|
|
@@ -21,4 +23,43 @@ export interface MeasureOptions {
|
|
|
21
23
|
*/
|
|
22
24
|
persist?: boolean;
|
|
23
25
|
}
|
|
26
|
+
/**
|
|
27
|
+
* Marks a startup-failure that was re-attempted from a cold package cache and
|
|
28
|
+
* failed the same way. Its absence on a startup-failure means the retry never
|
|
29
|
+
* ran (host mode, or a self-containerised command), not that it passed.
|
|
30
|
+
*/
|
|
31
|
+
export declare const RETRY_CONFIRMED_PREFIX = "reproduced with the shared package cache bypassed; ";
|
|
32
|
+
/**
|
|
33
|
+
* Whether a failed measurement gets a second attempt with the shared package
|
|
34
|
+
* caches bypassed (see `DockerOptions.noSharedCache`).
|
|
35
|
+
*
|
|
36
|
+
* A poisoned cache entry and a genuinely broken server produce the same exit
|
|
37
|
+
* code, so a `startup-failure` is not published until it reproduces from a cold
|
|
38
|
+
* cache. Deliberately only `startup-failure`: a `timeout` is usually the slow
|
|
39
|
+
* install a cold retry would only make slower (it has its own retry — see
|
|
40
|
+
* `retriesWithLongerTimeout`), `auth-required` is a real answer about the
|
|
41
|
+
* server, and a command that is already its own `docker run` has no cache mount
|
|
42
|
+
* to bypass.
|
|
43
|
+
*/
|
|
44
|
+
export declare function retriesWithoutSharedCache(status: Measurement['status'], docker: boolean, command: string): boolean;
|
|
45
|
+
/**
|
|
46
|
+
* Marks a `timeout` that was re-attempted on a larger budget and timed out
|
|
47
|
+
* again. Its absence on a timeout means the retry never ran, not that it passed.
|
|
48
|
+
*/
|
|
49
|
+
export declare const TIMEOUT_CONFIRMED_PREFIX = "reproduced on double the timeout budget; ";
|
|
50
|
+
/** A timed-out measurement is re-attempted on this multiple of its budget. */
|
|
51
|
+
export declare const TIMEOUT_RETRY_FACTOR = 2;
|
|
52
|
+
/**
|
|
53
|
+
* Whether a timed-out measurement gets a second attempt on a larger budget.
|
|
54
|
+
*
|
|
55
|
+
* Sweeps and `audit` both run a worker pool, and a server that starts slowly
|
|
56
|
+
* under that contention is indistinguishable from one that never starts: both
|
|
57
|
+
* come back `timeout`. Two servers were published as failures for exactly this
|
|
58
|
+
* reason (`puppeteer` and `kubernetes`, 2026-08-19) and measured normally when
|
|
59
|
+
* re-run alone, so a timeout is not published until it survives a second, wider
|
|
60
|
+
* budget. Unlike the cold-cache retry this is isolation-independent — contention
|
|
61
|
+
* is not a property of the package cache, so host runs and commands that are
|
|
62
|
+
* their own `docker run` are retried too.
|
|
63
|
+
*/
|
|
64
|
+
export declare function retriesWithLongerTimeout(status: Measurement['status']): boolean;
|
|
24
65
|
export declare function measureServer(name: string, command: string, opts?: MeasureOptions): Promise<Measurement>;
|