mcp-context-cost 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,59 @@
1
+ /**
2
+ * Audit orchestration: discover configs, measure each distinct server once,
3
+ * hand the results to `buildReport`.
4
+ *
5
+ * Kept apart from audit.ts so the arithmetic stays spawn-free and testable.
6
+ */
7
+ import { homedir } from 'node:os';
8
+ import { measureServer } from '../sweep/run.js';
9
+ import { buildReport, serverKey } from './audit.js';
10
+ import { configCandidates, loadConfigs } from './config.js';
11
+ export function discover(opts = {}) {
12
+ const cwd = opts.cwd ?? process.cwd();
13
+ const home = opts.home ?? homedir();
14
+ const candidates = opts.configPaths && opts.configPaths.length
15
+ ? opts.configPaths.map((path) => ({ client: 'explicit', path }))
16
+ : configCandidates({ home, cwd, platform: process.platform, appData: process.env.APPDATA });
17
+ return loadConfigs(candidates, cwd);
18
+ }
19
+ /** Measure every distinct stdio server across the given configs, once each. */
20
+ export async function measureAll(configs, opts = {}) {
21
+ const unique = new Map();
22
+ for (const cfg of configs) {
23
+ for (const s of cfg.servers) {
24
+ if (s.transport !== 'stdio')
25
+ continue;
26
+ // Two clients pointing at the same argv are one measurement, not two.
27
+ if (!unique.has(serverKey(s)))
28
+ unique.set(serverKey(s), s);
29
+ }
30
+ }
31
+ const queue = [...unique.entries()];
32
+ const total = queue.length;
33
+ const measured = new Map();
34
+ let done = 0;
35
+ const worker = async () => {
36
+ for (let next = queue.shift(); next; next = queue.shift()) {
37
+ const [key, s] = next;
38
+ const m = await measureServer(s.name, s.command ?? '', {
39
+ argv: s.argv,
40
+ env: s.env,
41
+ timeoutMs: opts.timeoutMs ?? 60_000,
42
+ docker: opts.docker,
43
+ persist: false,
44
+ });
45
+ measured.set(key, m);
46
+ opts.onProgress?.(s.name, ++done, total);
47
+ }
48
+ };
49
+ await Promise.all(Array.from({ length: Math.max(1, Math.min(opts.concurrency ?? 3, total || 1)) }, worker));
50
+ return measured;
51
+ }
52
+ export async function runAudit(opts = {}) {
53
+ const configs = discover(opts);
54
+ const measured = await measureAll(configs, opts);
55
+ return buildReport(configs, measured, {
56
+ contextWindow: opts.contextWindow,
57
+ budget: opts.budget,
58
+ });
59
+ }
package/dist/cli.js CHANGED
@@ -2,9 +2,14 @@
2
2
  /**
3
3
  * mcp-context-cost CLI — the dispute drill as a command.
4
4
  *
5
- * mcp-context-cost verify <measurement.json> re-derive the number from the
5
+ * mcp-context-cost audit [--budget N] [--json] measure the servers in your own
6
+ * MCP config; exit 1 if over budget
7
+ * mcp-context-cost verify <measurement.json> [--json] re-derive the number from the
6
8
  * published capture; exit 1 on mismatch
9
+ * mcp-context-cost verify --remote <url> [--json] same, fetched from a measurement URL
7
10
  * mcp-context-cost measure --name x --command "npx -y ..." one-off measurement
11
+ *
12
+ * Exit codes: 0 ok, 1 verification/measurement/budget failed, 2 usage error.
8
13
  */
9
14
  import { readFileSync } from 'node:fs';
10
15
  import { canonicalString, countTokens, sha256Hex } from './core/canonical.js';
@@ -26,14 +31,85 @@ export function verifyMeasurement(m) {
26
31
  return { ok: problems.length === 0, rederivedTokens: tokens, rederivedSha: sha, problems };
27
32
  }
28
33
  const [, , cmd, ...rest] = process.argv;
29
- if (cmd === 'verify') {
30
- const path = rest.find((a) => !a.startsWith('--'));
31
- if (!path) {
32
- console.error('usage: mcp-context-cost verify <measurement.json>');
34
+ if (cmd === 'audit') {
35
+ const argOf = (name) => {
36
+ const i = rest.indexOf(`--${name}`);
37
+ return i >= 0 ? rest[i + 1] : undefined;
38
+ };
39
+ const all = (name) => rest.flatMap((a, i) => (a === `--${name}` && rest[i + 1] ? [rest[i + 1]] : []));
40
+ const json = rest.includes('--json');
41
+ const numeric = (name) => {
42
+ const raw = argOf(name);
43
+ if (raw === undefined)
44
+ return undefined;
45
+ const v = Number(raw);
46
+ if (!Number.isFinite(v) || v <= 0) {
47
+ console.error(`--${name} must be a positive number, got '${raw}'`);
48
+ process.exit(2);
49
+ }
50
+ return v;
51
+ };
52
+ const budget = numeric('budget');
53
+ const { runAudit } = await import('./audit/run.js');
54
+ const { formatReport } = await import('./audit/audit.js');
55
+ const report = await runAudit({
56
+ configPaths: all('config'),
57
+ budget,
58
+ contextWindow: numeric('context'),
59
+ timeoutMs: numeric('timeout'),
60
+ concurrency: numeric('concurrency'),
61
+ docker: rest.includes('--docker'),
62
+ // Progress goes to stderr so `--json` stdout stays a single parseable object.
63
+ onProgress: json ? undefined : (name, done, total) => process.stderr.write(` [${done}/${total}] ${name}\n`),
64
+ });
65
+ if (report.configs.length === 0) {
66
+ const where = report.problems.length ? `\n${report.problems.map((p) => ` ${p}`).join('\n')}` : '';
67
+ if (json)
68
+ console.log(JSON.stringify(report));
69
+ else
70
+ console.error(`no MCP config found. Looked in the standard Claude Desktop / Claude Code / Cursor / VS Code / Windsurf locations.${where}\n` +
71
+ `Point at one explicitly: mcp-context-cost audit --config <path/to/mcp.json>`);
72
+ process.exit(1);
73
+ }
74
+ console.log(json ? JSON.stringify(report) : formatReport(report));
75
+ process.exit(report.budget?.over ? 1 : 0);
76
+ }
77
+ else if (cmd === 'verify') {
78
+ const json = rest.includes('--json');
79
+ const remoteIdx = rest.indexOf('--remote');
80
+ const remoteUrl = remoteIdx >= 0 ? rest[remoteIdx + 1] : undefined;
81
+ const path = rest.find((a) => !a.startsWith('--') && a !== remoteUrl);
82
+ if (!remoteUrl && !path) {
83
+ console.error('usage: mcp-context-cost verify <measurement.json> [--json]');
84
+ console.error(' mcp-context-cost verify --remote <url> [--json]');
33
85
  process.exit(2);
34
86
  }
35
- const m = JSON.parse(readFileSync(path, 'utf8'));
87
+ let raw;
88
+ if (remoteUrl) {
89
+ try {
90
+ const res = await fetch(remoteUrl, { signal: AbortSignal.timeout(15_000) });
91
+ if (!res.ok)
92
+ throw new Error(`HTTP ${res.status}`);
93
+ raw = await res.text();
94
+ }
95
+ catch (e) {
96
+ const problem = `failed to fetch ${remoteUrl}: ${e.message}`;
97
+ if (json)
98
+ console.log(JSON.stringify({ ok: false, rederivedTokens: null, rederivedSha: null, problems: [problem] }));
99
+ else
100
+ console.error(problem);
101
+ process.exit(1);
102
+ }
103
+ }
104
+ else {
105
+ raw = readFileSync(path, 'utf8');
106
+ }
107
+ const m = JSON.parse(raw);
36
108
  const r = verifyMeasurement(m);
109
+ if (json) {
110
+ console.log(JSON.stringify({ serverName: m.serverName, ...r, badge: r.ok ? toBadge(m) : undefined }));
111
+ process.exit(r.ok ? 0 : 1);
112
+ }
37
113
  if (r.ok) {
38
114
  console.log(`OK ${m.serverName}: ${r.rederivedTokens} tokens (${m.encoding}, methodology ${m.methodologyVersion}) — capture, hash, and count all agree`);
39
115
  console.log(`badge: ${JSON.stringify(toBadge(m))}`);
@@ -73,6 +149,10 @@ else if (cmd !== undefined && cmd !== '--help' && cmd !== '-h') {
73
149
  }
74
150
  else {
75
151
  console.log('mcp-context-cost — reproducible context-cost measurement for MCP servers');
76
- console.log(' verify <measurement.json> re-derive tokens+sha from the published capture');
152
+ console.log(' audit [--config <path>] [--budget N] measure the servers in your own MCP config');
153
+ console.log(' [--json] [--context N] [--timeout ms] [--concurrency N] [--docker]');
154
+ console.log(' verify <measurement.json> [--json] re-derive tokens+sha from the published capture');
155
+ console.log(' verify --remote <url> [--json] same, fetched from a measurement URL');
77
156
  console.log(' measure --name x --command "npx -y <server>" run a one-off measurement');
157
+ console.log('exit codes: 0 ok, 1 verification/measurement/budget failed, 2 usage error');
78
158
  }
@@ -5,3 +5,8 @@
5
5
  */
6
6
  export declare function bandColor(totalTokens: number): string;
7
7
  export declare const UNKNOWN_COLOR = "lightgrey";
8
+ /** Human-readable name + range for each band, shared by every generated view. */
9
+ export declare const BAND_META: Record<string, {
10
+ label: string;
11
+ range: string;
12
+ }>;
@@ -15,3 +15,11 @@ export function bandColor(totalTokens) {
15
15
  return 'red';
16
16
  }
17
17
  export const UNKNOWN_COLOR = 'lightgrey';
18
+ /** Human-readable name + range for each band, shared by every generated view. */
19
+ export const BAND_META = {
20
+ brightgreen: { label: 'lean', range: '< 1K' },
21
+ green: { label: 'light', range: '1–5K' },
22
+ yellow: { label: 'moderate', range: '5–15K' },
23
+ orange: { label: 'heavy', range: '15–30K' },
24
+ red: { label: 'very heavy', range: '≥ 30K' },
25
+ };
@@ -44,7 +44,9 @@ export function measureTools(tools, meta) {
44
44
  toolCount: tools.length,
45
45
  tools: perTool,
46
46
  canonicalSha256: sha256Hex(canonical),
47
- rawToolsCapture: tools,
47
+ // Snapshot from the canonical string, not the live `tools` reference — the whole
48
+ // point of a capture is that it can't change out from under its own hash later.
49
+ rawToolsCapture: JSON.parse(canonical),
48
50
  measuredAt: meta.measuredAt ?? new Date().toISOString(),
49
51
  serverName: meta.serverName,
50
52
  serverVersion: meta.serverVersion,
@@ -0,0 +1,71 @@
1
+ /** Method identifier, versioned independently of the o200k methodology. */
2
+ export declare const DIVERGENCE_METHOD = "tools-delta/v1";
3
+ /** The three fields an Anthropic tool definition carries — nothing else. */
4
+ export declare const ANTHROPIC_TOOL_FIELDS: readonly ['name', 'description', 'input_schema'];
5
+ export interface AnthropicTool {
6
+ name: string;
7
+ description: string;
8
+ input_schema: unknown;
9
+ }
10
+ export interface DivergenceRow {
11
+ /** o200k count of the full canonical capture — the published headline. */
12
+ o200kFull: number;
13
+ /** o200k count of the name/description/input_schema projection. */
14
+ o200kMapped: number;
15
+ /**
16
+ * Claude tokens the projection adds to a request: count_tokens(with tools)
17
+ * minus count_tokens(same request, no tools). Includes the one-time tool
18
+ * framework overhead, because attaching any server pays it once.
19
+ */
20
+ claudeDelta: number;
21
+ toolCount: number;
22
+ /**
23
+ * canonicalSha256 of the measurement this row was computed from. A re-sweep
24
+ * changes the hash, which marks the row stale instead of silently mismatched.
25
+ */
26
+ capturedSha256: string;
27
+ /** Set when count_tokens rejected the projection; no numbers are published. */
28
+ error?: string;
29
+ }
30
+ export interface DivergenceRun {
31
+ method: string;
32
+ /** Exact model id the counts are pinned to — Anthropic's tokenizer drifts. */
33
+ model: string;
34
+ /** UTC day the counts were taken (YYYY-MM-DD). */
35
+ measuredAt: string;
36
+ /** count_tokens for the probe request with no tools at all. */
37
+ baselineTokens: number;
38
+ /**
39
+ * A single minimal tool. Its delta is an upper bound on the fixed framework
40
+ * overhead included in every `claudeDelta` (the probe tool itself costs > 0,
41
+ * so the true overhead is strictly less).
42
+ */
43
+ probeDelta: number;
44
+ servers: Record<string, DivergenceRow>;
45
+ }
46
+ /**
47
+ * Project a `tools/list` capture onto the Anthropic tool shape. Tools without a
48
+ * usable name are dropped rather than renamed — a fabricated name would change
49
+ * the token count being compared.
50
+ */
51
+ export declare function toAnthropicTools(raw: unknown[]): AnthropicTool[];
52
+ /** o200k count of the projection — the same tokenizer as the headline number. */
53
+ export declare function mappedTokens(raw: unknown[]): number;
54
+ /**
55
+ * Share of the headline number that is MCP-only metadata (cause 1 above).
56
+ * Returns null when there is nothing to divide by.
57
+ */
58
+ export declare function fieldSelectionShare(row: DivergenceRow): number | null;
59
+ /**
60
+ * Claude tokens per o200k token of the headline — the ratio a reader wants when
61
+ * asking "does the badge over- or under-state what this costs me?".
62
+ */
63
+ export declare function claudeRatio(row: DivergenceRow): number | null;
64
+ /**
65
+ * A row is only reportable if it carries numbers and was computed from the
66
+ * capture currently on disk. Stale rows are hidden rather than shown with a
67
+ * caveat: a wrong number next to a fresh badge is worse than no number.
68
+ */
69
+ export declare function isCurrent(row: DivergenceRow | undefined, canonicalSha256: string | null): row is DivergenceRow;
70
+ /** Parse results/divergence.json; anything malformed yields null, never throws. */
71
+ export declare function parseDivergence(text: string): DivergenceRun | null;
@@ -0,0 +1,103 @@
1
+ /**
2
+ * Claude divergence — the gap between the published o200k index and what a
3
+ * server's tools actually cost in an Anthropic request.
4
+ *
5
+ * The gap has two independent causes, and reporting them as one number hides
6
+ * the larger of the two:
7
+ *
8
+ * 1. **Field selection.** An Anthropic tool definition carries exactly `name`,
9
+ * `description`, and `input_schema`. A `tools/list` result may additionally
10
+ * carry `title`, `annotations`, `outputSchema`, `execution`, `icons` — real
11
+ * bytes the server ships, counted by the canonical form, but never present
12
+ * in the `tools` array a client sends to the API. For some servers this is
13
+ * the majority of the payload.
14
+ * 2. **Tokenizer and framing.** o200k_base vs Anthropic's tokenizer, plus the
15
+ * fixed framework overhead the API adds once when any tool is present.
16
+ *
17
+ * So the divergence is published as a decomposition: the headline o200k count,
18
+ * the o200k count of the projection (isolating cause 1), and Claude's own count
19
+ * of that same projection (isolating cause 2).
20
+ */
21
+ import { countTokens } from './canonical.js';
22
+ /** Method identifier, versioned independently of the o200k methodology. */
23
+ export const DIVERGENCE_METHOD = 'tools-delta/v1';
24
+ /** The three fields an Anthropic tool definition carries — nothing else. */
25
+ export const ANTHROPIC_TOOL_FIELDS = ['name', 'description', 'input_schema'];
26
+ /**
27
+ * Project a `tools/list` capture onto the Anthropic tool shape. Tools without a
28
+ * usable name are dropped rather than renamed — a fabricated name would change
29
+ * the token count being compared.
30
+ */
31
+ export function toAnthropicTools(raw) {
32
+ const out = [];
33
+ for (const t of raw) {
34
+ const tool = (t ?? {});
35
+ if (typeof tool.name !== 'string' || tool.name === '')
36
+ continue;
37
+ out.push({
38
+ name: tool.name,
39
+ description: typeof tool.description === 'string' ? tool.description : '',
40
+ // An absent schema is sent as the empty object schema, which is what the
41
+ // API requires; every tool in the current sweep supplies one.
42
+ input_schema: tool.inputSchema ?? { type: 'object', properties: {} },
43
+ });
44
+ }
45
+ return out;
46
+ }
47
+ /** o200k count of the projection — the same tokenizer as the headline number. */
48
+ export function mappedTokens(raw) {
49
+ return countTokens(JSON.stringify(toAnthropicTools(raw)));
50
+ }
51
+ /**
52
+ * Share of the headline number that is MCP-only metadata (cause 1 above).
53
+ * Returns null when there is nothing to divide by.
54
+ */
55
+ export function fieldSelectionShare(row) {
56
+ if (row.o200kFull <= 0)
57
+ return null;
58
+ return (row.o200kFull - row.o200kMapped) / row.o200kFull;
59
+ }
60
+ /**
61
+ * Claude tokens per o200k token of the headline — the ratio a reader wants when
62
+ * asking "does the badge over- or under-state what this costs me?".
63
+ */
64
+ export function claudeRatio(row) {
65
+ if (row.o200kFull <= 0 || typeof row.claudeDelta !== 'number')
66
+ return null;
67
+ return row.claudeDelta / row.o200kFull;
68
+ }
69
+ /**
70
+ * A row is only reportable if it carries numbers and was computed from the
71
+ * capture currently on disk. Stale rows are hidden rather than shown with a
72
+ * caveat: a wrong number next to a fresh badge is worse than no number.
73
+ */
74
+ export function isCurrent(row, canonicalSha256) {
75
+ if (!row || row.error)
76
+ return false;
77
+ if (typeof row.claudeDelta !== 'number')
78
+ return false;
79
+ return !!canonicalSha256 && row.capturedSha256 === canonicalSha256;
80
+ }
81
+ /** Parse results/divergence.json; anything malformed yields null, never throws. */
82
+ export function parseDivergence(text) {
83
+ let run;
84
+ try {
85
+ run = JSON.parse(text);
86
+ }
87
+ catch {
88
+ return null;
89
+ }
90
+ const r = run;
91
+ if (!r || typeof r.model !== 'string' || typeof r.measuredAt !== 'string')
92
+ return null;
93
+ if (!r.servers || typeof r.servers !== 'object')
94
+ return null;
95
+ return {
96
+ method: typeof r.method === 'string' ? r.method : DIVERGENCE_METHOD,
97
+ model: r.model,
98
+ measuredAt: r.measuredAt,
99
+ baselineTokens: typeof r.baselineTokens === 'number' ? r.baselineTokens : 0,
100
+ probeDelta: typeof r.probeDelta === 'number' ? r.probeDelta : 0,
101
+ servers: r.servers,
102
+ };
103
+ }
@@ -1,5 +1,6 @@
1
1
  export * from './types.js';
2
2
  export * from './canonical.js';
3
+ export * from './divergence.js';
3
4
  export * from './bands.js';
4
5
  export * from './badge.js';
5
6
  export * from './snippet.js';
@@ -1,5 +1,6 @@
1
1
  export * from './types.js';
2
2
  export * from './canonical.js';
3
+ export * from './divergence.js';
3
4
  export * from './bands.js';
4
5
  export * from './badge.js';
5
6
  export * from './snippet.js';
@@ -6,17 +6,15 @@
6
6
  import { existsSync, readFileSync, writeFileSync, mkdirSync } from 'node:fs';
7
7
  import { dirname, join } from 'node:path';
8
8
  import { parse } from 'yaml';
9
- import { bandColor } from '../core/bands.js';
9
+ import { bandColor, BAND_META } from '../core/bands.js';
10
10
  const esc = (s) => String(s ?? '').replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;').replace(/"/g, '&quot;');
11
- const BAND_META = {
12
- brightgreen: { label: 'lean', range: '< 1K' },
13
- green: { label: 'light', range: '1–5K' },
14
- yellow: { label: 'moderate', range: '5–15K' },
15
- orange: { label: 'heavy', range: '15–30K' },
16
- red: { label: 'very heavy', range: '≥ 30K' },
17
- };
18
11
  export function generateDashboard(root = process.cwd()) {
19
12
  const doc = parse(readFileSync(join(root, 'servers.yaml'), 'utf8'));
13
+ const divergencePath = join(root, 'results', 'divergence.json');
14
+ const divergence = existsSync(divergencePath)
15
+ ? JSON.parse(readFileSync(divergencePath, 'utf8'))
16
+ : {};
17
+ const dSrv = divergence.servers ?? {};
20
18
  const rows = doc.servers.map((entry) => {
21
19
  const p = join(root, 'results', entry.name, 'measurement.json');
22
20
  return { entry, m: existsSync(p) ? JSON.parse(readFileSync(p, 'utf8')) : null };
@@ -39,12 +37,15 @@ export function generateDashboard(root = process.cwd()) {
39
37
  const meta = BAND_META[band];
40
38
  const largest = [...m.tools].sort((a, b) => b.tokens - a.tokens)[0];
41
39
  const pct = Math.max(1.2, (t / max) * 100);
42
- return `<div class="row" tabindex="0" data-tip="${esc(m.toolCount)} tools · largest: ${esc(largest?.name)} (${fmt(largest?.tokens ?? 0)} tok) · ${esc(r.entry.category)} · ${esc(m.status)}${m.serverVersion ? ' · v' + esc(String(m.serverVersion).replace(/^v/, '')) : ''}">
40
+ const div = dSrv[r.entry.name];
41
+ const claudeTip = div ? ` · in a Claude request: ${fmt(div.claudeDelta)} tok` : '';
42
+ // The whole row is the link to the server's detail page (docs/servers/).
43
+ return `<a class="row" href="servers/${encodeURIComponent(r.entry.name)}.html" data-tip="${esc(m.toolCount)} tools · largest: ${esc(largest?.name)} (${fmt(largest?.tokens ?? 0)} tok)${claudeTip} · ${esc(r.entry.category)} · ${esc(m.status)}${m.serverVersion ? ' · v' + esc(String(m.serverVersion).replace(/^v/, '')) : ''}">
43
44
  <span class="rank">${i + 1}</span>
44
45
  <span class="name">${esc(r.entry.name)}</span>
45
46
  <span class="track"><span class="bar" style="width:${pct.toFixed(1)}%"></span></span>
46
47
  <span class="val"><span class="dot dot-${band}" aria-hidden="true"></span>${fmt(t)}<span class="bandname">${meta.label}</span></span>
47
- </div>`;
48
+ </a>`;
48
49
  })
49
50
  .join('\n');
50
51
  const failRows = failed
@@ -57,7 +58,8 @@ export function generateDashboard(root = process.cwd()) {
57
58
  const tableRows = measured
58
59
  .map((r, i) => {
59
60
  const m = r.m;
60
- return `<tr><td>${i + 1}</td><td>${esc(r.entry.name)}</td><td class="num">${fmt(m.totalTokens)}</td><td class="num">${esc(m.toolCount)}</td><td>${esc(BAND_META[bandColor(m.totalTokens)].label)}</td><td>${esc(r.entry.category)}</td></tr>`;
61
+ const div = dSrv[r.entry.name];
62
+ return `<tr><td>${i + 1}</td><td>${esc(r.entry.name)}</td><td class="num">${fmt(m.totalTokens)}</td><td class="num">${div ? fmt(div.claudeDelta) : '—'}</td><td class="num">${esc(m.toolCount)}</td><td>${esc(BAND_META[bandColor(m.totalTokens)].label)}</td><td>${esc(r.entry.category)}</td></tr>`;
61
63
  })
62
64
  .join('\n');
63
65
  const specimens = [measured[measured.length - 1], measured[0]]
@@ -115,8 +117,9 @@ export function generateDashboard(root = process.cwd()) {
115
117
  .stat .l { font-size: 11px; letter-spacing: 0.08em; text-transform: uppercase; color: var(--muted); }
116
118
 
117
119
  .board { background: var(--surface); border: 1px solid var(--line); border-radius: 8px; padding: 14px 16px; }
118
- .row { display: grid; grid-template-columns: 2ch minmax(120px, 190px) 1fr max-content; gap: 10px; align-items: center; padding: 3px 4px; border-radius: 4px; outline: none; }
120
+ .row { display: grid; grid-template-columns: 2ch minmax(120px, 190px) 1fr max-content; gap: 10px; align-items: center; padding: 3px 4px; border-radius: 4px; outline: none; color: inherit; text-decoration: none; }
119
121
  .row:hover, .row:focus-visible { background: var(--accent-soft); }
122
+ .row:hover .name, .row:focus-visible .name { text-decoration: underline; }
120
123
  .rank { font-family: ui-monospace, Menlo, monospace; font-size: 11px; color: var(--muted); text-align: right; font-variant-numeric: tabular-nums; }
121
124
  .name { font-size: 0.86rem; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
122
125
  .track { background: var(--track); border-radius: 3px; height: 12px; overflow: hidden; }
@@ -169,7 +172,7 @@ export function generateDashboard(root = process.cwd()) {
169
172
  </div>
170
173
 
171
174
  <h2>Leaderboard</h2>
172
- <p class="h2sub">Tokens = o200k_base count of the canonical <code>tools/list</code> bytes. Hover or focus a row for detail.</p>
175
+ <p class="h2sub">Tokens = o200k_base count of the canonical <code>tools/list</code> bytes — the wire payload. What a <em>Claude request</em> actually carries can differ sharply (github: 54,422 on the wire, ${dSrv['github'] ? fmt(dSrv['github'].claudeDelta) : '…'} in a request — 80% of its schema bytes are fields no Anthropic request sends). Hover a row for both numbers; <a href="METHODOLOGY.html#claude-divergence">method</a>.</p>
173
176
  <div class="board">
174
177
  ${barRows || '<p class="h2sub">Sweep in progress — first results land shortly.</p>'}
175
178
  </div>
@@ -195,7 +198,7 @@ ${barRows || '<p class="h2sub">Sweep in progress — first results land shortly.
195
198
 
196
199
  <details><summary>Full data table</summary>
197
200
  <div class="tablewrap" style="margin-top:10px"><table>
198
- <thead><tr><th>#</th><th>server</th><th>tokens</th><th>tools</th><th>band</th><th>category</th></tr></thead>
201
+ <thead><tr><th>#</th><th>server</th><th>tokens (o200k)</th><th>claude req</th><th>tools</th><th>band</th><th>category</th></tr></thead>
199
202
  <tbody>${tableRows}</tbody>
200
203
  </table></div>
201
204
  </details>
@@ -0,0 +1,30 @@
1
+ import type { Measurement } from '../core/types.js';
2
+ export interface HistoryRow {
3
+ /** UTC calendar day of the measurement (YYYY-MM-DD). */
4
+ date: string;
5
+ server: string;
6
+ tokens: number;
7
+ toolCount: number;
8
+ /** 'measured' or 'dynamic' — dynamic means the tool set moved between captures. */
9
+ status: string;
10
+ }
11
+ export declare const HISTORY_HEADER = "date,server,tokens,toolCount,status";
12
+ /** Parse history.csv text; malformed or non-numeric rows are dropped, not thrown. */
13
+ export declare function parseHistory(text: string): HistoryRow[];
14
+ export declare function formatHistory(rows: HistoryRow[]): string;
15
+ /** Upsert one row by (date, server) — the newest write for a day wins. */
16
+ export declare function upsert(rows: HistoryRow[], row: HistoryRow): HistoryRow[];
17
+ /**
18
+ * A measurement contributes a row only if it produced a number: failures and
19
+ * auth walls are recorded in the leaderboard's "not measured" section, and
20
+ * writing them here as zeros would fabricate a drop to zero in the series.
21
+ */
22
+ export declare function rowFor(server: string, m: Measurement): HistoryRow | null;
23
+ /**
24
+ * Fold every results/<server>/measurement.json into results/history.csv.
25
+ * Idempotent: running twice over the same results is a no-op.
26
+ */
27
+ export declare function appendHistory(root?: string): {
28
+ rows: number;
29
+ added: number;
30
+ };
@@ -0,0 +1,120 @@
1
+ /**
2
+ * results/history.csv — the time series behind the leaderboard snapshot.
3
+ *
4
+ * One row per (date, server): a server's tokens on the day it was measured.
5
+ * Every sweep upserts by (date, server), so re-running a sweep on the same day
6
+ * corrects that day's row instead of appending a duplicate. Rows for earlier
7
+ * dates are never rewritten — history is append-only in practice.
8
+ */
9
+ import { existsSync, readFileSync, readdirSync, writeFileSync } from 'node:fs';
10
+ import { join } from 'node:path';
11
+ export const HISTORY_HEADER = 'date,server,tokens,toolCount,status';
12
+ function csvCell(s) {
13
+ const v = String(s ?? '');
14
+ return /[",\n]/.test(v) ? `"${v.replace(/"/g, '""')}"` : v;
15
+ }
16
+ /** Split one CSV line, honouring double-quoted fields. */
17
+ function splitCsvLine(line) {
18
+ const out = [];
19
+ let cur = '';
20
+ let quoted = false;
21
+ for (let i = 0; i < line.length; i++) {
22
+ const c = line[i];
23
+ if (quoted) {
24
+ if (c === '"' && line[i + 1] === '"') {
25
+ cur += '"';
26
+ i++;
27
+ }
28
+ else if (c === '"')
29
+ quoted = false;
30
+ else
31
+ cur += c;
32
+ }
33
+ else if (c === '"')
34
+ quoted = true;
35
+ else if (c === ',') {
36
+ out.push(cur);
37
+ cur = '';
38
+ }
39
+ else
40
+ cur += c;
41
+ }
42
+ out.push(cur);
43
+ return out;
44
+ }
45
+ /** Parse history.csv text; malformed or non-numeric rows are dropped, not thrown. */
46
+ export function parseHistory(text) {
47
+ const rows = [];
48
+ for (const line of text.split('\n')) {
49
+ if (!line.trim() || line.startsWith('date,'))
50
+ continue;
51
+ const [date, server, tokens, toolCount, status] = splitCsvLine(line);
52
+ if (!/^\d{4}-\d{2}-\d{2}$/.test(date ?? '') || !server)
53
+ continue;
54
+ if (!/^\d+$/.test(tokens ?? '') || !/^\d+$/.test(toolCount ?? ''))
55
+ continue;
56
+ rows.push({ date, server, tokens: Number(tokens), toolCount: Number(toolCount), status: status ?? '' });
57
+ }
58
+ return rows;
59
+ }
60
+ export function formatHistory(rows) {
61
+ const sorted = [...rows].sort((a, b) => a.date.localeCompare(b.date) || a.server.localeCompare(b.server));
62
+ const lines = sorted.map((r) => [csvCell(r.date), csvCell(r.server), r.tokens, r.toolCount, csvCell(r.status)].join(','));
63
+ return [HISTORY_HEADER, ...lines].join('\n') + '\n';
64
+ }
65
+ /** Upsert one row by (date, server) — the newest write for a day wins. */
66
+ export function upsert(rows, row) {
67
+ const i = rows.findIndex((r) => r.date === row.date && r.server === row.server);
68
+ if (i < 0)
69
+ return [...rows, row];
70
+ const next = [...rows];
71
+ next[i] = row;
72
+ return next;
73
+ }
74
+ /**
75
+ * A measurement contributes a row only if it produced a number: failures and
76
+ * auth walls are recorded in the leaderboard's "not measured" section, and
77
+ * writing them here as zeros would fabricate a drop to zero in the series.
78
+ */
79
+ export function rowFor(server, m) {
80
+ if (m.status !== 'measured' && m.status !== 'dynamic')
81
+ return null;
82
+ if (typeof m.totalTokens !== 'number' || typeof m.toolCount !== 'number')
83
+ return null;
84
+ const date = String(m.measuredAt ?? '').slice(0, 10);
85
+ if (!/^\d{4}-\d{2}-\d{2}$/.test(date))
86
+ return null;
87
+ return { date, server, tokens: m.totalTokens, toolCount: m.toolCount, status: m.status };
88
+ }
89
+ /**
90
+ * Fold every results/<server>/measurement.json into results/history.csv.
91
+ * Idempotent: running twice over the same results is a no-op.
92
+ */
93
+ export function appendHistory(root = process.cwd()) {
94
+ const resultsDir = join(root, 'results');
95
+ const path = join(resultsDir, 'history.csv');
96
+ const existing = existsSync(path) ? parseHistory(readFileSync(path, 'utf8')) : [];
97
+ let rows = existing;
98
+ if (!existsSync(resultsDir))
99
+ return { rows: 0, added: 0 };
100
+ for (const server of readdirSync(resultsDir, { withFileTypes: true })
101
+ .filter((d) => d.isDirectory())
102
+ .map((d) => d.name)
103
+ .sort()) {
104
+ const file = join(resultsDir, server, 'measurement.json');
105
+ if (!existsSync(file))
106
+ continue;
107
+ let m;
108
+ try {
109
+ m = JSON.parse(readFileSync(file, 'utf8'));
110
+ }
111
+ catch {
112
+ continue; // a half-written measurement should not abort the whole fold
113
+ }
114
+ const row = rowFor(server, m);
115
+ if (row)
116
+ rows = upsert(rows, row);
117
+ }
118
+ writeFileSync(path, formatHistory(rows));
119
+ return { rows: rows.length, added: rows.length - existing.length };
120
+ }
@@ -1,7 +1,14 @@
1
- /** Regenerate leaderboard + dashboard from existing results/: npx tsx src/sweep/regen.ts */
1
+ /** Regenerate leaderboard + history + server pages from results/: npx tsx src/sweep/regen.ts */
2
2
  import { readFileSync } from 'node:fs';
3
3
  import { parse } from 'yaml';
4
4
  import { writeLeaderboard, percentiles } from './report.js';
5
+ import { appendHistory } from './history.js';
6
+ import { writeServerPages } from './server-pages.js';
5
7
  const doc = parse(readFileSync('servers.yaml', 'utf8'));
6
8
  writeLeaderboard(doc.servers);
9
+ // History first: the server pages read history.csv for their over-time table.
10
+ const h = appendHistory();
11
+ const p = writeServerPages(doc.servers);
7
12
  console.log('leaderboard:', JSON.stringify(percentiles(doc.servers)));
13
+ console.log(`history: ${h.rows} rows (${h.added >= 0 ? '+' : ''}${h.added})`);
14
+ console.log(`server pages: ${p.pages}`);