mcp-context-cost 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,4 @@
1
+ import { type DivergenceRun } from '../core/divergence.js';
1
2
  export interface ServerEntry {
2
3
  name: string;
3
4
  command: string;
@@ -11,6 +12,10 @@ export interface ServerEntry {
11
12
  dockerImage?: string;
12
13
  timeoutSeconds?: number;
13
14
  }
15
+ /** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
16
+ export declare function mdCell(s: unknown): string;
17
+ /** results/divergence.json if a divergence run has been recorded, else null. */
18
+ export declare function loadDivergence(root?: string): DivergenceRun | null;
14
19
  export declare function writeLeaderboard(entries: ServerEntry[], root?: string): void;
15
20
  /** Percentile helper for freezing color bands against the observed distribution. */
16
21
  export declare function percentiles(entries: ServerEntry[], root?: string): Record<string, number>;
@@ -4,8 +4,9 @@
4
4
  */
5
5
  import { existsSync, readFileSync, writeFileSync } from 'node:fs';
6
6
  import { join } from 'node:path';
7
+ import { isCurrent, parseDivergence } from '../core/divergence.js';
7
8
  /** Neutralize markdown/table syntax in third-party strings (tool names, notes). */
8
- function mdCell(s) {
9
+ export function mdCell(s) {
9
10
  return String(s ?? '')
10
11
  .replace(/[|`[\]<>]/g, (c) => `\\${c}`)
11
12
  .replace(/\r?\n/g, ' ')
@@ -21,8 +22,21 @@ function loadRows(entries, root = process.cwd()) {
21
22
  return { entry, m: existsSync(p) ? JSON.parse(readFileSync(p, 'utf8')) : null };
22
23
  });
23
24
  }
25
+ /** results/divergence.json if a divergence run has been recorded, else null. */
26
+ export function loadDivergence(root = process.cwd()) {
27
+ const p = join(root, 'results', 'divergence.json');
28
+ return existsSync(p) ? parseDivergence(readFileSync(p, 'utf8')) : null;
29
+ }
24
30
  export function writeLeaderboard(entries, root = process.cwd()) {
25
31
  const rows = loadRows(entries, root);
32
+ const div = loadDivergence(root);
33
+ /** Claude tokens for a row, or null when not measured / stale / errored. */
34
+ const claude = (r) => {
35
+ if (!div || !r.m)
36
+ return null;
37
+ const d = div.servers[r.entry.name];
38
+ return isCurrent(d, r.m.canonicalSha256) ? d.claudeDelta : null;
39
+ };
26
40
  const measured = rows
27
41
  .filter((r) => r.m && (r.m.status === 'measured' || r.m.status === 'dynamic'))
28
42
  .sort((a, b) => (b.m.totalTokens ?? 0) - (a.m.totalTokens ?? 0));
@@ -31,14 +45,28 @@ export function writeLeaderboard(entries, root = process.cwd()) {
31
45
  md.push('# MCP server context-cost leaderboard');
32
46
  md.push('');
33
47
  md.push(`Tokens = o200k_base count of the canonical \`tools/list\` bytes ([methodology v1.0](../docs/METHODOLOGY.md)). ` +
34
- `Measured ${measured.length}/${rows.length} candidates; every candidate is listed — failures are findings, not omissions.`);
48
+ `Measured ${measured.length}/${rows.length} candidates; every candidate is listed — failures are findings, not omissions. ` +
49
+ `Server names link to their per-tool breakdown.`);
35
50
  md.push('');
36
- md.push('| # | server | tokens | tools | largest tool | status | category |');
37
- md.push('|---:|---|---:|---:|---|---|---|');
51
+ if (div) {
52
+ const n = measured.filter((r) => claude(r) !== null).length;
53
+ md.push(`The **claude** column is the same tools measured through Anthropic's \`count_tokens\` on ` +
54
+ `\`${mdCell(div.model)}\` (${mdCell(div.measuredAt)}, method \`${mdCell(div.method)}\`): the tokens the server's ` +
55
+ `tools add to a request, measured for the top ${n}. It is not a rescaling of the o200k column — ` +
56
+ `two effects pull in opposite directions, and the [per-server pages](../docs/servers/) break both out. ` +
57
+ `See [Claude divergence](../docs/METHODOLOGY.md#claude-divergence).`);
58
+ md.push('');
59
+ }
60
+ md.push(`| # | server | tokens |${div ? ' claude |' : ''} tools | largest tool | status | category |`);
61
+ md.push(`|---:|---|---:|${div ? '---:|' : ''}---:|---|---|---|`);
38
62
  measured.forEach((r, i) => {
39
63
  const m = r.m;
40
64
  const largest = [...m.tools].sort((a, b) => b.tokens - a.tokens)[0];
41
- md.push(`| ${i + 1} | ${mdCell(r.entry.name)} | ${m.totalTokens.toLocaleString('en-US')} | ${m.toolCount} | ` +
65
+ const link = `[${mdCell(r.entry.name)}](../docs/servers/${encodeURIComponent(r.entry.name)}.md)`;
66
+ const c = claude(r);
67
+ md.push(`| ${i + 1} | ${link} | ${m.totalTokens.toLocaleString('en-US')} |` +
68
+ (div ? ` ${c === null ? '—' : c.toLocaleString('en-US')} |` : '') +
69
+ ` ${m.toolCount} | ` +
42
70
  `${largest ? `${mdCell(largest.name)} (${largest.tokens.toLocaleString('en-US')})` : '—'} | ${m.status} | ${mdCell(r.entry.category)} |`);
43
71
  });
44
72
  md.push('');
@@ -54,9 +82,12 @@ export function writeLeaderboard(entries, root = process.cwd()) {
54
82
  md.push('');
55
83
  }
56
84
  writeFileSync(join(root, 'results', 'leaderboard.md'), md.join('\n') + '\n');
57
- const csv = ['name,tokens,toolCount,status,category,metric,metricSource'];
85
+ // Columns are append-only: consumers key off the header, so adding the Claude
86
+ // pair at the end leaves every existing parser working.
87
+ const csv = ['name,tokens,toolCount,status,category,metric,metricSource,claudeTokens,claudeModel'];
58
88
  for (const r of rows) {
59
89
  const m = r.m;
90
+ const c = claude(r);
60
91
  csv.push([
61
92
  csvCell(r.entry.name),
62
93
  m?.totalTokens ?? '',
@@ -65,6 +96,8 @@ export function writeLeaderboard(entries, root = process.cwd()) {
65
96
  csvCell(r.entry.category),
66
97
  r.entry.metric ?? '',
67
98
  csvCell(r.entry.metricSource),
99
+ c ?? '',
100
+ c === null ? '' : csvCell(div.model),
68
101
  ].join(','));
69
102
  }
70
103
  writeFileSync(join(root, 'results', 'leaderboard.csv'), csv.join('\n') + '\n');
@@ -7,5 +7,16 @@ export interface MeasureOptions {
7
7
  dockerImage?: string;
8
8
  /** env var NAMES to provide as dummy values (docker mode). */
9
9
  dummyEnv?: string[];
10
+ /**
11
+ * Exact argv, when the caller already has it (client configs store command and
12
+ * args separately). Avoids re-splitting a joined string on spaces, which would
13
+ * break any path containing one. Host path only — docker still wraps `command`.
14
+ */
15
+ argv?: string[];
16
+ /**
17
+ * Write results/<name>/measurement.json + badges/<name>.json (default true).
18
+ * `audit` runs in the user's own directory and must not litter it.
19
+ */
20
+ persist?: boolean;
10
21
  }
11
22
  export declare function measureServer(name: string, command: string, opts?: MeasureOptions): Promise<Measurement>;
package/dist/sweep/run.js CHANGED
@@ -5,7 +5,8 @@
5
5
  * Runs tools/list capture TWICE; differing tool sets -> status "dynamic".
6
6
  */
7
7
  import { mkdirSync, writeFileSync } from 'node:fs';
8
- import { join } from 'node:path';
8
+ import { join, resolve } from 'node:path';
9
+ import { fileURLToPath } from 'node:url';
9
10
  import { captureTools } from './client.js';
10
11
  import { dockerize } from './docker.js';
11
12
  import { measureTools, failedMeasurement, canonicalString } from '../core/canonical.js';
@@ -15,11 +16,14 @@ function arg(name) {
15
16
  return i >= 0 ? process.argv[i + 1] : undefined;
16
17
  }
17
18
  export async function measureServer(name, command, opts = {}) {
18
- if (!/^[a-z0-9][a-z0-9._-]*$/i.test(name) || name.includes('..')) {
19
+ const persist = opts.persist !== false;
20
+ // The name becomes a directory when persisting; that's the only reason it's
21
+ // constrained, so in-memory callers may use whatever the config called it.
22
+ if (persist && (!/^[a-z0-9][a-z0-9._-]*$/i.test(name) || name.includes('..'))) {
19
23
  throw new Error(`invalid server name '${name}' — letters/digits/dot/dash/underscore only`);
20
24
  }
21
25
  const root = opts.root ?? process.cwd();
22
- let spec = command;
26
+ let spec = opts.argv && opts.argv.length ? { command: opts.argv[0], argv: opts.argv.slice(1) } : command;
23
27
  let isolation = { docker: false };
24
28
  let containerName;
25
29
  if (opts.docker && command.trimStart().startsWith('docker ')) {
@@ -67,6 +71,8 @@ export async function measureServer(name, command, opts = {}) {
67
71
  spawn('docker', ['rm', '-f', containerName], { stdio: 'ignore' }).on('error', () => { });
68
72
  }
69
73
  }
74
+ if (!persist)
75
+ return m;
70
76
  const resultDir = join(root, 'results', name);
71
77
  mkdirSync(resultDir, { recursive: true });
72
78
  writeFileSync(join(resultDir, 'measurement.json'), JSON.stringify(m, null, 2) + '\n');
@@ -74,7 +80,10 @@ export async function measureServer(name, command, opts = {}) {
74
80
  writeFileSync(join(root, 'badges', `${name}.json`), JSON.stringify(toBadge(m)) + '\n');
75
81
  return m;
76
82
  }
77
- const isMain = process.argv[1]?.endsWith('run.ts') || process.argv[1]?.endsWith('run.js');
83
+ // Exact path match, not endsWith('run.ts'): any other file whose name happens to
84
+ // end in "run.ts" (src/audit/run.ts, a scratch dryrun.ts) would otherwise run this
85
+ // block and exit 2 on missing --name.
86
+ const isMain = process.argv[1] !== undefined && resolve(process.argv[1]) === fileURLToPath(import.meta.url);
78
87
  if (isMain) {
79
88
  const name = arg('name');
80
89
  const command = arg('command');
@@ -87,6 +96,10 @@ if (isMain) {
87
96
  docker: process.argv.includes('--docker'),
88
97
  dockerImage: arg('docker-image'),
89
98
  });
99
+ // CLI path only — measureServer itself stays history-free so concurrent
100
+ // sweep-all workers never race on the same file.
101
+ const { appendHistory } = await import('./history.js');
102
+ appendHistory();
90
103
  console.log(m.status === 'measured' || m.status === 'dynamic'
91
104
  ? `${name}: ${m.totalTokens} tokens across ${m.toolCount} tools (${m.status})`
92
105
  : `${name}: ${m.status} — ${m.notes ?? ''}`);
@@ -0,0 +1,17 @@
1
+ import type { Measurement } from '../core/types.js';
2
+ import { type ServerEntry } from './report.js';
3
+ import { type HistoryRow } from './history.js';
4
+ import { type DivergenceRun } from '../core/divergence.js';
5
+ /** One server's page. `history` is that server's rows, oldest first. */
6
+ export declare function renderServerPage(entry: ServerEntry, m: Measurement, history?: HistoryRow[], divergence?: DivergenceRun | null): string;
7
+ /** The index that lists every candidate — measured ones link to their page. */
8
+ export declare function renderServerIndex(rows: {
9
+ entry: ServerEntry;
10
+ m: Measurement | null;
11
+ }[]): string;
12
+ /** Write docs/servers/*.md for every measured server, plus the index. */
13
+ export declare function writeServerPages(entries: ServerEntry[], root?: string): {
14
+ pages: number;
15
+ };
16
+ /** Public URL of a server's page — used by the leaderboard and the badge snippet. */
17
+ export declare function serverPageUrl(name: string): string;
@@ -0,0 +1,228 @@
1
+ /**
2
+ * Per-server detail pages: docs/servers/<name>.md plus an index.
3
+ *
4
+ * These are the badge's click-through target. A badge says "12,430 tokens";
5
+ * the page behind it says which tools those tokens are in, what launched the
6
+ * server, the hash of the bytes counted, and the one command that re-derives
7
+ * the number. Generated from results/ only — no network, no timestamps beyond
8
+ * the measurement's own, so regenerating without a new sweep is a no-op diff.
9
+ */
10
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
11
+ import { join } from 'node:path';
12
+ import { bandColor, BAND_META } from '../core/bands.js';
13
+ import { loadDivergence, mdCell } from './report.js';
14
+ import { parseHistory } from './history.js';
15
+ import { claudeRatio, fieldSelectionShare, isCurrent } from '../core/divergence.js';
16
+ /**
17
+ * Pages are served from GitHub Pages (docs/), but results/ and badges/ are not
18
+ * published there — links into them must be absolute repo URLs.
19
+ */
20
+ const REPO_URL = 'https://github.com/athakur3/mcp-context-cost';
21
+ const BLOB = `${REPO_URL}/blob/main`;
22
+ const PAGES_URL = 'https://athakur3.github.io/mcp-context-cost';
23
+ /** Longest per-tool table we print inline; the rest live in the raw capture. */
24
+ const MAX_TOOL_ROWS = 30;
25
+ const fmt = (n) => n.toLocaleString('en-US');
26
+ /** True when a measurement produced a number we can stand behind. */
27
+ function isMeasured(m) {
28
+ return !!m && (m.status === 'measured' || m.status === 'dynamic') && typeof m.totalTokens === 'number';
29
+ }
30
+ function isolationText(m) {
31
+ const iso = m.isolation;
32
+ if (!iso)
33
+ return 'not recorded';
34
+ if (!iso.docker)
35
+ return 'host process (no container)';
36
+ return ['docker', iso.image, iso.network ? `network ${iso.network}` : '', iso.note]
37
+ .filter(Boolean)
38
+ .join(' · ');
39
+ }
40
+ /**
41
+ * The Claude divergence section: the headline, the projection onto the three
42
+ * fields an Anthropic tool definition carries, and Claude's own count of that
43
+ * projection. Printed only when a current divergence row exists for this server.
44
+ */
45
+ function divergenceSection(row, run) {
46
+ const share = fieldSelectionShare(row);
47
+ const ratio = claudeRatio(row);
48
+ const md = [];
49
+ md.push('## What this costs on Claude');
50
+ md.push('');
51
+ md.push(`Measured ${mdCell(run.measuredAt)} against \`${mdCell(run.model)}\` via Anthropic's \`count_tokens\` ` +
52
+ `(method \`${mdCell(run.method)}\`).`);
53
+ md.push('');
54
+ md.push('| | tokens | |');
55
+ md.push('|---|---:|---|');
56
+ md.push(`| o200k, full capture | ${fmt(row.o200kFull)} | the badge number — every byte \`tools/list\` returned |`);
57
+ md.push(`| o200k, Anthropic fields only | ${fmt(row.o200kMapped)} | ` +
58
+ `${share === null ? '—' : `${(share * 100).toFixed(1)}% of the capture is MCP-only metadata`} |`);
59
+ md.push(`| **Claude, same fields** | **${fmt(row.claudeDelta)}** | ` +
60
+ `${ratio === null ? '—' : `${ratio.toFixed(2)}× the badge number`} |`);
61
+ md.push('');
62
+ md.push(`An Anthropic tool definition carries \`name\`, \`description\`, and \`input_schema\` and nothing else, so ` +
63
+ `\`title\`, \`annotations\`, \`outputSchema\`, \`execution\`, and \`icons\` are dropped before the request — ` +
64
+ `that is the second row. The third row is the same tools counted by Anthropic, which is larger than the ` +
65
+ `second because Anthropic's tokenizer is denser on this content than o200k_base *and* the API adds its own ` +
66
+ `framing (at most ${fmt(run.probeDelta)} tokens of it fixed, measured against a single minimal tool). ` +
67
+ `The two effects run in opposite directions, which is why the Claude number is not a fixed multiple of the badge.`);
68
+ md.push('');
69
+ return md;
70
+ }
71
+ /** One server's page. `history` is that server's rows, oldest first. */
72
+ export function renderServerPage(entry, m, history = [], divergence = null) {
73
+ const total = m.totalTokens;
74
+ const band = BAND_META[bandColor(total)];
75
+ const tools = [...m.tools].sort((a, b) => b.tokens - a.tokens);
76
+ const shown = tools.slice(0, MAX_TOOL_ROWS);
77
+ const pct = (n) => (total > 0 ? `${((n / total) * 100).toFixed(1)}%` : '—');
78
+ const md = [];
79
+ md.push(`# ${mdCell(entry.name)} — context cost`);
80
+ md.push('');
81
+ md.push(`**${fmt(total)} tokens** across ${m.toolCount} tools — *${band.label}* (${band.range}). ` +
82
+ `Measured ${String(m.measuredAt).slice(0, 10)} under [methodology v${mdCell(m.methodologyVersion)}](../METHODOLOGY.html).`);
83
+ md.push('');
84
+ md.push('| | |');
85
+ md.push('|---|---|');
86
+ md.push(`| server (self-reported) | ${mdCell(m.serverName)}${m.serverVersion ? ` v${mdCell(String(m.serverVersion).replace(/^v/, ''))}` : ''} |`);
87
+ md.push(`| status | ${mdCell(m.status)} |`);
88
+ md.push(`| tokenizer | ${mdCell(m.provider)} / ${mdCell(m.encoding)} |`);
89
+ md.push(`| launch command | \`${mdCell(m.launchCommand ?? entry.command)}\` |`);
90
+ md.push(`| isolation | ${mdCell(isolationText(m))} |`);
91
+ md.push(`| env vars supplied | ${m.envVarNames?.length ? m.envVarNames.map(mdCell).join(', ') : 'none'} |`);
92
+ md.push(`| canonical SHA-256 | \`${mdCell(m.canonicalSha256)}\` |`);
93
+ if (entry.category)
94
+ md.push(`| category | ${mdCell(entry.category)} |`);
95
+ if (entry.repo)
96
+ md.push(`| source | ${mdCell(entry.repo)} |`);
97
+ md.push('');
98
+ if (m.status === 'dynamic') {
99
+ md.push('> This server\'s `tools/list` differed between two consecutive captures, so the number ' +
100
+ 'is the first capture and moves between sweeps. Treat it as a range, not a constant.');
101
+ md.push('');
102
+ }
103
+ md.push('## Where the tokens are');
104
+ md.push('');
105
+ md.push('| tool | tokens | share | description | schema |');
106
+ md.push('|---|---:|---:|---:|---:|');
107
+ for (const t of shown) {
108
+ md.push(`| ${mdCell(t.name)} | ${fmt(t.tokens)} | ${pct(t.tokens)} | ${fmt(t.descriptionTokens)} | ${fmt(t.inputSchemaTokens)} |`);
109
+ }
110
+ md.push('');
111
+ if (tools.length > shown.length) {
112
+ const rest = tools.slice(shown.length).reduce((s, t) => s + t.tokens, 0);
113
+ md.push(`*${tools.length - shown.length} smaller tools omitted (${fmt(rest)} tokens combined) — ` +
114
+ `all of them are in the [raw capture](${BLOB}/results/${encodeURIComponent(entry.name)}/measurement.json).*`);
115
+ md.push('');
116
+ }
117
+ md.push('Each tool is tokenized on its own, so the parts do not sum exactly to the whole: the ' +
118
+ 'array adds its own brackets and commas, and the tokenizer merges tokens across object ' +
119
+ 'boundaries. The badge number is always the count of the whole array, never a sum of parts.');
120
+ md.push('');
121
+ const divRow = divergence?.servers[entry.name];
122
+ if (divergence && isCurrent(divRow, m.canonicalSha256)) {
123
+ md.push(...divergenceSection(divRow, divergence));
124
+ }
125
+ if (history.length > 1) {
126
+ md.push('## Over time');
127
+ md.push('');
128
+ md.push('| date | tokens | tools | change |');
129
+ md.push('|---|---:|---:|---:|');
130
+ history.forEach((h, i) => {
131
+ const prev = history[i - 1];
132
+ const delta = prev ? h.tokens - prev.tokens : null;
133
+ const change = delta === null ? '—' : delta === 0 ? 'no change' : `${delta > 0 ? '+' : ''}${fmt(delta)}`;
134
+ md.push(`| ${mdCell(h.date)} | ${fmt(h.tokens)} | ${h.toolCount} | ${change} |`);
135
+ });
136
+ md.push('');
137
+ md.push(`Full series: [results/history.csv](${BLOB}/results/history.csv).`);
138
+ md.push('');
139
+ }
140
+ md.push('## Re-derive it');
141
+ md.push('');
142
+ md.push('```bash');
143
+ md.push(`npx -y mcp-context-cost verify results/${entry.name}/measurement.json`);
144
+ md.push('```');
145
+ md.push('');
146
+ md.push(`That re-tokenizes the [published capture](${BLOB}/results/${encodeURIComponent(entry.name)}/measurement.json) ` +
147
+ `and checks the count and the hash. If it disagrees with the badge, the badge is wrong — ` +
148
+ `[open an issue](${REPO_URL}/issues) and it gets corrected.`);
149
+ md.push('');
150
+ md.push(`[Badge JSON](${BLOB}/badges/${encodeURIComponent(entry.name)}.json) · ` +
151
+ `[All servers](index.html) · [Leaderboard](${BLOB}/results/leaderboard.md) · ` +
152
+ `[Methodology](../METHODOLOGY.html)`);
153
+ md.push('');
154
+ return md.join('\n');
155
+ }
156
+ /** The index that lists every candidate — measured ones link to their page. */
157
+ export function renderServerIndex(rows) {
158
+ const measured = rows.filter((r) => isMeasured(r.m)).sort((a, b) => b.m.totalTokens - a.m.totalTokens);
159
+ const rest = rows.filter((r) => !isMeasured(r.m));
160
+ const md = [];
161
+ md.push('# Server pages');
162
+ md.push('');
163
+ md.push(`One page per measured server: the per-tool breakdown behind the badge, the exact launch ` +
164
+ `command, and the command that re-derives the number. ${measured.length} of ${rows.length} ` +
165
+ `candidates measured.`);
166
+ md.push('');
167
+ md.push('| # | server | tokens | tools | band |');
168
+ md.push('|---:|---|---:|---:|---|');
169
+ measured.forEach((r, i) => {
170
+ const t = r.m.totalTokens;
171
+ md.push(`| ${i + 1} | [${mdCell(r.entry.name)}](${encodeURIComponent(r.entry.name)}.html) | ${fmt(t)} | ` +
172
+ `${r.m.toolCount} | ${BAND_META[bandColor(t)].label} |`);
173
+ });
174
+ md.push('');
175
+ if (rest.length > 0) {
176
+ md.push('## Not measured');
177
+ md.push('');
178
+ md.push('No page: there is no number to show. The reason is recorded per candidate.');
179
+ md.push('');
180
+ md.push('| server | status |');
181
+ md.push('|---|---|');
182
+ for (const r of rest) {
183
+ const status = r.entry.remote ? 'remote-auth-wall' : (r.m?.status ?? 'not-yet-run');
184
+ md.push(`| ${mdCell(r.entry.name)} | ${mdCell(status)} |`);
185
+ }
186
+ md.push('');
187
+ }
188
+ md.push(`[Leaderboard](${BLOB}/results/leaderboard.md) · [Methodology](../METHODOLOGY.html) · [Dashboard](../dashboard.html)`);
189
+ md.push('');
190
+ return md.join('\n');
191
+ }
192
+ /** Write docs/servers/*.md for every measured server, plus the index. */
193
+ export function writeServerPages(entries, root = process.cwd()) {
194
+ const outDir = join(root, 'docs', 'servers');
195
+ mkdirSync(outDir, { recursive: true });
196
+ const historyPath = join(root, 'results', 'history.csv');
197
+ const history = existsSync(historyPath) ? parseHistory(readFileSync(historyPath, 'utf8')) : [];
198
+ const divergence = loadDivergence(root);
199
+ const rows = entries.map((entry) => {
200
+ const p = join(root, 'results', entry.name, 'measurement.json');
201
+ let m = null;
202
+ if (existsSync(p)) {
203
+ try {
204
+ m = JSON.parse(readFileSync(p, 'utf8'));
205
+ }
206
+ catch {
207
+ m = null; // a half-written measurement should not abort the whole run
208
+ }
209
+ }
210
+ return { entry, m };
211
+ });
212
+ let pages = 0;
213
+ for (const { entry, m } of rows) {
214
+ if (!isMeasured(m))
215
+ continue;
216
+ const series = history
217
+ .filter((h) => h.server === entry.name)
218
+ .sort((a, b) => a.date.localeCompare(b.date));
219
+ writeFileSync(join(outDir, `${entry.name}.md`), renderServerPage(entry, m, series, divergence));
220
+ pages++;
221
+ }
222
+ writeFileSync(join(outDir, 'index.md'), renderServerIndex(rows));
223
+ return { pages };
224
+ }
225
+ /** Public URL of a server's page — used by the leaderboard and the badge snippet. */
226
+ export function serverPageUrl(name) {
227
+ return `${PAGES_URL}/servers/${encodeURIComponent(name)}.html`;
228
+ }
@@ -8,6 +8,7 @@ import { readFileSync } from 'node:fs';
8
8
  import { parse } from 'yaml';
9
9
  import { measureServer } from './run.js';
10
10
  import { writeLeaderboard } from './report.js';
11
+ import { appendHistory } from './history.js';
11
12
  function arg(name) {
12
13
  const i = process.argv.indexOf(`--${name}`);
13
14
  return i >= 0 ? process.argv[i + 1] : undefined;
@@ -45,6 +46,8 @@ async function worker() {
45
46
  }
46
47
  }
47
48
  await Promise.all(Array.from({ length: Math.max(1, concurrency) }, () => worker()));
49
+ // Serial, after every worker has finished: history.csv is a read-modify-write.
48
50
  writeLeaderboard(doc.servers);
51
+ const h = appendHistory();
49
52
  const measured = Object.values(summary).filter((s) => s.includes('tokens')).length;
50
- console.log(`done: ${measured}/${entries.length} measured; leaderboard regenerated`);
53
+ console.log(`done: ${measured}/${entries.length} measured; leaderboard + history (${h.rows} rows) regenerated`);
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "mcp-context-cost",
3
- "version": "0.1.0",
4
- "description": "Reproducible context-cost badges for MCP servers — measure what a server's tool schemas cost before the agent does any work",
3
+ "version": "0.3.0",
4
+ "description": "Measure what your MCP servers cost in context tokens — audit your own config, or badge the server you publish",
5
5
  "type": "module",
6
6
  "license": "MIT",
7
7
  "repository": {
@@ -15,6 +15,10 @@
15
15
  "model-context-protocol",
16
16
  "tokens",
17
17
  "context-window",
18
+ "audit",
19
+ "token-budget",
20
+ "claude",
21
+ "cursor",
18
22
  "badge",
19
23
  "shields",
20
24
  "developer-tools"
@@ -29,9 +33,11 @@
29
33
  ],
30
34
  "scripts": {
31
35
  "build": "tsc",
36
+ "prepublishOnly": "npm run build && npm test && npm run typecheck",
32
37
  "test": "vitest run",
33
38
  "test:badge": "./upstream/tests/badge-test.sh",
34
- "typecheck": "tsc --noEmit",
39
+ "typecheck": "tsc --noEmit && tsc -p tsconfig.tools.json",
40
+ "divergence": "tsx tools/measure-divergence.ts",
35
41
  "sweep": "tsx src/sweep/run.ts",
36
42
  "sweep:all": "tsx src/sweep/sweep-all.ts",
37
43
  "verify": "tsx src/cli.ts verify"
@@ -44,6 +50,7 @@
44
50
  "yaml": "^2.9.0"
45
51
  },
46
52
  "devDependencies": {
53
+ "@anthropic-ai/sdk": "^0.117.1",
47
54
  "@types/node": "^26.2.0",
48
55
  "tsx": "^4.23.12",
49
56
  "typescript": "^7.0.2",