champollion-mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.js ADDED
@@ -0,0 +1,1106 @@
1
+ /**
2
+ * Champollion MCP Server — core implementation.
3
+ *
4
+ * Exposes the Champollion benchmark queue, language metadata, and
5
+ * mt-eval harness controls to AI agents via the Model Context Protocol.
6
+ *
7
+ * Tools:
8
+ * - list_queue Read-only. Fetch and filter the public queue.
9
+ * - get_queue_item Read-only. Get details for a specific queue item.
10
+ * - estimate_cost Read-only. Estimate cost for a set of queue items.
11
+ * - search_languages Read-only. Search language cards by name/code/family.
12
+ * - get_project_info Read-only. Get Champollion project overview.
13
+ * - get_results Read-only. Read scored runs from the public leaderboard.
14
+ * - get_run_card Read-only. Get one run card (scores + metadata) by id.
15
+ * - run_benchmark Action. Start a benchmark via mt-eval (returns a job
16
+ * handle immediately; runs in the background).
17
+ * - get_run_status Read-only. Poll a benchmark job by id until it settles.
18
+ *
19
+ * Resources (read-only data exposed via URIs):
20
+ * - champollion://contributing-guide CONTRIBUTING.md content.
21
+ * - champollion://queue-schema Queue item field definitions.
22
+ *
23
+ * Prompts (reusable conversation starters):
24
+ * - contribute_compute "I want to help — what would $X buy?"
25
+ * - compete_for_prize "I want to win a prize — how do I compete?"
26
+ * - explore_language "Tell me about [language] in Champollion."
27
+ *
28
+ * Data sources:
29
+ * - Queue: fetched from champollion.dev/queue.json (cached 5 min)
30
+ * - Languages: loaded from the CLI's language-cards directory
31
+ * - Results: public Supabase leaderboard (run_cards), same anon read path
32
+ * the champollion.dev leaderboard uses
33
+ *
34
+ * The server is stateless between tool calls. Queue data is cached in
35
+ * memory with a 5-minute TTL to avoid hammering the endpoint.
36
+ */
37
+
38
+ import { McpServer } from '@modelcontextprotocol/sdk/server/mcp.js';
39
+ import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
40
+ import { z } from 'zod';
41
+ import { readFile } from 'node:fs/promises';
42
+ import { resolve, dirname } from 'node:path';
43
+ import { fileURLToPath } from 'node:url';
44
+
45
+ import { fetchQueue, filterQueue, getQueueItem, estimateCost } from './tools/queue.js';
46
+ import { searchLanguages, loadLanguageIndex } from './tools/languages.js';
47
+ import { runBenchmark, getRunStatus } from './tools/harness.js';
48
+ import { fetchResults, formatResults, fetchRunCard } from './tools/results.js';
49
+ import { metricReliability, formatReliability } from './tools/reliability.js';
50
+ import { translateTexts, formatTranslateResult, METHOD_ENV } from './tools/translate.js';
51
+ import { trainingGuardrails, formatTrainingGuardrails } from './tools/training.js';
52
+ import { forgeTool } from './tools/forge.js';
53
+
54
+ /**
55
+ * Read the agent behavioral guide from disk.
56
+ *
57
+ * @param {string} path Absolute path to instructions.md.
58
+ * @returns {Promise<string|null>} Trimmed contents, or null if unreadable.
59
+ */
60
+ async function loadInstructions(path) {
61
+ try {
62
+ const text = await readFile(path, 'utf-8');
63
+ return text.trim() || null;
64
+ } catch {
65
+ return null;
66
+ }
67
+ }
68
+
69
+ /**
70
+ * Create and configure the MCP server with all tool registrations.
71
+ *
72
+ * @returns {{ start: () => Promise<void> }} Server object with a start method.
73
+ */
74
+ export async function createServer() {
75
+ // Resolve paths relative to this file (for reading instructions.md,
76
+ // CONTRIBUTING.md, etc.). Computed before server creation because the
77
+ // server's `instructions` are loaded from disk at construction time.
78
+ const __dirname = dirname(fileURLToPath(import.meta.url));
79
+ const repoRoot = resolve(__dirname, '..', '..');
80
+
81
+ // Load the agent behavioral guide and hand it to the SDK as the server's
82
+ // `instructions`. The SDK surfaces these in the MCP `initialize` result so
83
+ // connecting clients can show the model server-level guidance. Falls back to
84
+ // no instructions if the file is missing rather than failing startup.
85
+ const instructions = await loadInstructions(resolve(__dirname, '..', 'instructions.md'));
86
+
87
+ const server = new McpServer(
88
+ { name: 'champollion', version: '0.1.0' },
89
+ instructions ? { instructions } : undefined,
90
+ );
91
+
92
+ // Pre-load language index (small, fast, stays in memory)
93
+ const languageIndex = await loadLanguageIndex();
94
+
95
+ // ==================================================================
96
+ // Resources — read-only data agents can pull on demand
97
+ //
98
+ // Resources expose public information only. Nothing from docs/AGENTS.md
99
+ // or other internal documents is exposed here.
100
+ // ==================================================================
101
+
102
+ // Resource: Contributing guide — how to help with the project
103
+ server.resource(
104
+ 'contributing-guide',
105
+ 'champollion://contributing-guide',
106
+ { description: 'How to contribute to Champollion — for language speakers, ML researchers, developers, and community organizations.' },
107
+ async () => {
108
+ try {
109
+ const content = await readFile(resolve(repoRoot, 'CONTRIBUTING.md'), 'utf-8');
110
+ return { contents: [{ uri: 'champollion://contributing-guide', mimeType: 'text/markdown', text: content }] };
111
+ } catch {
112
+ return { contents: [{ uri: 'champollion://contributing-guide', mimeType: 'text/plain', text: 'CONTRIBUTING.md not found. See https://champollion.dev/contribute for contribution guidelines.' }] };
113
+ }
114
+ }
115
+ );
116
+
117
+ // Resource: Queue item field schema — what each field in queue.json means
118
+ server.resource(
119
+ 'queue-schema',
120
+ 'champollion://queue-schema',
121
+ { description: 'Field definitions for queue.json items — explains every field an agent will encounter when reading the queue.' },
122
+ async () => {
123
+ const schemaDoc = [
124
+ '# Queue Item Schema',
125
+ '',
126
+ 'Each item in champollion.dev/queue.json represents an untested',
127
+ '(language pair, model, condition) combination.',
128
+ '',
129
+ '## Item Fields',
130
+ '',
131
+ '| Field | Type | Description |',
132
+ '|-------|------|-------------|',
133
+ '| `id` | string | Unique identifier: `{pair_id}__{model_slug}__{condition}` |',
134
+ '| `priority` | int | Rank in the queue (1 = highest ECV) |',
135
+ '| `language_pair` | string | `{source}>{target}` ISO 639-3 codes |',
136
+ '| `source_language` | string | Source language name |',
137
+ '| `target_language` | string | Target language name |',
138
+ '| `model` | string | Full OpenRouter model slug |',
139
+ '| `condition` | string | `naive` (zero-shot) or `coached` (with coaching prompt) |',
140
+ '| `est_cost_usd` | float\|null | Estimated API cost. null = unknown |',
141
+ '| `corpus_id` | string | Which evaluation corpus to use |',
142
+ '| `ecv_per_usd` | float | Expected Chain Value per dollar — the ranking metric |',
143
+ '| `predicted_strength` | float | Predicted chrF++ / 100 for this run |',
144
+ '| `exploration_bonus` | float | UCB bonus for under-tested combinations |',
145
+ '| `pair_prior` | float | Baseline expected quality for this language pair |',
146
+ '| `run_command` | string | The exact CLI command to execute this item |',
147
+ '',
148
+ '## Metadata Fields (top-level)',
149
+ '',
150
+ '| Field | Description |',
151
+ '|-------|-------------|',
152
+ '| `metadata.open_items` | Total number of open items |',
153
+ '| `metadata.models` | List of model slugs in the queue |',
154
+ '| `metadata.priority_model` | Prose description of the ranking algorithm |',
155
+ '| `metadata.cost_basis` | How cost estimates were derived |',
156
+ '| `metadata.how_to_run` | Setup and execution instructions |',
157
+ ].join('\n');
158
+ return { contents: [{ uri: 'champollion://queue-schema', mimeType: 'text/markdown', text: schemaDoc }] };
159
+ }
160
+ );
161
+
162
+ // Resource: Network data endpoints — the FULL machine-readable picture.
163
+ // The champollion.dev homepage map is an idealization; agents should
164
+ // read these sources, not the picture (founder directive 2026-07-19).
165
+ server.resource(
166
+ 'network-data',
167
+ 'champollion://network-data',
168
+ { description: 'Every public machine-readable data source behind the Champollion network map — queue, mesh, registry, provider coverage, leaderboard REST, language cards. The map is an idealization; agents should read these.' },
169
+ async () => {
170
+ const doc = [
171
+ '# Network Data Endpoints',
172
+ '',
173
+ 'The champollion.dev homepage map is an IDEALIZATION of this data —',
174
+ 'read the sources, not the picture.',
175
+ '',
176
+ '| Source | URL | What it is |',
177
+ '|--------|-----|------------|',
178
+ '| Sweep queue | https://champollion.dev/queue.json | Full public benchmark queue (tens of MB): every open item + run command + cost estimate + license/transmission stamps + re-derivable ranking metadata. Cached ~5 min. |',
179
+ '| Queue preview | https://champollion.dev/queue-preview.json | Small top-of-queue slice with the full metadata block — start here. |',
180
+ '| Language mesh | https://champollion.dev/mesh.json | The measured/registered pair network: edges with status, best chrF++, run references. |',
181
+ '| Corpus registry | https://champollion.dev/registry.json | Every registered eval corpus: license lane, attribution, checksum, fetch-from-source notes (~10 MB). |',
182
+ '| Provider coverage | shared/catalogue/method-coverage.json (repo) | Each MT provider’s published language list, cited + as-of + `tier` (service vs open). The map’s green has two tiers by exact ISO-639-3 code: bright = a DEPLOYED service lists it (Google/Microsoft/DeepL/LibreTranslate); dim = only an OPEN research model lists it (NLLB/OPUS/M2M-100/MADLAD-400 — a model-card code, not a usable service). "Covered" is a published-list claim, never a quality claim. |',
183
+ '| Scored runs (REST) | https://sjdomynysdljkbemupqa.supabase.co/rest/v1/run_cards | Public leaderboard as PostgREST, read-only via RLS (anon key sb_publishable_bV6CFNFnzxhQI0wlBx2J0A_5Vm5gFBp — publishable by design). Aggregates only; per-entry test sentences are license-gated and never exposed. Prefer the get_results / get_run_card tools. |',
184
+ '| Language cards | cli/shared/language-cards/ (repo) | 7,900+ per-language JSON cards, every value cited. Prefer the search_languages tool. |',
185
+ '| Queue ranking spec | https://champollion.dev/docs/network/specifications/queue-construction | Normative spec: lanes + map-value survey ordering + ECV. |',
186
+ '',
187
+ 'License note: most queued corpora are evaluation sets with',
188
+ 'do_not_train stamps — see each queue item’s license/transmission',
189
+ 'fields and the registry’s license lanes before any re-use.',
190
+ ].join('\n');
191
+ return { contents: [{ uri: 'champollion://network-data', mimeType: 'text/markdown', text: doc }] };
192
+ }
193
+ );
194
+
195
+ // ==================================================================
196
+ // Prompts — reusable conversation starters
197
+ //
198
+ // These give agents (and users) pre-built workflows they can invoke.
199
+ // ==================================================================
200
+
201
+ // Prompt: contribute_compute — "I want to help, what would $X buy?"
202
+ server.prompt(
203
+ 'contribute_compute',
204
+ 'Start a conversation about contributing compute to Champollion benchmarks. '
205
+ + 'Helps the user understand what their budget would fund and which languages '
206
+ + 'benefit most.',
207
+ {
208
+ budget: z.string().optional()
209
+ .describe('Budget in USD (e.g. "10" for $10). If omitted, the agent will ask.'),
210
+ language: z.string().optional()
211
+ .describe('Optional language preference (e.g. "Yoruba", "crk"). If omitted, shows highest-impact items.'),
212
+ },
213
+ async ({ budget, language }) => {
214
+ const parts = ['I want to help with Champollion.'];
215
+ if (budget) parts.push(`I have about $${budget} in API credits to contribute.`);
216
+ if (language) parts.push(`I\'m especially interested in ${language}.`);
217
+ parts.push(
218
+ '',
219
+ 'Can you show me what benchmark runs my budget would fund? ',
220
+ 'Which languages need the most help right now?'
221
+ );
222
+ return {
223
+ messages: [{
224
+ role: 'user',
225
+ content: { type: 'text', text: parts.join(' ') },
226
+ }],
227
+ };
228
+ }
229
+ );
230
+
231
+ // Prompt: explore_language — "Tell me about [language]"
232
+ server.prompt(
233
+ 'explore_language',
234
+ 'Look up a language in the Champollion database — metadata, benchmarks, and queue status.',
235
+ {
236
+ language: z.string()
237
+ .describe('Language name, ISO 639-3 code, or endonym (e.g. "Yoruba", "yor", "Èdè Yorùbá").'),
238
+ },
239
+ async ({ language }) => {
240
+ return {
241
+ messages: [{
242
+ role: 'user',
243
+ content: {
244
+ type: 'text',
245
+ text: `Tell me about ${language} in the Champollion project. `
246
+ + 'What metadata does Champollion have for this language? '
247
+ + 'Is it in the benchmark queue? '
248
+ + 'What would it cost to run all its pending benchmarks?',
249
+ },
250
+ }],
251
+ };
252
+ }
253
+ );
254
+
255
+ // Prompt: compete_for_prize — "I want to build a competitive method"
256
+ server.prompt(
257
+ 'compete_for_prize',
258
+ 'Start a conversation about competing in the Champollion Arena. '
259
+ + 'The Arena supports sponsored prize pools — check the prize spec for current status. '
260
+ + 'Also suggests contributing compute as an immediate way to help.',
261
+ {
262
+ language: z.string().optional()
263
+ .describe('Target language to compete for (e.g. "Plains Cree", "crk"). If omitted, discusses the general framework.'),
264
+ },
265
+ async ({ language }) => {
266
+ const lang = language || 'a low-resource language';
267
+ return {
268
+ messages: [{
269
+ role: 'user',
270
+ content: {
271
+ type: 'text',
272
+ text: [
273
+ `I'm interested in building a competitive translation method for ${lang} in the Champollion Arena.`,
274
+ '',
275
+ 'Can you help me understand:',
276
+ '1. Are there any active prize pools right now? Check the prize spec for current status.',
277
+ '2. What does it take to build a strong method? (approaches, data, tools)',
278
+ '3. How does the anti-gaming architecture work? (secret test sets, etc.)',
279
+ '4. Even if there\'s no active prize, I might contribute compute or benchmark runs — what would that look like?',
280
+ ].join(' '),
281
+ },
282
+ }],
283
+ };
284
+ }
285
+ );
286
+
287
+ server.tool(
288
+ 'list_queue',
289
+ 'List open benchmark items from the Champollion public queue. '
290
+ + 'Items are ranked by expected chain value (ECV) — the expected '
291
+ + 'improvement in translation mesh quality per dollar spent. '
292
+ + 'Filter by language, model, budget, or condition.',
293
+ {
294
+ // All parameters are optional — calling with no args returns the top items
295
+ budget: z.number().positive().optional()
296
+ .describe('Maximum budget in USD. Only items fitting within this budget are returned.'),
297
+ language: z.string().optional()
298
+ .describe('Filter by target language name or ISO 639-3 code (e.g. "Yoruba" or "yor"). Case-insensitive partial match.'),
299
+ source_language: z.string().optional()
300
+ .describe('Filter by source language code (e.g. "eng", "fra"). Exact match on the source side of the pair.'),
301
+ model: z.string().optional()
302
+ .describe('Filter by model name or substring (e.g. "haiku", "gpt-5.5", "gemini"). Case-insensitive.'),
303
+ condition: z.enum(['naive', 'coached']).optional()
304
+ .describe('Filter by run condition. "naive" = zero-shot, "coached" = with coaching prompt.'),
305
+ limit: z.number().int().min(1).max(100).default(20)
306
+ .describe('Maximum number of items to return (default 20, max 100).'),
307
+ },
308
+ async ({ budget, language, source_language, model, condition, limit }) => {
309
+ try {
310
+ const queue = await fetchQueue();
311
+ const items = filterQueue(queue.items, {
312
+ budget, language, source_language, model, condition, limit,
313
+ });
314
+
315
+ // Compute summary statistics for the filtered set
316
+ const totalCost = items.reduce((s, it) => s + (it.est_cost_usd || 0), 0);
317
+ const languages = [...new Set(items.map(it => it.target_language))];
318
+
319
+ // Format each item as a concise line for the agent
320
+ const lines = items.map(it =>
321
+ `#${it.priority} ${it.language_pair.replace('>', ' → ')} `
322
+ + `${it.target_language} ${it.model.split('/').pop()} `
323
+ + `$${(it.est_cost_usd || 0).toFixed(4)} `
324
+ + `[${it.condition}]`
325
+ );
326
+
327
+ const summary = [
328
+ `Found ${items.length} items (of ${queue.metadata.open_items} total open).`,
329
+ `Estimated cost: $${totalCost.toFixed(2)}`,
330
+ `Languages: ${languages.join(', ')}`,
331
+ `Models in queue: ${queue.metadata.models.map(m => m.split('/').pop()).join(', ')}`,
332
+ '',
333
+ ...lines,
334
+ ].join('\n');
335
+
336
+ return { content: [{ type: 'text', text: summary }] };
337
+ } catch (err) {
338
+ return {
339
+ content: [{ type: 'text', text: `Error fetching queue: ${err.message}` }],
340
+ isError: true,
341
+ };
342
+ }
343
+ }
344
+ );
345
+
346
+ // ------------------------------------------------------------------
347
+ // Tool: get_queue_item
348
+ // ------------------------------------------------------------------
349
+ server.tool(
350
+ 'get_queue_item',
351
+ 'Get full details for a specific queue item by its ID or priority rank.',
352
+ {
353
+ id: z.string().optional()
354
+ .describe('The queue item ID (e.g. "eng-zul-dev-v1__anthropic_claude-haiku-4.5__naive").'),
355
+ priority: z.number().int().positive().optional()
356
+ .describe('The priority rank number (1 = highest priority).'),
357
+ },
358
+ async ({ id, priority }) => {
359
+ if (!id && !priority) {
360
+ return {
361
+ content: [{ type: 'text', text: 'Provide either id or priority to look up a queue item.' }],
362
+ isError: true,
363
+ };
364
+ }
365
+ try {
366
+ const queue = await fetchQueue();
367
+ const item = getQueueItem(queue.items, { id, priority });
368
+ if (!item) {
369
+ return {
370
+ content: [{ type: 'text', text: `No queue item found for ${id ? `id="${id}"` : `priority=${priority}`}.` }],
371
+ isError: true,
372
+ };
373
+ }
374
+ return {
375
+ content: [{ type: 'text', text: JSON.stringify(item, null, 2) }],
376
+ };
377
+ } catch (err) {
378
+ return {
379
+ content: [{ type: 'text', text: `Error: ${err.message}` }],
380
+ isError: true,
381
+ };
382
+ }
383
+ }
384
+ );
385
+
386
+ // ------------------------------------------------------------------
387
+ // Tool: estimate_cost
388
+ // ------------------------------------------------------------------
389
+ server.tool(
390
+ 'estimate_cost',
391
+ 'Estimate the cost and item count for a hypothetical queue run. '
392
+ + 'Accepts the same filters as list_queue and returns how many items '
393
+ + 'would be selected and total estimated cost.',
394
+ {
395
+ budget: z.number().positive().optional()
396
+ .describe('Maximum budget in USD.'),
397
+ language: z.string().optional()
398
+ .describe('Filter by target language (name or code).'),
399
+ source_language: z.string().optional()
400
+ .describe('Filter by source language code.'),
401
+ model: z.string().optional()
402
+ .describe('Filter by model name.'),
403
+ condition: z.enum(['naive', 'coached']).optional()
404
+ .describe('Filter by condition.'),
405
+ },
406
+ async ({ budget, language, source_language, model, condition }) => {
407
+ try {
408
+ const queue = await fetchQueue();
409
+ const result = estimateCost(queue.items, {
410
+ budget, language, source_language, model, condition,
411
+ });
412
+ return {
413
+ content: [{
414
+ type: 'text',
415
+ text: [
416
+ `Items selected: ${result.count}${result.capped ? '+ (capped at 500 — totals below cover the first 500 matches only)' : ''}`,
417
+ `Estimated total cost: $${result.totalCost.toFixed(2)}`,
418
+ `Cheapest item: $${result.cheapest.toFixed(4)}`,
419
+ `Most expensive item: $${result.mostExpensive.toFixed(4)}`,
420
+ `Languages covered: ${result.languages.join(', ')}`,
421
+ budget ? `Budget remaining: $${(budget - result.totalCost).toFixed(2)}` : '',
422
+ ].filter(Boolean).join('\n'),
423
+ }],
424
+ };
425
+ } catch (err) {
426
+ return {
427
+ content: [{ type: 'text', text: `Error: ${err.message}` }],
428
+ isError: true,
429
+ };
430
+ }
431
+ }
432
+ );
433
+
434
+ // ------------------------------------------------------------------
435
+ // Tool: search_languages
436
+ // ------------------------------------------------------------------
437
+ server.tool(
438
+ 'search_languages',
439
+ 'Search Champollion language cards by name, ISO code, language family, '
440
+ + 'or region. Returns metadata about languages in the project.',
441
+ {
442
+ query: z.string()
443
+ .describe('Search term: language name, ISO 639-3 code, family name, or region. Case-insensitive.'),
444
+ limit: z.number().int().min(1).max(50).default(10)
445
+ .describe('Maximum results to return (default 10).'),
446
+ },
447
+ async ({ query, limit }) => {
448
+ // Guarded like every other handler here. This was the one without a
449
+ // try/catch, so when the card shape changed underneath it and
450
+ // `name.toLowerCase()` began throwing, the TypeError escaped to the SDK
451
+ // instead of becoming a message an agent could read and route around.
452
+ // An agent's first stop into this server should not be able to take the
453
+ // whole call down.
454
+ let results;
455
+ try {
456
+ results = searchLanguages(languageIndex, query, limit);
457
+ } catch (err) {
458
+ return {
459
+ content: [{
460
+ type: 'text',
461
+ text: `Language search failed for "${query}": ${err.message}. `
462
+ + 'This is a server-side fault, not a bad query — the language index '
463
+ + 'is loaded from card data and something in it did not have the '
464
+ + 'shape the index expects.',
465
+ }],
466
+ isError: true,
467
+ };
468
+ }
469
+ if (results.length === 0) {
470
+ return {
471
+ content: [{
472
+ type: 'text',
473
+ text: `No languages found matching "${query}". Try a different name, code, or family.`,
474
+ }],
475
+ };
476
+ }
477
+ const lines = results.map(lang =>
478
+ `${lang.code} ${lang.name} ${lang.endonym || ''} `
479
+ + `family: ${lang.family || 'unknown'} `
480
+ + `speakers: ${lang.speakers || 'unknown'}`
481
+ );
482
+ return {
483
+ content: [{
484
+ type: 'text',
485
+ text: `Found ${results.length} language(s):\n\n${lines.join('\n')}`,
486
+ }],
487
+ };
488
+ }
489
+ );
490
+
491
+ // ------------------------------------------------------------------
492
+ // Tool: get_project_info
493
+ // ------------------------------------------------------------------
494
+ server.tool(
495
+ 'get_project_info',
496
+ 'Get an overview of the Champollion project: what it is, how '
497
+ + 'contributions work, and current queue statistics.',
498
+ {},
499
+ async () => {
500
+ try {
501
+ const queue = await fetchQueue();
502
+ const meta = queue.metadata;
503
+ return {
504
+ content: [{
505
+ type: 'text',
506
+ text: [
507
+ '# Champollion — Open Translation Benchmarks',
508
+ '',
509
+ 'Champollion is building an open scoreboard for machine translation',
510
+ 'quality across 7,900+ languages. The project maintains:',
511
+ '',
512
+ '- A public **benchmark queue** of untested (language pair, model, condition)',
513
+ ' combinations ranked by expected improvement to the translation mesh',
514
+ '- An **evaluation harness** (mt-eval) that runs standardized benchmarks',
515
+ ' and publishes results to a public leaderboard',
516
+ '- **Language cards** with metadata for thousands of languages',
517
+ '',
518
+ '## Contributing Compute',
519
+ '',
520
+ 'The easiest way to help: donate some API tokens to run benchmarks',
521
+ 'from the public queue. Anyone with an API key can contribute:',
522
+ '',
523
+ '1. Install the harness: `pipx install mt-eval`',
524
+ '2. Set your API key: `export OPENROUTER_API_KEY=sk-or-...`',
525
+ '3. Run from the queue: `mt-eval queue --budget 5`',
526
+ ' (runs top items up to $5 estimated cost)',
527
+ '4. Results auto-publish to the leaderboard with your attribution',
528
+ '',
529
+ '## Prizes',
530
+ '',
531
+ 'The Arena supports sponsored prize pools for translation breakthroughs.',
532
+ 'Prizes are evaluated against secret test corpora (developers never see',
533
+ 'the test data) with community validation by bilingual speakers.',
534
+ '',
535
+ 'See the prize spec for current status and threshold conditions:',
536
+ 'https://champollion.dev/docs/network/specifications/prizes',
537
+ '',
538
+ '## Current Queue Stats',
539
+ '',
540
+ `- Open items: ${meta.open_items.toLocaleString()}`,
541
+ `- Corpora: ${meta.corpora}`,
542
+ `- Models: ${meta.models.map(m => m.split('/').pop()).join(', ')}`,
543
+ `- Conditions: ${meta.conditions.join(', ')}`,
544
+ `- Generated: ${meta.generated_at.slice(0, 10)}`,
545
+ '',
546
+ 'Most items cost under $0.55 (median ~$0.09). A $5 budget typically',
547
+ 'funds 50-100 benchmark runs.',
548
+ '',
549
+ 'Website: https://champollion.dev',
550
+ 'Leaderboard: https://champollion.dev/leaderboard',
551
+ 'Contribute: https://champollion.dev/contribute',
552
+ 'Prize framework: https://champollion.dev/docs/network/specifications/prizes',
553
+ 'Arena: https://champollion.dev/arena',
554
+ ].join('\n'),
555
+ }],
556
+ };
557
+ } catch (err) {
558
+ return {
559
+ content: [{ type: 'text', text: `Error: ${err.message}` }],
560
+ isError: true,
561
+ };
562
+ }
563
+ }
564
+ );
565
+
566
+ // ------------------------------------------------------------------
567
+ // Tool: get_results
568
+ // ------------------------------------------------------------------
569
+ server.tool(
570
+ 'get_results',
571
+ 'Read scored benchmark results from the public Champollion leaderboard. '
572
+ + 'This closes the loop after run_benchmark: see what you (or the '
573
+ + 'community) scored. Returns ranked, scored runs (composite, chrF++, '
574
+ + 'BLEU, COMET) with trust level and attribution — never raw test '
575
+ + 'sentences. Filter by language pair or model. Each row carries a '
576
+ + 'contamination score_lane: rows marked "relative-only" (HIGH/MEDIUM '
577
+ + 'contamination, FLORES, or unknown grade) are valid ONLY for comparing '
578
+ + 'methods on that same corpus — never co-rank them against absolute-quality '
579
+ + 'rows as if the scores were comparable.',
580
+ {
581
+ source_language: z.string().optional()
582
+ .describe('Source language ISO 639-3 code (e.g. "eng", "fra"). Matches the source side of the pair.'),
583
+ target_language: z.string().optional()
584
+ .describe('Target language ISO 639-3 code (e.g. "zul", "yor"). Matches the target side of the pair.'),
585
+ model: z.string().optional()
586
+ .describe('Model name or substring (e.g. "haiku", "gpt-5.5"). Case-insensitive.'),
587
+ sort: z.enum(['composite', 'chrf', 'bleu', 'comet', 'ter', 'cost', 'date']).default('composite')
588
+ .describe('Sort metric. Score metrics sort best-first; ter/cost lowest-first; date newest-first.'),
589
+ limit: z.number().int().min(1).max(100).default(20)
590
+ .describe('Maximum number of results to return (default 20, max 100).'),
591
+ },
592
+ async ({ source_language, target_language, model, sort, limit }) => {
593
+ try {
594
+ const rows = await fetchResults({ source_language, target_language, model, sort, limit });
595
+ return { content: [{ type: 'text', text: formatResults(rows, { sort }) }] };
596
+ } catch (err) {
597
+ return {
598
+ content: [{ type: 'text', text: `Error fetching results: ${err.message}` }],
599
+ isError: true,
600
+ };
601
+ }
602
+ }
603
+ );
604
+
605
+ // ------------------------------------------------------------------
606
+ // Tool: get_run_card
607
+ // ------------------------------------------------------------------
608
+ server.tool(
609
+ 'get_run_card',
610
+ 'Get the full run card for one leaderboard result by its run id — scores '
611
+ + 'plus method/config/provenance metadata (the same card the public '
612
+ + 'leaderboard shows on expand). No per-entry test sentences are returned.',
613
+ {
614
+ id: z.string()
615
+ .describe('The run_card id (the `id` field from a get_results row).'),
616
+ },
617
+ async ({ id }) => {
618
+ try {
619
+ const card = await fetchRunCard(id);
620
+ if (!card) {
621
+ return {
622
+ content: [{ type: 'text', text: `No run card found for id="${id}". Use get_results to list valid ids.` }],
623
+ isError: true,
624
+ };
625
+ }
626
+ return { content: [{ type: 'text', text: JSON.stringify(card, null, 2) }] };
627
+ } catch (err) {
628
+ return {
629
+ content: [{ type: 'text', text: `Error fetching run card: ${err.message}` }],
630
+ isError: true,
631
+ };
632
+ }
633
+ }
634
+ );
635
+
636
+ // ------------------------------------------------------------------
637
+ // Tool: get_metric_reliability
638
+ // ------------------------------------------------------------------
639
+ server.tool(
640
+ 'get_metric_reliability',
641
+ 'Which automatic MT metric can you TRUST for a target language? Returns '
642
+ + 'champollion-derived correlations between metrics (BLEU, spBLEU, chrF, '
643
+ + 'chrF++, COMET, MetricX) and WMT Metrics-task human judgments '
644
+ + '(DA/MQM/ESA, wmt19–wmt25), rolled up per target-language family. Use '
645
+ + 'this BEFORE trusting a benchmark score: for some language families '
646
+ + 'BLEU barely tracks human judgment (Inuktitut: r=0.16) while a learned '
647
+ + 'metric works, and for others the learned metric is the one that fails. '
648
+ + 'Honest by construction: languages no WMT campaign ever judged return '
649
+ + 'an explicit UNMEASURED answer (with what IS covered), and family-level '
650
+ + 'transfer carries a caveat. Research-lane evidence only (upstream data '
651
+ + 'license pending review) — never cite it in commercial claims.',
652
+ {
653
+ target: z.string()
654
+ .describe('Target language as an ISO 639 code ("iu", "iku", "kk") or '
655
+ + 'a language-family name ("Turkic", "Eskimo-Aleut"). This is the '
656
+ + 'language being TRANSLATED INTO — metric reliability is about '
657
+ + 'scoring output in that language.'),
658
+ },
659
+ async ({ target }) => {
660
+ try {
661
+ const answer = await metricReliability(target);
662
+ return {
663
+ content: [{ type: 'text', text: formatReliability(answer) }],
664
+ isError: answer.status === 'index-unavailable',
665
+ };
666
+ } catch (err) {
667
+ return {
668
+ content: [{ type: 'text', text: `Error reading metric reliability: ${err.message}` }],
669
+ isError: true,
670
+ };
671
+ }
672
+ }
673
+ );
674
+
675
+ // ------------------------------------------------------------------
676
+ // Tool: translate
677
+ // ------------------------------------------------------------------
678
+ server.tool(
679
+ 'translate',
680
+ 'Translate texts through the champollion pipeline — the tested, '
681
+ + 'deterministic alternative to improvising your own translation prompt. '
682
+ + 'You get: engine choice (LLM via OpenRouter, direct '
683
+ + 'OpenAI/Anthropic/Gemini, DeepL, Google Translate, Microsoft, '
684
+ + 'LibreTranslate), language-card register/formality conditioning, a '
685
+ + 'persistent Translation Memory (repeated texts cost ZERO tokens — the '
686
+ + 'response shows the savings), and a deterministic five-check quality '
687
+ + 'gate (empty/source-echo/hallucination-loop/length-inflation/'
688
+ + 'script-compliance) so a bad translation comes back as an explicit '
689
+ + 'failure, never as silent garbage. Spends real API tokens only for '
690
+ + 'texts the Translation Memory has not seen (cost estimate included in '
691
+ + 'the response). This is production translation — benchmark evidence '
692
+ + 'and quality claims still come from run_benchmark/get_results.',
693
+ {
694
+ texts: z.array(z.string()).min(1).max(50)
695
+ .describe('Source texts to translate (max 50 per call; the TM makes repeated calls cheap).'),
696
+ source_language: z.string()
697
+ .describe('Source language code (e.g. "en").'),
698
+ target_language: z.string()
699
+ .describe('Target language code (e.g. "fr", "crk", "zul").'),
700
+ method: z.enum(Object.keys(METHOD_ENV)).default('llm')
701
+ .describe('Translation engine. "llm" (OpenRouter) is the default; each engine needs its API key in the server environment.'),
702
+ model: z.string().optional()
703
+ .describe('Model override for LLM methods (e.g. "anthropic/claude-sonnet-5").'),
704
+ register: z.string().optional()
705
+ .describe('Style/register instruction (e.g. "formal", "casual-tu", or free-text guidance). Defaults to the language card\'s register.'),
706
+ use_tm: z.boolean().default(true)
707
+ .describe('Consult and populate the persistent Translation Memory (default true — identical requests become free).'),
708
+ validate: z.boolean().default(true)
709
+ .describe('Run the deterministic quality gate on fresh translations (default true).'),
710
+ },
711
+ async ({ texts, source_language, target_language, method, model, register, use_tm, validate }) => {
712
+ try {
713
+ const result = await translateTexts({
714
+ texts, source: source_language, target: target_language,
715
+ method, model, register, useTm: use_tm, validate,
716
+ });
717
+ return {
718
+ content: [{ type: 'text', text: formatTranslateResult(result) }],
719
+ isError: result.status !== 'ok',
720
+ };
721
+ } catch (err) {
722
+ return {
723
+ content: [{ type: 'text', text: `Translation error: ${err.message}` }],
724
+ isError: true,
725
+ };
726
+ }
727
+ }
728
+ );
729
+
730
+ // ------------------------------------------------------------------
731
+ // Tool: get_training_guardrails
732
+ // ------------------------------------------------------------------
733
+ server.tool(
734
+ 'get_training_guardrails',
735
+ 'How to TRAIN an NMT model without fooling yourself — call this BEFORE '
736
+ + 'building a training pipeline, splitting a corpus, generating '
737
+ + 'synthetic data, or reporting training results. Returns the guardrail '
738
+ + 'rules Champollion extracted from real, measured failures (the '
739
+ + '2026-07-12 Plains Cree mistake ledger): group-disjoint splits (never '
740
+ + 'row-level random on drill-heavy corpora), the dev-fence (checkpoint '
741
+ + 'selection must never see the test set), leak audits (exact + '
742
+ + 'near-dupe, both sides), funnel accounting, single-orthography '
743
+ + 'training, coverage vs cited grammar checklists, per-kind sampling '
744
+ + 'caps, bootstrap CIs on every number, the eval-read ledger, and '
745
+ + 'preregistration before test scoring. Each rule names the mistake it '
746
+ + 'kills and the enforcing tool in the monorepo forge/ package '
747
+ + '(nmt-forge). Also the non-negotiables: datasets marked do_not_train '
748
+ + 'or quarantined in the mt-eval registry NEVER enter training mixes, '
749
+ + 'and test sets are REAL DATA ONLY (no synthetic rows).',
750
+ {
751
+ topic: z.string().optional()
752
+ .describe('Optional filter: a guardrail id (split, dev-fence, '
753
+ + 'leak-audit, funnel, conventions, coverage, strata, ci, ledger, '
754
+ + 'prereg, synthesis, training) or a keyword. Omit for all.'),
755
+ },
756
+ async ({ topic }) => {
757
+ try {
758
+ const answer = trainingGuardrails(topic);
759
+ return {
760
+ content: [{ type: 'text', text: formatTrainingGuardrails(answer) }],
761
+ isError: false,
762
+ };
763
+ } catch (err) {
764
+ return {
765
+ content: [{ type: 'text', text: `Error rendering guardrails: ${err.message}` }],
766
+ isError: true,
767
+ };
768
+ }
769
+ }
770
+ );
771
+
772
+ // ------------------------------------------------------------------
773
+ // Tools: forge_* — drive nmt-forge one guarded step at a time.
774
+ //
775
+ // get_training_guardrails is the rulebook; these are the hands. The
776
+ // canonical loop an agent follows: forge_status (where am I? THE next
777
+ // command) → forge_preflight (will it refuse? every gate) → the suggested
778
+ // command → repeat. Discovery/scaffolding: forge_discover, forge_init.
779
+ // Data hygiene: forge_split, forge_leak_audit, forge_register_eval,
780
+ // forge_prereg. Results: forge_evaluate (closes the decode→battery loop),
781
+ // forge_report/forge_lint (read the diagnosis). `nmt-forge run` (the GPU
782
+ // job) is intentionally not a tool — forge_status returns its exact
783
+ // command and watcher guidance.
784
+ // ------------------------------------------------------------------
785
+ const forgeWs = z.string().optional()
786
+ .describe('Workspace directory (default .forge, or '
787
+ + 'CHAMPOLLION_FORGE_WORKSPACE). The project you are driving.');
788
+
789
+ server.tool(
790
+ 'forge_status',
791
+ 'WHERE AM I in an nmt-forge project, and WHAT DO I RUN NEXT? Call this '
792
+ + 'FIRST and after every step — it reads the actual workspace (registered '
793
+ + 'dev/test/sealed sets, preregistrations, completed runs, ledger) and '
794
+ + 'returns a state name, THE single next command with exact args, and any '
795
+ + 'blockers. Deterministic: same state → same advice. This is the driver '
796
+ + 'a weak agent follows mechanically instead of guessing.',
797
+ { workspace: forgeWs },
798
+ async ({ workspace }) => forgeTool(['status', '--json'], {
799
+ workspace,
800
+ nextHint: 'run result.advice.next_command; if it is `nmt-forge run` '
801
+ + '(a GPU job), run it in a terminal and watch the [schedule-sanity] '
802
+ + 'lines, then call forge_evaluate on the run manifest.',
803
+ }),
804
+ );
805
+
806
+ server.tool(
807
+ 'forge_preflight',
808
+ 'WILL THIS COMMAND REFUSE? Renders every gate a forge command will hit '
809
+ + '(✓/✗ with the fix for each ✗), so you stop trial-and-erroring through '
810
+ + 'refusals one at a time. Use it before run/score/split/prereg/'
811
+ + 'leak-audit when forge_status points you at one. Exit-2 semantics are '
812
+ + 'flattened into per-gate ok flags in the JSON.',
813
+ {
814
+ target: z.enum(['run', 'score', 'split', 'prereg', 'leak-audit'])
815
+ .describe('The command to preflight.'),
816
+ workspace: forgeWs,
817
+ },
818
+ async ({ target, workspace }) => forgeTool(['preflight', target, '--json'], {
819
+ workspace, nextHint: 'fix any gate with ok:false using its fix string, '
820
+ + 'then re-run forge_preflight until all gates pass.',
821
+ }),
822
+ );
823
+
824
+ server.tool(
825
+ 'forge_discover',
826
+ 'What does a language HAVE? Reads the SSOT language card and reports '
827
+ + 'scripts, analyzers, dictionaries, corpora, eval datasets (with '
828
+ + 'do_not_train/quarantine flags), LYSS referee plugins, and where the '
829
+ + 'language sits on the asset ladder. Absence means UNKNOWN, never zero. '
830
+ + 'Call this before init to know what training path is possible.',
831
+ {
832
+ code: z.string().describe('ISO 639-3 code (e.g. crk, nav, arb).'),
833
+ workspace: forgeWs,
834
+ },
835
+ async ({ code, workspace }) => forgeTool(['discover', code, '--json'], {
836
+ workspace, nextHint: 'scaffold the project with forge_init, then '
837
+ + 'forge_status.',
838
+ }),
839
+ );
840
+
841
+ server.tool(
842
+ 'forge_init',
843
+ 'Scaffold a forge project from a language card: workspace + starter '
844
+ + 'config + NEXT_STEPS brief. Run after forge_discover. Writes the '
845
+ + 'language block the training loop needs for plugin discovery.',
846
+ {
847
+ code: z.string().describe('ISO 639-3 code of the TARGET language.'),
848
+ dir: z.string().optional().describe('Project directory (default .).'),
849
+ pair: z.string().optional()
850
+ .describe('Language pair SRC-TGT (default eng-<code>).'),
851
+ workspace: forgeWs,
852
+ },
853
+ async ({ code, dir, pair, workspace }) => forgeTool(
854
+ ['init', code, ...(dir ? ['--dir', dir] : []), ...(pair ? ['--pair', pair] : [])],
855
+ { workspace, nextHint: 'call forge_status — it will tell you to split a '
856
+ + 'corpus and register a dev set next.' }),
857
+ );
858
+
859
+ server.tool(
860
+ 'forge_split',
861
+ 'Carve a parallel corpus into GROUP-DISJOINT train/dev/test — pairs '
862
+ + 'sharing a canonical source OR target land on one side, so answer-'
863
+ + 'sharing rows can never straddle the split (the split-guard). Use '
864
+ + '--register to also register the dev/test files in one step.',
865
+ {
866
+ corpus: z.string().describe('Path to the parallel corpus (.jsonl).'),
867
+ test: z.number().int().positive().describe('Test set size (rows).'),
868
+ dev: z.number().int().nonnegative().optional().describe('Dev set size.'),
869
+ seed: z.number().int().describe('Split seed (required, reproducible).'),
870
+ out: z.string().describe('Output directory for the split files.'),
871
+ register: z.string().optional()
872
+ .describe('Prefix: also register <prefix>-test / <prefix>-dev.'),
873
+ workspace: forgeWs,
874
+ },
875
+ async ({ corpus, test, dev, seed, out, register, workspace }) => forgeTool(
876
+ ['split', corpus, '--test', String(test), '--seed', String(seed),
877
+ '--out', out, ...(dev ? ['--dev', String(dev)] : []),
878
+ ...(register ? ['--register', register] : [])],
879
+ { workspace, nextHint: 'forge_status — you now have dev/test; next is '
880
+ + 'usually prereg then run.' }),
881
+ );
882
+
883
+ server.tool(
884
+ 'forge_leak_audit',
885
+ 'Screen a corpus against every registered eval set BEFORE training: '
886
+ + 'target-side exact/near-dupe is fatal (answer leakage); source-only '
887
+ + 'near-dupe is informational (different answer = legitimate minimal '
888
+ + 'contrast, kept). Pass clean_to to write the survivors instead of '
889
+ + 'failing. Runs automatically inside `run` too — use this to pre-clean.',
890
+ {
891
+ corpus: z.string().describe('Corpus to screen (.jsonl).'),
892
+ strict: z.boolean().optional().describe('Hard-fail on test/sealed hits.'),
893
+ clean_to: z.string().optional()
894
+ .describe('Write surviving rows here instead of failing.'),
895
+ workspace: forgeWs,
896
+ },
897
+ async ({ corpus, strict, clean_to, workspace }) => forgeTool(
898
+ ['leak-audit', corpus, ...(strict ? ['--strict'] : []),
899
+ ...(clean_to ? ['--clean-to', clean_to] : [])],
900
+ { workspace, nextHint: 'if rows were removed, train on the --clean-to '
901
+ + 'file; then forge_status.' }),
902
+ );
903
+
904
+ server.tool(
905
+ 'forge_register_eval',
906
+ 'Register an eval file in the workspace with a role: dev (fenced '
907
+ + 'checkpoint selection), test (prereg-gated scoring), or sealed '
908
+ + '(one-shot). Every downstream guard keys off these roles. Test sets '
909
+ + 'must be REAL data (synthetic rows are refused).',
910
+ {
911
+ name: z.string().describe('Registry name for the set.'),
912
+ path: z.string().describe('Path to the eval file (.jsonl).'),
913
+ role: z.enum(['dev', 'test', 'sealed']).describe('The set\'s role.'),
914
+ source_field: z.string().optional().describe('Source field (default source).'),
915
+ target_field: z.string().optional().describe('Target/reference field.'),
916
+ workspace: forgeWs,
917
+ },
918
+ async ({ name, path, role, source_field, target_field, workspace }) => forgeTool(
919
+ ['registry', 'add', name, path, '--role', role,
920
+ ...(source_field ? ['--source-field', source_field] : []),
921
+ ...(target_field ? ['--target-field', target_field] : [])],
922
+ { workspace, nextHint: 'forge_status; a test/sealed set needs a prereg '
923
+ + 'before it can be scored.' }),
924
+ );
925
+
926
+ server.tool(
927
+ 'forge_prereg',
928
+ 'Preregister falsifiable predictions for a test/sealed set BEFORE '
929
+ + 'scoring it. Scoring a test set is refused without a prereg that '
930
+ + 'predates the first scoring read — this is what makes a result honest '
931
+ + 'rather than results-first storytelling. Predictions is a JSON file: a '
932
+ + 'list of {metric, expect, rationale} objects.',
933
+ {
934
+ id: z.string().describe('Preregistration id.'),
935
+ eval_set: z.string().describe('The registered test/sealed set name.'),
936
+ predictions: z.string()
937
+ .describe('Path to a JSON file: list of prediction objects.'),
938
+ author: z.string().optional(),
939
+ workspace: forgeWs,
940
+ },
941
+ async ({ id, eval_set, predictions, author, workspace }) => forgeTool(
942
+ ['prereg', 'new', id, '--eval-set', eval_set, '--predictions', predictions,
943
+ ...(author ? ['--author', author] : [])],
944
+ { workspace, parseJson: false,
945
+ nextHint: 'forge_status — you should now be ready-to-train or '
946
+ + 'ready-to-score.' }),
947
+ );
948
+
949
+ server.tool(
950
+ 'forge_evaluate',
951
+ 'CLOSE THE LOOP after a training run: decode the config\'s battery with '
952
+ + 'the run\'s SELECTED checkpoint, score it (CIs, prereg-gated) and '
953
+ + 'auto-append a plain-language Diagnosis & Recommendations. This is the '
954
+ + 'step that used to require a manual checkpoint symlink + decoder. Call '
955
+ + 'it on the run-manifest.json that `nmt-forge run` wrote.',
956
+ {
957
+ run_manifest: z.string().describe('Path to run-manifest.json.'),
958
+ config: z.string().optional()
959
+ .describe('Config path (defaults to the config embedded in the '
960
+ + 'manifest); must carry an eval block.'),
961
+ out_hyps: z.string().optional().describe('Where to write decoded hyps.'),
962
+ workspace: forgeWs,
963
+ },
964
+ async ({ run_manifest, config, out_hyps, workspace }) => forgeTool(
965
+ ['evaluate', run_manifest, ...(config ? ['--config', config] : []),
966
+ ...(out_hyps ? ['--out-hyps', out_hyps] : [])],
967
+ { workspace, parseJson: false,
968
+ nextHint: 'read the Diagnosis section; call forge_lint on the battery '
969
+ + 'manifest for --json findings and the lever to pull next.' }),
970
+ );
971
+
972
+ server.tool(
973
+ 'forge_lint',
974
+ 'Diagnose a battery manifest: which registers are weak, the likeliest '
975
+ + 'cause given the co-occurring signals, and the exact LEVER to pull next '
976
+ + '(VOCABULARY / STRUCTURE / ORTHOGRAPHY / REAL-DATA / MEASUREMENT / '
977
+ + 'REFEREE). Returns rule-id\'d findings with evidence — recommendations '
978
+ + 'are explainable, not vibes. Pass the run manifest too for transfer-'
979
+ + 'plateau detection.',
980
+ {
981
+ manifest: z.string().describe('Battery manifest JSON path.'),
982
+ run_manifest: z.string().optional()
983
+ .describe('Run manifest for schedule/transfer-plateau signals.'),
984
+ workspace: forgeWs,
985
+ },
986
+ async ({ manifest, run_manifest, workspace }) => forgeTool(
987
+ ['lint', manifest, '--json',
988
+ ...(run_manifest ? ['--run-manifest', run_manifest] : [])],
989
+ { workspace, nextHint: 'act on the highest-severity finding\'s lever; '
990
+ + 're-run and re-evaluate.' }),
991
+ );
992
+
993
+ server.tool(
994
+ 'forge_report',
995
+ 'Re-render the plain-language report (with the Diagnosis section) from a '
996
+ + 'run manifest or a battery manifest. Read-only pretty-printer for a '
997
+ + 'result you already produced.',
998
+ { manifest: z.string().describe('Run or battery manifest path.'), workspace: forgeWs },
999
+ async ({ manifest, workspace }) => forgeTool(['report', manifest],
1000
+ { workspace, parseJson: false }),
1001
+ );
1002
+
1003
+ // ------------------------------------------------------------------
1004
+ // Tool: run_benchmark
1005
+ // ------------------------------------------------------------------
1006
+ server.tool(
1007
+ 'run_benchmark',
1008
+ 'Start one or more benchmark items from the queue using mt-eval. '
1009
+ + 'This executes the mt-eval harness on the user\'s machine and SPENDS '
1010
+ + 'real API tokens. Requires mt-eval to be installed and an API key set. '
1011
+ + 'Spending is gated: you MUST confirm with the user first and then pass '
1012
+ + 'confirm:true. Without confirm:true (or with dry_run:true) the tool '
1013
+ + 'returns a plan and spends nothing. '
1014
+ + 'A confirmed run does NOT block: the benchmark is launched in the '
1015
+ + 'background and this tool returns a JOB ID immediately (a real run takes '
1016
+ + 'minutes — far longer than a default 60s MCP client timeout). After it '
1017
+ + 'returns, poll get_run_status with that job id until the job reports '
1018
+ + 'COMPLETED or FAILED, then call get_results. '
1019
+ + 'Budget/top runs execute in deterministic top-of-queue order — the same '
1020
+ + 'order estimate_cost and list_queue preview — and auto-publish each result '
1021
+ + 'to the public leaderboard unless publish:false is passed. '
1022
+ + 'SCOPE IS MANDATORY: every real run must name exactly one bound — budget '
1023
+ + '(USD ceiling), top (item count), or item_id (one item). A call with none '
1024
+ + 'of them is REFUSED rather than run over the whole queue.',
1025
+ {
1026
+ budget: z.number().positive().optional()
1027
+ .describe('Run the top queue items up to this USD budget. One of '
1028
+ + 'budget/top/item_id is REQUIRED for a real run.'),
1029
+ top: z.number().int().positive().optional()
1030
+ .describe('Run the top N items from the queue. One of '
1031
+ + 'budget/top/item_id is REQUIRED for a real run.'),
1032
+ item_id: z.string().optional()
1033
+ .describe('Run a specific queue item by ID. One of '
1034
+ + 'budget/top/item_id is REQUIRED for a real run.'),
1035
+ dry_run: z.boolean().default(false)
1036
+ .describe('If true, show what would be run without executing anything.'),
1037
+ provider: z.enum(['openrouter', 'openai', 'anthropic', 'gemini']).optional()
1038
+ .describe('API provider. Auto-detected from environment if not specified.'),
1039
+ confirm: z.boolean().default(false)
1040
+ .describe('Must be true to actually spend tokens. Confirm with the '
1041
+ + 'user before setting this. Ignored for dry_run.'),
1042
+ publish: z.boolean().default(true)
1043
+ .describe('Auto-publish each result to the public leaderboard '
1044
+ + '(default true). Pass false for a scoring/validation run that '
1045
+ + 'spends tokens but makes NO leaderboard write. Applies to '
1046
+ + 'budget/top runs; a single item_id run is never auto-published.'),
1047
+ },
1048
+ async ({ budget, top, item_id, dry_run, provider, confirm, publish }) => {
1049
+ try {
1050
+ const result = await runBenchmark({
1051
+ budget, top, item_id, dry_run, provider, confirm, publish,
1052
+ });
1053
+ return {
1054
+ content: [{ type: 'text', text: result }],
1055
+ };
1056
+ } catch (err) {
1057
+ return {
1058
+ content: [{ type: 'text', text: `Error running benchmark: ${err.message}` }],
1059
+ isError: true,
1060
+ };
1061
+ }
1062
+ }
1063
+ );
1064
+
1065
+ // ------------------------------------------------------------------
1066
+ // Tool: get_run_status
1067
+ // ------------------------------------------------------------------
1068
+ server.tool(
1069
+ 'get_run_status',
1070
+ 'Poll the status of a benchmark launched by run_benchmark. run_benchmark '
1071
+ + 'returns immediately with a job id (the run continues in the background, '
1072
+ + 'so it never trips the default 60s MCP client request timeout); call this '
1073
+ + 'with that job id every ~15-30s until it reports COMPLETED or FAILED. '
1074
+ + 'Each poll returns instantly. On completion it returns the run output '
1075
+ + '(local score for an item run; per-item lines for a queue run) — then use '
1076
+ + 'get_results to see the published leaderboard entry. Call with no job_id '
1077
+ + 'to list all jobs started in this session.',
1078
+ {
1079
+ job_id: z.string().optional()
1080
+ .describe('The job id returned by run_benchmark (e.g. "run-1"). Omit to list all jobs started this session.'),
1081
+ },
1082
+ async ({ job_id }) => {
1083
+ try {
1084
+ return { content: [{ type: 'text', text: getRunStatus(job_id) }] };
1085
+ } catch (err) {
1086
+ return {
1087
+ content: [{ type: 'text', text: `Error reading run status: ${err.message}` }],
1088
+ isError: true,
1089
+ };
1090
+ }
1091
+ }
1092
+ );
1093
+
1094
+ // ------------------------------------------------------------------
1095
+ // Transport setup
1096
+ // ------------------------------------------------------------------
1097
+ return {
1098
+ /** Start the server on stdio transport. */
1099
+ async start() {
1100
+ const transport = new StdioServerTransport();
1101
+ await server.connect(transport);
1102
+ // stderr for diagnostics — stdout is reserved for JSON-RPC
1103
+ process.stderr.write('Champollion MCP server running on stdio\n');
1104
+ },
1105
+ };
1106
+ }