champollion-mcp-server 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +245 -0
- package/bin/server.js +19 -0
- package/instructions.md +234 -0
- package/package.json +50 -0
- package/src/index.js +1106 -0
- package/src/tools/forge.js +140 -0
- package/src/tools/harness.js +727 -0
- package/src/tools/languages.js +329 -0
- package/src/tools/queue.js +313 -0
- package/src/tools/reliability.js +190 -0
- package/src/tools/results.js +346 -0
- package/src/tools/training.js +349 -0
- package/src/tools/translate.js +385 -0
|
@@ -0,0 +1,727 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Harness tools — run mt-eval benchmarks via child process.
|
|
3
|
+
*
|
|
4
|
+
* This module wraps the mt-eval CLI to run benchmarks from the queue.
|
|
5
|
+
* The actual execution spawns mt-eval as a subprocess. If mt-eval
|
|
6
|
+
* isn't installed, the tool returns instructions for installation.
|
|
7
|
+
*
|
|
8
|
+
* Security:
|
|
9
|
+
* - The queue is fetched over HTTP and is therefore UNTRUSTED. We never
|
|
10
|
+
* execute an item's `run_command` string in a shell (`bash -c`). Instead
|
|
11
|
+
* we reconstruct a shell-free argv from the item's structured fields
|
|
12
|
+
* (corpus_id / model / target_language / condition) and spawn mt-eval
|
|
13
|
+
* directly (no shell). See buildRunArgv. A compromise of the static host,
|
|
14
|
+
* the queue-build pipeline, or the queue URL is therefore NOT arbitrary
|
|
15
|
+
* code execution on the agent's machine.
|
|
16
|
+
* - Spending tokens requires an EXPLICIT confirmation: runBenchmark refuses
|
|
17
|
+
* to spend unless `confirm: true` is passed. There is no TTY under MCP
|
|
18
|
+
* stdio, so the harness's own interactive prompt cannot be relied on — the
|
|
19
|
+
* confirmation gate lives here, at the MCP boundary, instead. dry_run
|
|
20
|
+
* spends nothing and needs no confirmation.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { spawn } from 'node:child_process';
|
|
24
|
+
import { appendFile } from 'node:fs/promises';
|
|
25
|
+
import { tmpdir } from 'node:os';
|
|
26
|
+
import { join } from 'node:path';
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Check if mt-eval is installed and accessible on the PATH.
|
|
30
|
+
*
|
|
31
|
+
* @returns {Promise<boolean>}
|
|
32
|
+
*/
|
|
33
|
+
export async function isMtEvalInstalled() {
|
|
34
|
+
return new Promise((resolve) => {
|
|
35
|
+
const proc = spawn('mt-eval', ['--version'], {
|
|
36
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
37
|
+
timeout: 5000,
|
|
38
|
+
});
|
|
39
|
+
proc.on('close', (code) => resolve(code === 0));
|
|
40
|
+
proc.on('error', () => resolve(false));
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Run a command and capture its output.
|
|
46
|
+
*
|
|
47
|
+
* Spawns WITHOUT `shell: true` — args are passed directly to the program, so
|
|
48
|
+
* no shell ever interprets them. stdin is `ignore` (not `inherit`): under MCP
|
|
49
|
+
* stdio the parent's stdin carries the JSON-RPC stream, and a child must never
|
|
50
|
+
* read or block on it.
|
|
51
|
+
*
|
|
52
|
+
* @param {string} cmd Command to run
|
|
53
|
+
* @param {string[]} args Arguments
|
|
54
|
+
* @param {object} opts Options
|
|
55
|
+
* @returns {Promise<{ code: number, stdout: string, stderr: string }>}
|
|
56
|
+
*/
|
|
57
|
+
export function execCapture(cmd, args, { timeout = 300_000 } = {}) {
|
|
58
|
+
return new Promise((resolve, reject) => {
|
|
59
|
+
const proc = spawn(cmd, args, {
|
|
60
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
61
|
+
timeout,
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
let stdout = '';
|
|
65
|
+
let stderr = '';
|
|
66
|
+
|
|
67
|
+
proc.stdout.on('data', (d) => { stdout += d; });
|
|
68
|
+
proc.stderr.on('data', (d) => { stderr += d; });
|
|
69
|
+
|
|
70
|
+
proc.on('close', (code) => {
|
|
71
|
+
resolve({ code: code ?? 1, stdout, stderr });
|
|
72
|
+
});
|
|
73
|
+
proc.on('error', (err) => {
|
|
74
|
+
reject(new Error(`Failed to start ${cmd}: ${err.message}`));
|
|
75
|
+
});
|
|
76
|
+
});
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// ---------------------------------------------------------------------------
|
|
80
|
+
// Subprocess error sanitization — never relay local absolute paths.
|
|
81
|
+
// ---------------------------------------------------------------------------
|
|
82
|
+
//
|
|
83
|
+
// mt-eval failures often arrive as full Python tracebacks whose `File "…"`
|
|
84
|
+
// frames carry local absolute paths (username, directory layout). Relaying
|
|
85
|
+
// those verbatim leaks the local filesystem into the agent conversation.
|
|
86
|
+
// Error relays therefore surface only a path-stripped summary plus a short
|
|
87
|
+
// hint; the complete, unedited output goes to a local debug log.
|
|
88
|
+
|
|
89
|
+
const DEBUG_LOG_BASENAME = 'champollion-mcp-debug.log';
|
|
90
|
+
|
|
91
|
+
/** Where the debug log lives. Env-overridable so tests can redirect it. */
|
|
92
|
+
function debugLogPath() {
|
|
93
|
+
return process.env.CHAMPOLLION_MCP_DEBUG_LOG
|
|
94
|
+
|| join(tmpdir(), DEBUG_LOG_BASENAME);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// The user-facing pointer to the log deliberately names the file, not its
|
|
98
|
+
// absolute path — surfacing the path would reintroduce the leak this exists
|
|
99
|
+
// to prevent.
|
|
100
|
+
const DEBUG_LOG_HINT = `${DEBUG_LOG_BASENAME} in the system temp directory`;
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Append full subprocess detail to the local debug log. Best-effort only —
|
|
104
|
+
* logging must never break a tool call, so write failures are swallowed.
|
|
105
|
+
*
|
|
106
|
+
* @param {string} title One-line summary of what failed.
|
|
107
|
+
* @param {string} detail Full, unedited output.
|
|
108
|
+
*/
|
|
109
|
+
async function writeDebugLog(title, detail) {
|
|
110
|
+
try {
|
|
111
|
+
await appendFile(
|
|
112
|
+
debugLogPath(),
|
|
113
|
+
`[${new Date().toISOString()}] ${title}\n${detail}\n\n`,
|
|
114
|
+
'utf-8',
|
|
115
|
+
);
|
|
116
|
+
} catch {
|
|
117
|
+
// Debug logging is diagnostics, not behavior.
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Full stdout/stderr block for a debug log entry. */
|
|
122
|
+
function debugDetail(stdout, stderr) {
|
|
123
|
+
return [
|
|
124
|
+
'--- stdout ---',
|
|
125
|
+
stdout || '(empty)',
|
|
126
|
+
'--- stderr ---',
|
|
127
|
+
stderr || '(empty)',
|
|
128
|
+
].join('\n');
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Strip absolute filesystem paths from relayed text, keeping basenames.
|
|
133
|
+
*
|
|
134
|
+
* Two passes: quoted paths first (Python traceback frames quote them, and
|
|
135
|
+
* they may contain spaces — `File "/Users/x/my dir/runner.py"`), then bare
|
|
136
|
+
* unquoted paths. The lookbehind on the second pass keeps URL paths
|
|
137
|
+
* (…dev/queue.json) and model slugs (anthropic/claude-…) intact.
|
|
138
|
+
*
|
|
139
|
+
* @param {string} text
|
|
140
|
+
* @returns {string}
|
|
141
|
+
*/
|
|
142
|
+
export function stripAbsolutePaths(text) {
|
|
143
|
+
if (!text) return text;
|
|
144
|
+
return text
|
|
145
|
+
.replace(/"(?:[A-Za-z]:)?[/\\][^"]*[/\\]([^"/\\]+)"/g, '"$1"')
|
|
146
|
+
.replace(/(?<![\w:/])\/(?:[^\s/'")\],]+\/)+([^\s/'")\],]+)/g, '$1');
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Reduce raw subprocess output to a single, path-stripped final error line.
|
|
151
|
+
* A Python traceback ends with the `ExceptionType: message` line — the only
|
|
152
|
+
* line worth relaying; the frames above it are local paths and stack noise.
|
|
153
|
+
*
|
|
154
|
+
* @param {string} stderr
|
|
155
|
+
* @param {string} stdout
|
|
156
|
+
* @returns {string}
|
|
157
|
+
*/
|
|
158
|
+
function finalErrorLine(stderr, stdout) {
|
|
159
|
+
const source = (stderr || '').trim() || (stdout || '').trim();
|
|
160
|
+
if (!source) return '(no error output captured)';
|
|
161
|
+
const lines = source.split('\n').map((l) => l.trim()).filter(Boolean);
|
|
162
|
+
return stripAbsolutePaths(lines[lines.length - 1]);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// ---------------------------------------------------------------------------
|
|
166
|
+
// Command construction — reconstruct argv locally, NEVER shell the queue.
|
|
167
|
+
// ---------------------------------------------------------------------------
|
|
168
|
+
//
|
|
169
|
+
// The queue is untrusted, so even though we never shell these values we still
|
|
170
|
+
// validate each structured field against its expected shape. The argv is built
|
|
171
|
+
// positionally (`--flag value`), so the residual risk is *argument* injection
|
|
172
|
+
// into mt-eval itself — chiefly a value that could be read as an option. We
|
|
173
|
+
// reject anything with a leading dash or control characters, plus anything
|
|
174
|
+
// outside the known id/slug character sets. (Mirror of the Python
|
|
175
|
+
// build_run_argv in arena/mt_eval_harness/queue_runner.py.)
|
|
176
|
+
const CORPUS_ID_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
|
|
177
|
+
const MODEL_RE = /^[A-Za-z0-9][A-Za-z0-9._/:@-]*$/;
|
|
178
|
+
// eslint-disable-next-line no-control-regex
|
|
179
|
+
const CTRL_RE = /[\u0000-\u001f\u007f]/;
|
|
180
|
+
|
|
181
|
+
function requireStrField(item, key) {
|
|
182
|
+
const val = item == null ? undefined : item[key];
|
|
183
|
+
if (typeof val !== 'string' || val.trim() === '') {
|
|
184
|
+
throw new Error(`queue item is missing or has an empty '${key}' field`);
|
|
185
|
+
}
|
|
186
|
+
return val;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Reconstruct the `mt-eval run` argv for a queue item from its STRUCTURED
|
|
191
|
+
* fields — never from the network-supplied `run_command` string.
|
|
192
|
+
*
|
|
193
|
+
* Returns a string[] (the argv AFTER the `mt-eval` program name) suitable for
|
|
194
|
+
* `spawn('mt-eval', argv)` with NO shell. Throws if a required field is
|
|
195
|
+
* missing or fails validation, or if the item is coached (the MCP cannot
|
|
196
|
+
* supply a per-contributor coaching file).
|
|
197
|
+
*
|
|
198
|
+
* @param {object} item
|
|
199
|
+
* @param {object} [opts]
|
|
200
|
+
* @param {string} [opts.provider]
|
|
201
|
+
* @returns {string[]}
|
|
202
|
+
*/
|
|
203
|
+
export function buildRunArgv(item, { provider } = {}) {
|
|
204
|
+
const corpusId = requireStrField(item, 'corpus_id');
|
|
205
|
+
const model = requireStrField(item, 'model');
|
|
206
|
+
const targetLanguage = requireStrField(item, 'target_language');
|
|
207
|
+
|
|
208
|
+
if (!CORPUS_ID_RE.test(corpusId)) {
|
|
209
|
+
throw new Error(`corpus_id failed validation: ${corpusId}`);
|
|
210
|
+
}
|
|
211
|
+
if (!MODEL_RE.test(model)) {
|
|
212
|
+
throw new Error(`model failed validation: ${model}`);
|
|
213
|
+
}
|
|
214
|
+
// Language names may contain spaces, parentheses, commas and non-ASCII
|
|
215
|
+
// letters (e.g. "Plains Cree (nêhiyawêwin, SRO)"). Reject only a leading
|
|
216
|
+
// dash (would parse as an option) and control characters.
|
|
217
|
+
if (targetLanguage.startsWith('-') || CTRL_RE.test(targetLanguage)) {
|
|
218
|
+
throw new Error(`target_language failed validation: ${targetLanguage}`);
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
if (item.condition === 'coached') {
|
|
222
|
+
throw new Error(
|
|
223
|
+
'coached items require a coaching file and cannot be run via the MCP; '
|
|
224
|
+
+ 'use the mt-eval CLI directly with --coaching-file.',
|
|
225
|
+
);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
const argv = [
|
|
229
|
+
'run',
|
|
230
|
+
'--corpus', corpusId,
|
|
231
|
+
'--model', model,
|
|
232
|
+
'--target-lang', targetLanguage,
|
|
233
|
+
'--yes',
|
|
234
|
+
];
|
|
235
|
+
// Provider is a local, schema-validated value (not network data). OpenRouter
|
|
236
|
+
// is the harness default, so only pass the flag for a non-default provider.
|
|
237
|
+
if (provider && provider !== 'openrouter') {
|
|
238
|
+
argv.push('--provider', provider);
|
|
239
|
+
}
|
|
240
|
+
return argv;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
function fmtCost(item) {
|
|
244
|
+
return `$${(item.est_cost_usd || 0).toFixed(4)}`;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
// ---------------------------------------------------------------------------
|
|
248
|
+
// Background job registry — the fix for the 60s client-timeout trap.
|
|
249
|
+
// ---------------------------------------------------------------------------
|
|
250
|
+
//
|
|
251
|
+
// A real benchmark (a non-trivial corpus through a live model) routinely takes
|
|
252
|
+
// several minutes. The MCP SDK's *client* request timeout defaults to 60s, so
|
|
253
|
+
// a tool that synchronously awaits the subprocess makes a correct run look like
|
|
254
|
+
// a failure: the client gives up at 60s even though mt-eval is still going.
|
|
255
|
+
//
|
|
256
|
+
// Instead, runBenchmark STARTS the subprocess in the background and returns a
|
|
257
|
+
// job handle immediately (well under any client timeout). The agent then polls
|
|
258
|
+
// get_run_status until the job settles. The registry is in-memory and lives for
|
|
259
|
+
// the MCP server process's lifetime — one process serves the agent across many
|
|
260
|
+
// tool calls, so finished records are retained for the agent to read the final
|
|
261
|
+
// output after completion.
|
|
262
|
+
const JOBS = new Map();
|
|
263
|
+
let JOB_SEQ = 0;
|
|
264
|
+
|
|
265
|
+
const MAX_OUTPUT_CHARS = 8000; // cap status output so a long queue run can't flood the channel
|
|
266
|
+
|
|
267
|
+
/** @returns {string} A fresh, process-unique job id. */
|
|
268
|
+
function newJobId() {
|
|
269
|
+
JOB_SEQ += 1;
|
|
270
|
+
return `run-${JOB_SEQ}`;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/** Elapsed (or total, if finished) seconds for a job, as an integer. */
|
|
274
|
+
function jobSeconds(job) {
|
|
275
|
+
const end = job.endedAt ?? Date.now();
|
|
276
|
+
return Math.max(0, Math.round((end - job.startedAt) / 1000));
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/** Tail-truncate long output for display, keeping the most recent lines. */
|
|
280
|
+
function tail(text) {
|
|
281
|
+
if (!text) return '';
|
|
282
|
+
if (text.length <= MAX_OUTPUT_CHARS) return text;
|
|
283
|
+
return '…(earlier output truncated; showing the tail)…\n'
|
|
284
|
+
+ text.slice(text.length - MAX_OUTPUT_CHARS);
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Launch a benchmark subprocess in the BACKGROUND and register it as a job.
|
|
289
|
+
*
|
|
290
|
+
* Returns the job record immediately — the subprocess is NOT awaited here. The
|
|
291
|
+
* background promise updates the record's status/output when mt-eval settles.
|
|
292
|
+
* A spawn failure (process never started) lands as status 'error'; a non-zero
|
|
293
|
+
* exit lands as 'failed'; success lands as 'completed'.
|
|
294
|
+
*
|
|
295
|
+
* @param {object} spec
|
|
296
|
+
* @param {string[]} spec.argv argv AFTER the `mt-eval` program name
|
|
297
|
+
* @param {'item'|'queue'} spec.mode
|
|
298
|
+
* @param {string} spec.label human description of what's running
|
|
299
|
+
* @param {string} [spec.estLabel] estimated-cost label for the start message
|
|
300
|
+
* @param {boolean} [spec.publish] queue mode: will results publish?
|
|
301
|
+
* @param {number} spec.timeout subprocess timeout (ms)
|
|
302
|
+
* @param {function} spec.exec execCapture (injectable for tests)
|
|
303
|
+
* @returns {object} the job record
|
|
304
|
+
*/
|
|
305
|
+
function launchJob({ argv, mode, label, estLabel, publish, timeout, exec }) {
|
|
306
|
+
const id = newJobId();
|
|
307
|
+
const job = {
|
|
308
|
+
id,
|
|
309
|
+
mode,
|
|
310
|
+
label,
|
|
311
|
+
estLabel: estLabel ?? null,
|
|
312
|
+
publish: publish !== false,
|
|
313
|
+
argv,
|
|
314
|
+
status: 'running',
|
|
315
|
+
startedAt: Date.now(),
|
|
316
|
+
endedAt: null,
|
|
317
|
+
exitCode: null,
|
|
318
|
+
stdout: '',
|
|
319
|
+
stderr: '',
|
|
320
|
+
error: null,
|
|
321
|
+
};
|
|
322
|
+
JOBS.set(id, job);
|
|
323
|
+
|
|
324
|
+
// Fire-and-forget. Wrapping the exec call in Promise.resolve().then(...) means
|
|
325
|
+
// even a *synchronous* throw from exec becomes a rejection the .catch handles,
|
|
326
|
+
// so a launch failure can never surface as an unhandled rejection.
|
|
327
|
+
job.promise = Promise.resolve()
|
|
328
|
+
.then(() => exec('mt-eval', argv, { timeout }))
|
|
329
|
+
.then(async ({ code, stdout, stderr }) => {
|
|
330
|
+
job.exitCode = code;
|
|
331
|
+
job.stdout = stdout;
|
|
332
|
+
job.stderr = stderr;
|
|
333
|
+
job.status = code === 0 ? 'completed' : 'failed';
|
|
334
|
+
job.endedAt = Date.now();
|
|
335
|
+
if (code !== 0) {
|
|
336
|
+
// Full unedited output goes to the debug log; get_run_status relays
|
|
337
|
+
// only a path-stripped version. Awaited inside the chain so
|
|
338
|
+
// awaitAllJobs() covers the write.
|
|
339
|
+
await writeDebugLog(
|
|
340
|
+
`job ${job.id} failed (exit ${code}) — argv: mt-eval ${argv.join(' ')}`,
|
|
341
|
+
debugDetail(stdout, stderr),
|
|
342
|
+
);
|
|
343
|
+
}
|
|
344
|
+
})
|
|
345
|
+
.catch(async (err) => {
|
|
346
|
+
job.error = err.message;
|
|
347
|
+
job.status = 'error';
|
|
348
|
+
job.endedAt = Date.now();
|
|
349
|
+
await writeDebugLog(
|
|
350
|
+
`job ${job.id} could not start — argv: mt-eval ${argv.join(' ')}`,
|
|
351
|
+
String((err && err.stack) || err),
|
|
352
|
+
);
|
|
353
|
+
});
|
|
354
|
+
|
|
355
|
+
return job;
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/** Build the "STARTED — poll get_run_status" message for a freshly launched job. */
|
|
359
|
+
function formatJobStarted(job) {
|
|
360
|
+
const publishLine = job.mode === 'queue'
|
|
361
|
+
? (job.publish
|
|
362
|
+
? 'Each result auto-publishes to the public leaderboard as it finishes.'
|
|
363
|
+
: 'Results will NOT be published (scoring/validation only — no leaderboard write).')
|
|
364
|
+
: 'Scored locally — a single-item run is not auto-published.';
|
|
365
|
+
return [
|
|
366
|
+
'STARTED — the benchmark is now running in the background.',
|
|
367
|
+
'',
|
|
368
|
+
`Job id: ${job.id}`,
|
|
369
|
+
`Running: ${job.label}`,
|
|
370
|
+
job.estLabel ? `Est. cost: ${job.estLabel}` : '',
|
|
371
|
+
publishLine,
|
|
372
|
+
'',
|
|
373
|
+
'This call returned immediately and did NOT block. A real benchmark can take',
|
|
374
|
+
'several minutes — longer than a default 60-second MCP client request',
|
|
375
|
+
'timeout — so it runs detached from this tool call.',
|
|
376
|
+
'',
|
|
377
|
+
`Next: poll get_run_status with { "job_id": "${job.id}" } every ~15-30s until`,
|
|
378
|
+
'it reports COMPLETED or FAILED. Each poll returns instantly. After it',
|
|
379
|
+
'completes, call get_results to see the scored run on the leaderboard.',
|
|
380
|
+
].filter((line) => line !== '').join('\n');
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* Format a job's current status for the agent. With no id, lists all jobs
|
|
385
|
+
* started in this session.
|
|
386
|
+
*
|
|
387
|
+
* @param {string} [jobId]
|
|
388
|
+
* @returns {string}
|
|
389
|
+
*/
|
|
390
|
+
export function getRunStatus(jobId) {
|
|
391
|
+
if (!jobId) {
|
|
392
|
+
if (JOBS.size === 0) {
|
|
393
|
+
return 'No benchmark jobs have been started in this session yet. '
|
|
394
|
+
+ 'Start one with run_benchmark (confirm: true), then poll its job id here.';
|
|
395
|
+
}
|
|
396
|
+
const lines = [...JOBS.values()].map(
|
|
397
|
+
(j) => ` ${j.id} [${j.status.toUpperCase()}] ${j.label} (${jobSeconds(j)}s)`,
|
|
398
|
+
);
|
|
399
|
+
return [
|
|
400
|
+
`Benchmark jobs this session (${JOBS.size}):`,
|
|
401
|
+
'',
|
|
402
|
+
...lines,
|
|
403
|
+
'',
|
|
404
|
+
'Call get_run_status with a specific job_id for full output.',
|
|
405
|
+
].join('\n');
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
const job = JOBS.get(jobId);
|
|
409
|
+
if (!job) {
|
|
410
|
+
const known = [...JOBS.keys()];
|
|
411
|
+
return `No benchmark job with id "${jobId}". `
|
|
412
|
+
+ (known.length
|
|
413
|
+
? `Known job ids: ${known.join(', ')}.`
|
|
414
|
+
: 'No jobs have been started in this session yet.');
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
const secs = jobSeconds(job);
|
|
418
|
+
|
|
419
|
+
if (job.status === 'running') {
|
|
420
|
+
return [
|
|
421
|
+
`RUNNING — job ${job.id} (${secs}s elapsed)`,
|
|
422
|
+
job.label,
|
|
423
|
+
'',
|
|
424
|
+
'Still working. This is expected for a real run — poll get_run_status '
|
|
425
|
+
+ 'again in ~15-30s.',
|
|
426
|
+
].join('\n');
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
if (job.status === 'completed') {
|
|
430
|
+
const closing = job.mode === 'queue'
|
|
431
|
+
? (job.publish
|
|
432
|
+
? 'Each result was published to the public leaderboard — call get_results '
|
|
433
|
+
+ '(filtered to this pair/model) to see it.'
|
|
434
|
+
: 'Results were scored but NOT published (validation run).')
|
|
435
|
+
: 'Scored locally (single-item runs are not auto-published).';
|
|
436
|
+
return [
|
|
437
|
+
`COMPLETED — job ${job.id} (took ${secs}s)`,
|
|
438
|
+
job.label,
|
|
439
|
+
'',
|
|
440
|
+
tail(job.stdout) || '(no output captured)',
|
|
441
|
+
'',
|
|
442
|
+
closing,
|
|
443
|
+
].join('\n');
|
|
444
|
+
}
|
|
445
|
+
|
|
446
|
+
if (job.status === 'failed') {
|
|
447
|
+
// Keep the output tail for context (a long run's progress lines are
|
|
448
|
+
// useful), but path-stripped — traceback frames must not leak local
|
|
449
|
+
// absolute paths. The unedited output is in the debug log.
|
|
450
|
+
return [
|
|
451
|
+
`FAILED (exit ${job.exitCode}) — job ${job.id} (after ${secs}s)`,
|
|
452
|
+
job.label,
|
|
453
|
+
'',
|
|
454
|
+
stripAbsolutePaths(tail(job.stderr || job.stdout)) || '(no output captured)',
|
|
455
|
+
'',
|
|
456
|
+
`Full unedited output is in the debug log (${DEBUG_LOG_HINT}).`,
|
|
457
|
+
].join('\n');
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
// status === 'error' — the subprocess never started.
|
|
461
|
+
return [
|
|
462
|
+
`ERROR — job ${job.id} could not start (after ${secs}s)`,
|
|
463
|
+
job.label,
|
|
464
|
+
'',
|
|
465
|
+
stripAbsolutePaths(job.error || 'unknown error'),
|
|
466
|
+
].join('\n');
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
// --- Test/introspection helpers (not part of the agent-facing surface) -------
|
|
470
|
+
|
|
471
|
+
/** Await every job's background promise to settle. For deterministic tests. */
|
|
472
|
+
export async function awaitAllJobs() {
|
|
473
|
+
await Promise.all([...JOBS.values()].map((j) => j.promise).filter(Boolean));
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
/** Snapshot of all job records (most recent last). For tests/introspection. */
|
|
477
|
+
export function listJobs() {
|
|
478
|
+
return [...JOBS.values()];
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
/** Clear the registry. For test isolation. */
|
|
482
|
+
export function resetJobs() {
|
|
483
|
+
JOBS.clear();
|
|
484
|
+
JOB_SEQ = 0;
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
/**
|
|
488
|
+
* Run one or more benchmark items from the queue.
|
|
489
|
+
*
|
|
490
|
+
* Reconstructs and executes an mt-eval command WITHOUT a shell. Spending
|
|
491
|
+
* tokens requires `confirm: true` — without it (or with `dry_run`), the tool
|
|
492
|
+
* returns a plan and spends nothing.
|
|
493
|
+
*
|
|
494
|
+
* ASYNC EXECUTION: a confirmed run does NOT block this call. The mt-eval
|
|
495
|
+
* subprocess is launched in the BACKGROUND and a job handle is returned
|
|
496
|
+
* immediately, so the call returns well under any MCP client request timeout
|
|
497
|
+
* (the SDK default is 60s, far shorter than a real run). The agent then polls
|
|
498
|
+
* get_run_status with the returned job id until the job settles. dry_run and
|
|
499
|
+
* the planner stay synchronous (they are fast, no model calls).
|
|
500
|
+
*
|
|
501
|
+
* @param {object} params
|
|
502
|
+
* @param {number} [params.budget] Budget cap in USD
|
|
503
|
+
* @param {number} [params.top] Number of top items to run
|
|
504
|
+
* @param {string} [params.item_id] Specific item ID to run
|
|
505
|
+
* @param {boolean} [params.dry_run] Show plan without executing
|
|
506
|
+
* @param {string} [params.provider] API provider override
|
|
507
|
+
* @param {boolean} [params.confirm] Must be true to actually spend tokens
|
|
508
|
+
* @param {boolean} [params.publish] Auto-publish to the public leaderboard
|
|
509
|
+
* (default true). Pass false for a
|
|
510
|
+
* scoring/validation run with NO prod write.
|
|
511
|
+
* @param {object} [deps] Injected dependencies (for testing)
|
|
512
|
+
* @returns {Promise<string>} Human-readable result. For a confirmed
|
|
513
|
+
* run this is a "STARTED" message carrying
|
|
514
|
+
* the job id to poll; otherwise a plan or
|
|
515
|
+
* confirmation prompt.
|
|
516
|
+
*/
|
|
517
|
+
export async function runBenchmark(
|
|
518
|
+
{ budget, top, item_id, dry_run = false, provider, confirm = false,
|
|
519
|
+
publish = true },
|
|
520
|
+
deps = {},
|
|
521
|
+
) {
|
|
522
|
+
const {
|
|
523
|
+
isMtEvalInstalled: checkInstalled = isMtEvalInstalled,
|
|
524
|
+
fetchQueue = null,
|
|
525
|
+
execCapture: exec = execCapture,
|
|
526
|
+
} = deps;
|
|
527
|
+
|
|
528
|
+
// Pre-flight check: is mt-eval installed?
|
|
529
|
+
const installed = await checkInstalled();
|
|
530
|
+
if (!installed) {
|
|
531
|
+
return [
|
|
532
|
+
'mt-eval is not installed on this machine.',
|
|
533
|
+
'',
|
|
534
|
+
'To install it, run:',
|
|
535
|
+
' pipx install mt-eval',
|
|
536
|
+
'',
|
|
537
|
+
'After installation, set your API key:',
|
|
538
|
+
' export OPENROUTER_API_KEY=sk-or-...',
|
|
539
|
+
'',
|
|
540
|
+
'Then try again.',
|
|
541
|
+
].join('\n');
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
// ----- Specific item -----------------------------------------------------
|
|
545
|
+
if (item_id) {
|
|
546
|
+
const getQueue = fetchQueue
|
|
547
|
+
?? (await import('./queue.js')).fetchQueue;
|
|
548
|
+
const queue = await getQueue();
|
|
549
|
+
const item = queue.items.find((it) => it.id === item_id);
|
|
550
|
+
if (!item) {
|
|
551
|
+
return `Queue item "${item_id}" not found. Use list_queue to see available items.`;
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
// Reconstruct a shell-free argv from STRUCTURED fields. The item's
|
|
555
|
+
// network-supplied run_command is NEVER executed.
|
|
556
|
+
let argv;
|
|
557
|
+
try {
|
|
558
|
+
argv = buildRunArgv(item, { provider });
|
|
559
|
+
} catch (err) {
|
|
560
|
+
return `Cannot run "${item_id}": ${err.message}`;
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
if (dry_run) {
|
|
564
|
+
return [
|
|
565
|
+
'DRY RUN — would execute (no shell):',
|
|
566
|
+
'',
|
|
567
|
+
` mt-eval ${argv.join(' ')}`,
|
|
568
|
+
'',
|
|
569
|
+
`Language pair: ${item.language_pair.replace('>', ' → ')}`,
|
|
570
|
+
`Target: ${item.target_language}`,
|
|
571
|
+
`Model: ${item.model}`,
|
|
572
|
+
`Condition: ${item.condition}`,
|
|
573
|
+
`Estimated cost: ${fmtCost(item)}`,
|
|
574
|
+
'',
|
|
575
|
+
'No tokens were spent.',
|
|
576
|
+
].join('\n');
|
|
577
|
+
}
|
|
578
|
+
|
|
579
|
+
// ENFORCED confirmation gate — spending requires confirm: true.
|
|
580
|
+
if (confirm !== true) {
|
|
581
|
+
return [
|
|
582
|
+
'CONFIRMATION REQUIRED — this will spend real tokens.',
|
|
583
|
+
'',
|
|
584
|
+
`Item: ${item_id}`,
|
|
585
|
+
`Pair: ${item.language_pair.replace('>', ' → ')}`,
|
|
586
|
+
`Model: ${item.model}`,
|
|
587
|
+
`Est. cost: ${fmtCost(item)} (actual depends on provider pricing)`,
|
|
588
|
+
'',
|
|
589
|
+
// A single-item `mt-eval run` is scored locally and is NOT
|
|
590
|
+
// auto-published under the MCP (publishing is the budget/top queue
|
|
591
|
+
// flow). State it so the agent isn't surprised either way; the
|
|
592
|
+
// publish:false param applies to budget/top runs.
|
|
593
|
+
'Note: a single-item run is scored locally and is NOT auto-published '
|
|
594
|
+
+ 'to the leaderboard — publish it later with `mt-eval publish`, or use '
|
|
595
|
+
+ 'budget/top queue mode to auto-publish.',
|
|
596
|
+
'',
|
|
597
|
+
'Confirm with the user first, then call run_benchmark again with '
|
|
598
|
+
+ 'confirm: true to proceed. Or pass dry_run: true to preview without '
|
|
599
|
+
+ 'spending.',
|
|
600
|
+
].join('\n');
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
// Launch in the BACKGROUND and return a job handle immediately. A real run
|
|
604
|
+
// would otherwise blow past the client's 60s request timeout. The agent
|
|
605
|
+
// polls get_run_status with the returned job id.
|
|
606
|
+
const job = launchJob({
|
|
607
|
+
argv,
|
|
608
|
+
mode: 'item',
|
|
609
|
+
label: `${item.language_pair.replace('>', ' → ')} ${item.model} [${item.condition}]`,
|
|
610
|
+
estLabel: fmtCost(item),
|
|
611
|
+
timeout: 600_000, // 10 minute subprocess timeout for a single run
|
|
612
|
+
exec,
|
|
613
|
+
});
|
|
614
|
+
return formatJobStarted(job);
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
// ----- Queue mode (budget / top) -----------------------------------------
|
|
618
|
+
const args = ['queue'];
|
|
619
|
+
if (budget != null) args.push('--budget', String(budget));
|
|
620
|
+
if (top != null) args.push('--top', String(top));
|
|
621
|
+
if (provider) args.push('--provider', provider);
|
|
622
|
+
|
|
623
|
+
// Determinism: estimate_cost / list_queue preview the queue with the
|
|
624
|
+
// deterministic top-order filterQueue (no anti-collision spread). The
|
|
625
|
+
// harness, however, turns spread ON by default for --budget (the mass
|
|
626
|
+
// curl|bash donate flow, where many workers must fan out). Under the MCP a
|
|
627
|
+
// single agent-mediated run just executed a previewed plan, so pass
|
|
628
|
+
// --no-spread to make the EXECUTED selection match that preview item-for-item
|
|
629
|
+
// instead of a spread-permuted set the user never saw. (No-op for --top,
|
|
630
|
+
// which is already deterministic; harmless to pass.)
|
|
631
|
+
args.push('--no-spread');
|
|
632
|
+
|
|
633
|
+
// Prod-write opt-out. Queue runs auto-publish each result to the public
|
|
634
|
+
// leaderboard by default. publish:false threads --no-publish so an agent can
|
|
635
|
+
// run a scoring/validation pass with NO leaderboard write (and the harness
|
|
636
|
+
// then skips OAuth entirely).
|
|
637
|
+
if (publish === false) args.push('--no-publish');
|
|
638
|
+
|
|
639
|
+
if (dry_run) {
|
|
640
|
+
args.push('--dry-run');
|
|
641
|
+
const { code, stdout, stderr } = await exec('mt-eval', args, {
|
|
642
|
+
timeout: 1800_000,
|
|
643
|
+
});
|
|
644
|
+
if (code !== 0) {
|
|
645
|
+
// Never relay the raw harness output here — a Python traceback carries
|
|
646
|
+
// local absolute paths. Surface the final error line (path-stripped)
|
|
647
|
+
// and keep the full detail in the debug log only.
|
|
648
|
+
await writeDebugLog(
|
|
649
|
+
`queue dry-run failed (exit ${code}) — argv: mt-eval ${args.join(' ')}`,
|
|
650
|
+
debugDetail(stdout, stderr),
|
|
651
|
+
);
|
|
652
|
+
return [
|
|
653
|
+
`Queue dry-run failed (exit code ${code}): ${finalErrorLine(stderr, stdout)}`,
|
|
654
|
+
'',
|
|
655
|
+
'Hint: this error came from the local mt-eval harness, not the queue. '
|
|
656
|
+
+ 'Run the same mt-eval command in a terminal to reproduce; the full '
|
|
657
|
+
+ `unedited output was saved to the debug log (${DEBUG_LOG_HINT}).`,
|
|
658
|
+
].join('\n');
|
|
659
|
+
}
|
|
660
|
+
return stdout || 'Queue dry-run completed (no output captured).';
|
|
661
|
+
}
|
|
662
|
+
|
|
663
|
+
// ENFORCED bound on scope. Without a selector, argv is
|
|
664
|
+
// `mt-eval queue --no-spread --yes` — the WHOLE queue (thousands of items),
|
|
665
|
+
// with no dollar figure anywhere in the confirmation text to warn the agent
|
|
666
|
+
// or the user. A run must always name how much it may spend (--budget) or
|
|
667
|
+
// how many items it may take (--top). dry_run is exempt: previewing the full
|
|
668
|
+
// queue plan costs nothing.
|
|
669
|
+
if (budget == null && top == null) {
|
|
670
|
+
return [
|
|
671
|
+
'REFUSED — an unbounded queue run is not allowed.',
|
|
672
|
+
'',
|
|
673
|
+
'Calling run_benchmark with no budget, no top and no item_id would run '
|
|
674
|
+
+ 'the ENTIRE queue (thousands of paid model calls).',
|
|
675
|
+
'',
|
|
676
|
+
'Pass exactly one bound:',
|
|
677
|
+
' • budget: 5 — spend at most $5.00, harness picks the items',
|
|
678
|
+
' • top: 10 — run the top 10 queue items',
|
|
679
|
+
' • item_id: "<id>" — run one specific item',
|
|
680
|
+
'',
|
|
681
|
+
'Use estimate_cost first to see what a given bound would cost, or '
|
|
682
|
+
+ 'dry_run: true to preview the full queue plan without spending.',
|
|
683
|
+
].join('\n');
|
|
684
|
+
}
|
|
685
|
+
|
|
686
|
+
// ENFORCED confirmation gate for spend.
|
|
687
|
+
if (confirm !== true) {
|
|
688
|
+
const scope = budget != null
|
|
689
|
+
? `up to $${Number(budget).toFixed(2)} of estimated spend`
|
|
690
|
+
: `the top ${top} item(s)`;
|
|
691
|
+
const publishLine = publish === false
|
|
692
|
+
? 'Results will NOT be published (scoring/validation only — no leaderboard write).'
|
|
693
|
+
: 'Each result will be published to the public leaderboard.';
|
|
694
|
+
return [
|
|
695
|
+
'CONFIRMATION REQUIRED — this will spend real tokens.',
|
|
696
|
+
'',
|
|
697
|
+
`This would run ${scope} from the top of the queue.`,
|
|
698
|
+
publishLine,
|
|
699
|
+
'',
|
|
700
|
+
'Confirm with the user first, then call run_benchmark again with '
|
|
701
|
+
+ 'confirm: true to proceed. Or pass dry_run: true to preview the plan '
|
|
702
|
+
+ 'without spending.',
|
|
703
|
+
].join('\n');
|
|
704
|
+
}
|
|
705
|
+
|
|
706
|
+
// Confirmation happened here, at the MCP boundary. Pass --yes so mt-eval
|
|
707
|
+
// does not block on its own interactive prompt — there is no TTY under MCP
|
|
708
|
+
// stdio, so that prompt would hang (or, with an inherited stdin, consume the
|
|
709
|
+
// JSON-RPC stream).
|
|
710
|
+
args.push('--yes');
|
|
711
|
+
|
|
712
|
+
// Launch in the BACKGROUND and return a job handle immediately — a queue run
|
|
713
|
+
// can span many model calls and minutes, far past the client's 60s request
|
|
714
|
+
// timeout. The agent polls get_run_status with the returned job id.
|
|
715
|
+
const scope = budget != null
|
|
716
|
+
? `up to $${Number(budget).toFixed(2)} from the top of the queue`
|
|
717
|
+
: `the top ${top} item(s) from the queue`;
|
|
718
|
+
const job = launchJob({
|
|
719
|
+
argv: args,
|
|
720
|
+
mode: 'queue',
|
|
721
|
+
label: scope,
|
|
722
|
+
publish,
|
|
723
|
+
timeout: 1800_000, // 30 minute subprocess timeout for queue runs
|
|
724
|
+
exec,
|
|
725
|
+
});
|
|
726
|
+
return formatJobStarted(job);
|
|
727
|
+
}
|