champollion-mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,727 @@
1
+ /**
2
+ * Harness tools — run mt-eval benchmarks via child process.
3
+ *
4
+ * This module wraps the mt-eval CLI to run benchmarks from the queue.
5
+ * The actual execution spawns mt-eval as a subprocess. If mt-eval
6
+ * isn't installed, the tool returns instructions for installation.
7
+ *
8
+ * Security:
9
+ * - The queue is fetched over HTTP and is therefore UNTRUSTED. We never
10
+ * execute an item's `run_command` string in a shell (`bash -c`). Instead
11
+ * we reconstruct a shell-free argv from the item's structured fields
12
+ * (corpus_id / model / target_language / condition) and spawn mt-eval
13
+ * directly (no shell). See buildRunArgv. A compromise of the static host,
14
+ * the queue-build pipeline, or the queue URL is therefore NOT arbitrary
15
+ * code execution on the agent's machine.
16
+ * - Spending tokens requires an EXPLICIT confirmation: runBenchmark refuses
17
+ * to spend unless `confirm: true` is passed. There is no TTY under MCP
18
+ * stdio, so the harness's own interactive prompt cannot be relied on — the
19
+ * confirmation gate lives here, at the MCP boundary, instead. dry_run
20
+ * spends nothing and needs no confirmation.
21
+ */
22
+
23
+ import { spawn } from 'node:child_process';
24
+ import { appendFile } from 'node:fs/promises';
25
+ import { tmpdir } from 'node:os';
26
+ import { join } from 'node:path';
27
+
28
+ /**
29
+ * Check if mt-eval is installed and accessible on the PATH.
30
+ *
31
+ * @returns {Promise<boolean>}
32
+ */
33
+ export async function isMtEvalInstalled() {
34
+ return new Promise((resolve) => {
35
+ const proc = spawn('mt-eval', ['--version'], {
36
+ stdio: ['ignore', 'pipe', 'pipe'],
37
+ timeout: 5000,
38
+ });
39
+ proc.on('close', (code) => resolve(code === 0));
40
+ proc.on('error', () => resolve(false));
41
+ });
42
+ }
43
+
44
+ /**
45
+ * Run a command and capture its output.
46
+ *
47
+ * Spawns WITHOUT `shell: true` — args are passed directly to the program, so
48
+ * no shell ever interprets them. stdin is `ignore` (not `inherit`): under MCP
49
+ * stdio the parent's stdin carries the JSON-RPC stream, and a child must never
50
+ * read or block on it.
51
+ *
52
+ * @param {string} cmd Command to run
53
+ * @param {string[]} args Arguments
54
+ * @param {object} opts Options
55
+ * @returns {Promise<{ code: number, stdout: string, stderr: string }>}
56
+ */
57
+ export function execCapture(cmd, args, { timeout = 300_000 } = {}) {
58
+ return new Promise((resolve, reject) => {
59
+ const proc = spawn(cmd, args, {
60
+ stdio: ['ignore', 'pipe', 'pipe'],
61
+ timeout,
62
+ });
63
+
64
+ let stdout = '';
65
+ let stderr = '';
66
+
67
+ proc.stdout.on('data', (d) => { stdout += d; });
68
+ proc.stderr.on('data', (d) => { stderr += d; });
69
+
70
+ proc.on('close', (code) => {
71
+ resolve({ code: code ?? 1, stdout, stderr });
72
+ });
73
+ proc.on('error', (err) => {
74
+ reject(new Error(`Failed to start ${cmd}: ${err.message}`));
75
+ });
76
+ });
77
+ }
78
+
79
+ // ---------------------------------------------------------------------------
80
+ // Subprocess error sanitization — never relay local absolute paths.
81
+ // ---------------------------------------------------------------------------
82
+ //
83
+ // mt-eval failures often arrive as full Python tracebacks whose `File "…"`
84
+ // frames carry local absolute paths (username, directory layout). Relaying
85
+ // those verbatim leaks the local filesystem into the agent conversation.
86
+ // Error relays therefore surface only a path-stripped summary plus a short
87
+ // hint; the complete, unedited output goes to a local debug log.
88
+
89
+ const DEBUG_LOG_BASENAME = 'champollion-mcp-debug.log';
90
+
91
+ /** Where the debug log lives. Env-overridable so tests can redirect it. */
92
+ function debugLogPath() {
93
+ return process.env.CHAMPOLLION_MCP_DEBUG_LOG
94
+ || join(tmpdir(), DEBUG_LOG_BASENAME);
95
+ }
96
+
97
+ // The user-facing pointer to the log deliberately names the file, not its
98
+ // absolute path — surfacing the path would reintroduce the leak this exists
99
+ // to prevent.
100
+ const DEBUG_LOG_HINT = `${DEBUG_LOG_BASENAME} in the system temp directory`;
101
+
102
+ /**
103
+ * Append full subprocess detail to the local debug log. Best-effort only —
104
+ * logging must never break a tool call, so write failures are swallowed.
105
+ *
106
+ * @param {string} title One-line summary of what failed.
107
+ * @param {string} detail Full, unedited output.
108
+ */
109
+ async function writeDebugLog(title, detail) {
110
+ try {
111
+ await appendFile(
112
+ debugLogPath(),
113
+ `[${new Date().toISOString()}] ${title}\n${detail}\n\n`,
114
+ 'utf-8',
115
+ );
116
+ } catch {
117
+ // Debug logging is diagnostics, not behavior.
118
+ }
119
+ }
120
+
121
+ /** Full stdout/stderr block for a debug log entry. */
122
+ function debugDetail(stdout, stderr) {
123
+ return [
124
+ '--- stdout ---',
125
+ stdout || '(empty)',
126
+ '--- stderr ---',
127
+ stderr || '(empty)',
128
+ ].join('\n');
129
+ }
130
+
131
+ /**
132
+ * Strip absolute filesystem paths from relayed text, keeping basenames.
133
+ *
134
+ * Two passes: quoted paths first (Python traceback frames quote them, and
135
+ * they may contain spaces — `File "/Users/x/my dir/runner.py"`), then bare
136
+ * unquoted paths. The lookbehind on the second pass keeps URL paths
137
+ * (…dev/queue.json) and model slugs (anthropic/claude-…) intact.
138
+ *
139
+ * @param {string} text
140
+ * @returns {string}
141
+ */
142
+ export function stripAbsolutePaths(text) {
143
+ if (!text) return text;
144
+ return text
145
+ .replace(/"(?:[A-Za-z]:)?[/\\][^"]*[/\\]([^"/\\]+)"/g, '"$1"')
146
+ .replace(/(?<![\w:/])\/(?:[^\s/'")\],]+\/)+([^\s/'")\],]+)/g, '$1');
147
+ }
148
+
149
+ /**
150
+ * Reduce raw subprocess output to a single, path-stripped final error line.
151
+ * A Python traceback ends with the `ExceptionType: message` line — the only
152
+ * line worth relaying; the frames above it are local paths and stack noise.
153
+ *
154
+ * @param {string} stderr
155
+ * @param {string} stdout
156
+ * @returns {string}
157
+ */
158
+ function finalErrorLine(stderr, stdout) {
159
+ const source = (stderr || '').trim() || (stdout || '').trim();
160
+ if (!source) return '(no error output captured)';
161
+ const lines = source.split('\n').map((l) => l.trim()).filter(Boolean);
162
+ return stripAbsolutePaths(lines[lines.length - 1]);
163
+ }
164
+
165
+ // ---------------------------------------------------------------------------
166
+ // Command construction — reconstruct argv locally, NEVER shell the queue.
167
+ // ---------------------------------------------------------------------------
168
+ //
169
+ // The queue is untrusted, so even though we never shell these values we still
170
+ // validate each structured field against its expected shape. The argv is built
171
+ // positionally (`--flag value`), so the residual risk is *argument* injection
172
+ // into mt-eval itself — chiefly a value that could be read as an option. We
173
+ // reject anything with a leading dash or control characters, plus anything
174
+ // outside the known id/slug character sets. (Mirror of the Python
175
+ // build_run_argv in arena/mt_eval_harness/queue_runner.py.)
176
+ const CORPUS_ID_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
177
+ const MODEL_RE = /^[A-Za-z0-9][A-Za-z0-9._/:@-]*$/;
178
+ // eslint-disable-next-line no-control-regex
179
+ const CTRL_RE = /[\u0000-\u001f\u007f]/;
180
+
181
+ function requireStrField(item, key) {
182
+ const val = item == null ? undefined : item[key];
183
+ if (typeof val !== 'string' || val.trim() === '') {
184
+ throw new Error(`queue item is missing or has an empty '${key}' field`);
185
+ }
186
+ return val;
187
+ }
188
+
189
+ /**
190
+ * Reconstruct the `mt-eval run` argv for a queue item from its STRUCTURED
191
+ * fields — never from the network-supplied `run_command` string.
192
+ *
193
+ * Returns a string[] (the argv AFTER the `mt-eval` program name) suitable for
194
+ * `spawn('mt-eval', argv)` with NO shell. Throws if a required field is
195
+ * missing or fails validation, or if the item is coached (the MCP cannot
196
+ * supply a per-contributor coaching file).
197
+ *
198
+ * @param {object} item
199
+ * @param {object} [opts]
200
+ * @param {string} [opts.provider]
201
+ * @returns {string[]}
202
+ */
203
+ export function buildRunArgv(item, { provider } = {}) {
204
+ const corpusId = requireStrField(item, 'corpus_id');
205
+ const model = requireStrField(item, 'model');
206
+ const targetLanguage = requireStrField(item, 'target_language');
207
+
208
+ if (!CORPUS_ID_RE.test(corpusId)) {
209
+ throw new Error(`corpus_id failed validation: ${corpusId}`);
210
+ }
211
+ if (!MODEL_RE.test(model)) {
212
+ throw new Error(`model failed validation: ${model}`);
213
+ }
214
+ // Language names may contain spaces, parentheses, commas and non-ASCII
215
+ // letters (e.g. "Plains Cree (nêhiyawêwin, SRO)"). Reject only a leading
216
+ // dash (would parse as an option) and control characters.
217
+ if (targetLanguage.startsWith('-') || CTRL_RE.test(targetLanguage)) {
218
+ throw new Error(`target_language failed validation: ${targetLanguage}`);
219
+ }
220
+
221
+ if (item.condition === 'coached') {
222
+ throw new Error(
223
+ 'coached items require a coaching file and cannot be run via the MCP; '
224
+ + 'use the mt-eval CLI directly with --coaching-file.',
225
+ );
226
+ }
227
+
228
+ const argv = [
229
+ 'run',
230
+ '--corpus', corpusId,
231
+ '--model', model,
232
+ '--target-lang', targetLanguage,
233
+ '--yes',
234
+ ];
235
+ // Provider is a local, schema-validated value (not network data). OpenRouter
236
+ // is the harness default, so only pass the flag for a non-default provider.
237
+ if (provider && provider !== 'openrouter') {
238
+ argv.push('--provider', provider);
239
+ }
240
+ return argv;
241
+ }
242
+
243
+ function fmtCost(item) {
244
+ return `$${(item.est_cost_usd || 0).toFixed(4)}`;
245
+ }
246
+
247
+ // ---------------------------------------------------------------------------
248
+ // Background job registry — the fix for the 60s client-timeout trap.
249
+ // ---------------------------------------------------------------------------
250
+ //
251
+ // A real benchmark (a non-trivial corpus through a live model) routinely takes
252
+ // several minutes. The MCP SDK's *client* request timeout defaults to 60s, so
253
+ // a tool that synchronously awaits the subprocess makes a correct run look like
254
+ // a failure: the client gives up at 60s even though mt-eval is still going.
255
+ //
256
+ // Instead, runBenchmark STARTS the subprocess in the background and returns a
257
+ // job handle immediately (well under any client timeout). The agent then polls
258
+ // get_run_status until the job settles. The registry is in-memory and lives for
259
+ // the MCP server process's lifetime — one process serves the agent across many
260
+ // tool calls, so finished records are retained for the agent to read the final
261
+ // output after completion.
262
+ const JOBS = new Map();
263
+ let JOB_SEQ = 0;
264
+
265
+ const MAX_OUTPUT_CHARS = 8000; // cap status output so a long queue run can't flood the channel
266
+
267
+ /** @returns {string} A fresh, process-unique job id. */
268
+ function newJobId() {
269
+ JOB_SEQ += 1;
270
+ return `run-${JOB_SEQ}`;
271
+ }
272
+
273
+ /** Elapsed (or total, if finished) seconds for a job, as an integer. */
274
+ function jobSeconds(job) {
275
+ const end = job.endedAt ?? Date.now();
276
+ return Math.max(0, Math.round((end - job.startedAt) / 1000));
277
+ }
278
+
279
+ /** Tail-truncate long output for display, keeping the most recent lines. */
280
+ function tail(text) {
281
+ if (!text) return '';
282
+ if (text.length <= MAX_OUTPUT_CHARS) return text;
283
+ return '…(earlier output truncated; showing the tail)…\n'
284
+ + text.slice(text.length - MAX_OUTPUT_CHARS);
285
+ }
286
+
287
+ /**
288
+ * Launch a benchmark subprocess in the BACKGROUND and register it as a job.
289
+ *
290
+ * Returns the job record immediately — the subprocess is NOT awaited here. The
291
+ * background promise updates the record's status/output when mt-eval settles.
292
+ * A spawn failure (process never started) lands as status 'error'; a non-zero
293
+ * exit lands as 'failed'; success lands as 'completed'.
294
+ *
295
+ * @param {object} spec
296
+ * @param {string[]} spec.argv argv AFTER the `mt-eval` program name
297
+ * @param {'item'|'queue'} spec.mode
298
+ * @param {string} spec.label human description of what's running
299
+ * @param {string} [spec.estLabel] estimated-cost label for the start message
300
+ * @param {boolean} [spec.publish] queue mode: will results publish?
301
+ * @param {number} spec.timeout subprocess timeout (ms)
302
+ * @param {function} spec.exec execCapture (injectable for tests)
303
+ * @returns {object} the job record
304
+ */
305
+ function launchJob({ argv, mode, label, estLabel, publish, timeout, exec }) {
306
+ const id = newJobId();
307
+ const job = {
308
+ id,
309
+ mode,
310
+ label,
311
+ estLabel: estLabel ?? null,
312
+ publish: publish !== false,
313
+ argv,
314
+ status: 'running',
315
+ startedAt: Date.now(),
316
+ endedAt: null,
317
+ exitCode: null,
318
+ stdout: '',
319
+ stderr: '',
320
+ error: null,
321
+ };
322
+ JOBS.set(id, job);
323
+
324
+ // Fire-and-forget. Wrapping the exec call in Promise.resolve().then(...) means
325
+ // even a *synchronous* throw from exec becomes a rejection the .catch handles,
326
+ // so a launch failure can never surface as an unhandled rejection.
327
+ job.promise = Promise.resolve()
328
+ .then(() => exec('mt-eval', argv, { timeout }))
329
+ .then(async ({ code, stdout, stderr }) => {
330
+ job.exitCode = code;
331
+ job.stdout = stdout;
332
+ job.stderr = stderr;
333
+ job.status = code === 0 ? 'completed' : 'failed';
334
+ job.endedAt = Date.now();
335
+ if (code !== 0) {
336
+ // Full unedited output goes to the debug log; get_run_status relays
337
+ // only a path-stripped version. Awaited inside the chain so
338
+ // awaitAllJobs() covers the write.
339
+ await writeDebugLog(
340
+ `job ${job.id} failed (exit ${code}) — argv: mt-eval ${argv.join(' ')}`,
341
+ debugDetail(stdout, stderr),
342
+ );
343
+ }
344
+ })
345
+ .catch(async (err) => {
346
+ job.error = err.message;
347
+ job.status = 'error';
348
+ job.endedAt = Date.now();
349
+ await writeDebugLog(
350
+ `job ${job.id} could not start — argv: mt-eval ${argv.join(' ')}`,
351
+ String((err && err.stack) || err),
352
+ );
353
+ });
354
+
355
+ return job;
356
+ }
357
+
358
+ /** Build the "STARTED — poll get_run_status" message for a freshly launched job. */
359
+ function formatJobStarted(job) {
360
+ const publishLine = job.mode === 'queue'
361
+ ? (job.publish
362
+ ? 'Each result auto-publishes to the public leaderboard as it finishes.'
363
+ : 'Results will NOT be published (scoring/validation only — no leaderboard write).')
364
+ : 'Scored locally — a single-item run is not auto-published.';
365
+ return [
366
+ 'STARTED — the benchmark is now running in the background.',
367
+ '',
368
+ `Job id: ${job.id}`,
369
+ `Running: ${job.label}`,
370
+ job.estLabel ? `Est. cost: ${job.estLabel}` : '',
371
+ publishLine,
372
+ '',
373
+ 'This call returned immediately and did NOT block. A real benchmark can take',
374
+ 'several minutes — longer than a default 60-second MCP client request',
375
+ 'timeout — so it runs detached from this tool call.',
376
+ '',
377
+ `Next: poll get_run_status with { "job_id": "${job.id}" } every ~15-30s until`,
378
+ 'it reports COMPLETED or FAILED. Each poll returns instantly. After it',
379
+ 'completes, call get_results to see the scored run on the leaderboard.',
380
+ ].filter((line) => line !== '').join('\n');
381
+ }
382
+
383
+ /**
384
+ * Format a job's current status for the agent. With no id, lists all jobs
385
+ * started in this session.
386
+ *
387
+ * @param {string} [jobId]
388
+ * @returns {string}
389
+ */
390
+ export function getRunStatus(jobId) {
391
+ if (!jobId) {
392
+ if (JOBS.size === 0) {
393
+ return 'No benchmark jobs have been started in this session yet. '
394
+ + 'Start one with run_benchmark (confirm: true), then poll its job id here.';
395
+ }
396
+ const lines = [...JOBS.values()].map(
397
+ (j) => ` ${j.id} [${j.status.toUpperCase()}] ${j.label} (${jobSeconds(j)}s)`,
398
+ );
399
+ return [
400
+ `Benchmark jobs this session (${JOBS.size}):`,
401
+ '',
402
+ ...lines,
403
+ '',
404
+ 'Call get_run_status with a specific job_id for full output.',
405
+ ].join('\n');
406
+ }
407
+
408
+ const job = JOBS.get(jobId);
409
+ if (!job) {
410
+ const known = [...JOBS.keys()];
411
+ return `No benchmark job with id "${jobId}". `
412
+ + (known.length
413
+ ? `Known job ids: ${known.join(', ')}.`
414
+ : 'No jobs have been started in this session yet.');
415
+ }
416
+
417
+ const secs = jobSeconds(job);
418
+
419
+ if (job.status === 'running') {
420
+ return [
421
+ `RUNNING — job ${job.id} (${secs}s elapsed)`,
422
+ job.label,
423
+ '',
424
+ 'Still working. This is expected for a real run — poll get_run_status '
425
+ + 'again in ~15-30s.',
426
+ ].join('\n');
427
+ }
428
+
429
+ if (job.status === 'completed') {
430
+ const closing = job.mode === 'queue'
431
+ ? (job.publish
432
+ ? 'Each result was published to the public leaderboard — call get_results '
433
+ + '(filtered to this pair/model) to see it.'
434
+ : 'Results were scored but NOT published (validation run).')
435
+ : 'Scored locally (single-item runs are not auto-published).';
436
+ return [
437
+ `COMPLETED — job ${job.id} (took ${secs}s)`,
438
+ job.label,
439
+ '',
440
+ tail(job.stdout) || '(no output captured)',
441
+ '',
442
+ closing,
443
+ ].join('\n');
444
+ }
445
+
446
+ if (job.status === 'failed') {
447
+ // Keep the output tail for context (a long run's progress lines are
448
+ // useful), but path-stripped — traceback frames must not leak local
449
+ // absolute paths. The unedited output is in the debug log.
450
+ return [
451
+ `FAILED (exit ${job.exitCode}) — job ${job.id} (after ${secs}s)`,
452
+ job.label,
453
+ '',
454
+ stripAbsolutePaths(tail(job.stderr || job.stdout)) || '(no output captured)',
455
+ '',
456
+ `Full unedited output is in the debug log (${DEBUG_LOG_HINT}).`,
457
+ ].join('\n');
458
+ }
459
+
460
+ // status === 'error' — the subprocess never started.
461
+ return [
462
+ `ERROR — job ${job.id} could not start (after ${secs}s)`,
463
+ job.label,
464
+ '',
465
+ stripAbsolutePaths(job.error || 'unknown error'),
466
+ ].join('\n');
467
+ }
468
+
469
+ // --- Test/introspection helpers (not part of the agent-facing surface) -------
470
+
471
+ /** Await every job's background promise to settle. For deterministic tests. */
472
+ export async function awaitAllJobs() {
473
+ await Promise.all([...JOBS.values()].map((j) => j.promise).filter(Boolean));
474
+ }
475
+
476
+ /** Snapshot of all job records (most recent last). For tests/introspection. */
477
+ export function listJobs() {
478
+ return [...JOBS.values()];
479
+ }
480
+
481
+ /** Clear the registry. For test isolation. */
482
+ export function resetJobs() {
483
+ JOBS.clear();
484
+ JOB_SEQ = 0;
485
+ }
486
+
487
+ /**
488
+ * Run one or more benchmark items from the queue.
489
+ *
490
+ * Reconstructs and executes an mt-eval command WITHOUT a shell. Spending
491
+ * tokens requires `confirm: true` — without it (or with `dry_run`), the tool
492
+ * returns a plan and spends nothing.
493
+ *
494
+ * ASYNC EXECUTION: a confirmed run does NOT block this call. The mt-eval
495
+ * subprocess is launched in the BACKGROUND and a job handle is returned
496
+ * immediately, so the call returns well under any MCP client request timeout
497
+ * (the SDK default is 60s, far shorter than a real run). The agent then polls
498
+ * get_run_status with the returned job id until the job settles. dry_run and
499
+ * the planner stay synchronous (they are fast, no model calls).
500
+ *
501
+ * @param {object} params
502
+ * @param {number} [params.budget] Budget cap in USD
503
+ * @param {number} [params.top] Number of top items to run
504
+ * @param {string} [params.item_id] Specific item ID to run
505
+ * @param {boolean} [params.dry_run] Show plan without executing
506
+ * @param {string} [params.provider] API provider override
507
+ * @param {boolean} [params.confirm] Must be true to actually spend tokens
508
+ * @param {boolean} [params.publish] Auto-publish to the public leaderboard
509
+ * (default true). Pass false for a
510
+ * scoring/validation run with NO prod write.
511
+ * @param {object} [deps] Injected dependencies (for testing)
512
+ * @returns {Promise<string>} Human-readable result. For a confirmed
513
+ * run this is a "STARTED" message carrying
514
+ * the job id to poll; otherwise a plan or
515
+ * confirmation prompt.
516
+ */
517
+ export async function runBenchmark(
518
+ { budget, top, item_id, dry_run = false, provider, confirm = false,
519
+ publish = true },
520
+ deps = {},
521
+ ) {
522
+ const {
523
+ isMtEvalInstalled: checkInstalled = isMtEvalInstalled,
524
+ fetchQueue = null,
525
+ execCapture: exec = execCapture,
526
+ } = deps;
527
+
528
+ // Pre-flight check: is mt-eval installed?
529
+ const installed = await checkInstalled();
530
+ if (!installed) {
531
+ return [
532
+ 'mt-eval is not installed on this machine.',
533
+ '',
534
+ 'To install it, run:',
535
+ ' pipx install mt-eval',
536
+ '',
537
+ 'After installation, set your API key:',
538
+ ' export OPENROUTER_API_KEY=sk-or-...',
539
+ '',
540
+ 'Then try again.',
541
+ ].join('\n');
542
+ }
543
+
544
+ // ----- Specific item -----------------------------------------------------
545
+ if (item_id) {
546
+ const getQueue = fetchQueue
547
+ ?? (await import('./queue.js')).fetchQueue;
548
+ const queue = await getQueue();
549
+ const item = queue.items.find((it) => it.id === item_id);
550
+ if (!item) {
551
+ return `Queue item "${item_id}" not found. Use list_queue to see available items.`;
552
+ }
553
+
554
+ // Reconstruct a shell-free argv from STRUCTURED fields. The item's
555
+ // network-supplied run_command is NEVER executed.
556
+ let argv;
557
+ try {
558
+ argv = buildRunArgv(item, { provider });
559
+ } catch (err) {
560
+ return `Cannot run "${item_id}": ${err.message}`;
561
+ }
562
+
563
+ if (dry_run) {
564
+ return [
565
+ 'DRY RUN — would execute (no shell):',
566
+ '',
567
+ ` mt-eval ${argv.join(' ')}`,
568
+ '',
569
+ `Language pair: ${item.language_pair.replace('>', ' → ')}`,
570
+ `Target: ${item.target_language}`,
571
+ `Model: ${item.model}`,
572
+ `Condition: ${item.condition}`,
573
+ `Estimated cost: ${fmtCost(item)}`,
574
+ '',
575
+ 'No tokens were spent.',
576
+ ].join('\n');
577
+ }
578
+
579
+ // ENFORCED confirmation gate — spending requires confirm: true.
580
+ if (confirm !== true) {
581
+ return [
582
+ 'CONFIRMATION REQUIRED — this will spend real tokens.',
583
+ '',
584
+ `Item: ${item_id}`,
585
+ `Pair: ${item.language_pair.replace('>', ' → ')}`,
586
+ `Model: ${item.model}`,
587
+ `Est. cost: ${fmtCost(item)} (actual depends on provider pricing)`,
588
+ '',
589
+ // A single-item `mt-eval run` is scored locally and is NOT
590
+ // auto-published under the MCP (publishing is the budget/top queue
591
+ // flow). State it so the agent isn't surprised either way; the
592
+ // publish:false param applies to budget/top runs.
593
+ 'Note: a single-item run is scored locally and is NOT auto-published '
594
+ + 'to the leaderboard — publish it later with `mt-eval publish`, or use '
595
+ + 'budget/top queue mode to auto-publish.',
596
+ '',
597
+ 'Confirm with the user first, then call run_benchmark again with '
598
+ + 'confirm: true to proceed. Or pass dry_run: true to preview without '
599
+ + 'spending.',
600
+ ].join('\n');
601
+ }
602
+
603
+ // Launch in the BACKGROUND and return a job handle immediately. A real run
604
+ // would otherwise blow past the client's 60s request timeout. The agent
605
+ // polls get_run_status with the returned job id.
606
+ const job = launchJob({
607
+ argv,
608
+ mode: 'item',
609
+ label: `${item.language_pair.replace('>', ' → ')} ${item.model} [${item.condition}]`,
610
+ estLabel: fmtCost(item),
611
+ timeout: 600_000, // 10 minute subprocess timeout for a single run
612
+ exec,
613
+ });
614
+ return formatJobStarted(job);
615
+ }
616
+
617
+ // ----- Queue mode (budget / top) -----------------------------------------
618
+ const args = ['queue'];
619
+ if (budget != null) args.push('--budget', String(budget));
620
+ if (top != null) args.push('--top', String(top));
621
+ if (provider) args.push('--provider', provider);
622
+
623
+ // Determinism: estimate_cost / list_queue preview the queue with the
624
+ // deterministic top-order filterQueue (no anti-collision spread). The
625
+ // harness, however, turns spread ON by default for --budget (the mass
626
+ // curl|bash donate flow, where many workers must fan out). Under the MCP a
627
+ // single agent-mediated run just executed a previewed plan, so pass
628
+ // --no-spread to make the EXECUTED selection match that preview item-for-item
629
+ // instead of a spread-permuted set the user never saw. (No-op for --top,
630
+ // which is already deterministic; harmless to pass.)
631
+ args.push('--no-spread');
632
+
633
+ // Prod-write opt-out. Queue runs auto-publish each result to the public
634
+ // leaderboard by default. publish:false threads --no-publish so an agent can
635
+ // run a scoring/validation pass with NO leaderboard write (and the harness
636
+ // then skips OAuth entirely).
637
+ if (publish === false) args.push('--no-publish');
638
+
639
+ if (dry_run) {
640
+ args.push('--dry-run');
641
+ const { code, stdout, stderr } = await exec('mt-eval', args, {
642
+ timeout: 1800_000,
643
+ });
644
+ if (code !== 0) {
645
+ // Never relay the raw harness output here — a Python traceback carries
646
+ // local absolute paths. Surface the final error line (path-stripped)
647
+ // and keep the full detail in the debug log only.
648
+ await writeDebugLog(
649
+ `queue dry-run failed (exit ${code}) — argv: mt-eval ${args.join(' ')}`,
650
+ debugDetail(stdout, stderr),
651
+ );
652
+ return [
653
+ `Queue dry-run failed (exit code ${code}): ${finalErrorLine(stderr, stdout)}`,
654
+ '',
655
+ 'Hint: this error came from the local mt-eval harness, not the queue. '
656
+ + 'Run the same mt-eval command in a terminal to reproduce; the full '
657
+ + `unedited output was saved to the debug log (${DEBUG_LOG_HINT}).`,
658
+ ].join('\n');
659
+ }
660
+ return stdout || 'Queue dry-run completed (no output captured).';
661
+ }
662
+
663
+ // ENFORCED bound on scope. Without a selector, argv is
664
+ // `mt-eval queue --no-spread --yes` — the WHOLE queue (thousands of items),
665
+ // with no dollar figure anywhere in the confirmation text to warn the agent
666
+ // or the user. A run must always name how much it may spend (--budget) or
667
+ // how many items it may take (--top). dry_run is exempt: previewing the full
668
+ // queue plan costs nothing.
669
+ if (budget == null && top == null) {
670
+ return [
671
+ 'REFUSED — an unbounded queue run is not allowed.',
672
+ '',
673
+ 'Calling run_benchmark with no budget, no top and no item_id would run '
674
+ + 'the ENTIRE queue (thousands of paid model calls).',
675
+ '',
676
+ 'Pass exactly one bound:',
677
+ ' • budget: 5 — spend at most $5.00, harness picks the items',
678
+ ' • top: 10 — run the top 10 queue items',
679
+ ' • item_id: "<id>" — run one specific item',
680
+ '',
681
+ 'Use estimate_cost first to see what a given bound would cost, or '
682
+ + 'dry_run: true to preview the full queue plan without spending.',
683
+ ].join('\n');
684
+ }
685
+
686
+ // ENFORCED confirmation gate for spend.
687
+ if (confirm !== true) {
688
+ const scope = budget != null
689
+ ? `up to $${Number(budget).toFixed(2)} of estimated spend`
690
+ : `the top ${top} item(s)`;
691
+ const publishLine = publish === false
692
+ ? 'Results will NOT be published (scoring/validation only — no leaderboard write).'
693
+ : 'Each result will be published to the public leaderboard.';
694
+ return [
695
+ 'CONFIRMATION REQUIRED — this will spend real tokens.',
696
+ '',
697
+ `This would run ${scope} from the top of the queue.`,
698
+ publishLine,
699
+ '',
700
+ 'Confirm with the user first, then call run_benchmark again with '
701
+ + 'confirm: true to proceed. Or pass dry_run: true to preview the plan '
702
+ + 'without spending.',
703
+ ].join('\n');
704
+ }
705
+
706
+ // Confirmation happened here, at the MCP boundary. Pass --yes so mt-eval
707
+ // does not block on its own interactive prompt — there is no TTY under MCP
708
+ // stdio, so that prompt would hang (or, with an inherited stdin, consume the
709
+ // JSON-RPC stream).
710
+ args.push('--yes');
711
+
712
+ // Launch in the BACKGROUND and return a job handle immediately — a queue run
713
+ // can span many model calls and minutes, far past the client's 60s request
714
+ // timeout. The agent polls get_run_status with the returned job id.
715
+ const scope = budget != null
716
+ ? `up to $${Number(budget).toFixed(2)} from the top of the queue`
717
+ : `the top ${top} item(s) from the queue`;
718
+ const job = launchJob({
719
+ argv: args,
720
+ mode: 'queue',
721
+ label: scope,
722
+ publish,
723
+ timeout: 1800_000, // 30 minute subprocess timeout for queue runs
724
+ exec,
725
+ });
726
+ return formatJobStarted(job);
727
+ }