champollion-mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,190 @@
1
+ /**
2
+ * Metric-reliability tool — which automatic metric can you TRUST for a
3
+ * target language?
4
+ *
5
+ * Backed by shared/catalogue/metric-reliability.json: champollion-derived
6
+ * correlations between automatic MT metrics (BLEU, spBLEU, chrF, chrF++,
7
+ * COMET, MetricX) and the WMT Metrics-task human judgments (DA/MQM/ESA,
8
+ * wmt19–wmt25), rolled up per TARGET-language family. Methodology spec:
9
+ * https://champollion.dev/docs/network/specifications/metric-reliability
10
+ *
11
+ * Honesty contract (mirrors mt_eval_harness.recommend tier 4):
12
+ * - A language no WMT campaign ever judged returns an explicit
13
+ * "unmeasured" answer listing what IS covered — never borrowed numbers.
14
+ * - Family-level evidence that doesn't include the exact language carries
15
+ * a transfer caveat: within-family transfer is an assumption.
16
+ * - The index rides a non-commercial hold (upstream data license
17
+ * unstated, founder review pending) — every answer says so.
18
+ *
19
+ * The index is monorepo-tracked (deliberately NOT npm-bundled); outside a
20
+ * monorepo checkout the tool degrades to an explicit "not available" answer.
21
+ */
22
+
23
+ import { readFile } from 'node:fs/promises';
24
+ import { resolve } from 'node:path';
25
+ import { fileURLToPath } from 'node:url';
26
+
27
+ const __dirname = fileURLToPath(new URL('.', import.meta.url));
28
+
29
+ /** Candidate paths to the reliability index; first hit wins. */
30
+ const INDEX_PATHS = [
31
+ resolve(__dirname, '../../../shared/catalogue/metric-reliability.json'),
32
+ resolve(__dirname, '../../shared/catalogue/metric-reliability.json'),
33
+ ];
34
+
35
+ let _cache;
36
+
37
+ /** Load (and cache) the reliability index; null when not available. */
38
+ export async function loadReliabilityIndex() {
39
+ if (_cache !== undefined) return _cache;
40
+ for (const p of INDEX_PATHS) {
41
+ try {
42
+ _cache = JSON.parse(await readFile(p, 'utf8'));
43
+ return _cache;
44
+ } catch {
45
+ // try the next candidate
46
+ }
47
+ }
48
+ _cache = null;
49
+ return _cache;
50
+ }
51
+
52
+ /** Test hook: override / clear the cached index. */
53
+ export function _setReliabilityIndexForTests(value) {
54
+ _cache = value;
55
+ }
56
+
57
+ /**
58
+ * Resolve a user query (ISO code, WMT code, or family name) against the
59
+ * index. Returns { kind: 'language'|'family'|'none', code?, info?, family? }.
60
+ */
61
+ export function resolveReliabilityQuery(index, query) {
62
+ const q = String(query || '').trim();
63
+ const languages = index.languages || {};
64
+ for (const [key, entry] of Object.entries(languages)) {
65
+ if (q.toLowerCase() === key.toLowerCase()
66
+ || q.toLowerCase() === String((entry || {}).iso639_3 || '').toLowerCase()) {
67
+ return { kind: 'language', code: key, info: entry, family: (entry || {}).family || 'Unclassified' };
68
+ }
69
+ }
70
+ for (const family of Object.keys(index.families || {})) {
71
+ if (q.toLowerCase() === family.toLowerCase()) {
72
+ return { kind: 'family', family };
73
+ }
74
+ }
75
+ return { kind: 'none' };
76
+ }
77
+
78
+ /**
79
+ * Build the reliability answer for a target language or family query.
80
+ *
81
+ * @param {string} target - ISO 639 code, WMT pair code, or family name.
82
+ * @param {object|null} [index] - Fixture for tests; defaults to the tracked index.
83
+ * @returns {Promise<object>} structured answer (see formatReliability).
84
+ */
85
+ export async function metricReliability(target, index = undefined) {
86
+ const idx = index !== undefined ? index : await loadReliabilityIndex();
87
+ if (!idx) {
88
+ return {
89
+ status: 'index-unavailable',
90
+ note: 'metric-reliability.json is monorepo-tracked and not reachable '
91
+ + 'from this install — no metric-trust evidence available here. '
92
+ + 'Methodology + data: '
93
+ + 'https://champollion.dev/docs/network/specifications/metric-reliability',
94
+ };
95
+ }
96
+ const hit = resolveReliabilityQuery(idx, target);
97
+ if (hit.kind === 'none') {
98
+ return {
99
+ status: 'unmeasured',
100
+ target,
101
+ note: `No WMT human-judgment meta-evaluation covers '${target}' — `
102
+ + 'metric choice for this language is UNMEASURED. Treat every metric '
103
+ + 'as unvalidated there and validate locally where possible.',
104
+ measured_families: Object.keys(idx.families || {}).sort(),
105
+ measured_languages: Object.keys(idx.languages || {}).sort(),
106
+ };
107
+ }
108
+ const family = hit.family;
109
+ const famBlock = (idx.families || {})[family] || {};
110
+ const metrics = [];
111
+ for (const [metricId, levels] of Object.entries(famBlock.metrics || {})) {
112
+ const sysE = levels.sys || {};
113
+ const segE = levels.seg || {};
114
+ metrics.push({
115
+ metric: metricId,
116
+ sys_pearson: sysE.pearson_weighted_mean ?? null,
117
+ sys_pairwise_accuracy: sysE.pairwise_accuracy_weighted_mean ?? null,
118
+ sys_n_pairs: sysE.n_pairs ?? null,
119
+ seg_kendall: segE.kendall_tau_b_weighted_mean ?? null,
120
+ seg_n_pairs: segE.n_pairs ?? null,
121
+ });
122
+ }
123
+ metrics.sort((a, b) =>
124
+ (b.sys_pearson ?? -2) - (a.sys_pearson ?? -2)
125
+ || a.metric.localeCompare(b.metric));
126
+ const exactPairs = hit.kind === 'language'
127
+ ? [...new Set((idx.cells || [])
128
+ .filter((c) => c.tgt === hit.code && c.preferred)
129
+ .map((c) => c.pair))].sort()
130
+ : [];
131
+ return {
132
+ status: 'ok',
133
+ query_kind: hit.kind,
134
+ target,
135
+ target_code: hit.code ?? null,
136
+ target_family: family,
137
+ exact_pairs_measured: exactPairs,
138
+ metrics,
139
+ family_pairs: famBlock.n_pairs ?? null,
140
+ license_note: (idx.license_lane || {}).commercial_ok === false
141
+ ? 'Non-commercial evidence lane: the upstream WMT judgment data '
142
+ + 'license is unstated (founder review pending) — cite in research '
143
+ + 'lanes only.'
144
+ : null,
145
+ provenance: idx.provenance ?? null,
146
+ };
147
+ }
148
+
149
+ /** Human-readable rendering of a metricReliability() answer. */
150
+ export function formatReliability(r) {
151
+ if (r.status === 'index-unavailable') return r.note;
152
+ if (r.status === 'unmeasured') {
153
+ return [
154
+ r.note,
155
+ '',
156
+ `Families with WMT human-judgment evidence: ${r.measured_families.join(', ')}.`,
157
+ `Directly judged target languages (WMT codes): ${r.measured_languages.join(', ')}.`,
158
+ ].join('\n');
159
+ }
160
+ const fmt = (v) => (v === null || v === undefined
161
+ ? ' — '
162
+ : `${v >= 0 ? '+' : ''}${v.toFixed(2)}`);
163
+ const out = [];
164
+ const scope = r.query_kind === 'family'
165
+ ? `family '${r.target_family}'`
166
+ : `'${r.target}' (family: ${r.target_family})`;
167
+ out.push(`Metric trust for ${scope} — correlation with WMT human judgment `
168
+ + '(higher = the metric agrees with human raters more; sys = ranking '
169
+ + 'whole systems, seg = scoring individual sentences):');
170
+ if (r.metrics.length === 0) {
171
+ out.push(' (no metric cells for this family — evidence gap)');
172
+ }
173
+ for (const m of r.metrics) {
174
+ const n = m.sys_n_pairs ?? m.seg_n_pairs ?? 0;
175
+ out.push(` ${m.metric.padEnd(18)} sys-Pearson ${fmt(m.sys_pearson)} `
176
+ + `seg-Kendall ${fmt(m.seg_kendall)} (${n} language pair(s))`);
177
+ }
178
+ if (r.query_kind === 'language') {
179
+ out.push(r.exact_pairs_measured.length > 0
180
+ ? ` Directly judged pairs for this language: ${r.exact_pairs_measured.join(', ')}.`
181
+ : ' ⚠ No WMT campaign judged this exact language — the numbers above '
182
+ + 'come from other languages in the same family; within-family '
183
+ + 'transfer is an assumption, not a measurement.');
184
+ }
185
+ out.push('');
186
+ out.push('Methodology (definitions, inclusions/exclusions, reproduction): '
187
+ + 'https://champollion.dev/docs/network/specifications/metric-reliability');
188
+ if (r.license_note) out.push(`⚠ ${r.license_note}`);
189
+ return out.join('\n');
190
+ }
@@ -0,0 +1,346 @@
1
+ /**
2
+ * Results tools — read the public Champollion leaderboard (scored run_cards).
3
+ *
4
+ * This is the read side that closes the contribute → run → see-impact loop:
5
+ * an agent can run_benchmark (which spends real API credits) and then query
6
+ * what it scored, or browse what the community has already benchmarked.
7
+ *
8
+ * Public read path: the SAME Supabase project + anon key the public
9
+ * leaderboard embeds (cli/website/src/pages/leaderboard.js). The anon key is
10
+ * public and RLS makes run_cards read-only — there is no secret here. Override
11
+ * with CHAMPOLLION_SUPABASE_URL / CHAMPOLLION_SUPABASE_ANON_KEY to point at a
12
+ * dev/staging branch (CHAMPOLLION_-prefixed so an unrelated SUPABASE_URL in
13
+ * the user's environment can't accidentally repoint the MCP server).
14
+ *
15
+ * Sovereignty note: these tools expose only the scored AGGREGATE columns
16
+ * (composite, chrF++, BLEU, COMET, cost, trust) plus the run_card metadata
17
+ * card — the same fields the public leaderboard already serves. Per-entry
18
+ * sentence text lives in the separately license/quarantine-gated
19
+ * run_card_entries table and is never read here.
20
+ */
21
+
22
+ const SUPABASE_URL =
23
+ process.env.CHAMPOLLION_SUPABASE_URL ||
24
+ 'https://sjdomynysdljkbemupqa.supabase.co';
25
+ const SUPABASE_ANON_KEY =
26
+ process.env.CHAMPOLLION_SUPABASE_ANON_KEY ||
27
+ 'sb_publishable_bV6CFNFnzxhQI0wlBx2J0A_5Vm5gFBp';
28
+
29
+ const SB_HEADERS = {
30
+ apikey: SUPABASE_ANON_KEY,
31
+ Authorization: `Bearer ${SUPABASE_ANON_KEY}`,
32
+ Accept: 'application/json',
33
+ };
34
+
35
+ /**
36
+ * Scored aggregate columns only — mirrors leaderboard.js LISTING_SELECT,
37
+ * trimmed to what an agent needs. NO per-entry sentence text.
38
+ *
39
+ * `contamination:run_card->>contamination` extracts the grade publish.py stamps
40
+ * into the run_card JSONB (there is no top-level column). Without it an agent
41
+ * gets no contamination signal at all and could co-rank a relative-only
42
+ * (HIGH/FLORES) corpus against absolute-quality corpora as if they were equal.
43
+ */
44
+ export const RESULTS_SELECT = [
45
+ 'id', 'submitter', 'model_slug', 'condition', 'dataset_id',
46
+ 'language_pair', 'composite_score', 'quality_tier', 'trust',
47
+ 'chrf_plus_plus', 'corpus_bleu', 'comet_score', 'ter',
48
+ 'total_cost_usd', 'cost_per_entry_usd', 'run_timestamp', 'submitted_at',
49
+ 'contamination:run_card->>contamination',
50
+ ].join(',');
51
+
52
+ // ---------------------------------------------------------------------------
53
+ // Contamination lane — FAIL SAFE. Self-contained mirror of the website's
54
+ // cli/website/src/utils/contaminationBadge.js (mcp-server is a separate npm
55
+ // package, so it can't import it) and the SSOT lane policy in
56
+ // arena/mt_eval_harness/contamination.py.
57
+ //
58
+ // Policy: a corpus is RELATIVE-COMPARISON-ONLY (its scores rank methods against
59
+ // each other on THAT corpus, never as absolute quality) when its contamination
60
+ // grade is HIGH or MEDIUM, OR when the grade is unknown/unset. Only a positively
61
+ // LOW grade is rankable on absolute quality. Defaulting unknown to the absolute
62
+ // lane is the "fails open" bug this guards against: a prod-missing or ungraded
63
+ // HIGH corpus must never be co-ranked as trustworthy absolute quality.
64
+ // ---------------------------------------------------------------------------
65
+
66
+ export const LANE_ABSOLUTE = 'absolute-quality';
67
+ export const LANE_RELATIVE_ONLY = 'relative-comparison-only';
68
+
69
+ /** Grades that force a corpus into the relative-only lane. Mirrors
70
+ * contamination.RELATIVE_ONLY_GRADES (HIGH + MEDIUM). */
71
+ const RELATIVE_ONLY_GRADES = new Set(['HIGH', 'MEDIUM']);
72
+
73
+ /** Upper-case a contamination grade; map empty / "NONE" to null (unknown). */
74
+ export function normalizeContamination(grade) {
75
+ if (grade == null) return null;
76
+ const g = String(grade).trim().toUpperCase();
77
+ if (!g || g === 'NONE') return null;
78
+ return g;
79
+ }
80
+
81
+ /**
82
+ * Lane gate — FAIL SAFE. True ⇒ relative-comparison-only (must NOT be co-ranked
83
+ * with absolute-quality corpora). Known HIGH/MEDIUM or UNKNOWN → true; only a
84
+ * positively LOW grade → false. Mirrors isRelativeOnlyLane in the website util.
85
+ *
86
+ * @param {string|null|undefined} grade run_card->>contamination for the row
87
+ * @returns {boolean}
88
+ */
89
+ export function isRelativeOnlyLane(grade) {
90
+ const g = normalizeContamination(grade);
91
+ if (g == null) return true; // fail safe: unknown contamination → relative-only
92
+ return RELATIVE_ONLY_GRADES.has(g);
93
+ }
94
+
95
+ /**
96
+ * Sort key → (PostgREST column, direction). Higher-is-better metrics sort
97
+ * descending; TER, cost, and "lowest" sort ascending; date sorts newest-first.
98
+ */
99
+ export const RESULTS_SORT = {
100
+ composite: { column: 'composite_score', dir: 'desc' },
101
+ chrf: { column: 'chrf_plus_plus', dir: 'desc' },
102
+ bleu: { column: 'corpus_bleu', dir: 'desc' },
103
+ comet: { column: 'comet_score', dir: 'desc' },
104
+ ter: { column: 'ter', dir: 'asc' },
105
+ cost: { column: 'cost_per_entry_usd', dir: 'asc' },
106
+ date: { column: 'run_timestamp', dir: 'desc' },
107
+ };
108
+
109
+ /**
110
+ * Trust vocabulary — maps the DB trust enum to display keys. Mirrors
111
+ * DB_TRUST_TO_DISPLAY in cli/website (kept in sync; the website is a separate
112
+ * package so we can't import it). Every CLI submission is 'unverified'
113
+ * (self-reported); 'verified' is set only by the server-side re-scoring
114
+ * verifier; 'disqualified' is filtered from ranked views.
115
+ */
116
+ export const TRUST_DISPLAY = {
117
+ unverified: 'self-benchmarked',
118
+ verified: 'champollion-verified',
119
+ disqualified: 'disqualified',
120
+ };
121
+
122
+ /**
123
+ * Strip characters significant to PostgREST filter syntax so a user-supplied
124
+ * value can't inject extra filters. Returns null for empty/blank input.
125
+ *
126
+ * @param {string|null|undefined} s
127
+ * @returns {string|null}
128
+ */
129
+ export function sanitize(s) {
130
+ if (s == null) return null;
131
+ const cleaned = String(s).replace(/[%,()*]/g, '').trim();
132
+ return cleaned.length ? cleaned : null;
133
+ }
134
+
135
+ /**
136
+ * Build the PostgREST URL for a leaderboard listing query.
137
+ *
138
+ * Mirrors buildListingUrl in cli/website/src/utils/leaderboardUtils.js:
139
+ * server-side filtering + `trust=neq.disqualified` so disqualified runs never
140
+ * surface, ordered by the chosen metric with nulls last.
141
+ *
142
+ * @param {object} params
143
+ * @param {string} [params.supabaseUrl] Override Supabase project URL.
144
+ * @param {string} [params.select] Comma-separated column list.
145
+ * @param {string} [params.source_language] Source ISO 639-3 code (or null).
146
+ * @param {string} [params.target_language] Target ISO 639-3 code (or null).
147
+ * @param {string} [params.model] Model slug substring (ilike).
148
+ * @param {string} [params.sort] Sort key (see RESULTS_SORT).
149
+ * @param {number} [params.limit] Max rows (default 20).
150
+ * @returns {string} Full PostgREST URL.
151
+ */
152
+ export function buildResultsUrl({
153
+ supabaseUrl = SUPABASE_URL,
154
+ select = RESULTS_SELECT,
155
+ source_language = null,
156
+ target_language = null,
157
+ model = null,
158
+ sort = 'composite',
159
+ limit = 20,
160
+ } = {}) {
161
+ const params = new URLSearchParams();
162
+ params.set('select', select);
163
+ params.set('trust', 'neq.disqualified');
164
+
165
+ // Language pair — Supabase stores "src>tgt" lowercase ISO 639-3.
166
+ const src = sanitize(source_language)?.toLowerCase();
167
+ const tgt = sanitize(target_language)?.toLowerCase();
168
+ if (src && tgt) {
169
+ params.set('language_pair', `eq.${src}>${tgt}`);
170
+ } else if (src) {
171
+ params.set('language_pair', `like.${src}>%`);
172
+ } else if (tgt) {
173
+ params.set('language_pair', `like.%>${tgt}`);
174
+ }
175
+
176
+ // Model — case-insensitive substring match (agents pass "haiku", not the
177
+ // full slug). ilike treats * as the wildcard.
178
+ const m = sanitize(model);
179
+ if (m) params.set('model_slug', `ilike.*${m}*`);
180
+
181
+ const { column, dir } = RESULTS_SORT[sort] || RESULTS_SORT.composite;
182
+ params.set('order', `${column}.${dir}.nullslast`);
183
+ params.set('limit', String(limit));
184
+
185
+ return `${supabaseUrl}/rest/v1/run_cards?${params.toString()}`;
186
+ }
187
+
188
+ /**
189
+ * Map a raw run_cards row to the compact scored shape the tool returns.
190
+ *
191
+ * @param {object} row PostgREST row
192
+ * @returns {object} Compact result entry
193
+ */
194
+ export function mapResultRow(row) {
195
+ const pair = (row.language_pair || '?').trim().toLowerCase();
196
+ const contamination = normalizeContamination(row.contamination);
197
+ const relativeOnly = isRelativeOnlyLane(row.contamination);
198
+ return {
199
+ id: row.id,
200
+ model: row.model_slug || '?',
201
+ pair,
202
+ condition: row.condition || '?',
203
+ composite: row.composite_score ?? null,
204
+ chrf: row.chrf_plus_plus ?? null,
205
+ bleu: row.corpus_bleu ?? null,
206
+ comet: row.comet_score ?? null,
207
+ ter: row.ter ?? null,
208
+ qualityTier: row.quality_tier ?? null,
209
+ trust: TRUST_DISPLAY[row.trust] || 'self-benchmarked',
210
+ author: row.submitter || 'anonymous',
211
+ dataset: row.dataset_id || null,
212
+ cost_usd: row.total_cost_usd ?? null,
213
+ date: (row.run_timestamp || row.submitted_at || '').slice(0, 10) || null,
214
+ // Contamination lane (FAIL SAFE). `relative_only` rows are NOT comparable
215
+ // on absolute quality against absolute-lane rows — only against each other
216
+ // on the same corpus. An agent that ignores this would co-rank a corpus
217
+ // that's in models' training data (or whose grade is unknown) as if it
218
+ // measured real quality.
219
+ contamination: contamination, // normalized grade, or null when unknown
220
+ relative_only: relativeOnly,
221
+ score_lane: relativeOnly ? LANE_RELATIVE_ONLY : LANE_ABSOLUTE,
222
+ };
223
+ }
224
+
225
+ /**
226
+ * Format mapped result rows as a concise, agent-readable block.
227
+ *
228
+ * @param {object[]} rows Mapped rows (from mapResultRow)
229
+ * @param {object} [opts]
230
+ * @param {string} [opts.sort] Sort key used (for the header line)
231
+ * @returns {string}
232
+ */
233
+ export function formatResults(rows, { sort = 'composite' } = {}) {
234
+ if (!rows || rows.length === 0) {
235
+ return [
236
+ 'No scored results on the public leaderboard yet.',
237
+ '',
238
+ 'The board is filled by people running benchmarks from the queue.',
239
+ 'Use list_queue to see what is pending, then run_benchmark to score an',
240
+ 'item — your run publishes here with your attribution.',
241
+ '',
242
+ 'Leaderboard: https://champollion.dev/leaderboard',
243
+ ].join('\n');
244
+ }
245
+
246
+ const fmt = (v, d = 3) => (v == null ? '—' : Number(v).toFixed(d));
247
+ const lines = rows.map((r, i) => {
248
+ const arrow = r.pair.includes('>') ? r.pair.replace('>', '→') : r.pair;
249
+ // Mark relative-only rows so the score is never read as absolute quality.
250
+ const lane = r.relative_only
251
+ ? ` ⚠ relative-only${r.contamination ? ` (${r.contamination})` : ' (grade unknown)'}`
252
+ : '';
253
+ // The id closes the loop with get_run_card, whose not-found error
254
+ // points agents back here for valid ids.
255
+ const idTag = r.id != null ? ` id ${r.id}` : '';
256
+ return (
257
+ `#${i + 1} ${arrow} composite ${fmt(r.composite)} chrF++ ${fmt(r.chrf, 1)} `
258
+ + `${r.model.split('/').pop()} [${r.condition}] ${r.trust} by ${r.author}${idTag}${lane}`
259
+ );
260
+ });
261
+
262
+ // Lane-mixing guard: relative-only and absolute-quality scores are NOT
263
+ // comparable across lanes — only within the same corpus. Warn loudly when a
264
+ // single ranked list spans both lanes, or when every row is relative-only.
265
+ const relCount = rows.filter((r) => r.relative_only).length;
266
+ const absCount = rows.length - relCount;
267
+ const laneNotes = [];
268
+ if (relCount > 0 && absCount > 0) {
269
+ laneNotes.push(
270
+ '',
271
+ `⚠ This list mixes ${absCount} absolute-quality row(s) and ${relCount} `
272
+ + 'relative-comparison-only row(s) (marked above). Do NOT rank them against '
273
+ + 'each other: a relative-only corpus (HIGH/MEDIUM contamination, FLORES, or '
274
+ + 'an unknown grade) is in models\' training data — its score is valid only '
275
+ + 'for comparing methods on THAT corpus, not as absolute quality.',
276
+ );
277
+ } else if (relCount > 0 && absCount === 0) {
278
+ laneNotes.push(
279
+ '',
280
+ '⚠ Every row above is relative-comparison-only (HIGH/MEDIUM contamination, '
281
+ + 'FLORES, or unknown grade). These scores compare methods on the same '
282
+ + 'corpus — they are NOT absolute quality and must not be ranked against '
283
+ + 'other corpora.',
284
+ );
285
+ }
286
+
287
+ return [
288
+ `Top ${rows.length} scored result(s) by ${sort}:`,
289
+ '',
290
+ ...lines,
291
+ ...laneNotes,
292
+ '',
293
+ 'Full leaderboard: https://champollion.dev/leaderboard',
294
+ ].join('\n');
295
+ }
296
+
297
+ /**
298
+ * Fetch scored leaderboard results from the public Supabase endpoint.
299
+ *
300
+ * @param {object} [params] Same shape as buildResultsUrl params.
301
+ * @returns {Promise<object[]>} Mapped result rows.
302
+ */
303
+ export async function fetchResults(params = {}) {
304
+ const url = buildResultsUrl(params);
305
+ const resp = await fetch(url, {
306
+ headers: SB_HEADERS,
307
+ signal: AbortSignal.timeout(30_000),
308
+ });
309
+ if (!resp.ok) {
310
+ throw new Error(
311
+ `Leaderboard fetch failed: HTTP ${resp.status} ${(await resp.text()).slice(0, 120)}`,
312
+ );
313
+ }
314
+ const rows = await resp.json();
315
+ return rows.map(mapResultRow);
316
+ }
317
+
318
+ /**
319
+ * Fetch the full run card (scores + method/config metadata) for one run id.
320
+ * Returns the same `run_card` JSON the public leaderboard serves on expand;
321
+ * no sentence text is read. Returns null if no run matches.
322
+ *
323
+ * @param {string} id run_cards.id
324
+ * @param {object} [opts]
325
+ * @param {string} [opts.supabaseUrl] Override Supabase project URL.
326
+ * @returns {Promise<object|null>}
327
+ */
328
+ export async function fetchRunCard(id, { supabaseUrl = SUPABASE_URL } = {}) {
329
+ const safe = sanitize(id);
330
+ if (!safe) return null;
331
+ const select =
332
+ 'id,model_slug,language_pair,condition,trust,submitter,' +
333
+ 'composite_score,chrf_plus_plus,dataset_id,run_timestamp,run_card';
334
+ const url =
335
+ `${supabaseUrl}/rest/v1/run_cards?select=${select}` +
336
+ `&trust=neq.disqualified&id=eq.${encodeURIComponent(safe)}&limit=1`;
337
+ const resp = await fetch(url, {
338
+ headers: SB_HEADERS,
339
+ signal: AbortSignal.timeout(30_000),
340
+ });
341
+ if (!resp.ok) {
342
+ throw new Error(`Run-card fetch failed: HTTP ${resp.status}`);
343
+ }
344
+ const rows = await resp.json();
345
+ return rows[0] || null;
346
+ }