champollion-mcp-server 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +245 -0
- package/bin/server.js +19 -0
- package/instructions.md +234 -0
- package/package.json +50 -0
- package/src/index.js +1106 -0
- package/src/tools/forge.js +140 -0
- package/src/tools/harness.js +727 -0
- package/src/tools/languages.js +329 -0
- package/src/tools/queue.js +313 -0
- package/src/tools/reliability.js +190 -0
- package/src/tools/results.js +346 -0
- package/src/tools/training.js +349 -0
- package/src/tools/translate.js +385 -0
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Metric-reliability tool — which automatic metric can you TRUST for a
|
|
3
|
+
* target language?
|
|
4
|
+
*
|
|
5
|
+
* Backed by shared/catalogue/metric-reliability.json: champollion-derived
|
|
6
|
+
* correlations between automatic MT metrics (BLEU, spBLEU, chrF, chrF++,
|
|
7
|
+
* COMET, MetricX) and the WMT Metrics-task human judgments (DA/MQM/ESA,
|
|
8
|
+
* wmt19–wmt25), rolled up per TARGET-language family. Methodology spec:
|
|
9
|
+
* https://champollion.dev/docs/network/specifications/metric-reliability
|
|
10
|
+
*
|
|
11
|
+
* Honesty contract (mirrors mt_eval_harness.recommend tier 4):
|
|
12
|
+
* - A language no WMT campaign ever judged returns an explicit
|
|
13
|
+
* "unmeasured" answer listing what IS covered — never borrowed numbers.
|
|
14
|
+
* - Family-level evidence that doesn't include the exact language carries
|
|
15
|
+
* a transfer caveat: within-family transfer is an assumption.
|
|
16
|
+
* - The index rides a non-commercial hold (upstream data license
|
|
17
|
+
* unstated, founder review pending) — every answer says so.
|
|
18
|
+
*
|
|
19
|
+
* The index is monorepo-tracked (deliberately NOT npm-bundled); outside a
|
|
20
|
+
* monorepo checkout the tool degrades to an explicit "not available" answer.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { readFile } from 'node:fs/promises';
|
|
24
|
+
import { resolve } from 'node:path';
|
|
25
|
+
import { fileURLToPath } from 'node:url';
|
|
26
|
+
|
|
27
|
+
const __dirname = fileURLToPath(new URL('.', import.meta.url));
|
|
28
|
+
|
|
29
|
+
/** Candidate paths to the reliability index; first hit wins. */
|
|
30
|
+
const INDEX_PATHS = [
|
|
31
|
+
resolve(__dirname, '../../../shared/catalogue/metric-reliability.json'),
|
|
32
|
+
resolve(__dirname, '../../shared/catalogue/metric-reliability.json'),
|
|
33
|
+
];
|
|
34
|
+
|
|
35
|
+
let _cache;
|
|
36
|
+
|
|
37
|
+
/** Load (and cache) the reliability index; null when not available. */
|
|
38
|
+
export async function loadReliabilityIndex() {
|
|
39
|
+
if (_cache !== undefined) return _cache;
|
|
40
|
+
for (const p of INDEX_PATHS) {
|
|
41
|
+
try {
|
|
42
|
+
_cache = JSON.parse(await readFile(p, 'utf8'));
|
|
43
|
+
return _cache;
|
|
44
|
+
} catch {
|
|
45
|
+
// try the next candidate
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
_cache = null;
|
|
49
|
+
return _cache;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Test hook: override / clear the cached index. */
|
|
53
|
+
export function _setReliabilityIndexForTests(value) {
|
|
54
|
+
_cache = value;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Resolve a user query (ISO code, WMT code, or family name) against the
|
|
59
|
+
* index. Returns { kind: 'language'|'family'|'none', code?, info?, family? }.
|
|
60
|
+
*/
|
|
61
|
+
export function resolveReliabilityQuery(index, query) {
|
|
62
|
+
const q = String(query || '').trim();
|
|
63
|
+
const languages = index.languages || {};
|
|
64
|
+
for (const [key, entry] of Object.entries(languages)) {
|
|
65
|
+
if (q.toLowerCase() === key.toLowerCase()
|
|
66
|
+
|| q.toLowerCase() === String((entry || {}).iso639_3 || '').toLowerCase()) {
|
|
67
|
+
return { kind: 'language', code: key, info: entry, family: (entry || {}).family || 'Unclassified' };
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
for (const family of Object.keys(index.families || {})) {
|
|
71
|
+
if (q.toLowerCase() === family.toLowerCase()) {
|
|
72
|
+
return { kind: 'family', family };
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return { kind: 'none' };
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Build the reliability answer for a target language or family query.
|
|
80
|
+
*
|
|
81
|
+
* @param {string} target - ISO 639 code, WMT pair code, or family name.
|
|
82
|
+
* @param {object|null} [index] - Fixture for tests; defaults to the tracked index.
|
|
83
|
+
* @returns {Promise<object>} structured answer (see formatReliability).
|
|
84
|
+
*/
|
|
85
|
+
export async function metricReliability(target, index = undefined) {
|
|
86
|
+
const idx = index !== undefined ? index : await loadReliabilityIndex();
|
|
87
|
+
if (!idx) {
|
|
88
|
+
return {
|
|
89
|
+
status: 'index-unavailable',
|
|
90
|
+
note: 'metric-reliability.json is monorepo-tracked and not reachable '
|
|
91
|
+
+ 'from this install — no metric-trust evidence available here. '
|
|
92
|
+
+ 'Methodology + data: '
|
|
93
|
+
+ 'https://champollion.dev/docs/network/specifications/metric-reliability',
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
const hit = resolveReliabilityQuery(idx, target);
|
|
97
|
+
if (hit.kind === 'none') {
|
|
98
|
+
return {
|
|
99
|
+
status: 'unmeasured',
|
|
100
|
+
target,
|
|
101
|
+
note: `No WMT human-judgment meta-evaluation covers '${target}' — `
|
|
102
|
+
+ 'metric choice for this language is UNMEASURED. Treat every metric '
|
|
103
|
+
+ 'as unvalidated there and validate locally where possible.',
|
|
104
|
+
measured_families: Object.keys(idx.families || {}).sort(),
|
|
105
|
+
measured_languages: Object.keys(idx.languages || {}).sort(),
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
const family = hit.family;
|
|
109
|
+
const famBlock = (idx.families || {})[family] || {};
|
|
110
|
+
const metrics = [];
|
|
111
|
+
for (const [metricId, levels] of Object.entries(famBlock.metrics || {})) {
|
|
112
|
+
const sysE = levels.sys || {};
|
|
113
|
+
const segE = levels.seg || {};
|
|
114
|
+
metrics.push({
|
|
115
|
+
metric: metricId,
|
|
116
|
+
sys_pearson: sysE.pearson_weighted_mean ?? null,
|
|
117
|
+
sys_pairwise_accuracy: sysE.pairwise_accuracy_weighted_mean ?? null,
|
|
118
|
+
sys_n_pairs: sysE.n_pairs ?? null,
|
|
119
|
+
seg_kendall: segE.kendall_tau_b_weighted_mean ?? null,
|
|
120
|
+
seg_n_pairs: segE.n_pairs ?? null,
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
metrics.sort((a, b) =>
|
|
124
|
+
(b.sys_pearson ?? -2) - (a.sys_pearson ?? -2)
|
|
125
|
+
|| a.metric.localeCompare(b.metric));
|
|
126
|
+
const exactPairs = hit.kind === 'language'
|
|
127
|
+
? [...new Set((idx.cells || [])
|
|
128
|
+
.filter((c) => c.tgt === hit.code && c.preferred)
|
|
129
|
+
.map((c) => c.pair))].sort()
|
|
130
|
+
: [];
|
|
131
|
+
return {
|
|
132
|
+
status: 'ok',
|
|
133
|
+
query_kind: hit.kind,
|
|
134
|
+
target,
|
|
135
|
+
target_code: hit.code ?? null,
|
|
136
|
+
target_family: family,
|
|
137
|
+
exact_pairs_measured: exactPairs,
|
|
138
|
+
metrics,
|
|
139
|
+
family_pairs: famBlock.n_pairs ?? null,
|
|
140
|
+
license_note: (idx.license_lane || {}).commercial_ok === false
|
|
141
|
+
? 'Non-commercial evidence lane: the upstream WMT judgment data '
|
|
142
|
+
+ 'license is unstated (founder review pending) — cite in research '
|
|
143
|
+
+ 'lanes only.'
|
|
144
|
+
: null,
|
|
145
|
+
provenance: idx.provenance ?? null,
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/** Human-readable rendering of a metricReliability() answer. */
|
|
150
|
+
export function formatReliability(r) {
|
|
151
|
+
if (r.status === 'index-unavailable') return r.note;
|
|
152
|
+
if (r.status === 'unmeasured') {
|
|
153
|
+
return [
|
|
154
|
+
r.note,
|
|
155
|
+
'',
|
|
156
|
+
`Families with WMT human-judgment evidence: ${r.measured_families.join(', ')}.`,
|
|
157
|
+
`Directly judged target languages (WMT codes): ${r.measured_languages.join(', ')}.`,
|
|
158
|
+
].join('\n');
|
|
159
|
+
}
|
|
160
|
+
const fmt = (v) => (v === null || v === undefined
|
|
161
|
+
? ' — '
|
|
162
|
+
: `${v >= 0 ? '+' : ''}${v.toFixed(2)}`);
|
|
163
|
+
const out = [];
|
|
164
|
+
const scope = r.query_kind === 'family'
|
|
165
|
+
? `family '${r.target_family}'`
|
|
166
|
+
: `'${r.target}' (family: ${r.target_family})`;
|
|
167
|
+
out.push(`Metric trust for ${scope} — correlation with WMT human judgment `
|
|
168
|
+
+ '(higher = the metric agrees with human raters more; sys = ranking '
|
|
169
|
+
+ 'whole systems, seg = scoring individual sentences):');
|
|
170
|
+
if (r.metrics.length === 0) {
|
|
171
|
+
out.push(' (no metric cells for this family — evidence gap)');
|
|
172
|
+
}
|
|
173
|
+
for (const m of r.metrics) {
|
|
174
|
+
const n = m.sys_n_pairs ?? m.seg_n_pairs ?? 0;
|
|
175
|
+
out.push(` ${m.metric.padEnd(18)} sys-Pearson ${fmt(m.sys_pearson)} `
|
|
176
|
+
+ `seg-Kendall ${fmt(m.seg_kendall)} (${n} language pair(s))`);
|
|
177
|
+
}
|
|
178
|
+
if (r.query_kind === 'language') {
|
|
179
|
+
out.push(r.exact_pairs_measured.length > 0
|
|
180
|
+
? ` Directly judged pairs for this language: ${r.exact_pairs_measured.join(', ')}.`
|
|
181
|
+
: ' ⚠ No WMT campaign judged this exact language — the numbers above '
|
|
182
|
+
+ 'come from other languages in the same family; within-family '
|
|
183
|
+
+ 'transfer is an assumption, not a measurement.');
|
|
184
|
+
}
|
|
185
|
+
out.push('');
|
|
186
|
+
out.push('Methodology (definitions, inclusions/exclusions, reproduction): '
|
|
187
|
+
+ 'https://champollion.dev/docs/network/specifications/metric-reliability');
|
|
188
|
+
if (r.license_note) out.push(`⚠ ${r.license_note}`);
|
|
189
|
+
return out.join('\n');
|
|
190
|
+
}
|
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Results tools — read the public Champollion leaderboard (scored run_cards).
|
|
3
|
+
*
|
|
4
|
+
* This is the read side that closes the contribute → run → see-impact loop:
|
|
5
|
+
* an agent can run_benchmark (which spends real API credits) and then query
|
|
6
|
+
* what it scored, or browse what the community has already benchmarked.
|
|
7
|
+
*
|
|
8
|
+
* Public read path: the SAME Supabase project + anon key the public
|
|
9
|
+
* leaderboard embeds (cli/website/src/pages/leaderboard.js). The anon key is
|
|
10
|
+
* public and RLS makes run_cards read-only — there is no secret here. Override
|
|
11
|
+
* with CHAMPOLLION_SUPABASE_URL / CHAMPOLLION_SUPABASE_ANON_KEY to point at a
|
|
12
|
+
* dev/staging branch (CHAMPOLLION_-prefixed so an unrelated SUPABASE_URL in
|
|
13
|
+
* the user's environment can't accidentally repoint the MCP server).
|
|
14
|
+
*
|
|
15
|
+
* Sovereignty note: these tools expose only the scored AGGREGATE columns
|
|
16
|
+
* (composite, chrF++, BLEU, COMET, cost, trust) plus the run_card metadata
|
|
17
|
+
* card — the same fields the public leaderboard already serves. Per-entry
|
|
18
|
+
* sentence text lives in the separately license/quarantine-gated
|
|
19
|
+
* run_card_entries table and is never read here.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
const SUPABASE_URL =
|
|
23
|
+
process.env.CHAMPOLLION_SUPABASE_URL ||
|
|
24
|
+
'https://sjdomynysdljkbemupqa.supabase.co';
|
|
25
|
+
const SUPABASE_ANON_KEY =
|
|
26
|
+
process.env.CHAMPOLLION_SUPABASE_ANON_KEY ||
|
|
27
|
+
'sb_publishable_bV6CFNFnzxhQI0wlBx2J0A_5Vm5gFBp';
|
|
28
|
+
|
|
29
|
+
const SB_HEADERS = {
|
|
30
|
+
apikey: SUPABASE_ANON_KEY,
|
|
31
|
+
Authorization: `Bearer ${SUPABASE_ANON_KEY}`,
|
|
32
|
+
Accept: 'application/json',
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Scored aggregate columns only — mirrors leaderboard.js LISTING_SELECT,
|
|
37
|
+
* trimmed to what an agent needs. NO per-entry sentence text.
|
|
38
|
+
*
|
|
39
|
+
* `contamination:run_card->>contamination` extracts the grade publish.py stamps
|
|
40
|
+
* into the run_card JSONB (there is no top-level column). Without it an agent
|
|
41
|
+
* gets no contamination signal at all and could co-rank a relative-only
|
|
42
|
+
* (HIGH/FLORES) corpus against absolute-quality corpora as if they were equal.
|
|
43
|
+
*/
|
|
44
|
+
export const RESULTS_SELECT = [
|
|
45
|
+
'id', 'submitter', 'model_slug', 'condition', 'dataset_id',
|
|
46
|
+
'language_pair', 'composite_score', 'quality_tier', 'trust',
|
|
47
|
+
'chrf_plus_plus', 'corpus_bleu', 'comet_score', 'ter',
|
|
48
|
+
'total_cost_usd', 'cost_per_entry_usd', 'run_timestamp', 'submitted_at',
|
|
49
|
+
'contamination:run_card->>contamination',
|
|
50
|
+
].join(',');
|
|
51
|
+
|
|
52
|
+
// ---------------------------------------------------------------------------
|
|
53
|
+
// Contamination lane — FAIL SAFE. Self-contained mirror of the website's
|
|
54
|
+
// cli/website/src/utils/contaminationBadge.js (mcp-server is a separate npm
|
|
55
|
+
// package, so it can't import it) and the SSOT lane policy in
|
|
56
|
+
// arena/mt_eval_harness/contamination.py.
|
|
57
|
+
//
|
|
58
|
+
// Policy: a corpus is RELATIVE-COMPARISON-ONLY (its scores rank methods against
|
|
59
|
+
// each other on THAT corpus, never as absolute quality) when its contamination
|
|
60
|
+
// grade is HIGH or MEDIUM, OR when the grade is unknown/unset. Only a positively
|
|
61
|
+
// LOW grade is rankable on absolute quality. Defaulting unknown to the absolute
|
|
62
|
+
// lane is the "fails open" bug this guards against: a prod-missing or ungraded
|
|
63
|
+
// HIGH corpus must never be co-ranked as trustworthy absolute quality.
|
|
64
|
+
// ---------------------------------------------------------------------------
|
|
65
|
+
|
|
66
|
+
export const LANE_ABSOLUTE = 'absolute-quality';
|
|
67
|
+
export const LANE_RELATIVE_ONLY = 'relative-comparison-only';
|
|
68
|
+
|
|
69
|
+
/** Grades that force a corpus into the relative-only lane. Mirrors
|
|
70
|
+
* contamination.RELATIVE_ONLY_GRADES (HIGH + MEDIUM). */
|
|
71
|
+
const RELATIVE_ONLY_GRADES = new Set(['HIGH', 'MEDIUM']);
|
|
72
|
+
|
|
73
|
+
/** Upper-case a contamination grade; map empty / "NONE" to null (unknown). */
|
|
74
|
+
export function normalizeContamination(grade) {
|
|
75
|
+
if (grade == null) return null;
|
|
76
|
+
const g = String(grade).trim().toUpperCase();
|
|
77
|
+
if (!g || g === 'NONE') return null;
|
|
78
|
+
return g;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Lane gate — FAIL SAFE. True ⇒ relative-comparison-only (must NOT be co-ranked
|
|
83
|
+
* with absolute-quality corpora). Known HIGH/MEDIUM or UNKNOWN → true; only a
|
|
84
|
+
* positively LOW grade → false. Mirrors isRelativeOnlyLane in the website util.
|
|
85
|
+
*
|
|
86
|
+
* @param {string|null|undefined} grade run_card->>contamination for the row
|
|
87
|
+
* @returns {boolean}
|
|
88
|
+
*/
|
|
89
|
+
export function isRelativeOnlyLane(grade) {
|
|
90
|
+
const g = normalizeContamination(grade);
|
|
91
|
+
if (g == null) return true; // fail safe: unknown contamination → relative-only
|
|
92
|
+
return RELATIVE_ONLY_GRADES.has(g);
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Sort key → (PostgREST column, direction). Higher-is-better metrics sort
|
|
97
|
+
* descending; TER, cost, and "lowest" sort ascending; date sorts newest-first.
|
|
98
|
+
*/
|
|
99
|
+
export const RESULTS_SORT = {
|
|
100
|
+
composite: { column: 'composite_score', dir: 'desc' },
|
|
101
|
+
chrf: { column: 'chrf_plus_plus', dir: 'desc' },
|
|
102
|
+
bleu: { column: 'corpus_bleu', dir: 'desc' },
|
|
103
|
+
comet: { column: 'comet_score', dir: 'desc' },
|
|
104
|
+
ter: { column: 'ter', dir: 'asc' },
|
|
105
|
+
cost: { column: 'cost_per_entry_usd', dir: 'asc' },
|
|
106
|
+
date: { column: 'run_timestamp', dir: 'desc' },
|
|
107
|
+
};
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Trust vocabulary — maps the DB trust enum to display keys. Mirrors
|
|
111
|
+
* DB_TRUST_TO_DISPLAY in cli/website (kept in sync; the website is a separate
|
|
112
|
+
* package so we can't import it). Every CLI submission is 'unverified'
|
|
113
|
+
* (self-reported); 'verified' is set only by the server-side re-scoring
|
|
114
|
+
* verifier; 'disqualified' is filtered from ranked views.
|
|
115
|
+
*/
|
|
116
|
+
export const TRUST_DISPLAY = {
|
|
117
|
+
unverified: 'self-benchmarked',
|
|
118
|
+
verified: 'champollion-verified',
|
|
119
|
+
disqualified: 'disqualified',
|
|
120
|
+
};
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Strip characters significant to PostgREST filter syntax so a user-supplied
|
|
124
|
+
* value can't inject extra filters. Returns null for empty/blank input.
|
|
125
|
+
*
|
|
126
|
+
* @param {string|null|undefined} s
|
|
127
|
+
* @returns {string|null}
|
|
128
|
+
*/
|
|
129
|
+
export function sanitize(s) {
|
|
130
|
+
if (s == null) return null;
|
|
131
|
+
const cleaned = String(s).replace(/[%,()*]/g, '').trim();
|
|
132
|
+
return cleaned.length ? cleaned : null;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Build the PostgREST URL for a leaderboard listing query.
|
|
137
|
+
*
|
|
138
|
+
* Mirrors buildListingUrl in cli/website/src/utils/leaderboardUtils.js:
|
|
139
|
+
* server-side filtering + `trust=neq.disqualified` so disqualified runs never
|
|
140
|
+
* surface, ordered by the chosen metric with nulls last.
|
|
141
|
+
*
|
|
142
|
+
* @param {object} params
|
|
143
|
+
* @param {string} [params.supabaseUrl] Override Supabase project URL.
|
|
144
|
+
* @param {string} [params.select] Comma-separated column list.
|
|
145
|
+
* @param {string} [params.source_language] Source ISO 639-3 code (or null).
|
|
146
|
+
* @param {string} [params.target_language] Target ISO 639-3 code (or null).
|
|
147
|
+
* @param {string} [params.model] Model slug substring (ilike).
|
|
148
|
+
* @param {string} [params.sort] Sort key (see RESULTS_SORT).
|
|
149
|
+
* @param {number} [params.limit] Max rows (default 20).
|
|
150
|
+
* @returns {string} Full PostgREST URL.
|
|
151
|
+
*/
|
|
152
|
+
export function buildResultsUrl({
|
|
153
|
+
supabaseUrl = SUPABASE_URL,
|
|
154
|
+
select = RESULTS_SELECT,
|
|
155
|
+
source_language = null,
|
|
156
|
+
target_language = null,
|
|
157
|
+
model = null,
|
|
158
|
+
sort = 'composite',
|
|
159
|
+
limit = 20,
|
|
160
|
+
} = {}) {
|
|
161
|
+
const params = new URLSearchParams();
|
|
162
|
+
params.set('select', select);
|
|
163
|
+
params.set('trust', 'neq.disqualified');
|
|
164
|
+
|
|
165
|
+
// Language pair — Supabase stores "src>tgt" lowercase ISO 639-3.
|
|
166
|
+
const src = sanitize(source_language)?.toLowerCase();
|
|
167
|
+
const tgt = sanitize(target_language)?.toLowerCase();
|
|
168
|
+
if (src && tgt) {
|
|
169
|
+
params.set('language_pair', `eq.${src}>${tgt}`);
|
|
170
|
+
} else if (src) {
|
|
171
|
+
params.set('language_pair', `like.${src}>%`);
|
|
172
|
+
} else if (tgt) {
|
|
173
|
+
params.set('language_pair', `like.%>${tgt}`);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// Model — case-insensitive substring match (agents pass "haiku", not the
|
|
177
|
+
// full slug). ilike treats * as the wildcard.
|
|
178
|
+
const m = sanitize(model);
|
|
179
|
+
if (m) params.set('model_slug', `ilike.*${m}*`);
|
|
180
|
+
|
|
181
|
+
const { column, dir } = RESULTS_SORT[sort] || RESULTS_SORT.composite;
|
|
182
|
+
params.set('order', `${column}.${dir}.nullslast`);
|
|
183
|
+
params.set('limit', String(limit));
|
|
184
|
+
|
|
185
|
+
return `${supabaseUrl}/rest/v1/run_cards?${params.toString()}`;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Map a raw run_cards row to the compact scored shape the tool returns.
|
|
190
|
+
*
|
|
191
|
+
* @param {object} row PostgREST row
|
|
192
|
+
* @returns {object} Compact result entry
|
|
193
|
+
*/
|
|
194
|
+
export function mapResultRow(row) {
|
|
195
|
+
const pair = (row.language_pair || '?').trim().toLowerCase();
|
|
196
|
+
const contamination = normalizeContamination(row.contamination);
|
|
197
|
+
const relativeOnly = isRelativeOnlyLane(row.contamination);
|
|
198
|
+
return {
|
|
199
|
+
id: row.id,
|
|
200
|
+
model: row.model_slug || '?',
|
|
201
|
+
pair,
|
|
202
|
+
condition: row.condition || '?',
|
|
203
|
+
composite: row.composite_score ?? null,
|
|
204
|
+
chrf: row.chrf_plus_plus ?? null,
|
|
205
|
+
bleu: row.corpus_bleu ?? null,
|
|
206
|
+
comet: row.comet_score ?? null,
|
|
207
|
+
ter: row.ter ?? null,
|
|
208
|
+
qualityTier: row.quality_tier ?? null,
|
|
209
|
+
trust: TRUST_DISPLAY[row.trust] || 'self-benchmarked',
|
|
210
|
+
author: row.submitter || 'anonymous',
|
|
211
|
+
dataset: row.dataset_id || null,
|
|
212
|
+
cost_usd: row.total_cost_usd ?? null,
|
|
213
|
+
date: (row.run_timestamp || row.submitted_at || '').slice(0, 10) || null,
|
|
214
|
+
// Contamination lane (FAIL SAFE). `relative_only` rows are NOT comparable
|
|
215
|
+
// on absolute quality against absolute-lane rows — only against each other
|
|
216
|
+
// on the same corpus. An agent that ignores this would co-rank a corpus
|
|
217
|
+
// that's in models' training data (or whose grade is unknown) as if it
|
|
218
|
+
// measured real quality.
|
|
219
|
+
contamination: contamination, // normalized grade, or null when unknown
|
|
220
|
+
relative_only: relativeOnly,
|
|
221
|
+
score_lane: relativeOnly ? LANE_RELATIVE_ONLY : LANE_ABSOLUTE,
|
|
222
|
+
};
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* Format mapped result rows as a concise, agent-readable block.
|
|
227
|
+
*
|
|
228
|
+
* @param {object[]} rows Mapped rows (from mapResultRow)
|
|
229
|
+
* @param {object} [opts]
|
|
230
|
+
* @param {string} [opts.sort] Sort key used (for the header line)
|
|
231
|
+
* @returns {string}
|
|
232
|
+
*/
|
|
233
|
+
export function formatResults(rows, { sort = 'composite' } = {}) {
|
|
234
|
+
if (!rows || rows.length === 0) {
|
|
235
|
+
return [
|
|
236
|
+
'No scored results on the public leaderboard yet.',
|
|
237
|
+
'',
|
|
238
|
+
'The board is filled by people running benchmarks from the queue.',
|
|
239
|
+
'Use list_queue to see what is pending, then run_benchmark to score an',
|
|
240
|
+
'item — your run publishes here with your attribution.',
|
|
241
|
+
'',
|
|
242
|
+
'Leaderboard: https://champollion.dev/leaderboard',
|
|
243
|
+
].join('\n');
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
const fmt = (v, d = 3) => (v == null ? '—' : Number(v).toFixed(d));
|
|
247
|
+
const lines = rows.map((r, i) => {
|
|
248
|
+
const arrow = r.pair.includes('>') ? r.pair.replace('>', '→') : r.pair;
|
|
249
|
+
// Mark relative-only rows so the score is never read as absolute quality.
|
|
250
|
+
const lane = r.relative_only
|
|
251
|
+
? ` ⚠ relative-only${r.contamination ? ` (${r.contamination})` : ' (grade unknown)'}`
|
|
252
|
+
: '';
|
|
253
|
+
// The id closes the loop with get_run_card, whose not-found error
|
|
254
|
+
// points agents back here for valid ids.
|
|
255
|
+
const idTag = r.id != null ? ` id ${r.id}` : '';
|
|
256
|
+
return (
|
|
257
|
+
`#${i + 1} ${arrow} composite ${fmt(r.composite)} chrF++ ${fmt(r.chrf, 1)} `
|
|
258
|
+
+ `${r.model.split('/').pop()} [${r.condition}] ${r.trust} by ${r.author}${idTag}${lane}`
|
|
259
|
+
);
|
|
260
|
+
});
|
|
261
|
+
|
|
262
|
+
// Lane-mixing guard: relative-only and absolute-quality scores are NOT
|
|
263
|
+
// comparable across lanes — only within the same corpus. Warn loudly when a
|
|
264
|
+
// single ranked list spans both lanes, or when every row is relative-only.
|
|
265
|
+
const relCount = rows.filter((r) => r.relative_only).length;
|
|
266
|
+
const absCount = rows.length - relCount;
|
|
267
|
+
const laneNotes = [];
|
|
268
|
+
if (relCount > 0 && absCount > 0) {
|
|
269
|
+
laneNotes.push(
|
|
270
|
+
'',
|
|
271
|
+
`⚠ This list mixes ${absCount} absolute-quality row(s) and ${relCount} `
|
|
272
|
+
+ 'relative-comparison-only row(s) (marked above). Do NOT rank them against '
|
|
273
|
+
+ 'each other: a relative-only corpus (HIGH/MEDIUM contamination, FLORES, or '
|
|
274
|
+
+ 'an unknown grade) is in models\' training data — its score is valid only '
|
|
275
|
+
+ 'for comparing methods on THAT corpus, not as absolute quality.',
|
|
276
|
+
);
|
|
277
|
+
} else if (relCount > 0 && absCount === 0) {
|
|
278
|
+
laneNotes.push(
|
|
279
|
+
'',
|
|
280
|
+
'⚠ Every row above is relative-comparison-only (HIGH/MEDIUM contamination, '
|
|
281
|
+
+ 'FLORES, or unknown grade). These scores compare methods on the same '
|
|
282
|
+
+ 'corpus — they are NOT absolute quality and must not be ranked against '
|
|
283
|
+
+ 'other corpora.',
|
|
284
|
+
);
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
return [
|
|
288
|
+
`Top ${rows.length} scored result(s) by ${sort}:`,
|
|
289
|
+
'',
|
|
290
|
+
...lines,
|
|
291
|
+
...laneNotes,
|
|
292
|
+
'',
|
|
293
|
+
'Full leaderboard: https://champollion.dev/leaderboard',
|
|
294
|
+
].join('\n');
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
/**
|
|
298
|
+
* Fetch scored leaderboard results from the public Supabase endpoint.
|
|
299
|
+
*
|
|
300
|
+
* @param {object} [params] Same shape as buildResultsUrl params.
|
|
301
|
+
* @returns {Promise<object[]>} Mapped result rows.
|
|
302
|
+
*/
|
|
303
|
+
export async function fetchResults(params = {}) {
|
|
304
|
+
const url = buildResultsUrl(params);
|
|
305
|
+
const resp = await fetch(url, {
|
|
306
|
+
headers: SB_HEADERS,
|
|
307
|
+
signal: AbortSignal.timeout(30_000),
|
|
308
|
+
});
|
|
309
|
+
if (!resp.ok) {
|
|
310
|
+
throw new Error(
|
|
311
|
+
`Leaderboard fetch failed: HTTP ${resp.status} ${(await resp.text()).slice(0, 120)}`,
|
|
312
|
+
);
|
|
313
|
+
}
|
|
314
|
+
const rows = await resp.json();
|
|
315
|
+
return rows.map(mapResultRow);
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Fetch the full run card (scores + method/config metadata) for one run id.
|
|
320
|
+
* Returns the same `run_card` JSON the public leaderboard serves on expand;
|
|
321
|
+
* no sentence text is read. Returns null if no run matches.
|
|
322
|
+
*
|
|
323
|
+
* @param {string} id run_cards.id
|
|
324
|
+
* @param {object} [opts]
|
|
325
|
+
* @param {string} [opts.supabaseUrl] Override Supabase project URL.
|
|
326
|
+
* @returns {Promise<object|null>}
|
|
327
|
+
*/
|
|
328
|
+
export async function fetchRunCard(id, { supabaseUrl = SUPABASE_URL } = {}) {
|
|
329
|
+
const safe = sanitize(id);
|
|
330
|
+
if (!safe) return null;
|
|
331
|
+
const select =
|
|
332
|
+
'id,model_slug,language_pair,condition,trust,submitter,' +
|
|
333
|
+
'composite_score,chrf_plus_plus,dataset_id,run_timestamp,run_card';
|
|
334
|
+
const url =
|
|
335
|
+
`${supabaseUrl}/rest/v1/run_cards?select=${select}` +
|
|
336
|
+
`&trust=neq.disqualified&id=eq.${encodeURIComponent(safe)}&limit=1`;
|
|
337
|
+
const resp = await fetch(url, {
|
|
338
|
+
headers: SB_HEADERS,
|
|
339
|
+
signal: AbortSignal.timeout(30_000),
|
|
340
|
+
});
|
|
341
|
+
if (!resp.ok) {
|
|
342
|
+
throw new Error(`Run-card fetch failed: HTTP ${resp.status}`);
|
|
343
|
+
}
|
|
344
|
+
const rows = await resp.json();
|
|
345
|
+
return rows[0] || null;
|
|
346
|
+
}
|