champollion 0.3.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/README.md +41 -26
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +34 -0
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +286 -85
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +632 -125
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +15 -9
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +194 -35
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +6 -1
  100. package/lib/seal.mjs +4 -3
  101. package/lib/sealed-qualifier.mjs +1 -1
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +1 -1
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/cards-fallback.json +1 -1
  123. package/shared/curated-orthography-conventions.json +26 -8
  124. package/shared/gettext-plural-forms.json +45 -0
  125. package/shared/method-registry.json +2 -0
  126. package/shared/metric-registry.json +96 -18
  127. package/shared/schemas/champollion-plugin.schema.json +4 -0
  128. package/shared/schemas/corpora-card.schema.json +8 -2
  129. package/shared/schemas/method-index-record.schema.json +67 -0
  130. package/shared/schemas/method-registry.schema.json +4 -0
  131. package/shared/schemas/metric-registry.schema.json +55 -1
  132. package/shared/docent/corpus.json +0 -11739
@@ -9,18 +9,47 @@
9
9
  */
10
10
 
11
11
  import fs from 'node:fs';
12
- import path from 'node:path';
13
12
  import { resolveConfig } from '../config.js';
14
- import { auditLocalePair, formatIntegrityReport } from '../integrity.js';
13
+ import { auditLocalePair, formatIntegrityReport, auditARBDocument } from '../integrity.js';
15
14
  import { compileNoTranslate } from '../no-translate.js';
16
15
  import { resolvePairs } from '../pairs.js';
17
- import { loadTM, lookupTM, tmMethodKey } from '../tm.js';
16
+ import { loadTM, saveTM, isTMDirty } from '../tm.js';
17
+ import { tmTextFor, tmProofTextsFor, createTMEvictor } from '../tm-evict.js';
18
+ import { poPluralSlots } from '../po.js';
19
+ import { tmKeysForPair, tmHoldsValue } from '../fallback.js';
18
20
  import { getLanguageCard } from '../registers.js';
19
21
  import { SCRIPT_CONVERTERS, converterKeyForLocale } from '../scripts.js';
20
- import { flattenKeys } from '../flatten.js';
21
- import { readLocaleFile, detectFormatFromDir, getExtension } from '../format.js';
22
+ import {
23
+ discoverLocaleLayout, loadSourceUnits, expectedForTarget, readLocaleFlat, NS_SEPARATOR,
24
+ } from '../locale-layout.js';
22
25
  import { output } from '../output.js';
23
26
 
27
+ /**
28
+ * Fold one file's audit into a locale's running audit. In a namespaced
29
+ * layout every key is reported as "<ns>::<key>" so two files' findings for
30
+ * "title" stay distinguishable (same convention as the lock manifest).
31
+ *
32
+ * @param {object|null} into - Accumulated audit (null to start one)
33
+ * @param {object} audit - auditLocalePair() result for one file
34
+ * @param {string|null} ns - Namespace to prefix, or null for single-file layouts
35
+ * @returns {object} Merged audit
36
+ */
37
+ function mergeAudits(into, audit, ns) {
38
+ const tag = (k) => (ns === null ? k : `${ns}${NS_SEPARATOR}${k}`);
39
+ const out = into || {
40
+ placeholderIssues: [], encodingIssues: [], copies: [], orphans: [], noTranslateDrift: [],
41
+ unexpectedPua: [], hollowedValues: [], icuIssues: [], documentIssues: [], pluralIssues: [], pluralNotes: [], bomFiles: [],
42
+ };
43
+ for (const field of ['placeholderIssues', 'encodingIssues', 'noTranslateDrift', 'unexpectedPua', 'hollowedValues', 'icuIssues', 'documentIssues', 'pluralIssues', 'pluralNotes']) {
44
+ for (const item of audit[field] || []) out[field].push({ ...item, key: tag(item.key) });
45
+ }
46
+ for (const field of ['copies', 'orphans']) {
47
+ for (const key of audit[field] || []) out[field].push(tag(key));
48
+ }
49
+ out.bomFiles.push(...(audit.bomFiles || []));
50
+ return out;
51
+ }
52
+
24
53
  /**
25
54
  * @param {import('../types.js').CLIArgs} args - Parsed CLI arguments
26
55
  * @param {string} cwd - Working directory
@@ -46,9 +75,17 @@ async function run(args, cwd) {
46
75
  // applies; without it, integrity reported thousands of "untranslated
47
76
  // copies" on a project sync calls fully synced. Read-only TM load.
48
77
  const tm = loadTM(cwd);
78
+ // Damaged values the TM proves the pipeline wrote are evicted from it, so
79
+ // the `sync --force-keys` the report names re-translates them instead of
80
+ // re-serving the damage for free (lib/tm-evict.js). Hand-written values
81
+ // have no equal cache entry and are never touched.
82
+ const evictor = createTMEvictor(tm);
83
+ let tmEvicted = 0;
49
84
  const tmKeys = new Map();
50
- for (const [, pc] of resolvePairs(config)) {
51
- tmKeys.set(pc.target, tmMethodKey(pc));
85
+ for (const [, pc] of resolvePairs(config, { cwd })) {
86
+ // The pair's own key, then its fallback's: a value the fallback
87
+ // produced is cached under the fallback (lib/fallback.js).
88
+ tmKeys.set(pc.target, tmKeysForPair(pc));
52
89
  const key = converterKeyForLocale(pc.target, getLanguageCard(pc.target));
53
90
  if (!key || !SCRIPT_CONVERTERS[key].puaRange) continue;
54
91
  scriptExpectations.set(pc.target, {
@@ -57,57 +94,93 @@ async function run(args, cwd) {
57
94
  });
58
95
  }
59
96
 
60
- const format = config.format !== 'auto'
61
- ? config.format
62
- : detectFormatFromDir(config.localesDir);
63
- const ext = getExtension(format);
64
- const sourcePath = path.join(config.localesDir, `${config.inputLocale}${ext}`);
65
-
66
- if (!fs.existsSync(sourcePath)) {
97
+ // The project's files, per locale, from the ONE layout module (flat,
98
+ // folder per locale, or localesPattern).
99
+ const layout = discoverLocaleLayout(config, { cwd });
100
+ const sourceMissing = layout.namespaced
101
+ ? layout.sourceFiles.length === 0
102
+ : !fs.existsSync(layout.sourceFiles[0].path);
103
+ if (sourceMissing) {
104
+ const where = layout.namespaced
105
+ ? `${layout.display} (no ${layout.format} files for ${config.inputLocale})`
106
+ : layout.sourceFiles[0].path;
67
107
  if (json) {
68
108
  console.log(JSON.stringify({
69
109
  command: 'integrity',
70
- error: `Source locale file not found: ${sourcePath}`,
110
+ error: `Source locale file not found: ${where}`,
71
111
  }, null, 2));
72
112
  return 1;
73
113
  }
74
- output.error(`Source locale file not found: ${sourcePath}`);
114
+ output.error(`Source locale file not found: ${where}`);
75
115
  return 1;
76
116
  }
77
117
 
78
- const sourceRaw = readLocaleFile(sourcePath, format);
79
- const sourceFlat = format === 'json' ? flattenKeys(sourceRaw) : sourceRaw;
118
+ const units = loadSourceUnits(layout);
119
+ const sourceKeyCount = units.reduce((n, u) => n + Object.keys(u.flat).length, 0);
80
120
 
81
- // Detect target locales from directory listing
82
- const files = fs.readdirSync(config.localesDir);
83
- const targetLocales = files
84
- .filter(f => f.endsWith(ext) && !f.startsWith(config.inputLocale))
85
- .map(f => f.replace(ext, ''));
121
+ // Target locales present on disk — EXACT code match. The old prefix test
122
+ // (`!f.startsWith(inputLocale)`) silently skipped en-GB/en-AU when the
123
+ // source was en, so those files were never audited.
124
+ const targetLocales = layout.listLocales();
86
125
 
87
126
  output.raw('\n champollion integrity — Locale File Audit\n');
88
- output.raw(` Source: ${config.inputLocale} (${Object.keys(sourceFlat).length} keys)`);
127
+ output.raw(` Source: ${config.inputLocale} (${sourceKeyCount} keys)`);
89
128
  output.raw(` Targets: ${targetLocales.join(', ')}\n`);
90
129
 
91
130
  let totalIssues = 0;
131
+
132
+ let totalAdvisory = 0;
92
133
  const localeReports = [];
93
134
 
94
135
  for (const locale of targetLocales) {
95
- const targetPath = path.join(config.localesDir, `${locale}${ext}`);
96
- const targetRaw = readLocaleFile(targetPath, format);
97
- const targetFlat = format === 'json' ? flattenKeys(targetRaw) : targetRaw;
98
-
99
- const tmKey = tmKeys.get(locale) || null;
100
- const audit = auditLocalePair(sourceFlat, targetFlat, locale, {
101
- noTranslate,
102
- scriptExpectation: scriptExpectations.get(locale) || null,
103
- isConfirmedEcho: tmKey
104
- ? (key, sourceValue) => lookupTM(tm, sourceValue, locale, tmKey) === sourceValue
105
- : null,
106
- });
136
+ const localeTmKeys = tmKeys.get(locale) || null;
137
+ // A borrowed i18next plural form is cached under its own text (tmTextFor).
138
+ const confirmedEchoFor = (expansion) => (localeTmKeys
139
+ ? (key, sourceValue) => tmHoldsValue(tm, tmTextFor(key, sourceValue, expansion), locale, localeTmKeys, sourceValue)
140
+ : null);
141
+ // One audit per file; namespaced keys ("common::nav.home") keep the
142
+ // per-file findings apart in the merged locale report.
143
+ let audit = null;
144
+ for (const unit of units) {
145
+ const file = layout.fileFor(locale, unit.ns);
146
+ if (!fs.existsSync(file.path) && !layout.namespaced) continue;
147
+ const targetFlat = fs.existsSync(file.path) ? readLocaleFlat(file) : {};
148
+ // Judge the file by the keys this locale should have — i18next
149
+ // plurals expand to the target's own CLDR categories.
150
+ const { flat: expected, expansion } = expectedForTarget(unit, config.inputLocale, locale);
151
+ const fileAudit = auditLocalePair(expected, targetFlat, locale, {
152
+ noTranslate,
153
+ scriptExpectation: scriptExpectations.get(locale) || null,
154
+ isConfirmedEcho: confirmedEchoFor(expansion),
155
+ // A gettext catalog holds only the plural forms its header has slots for.
156
+ pluralSlots: file.format === 'po' && fs.existsSync(file.path)
157
+ ? poPluralSlots(fs.readFileSync(file.path, 'utf-8'), locale) : null,
158
+ });
159
+ fileAudit.documentIssues = file.format === 'arb' && fs.existsSync(file.path)
160
+ ? auditARBDocument(unit.file.path, file.path, locale)
161
+ : [];
162
+ // Values damaged in a way the quality gate now refuses: evict the TM
163
+ // entries that produced them.
164
+ const damaged = [
165
+ ...fileAudit.icuIssues.map(i => [i.key, i.actual]),
166
+ ...fileAudit.placeholderIssues.map(i => [i.key, i.targetVal]),
167
+ ...fileAudit.hollowedValues.map(i => [i.key, i.actual]),
168
+ ];
169
+ for (const [key, value] of damaged) {
170
+ if (typeof expected[key] !== 'string') continue;
171
+ for (const text of tmProofTextsFor(key, expected[key], expansion)) {
172
+ tmEvicted += evictor.evictProducing(text, locale, value, localeTmKeys || []);
173
+ }
174
+ }
175
+ // Single-file layouts keep auditLocalePair's own result object (and
176
+ // its exact shape in --json); namespaced ones merge with ns tags.
177
+ audit = layout.namespaced ? mergeAudits(audit, fileAudit, unit.ns) : fileAudit;
178
+ }
179
+ if (!audit) continue;
107
180
  const report = formatIntegrityReport(locale, audit);
108
181
  output.raw(report);
109
182
 
110
- // Only these five categories drive the exit code (pluralIssues and
183
+ // Only these categories drive the exit code (pluralIssues and
111
184
  // bomFiles are reported but advisory) — issueCount matches that.
112
185
  //
113
186
  // noTranslateDrift is an error, not advisory: a declared-verbatim key
@@ -124,21 +197,35 @@ async function run(args, cwd) {
124
197
  audit.orphans.length +
125
198
  audit.noTranslateDrift.length +
126
199
  audit.unexpectedPua.length +
127
- audit.hollowedValues.length;
200
+ audit.hollowedValues.length +
201
+ // ICU structure damage and ARB document damage break the app at
202
+ // runtime / build time — errors, like the hollowed values.
203
+ (audit.icuIssues?.length || 0) +
204
+ (audit.documentIssues?.length || 0);
128
205
  totalIssues += issueCount;
206
+ totalAdvisory += (audit.pluralIssues?.length || 0) + (audit.bomFiles?.length || 0);
129
207
  localeReports.push({ locale, issues: audit, issueCount });
130
208
  }
131
209
 
132
- output.raw(` Total issues: ${totalIssues}`);
210
+ // Advisory findings (plural categories, BOMs) don't fail the run, but a
211
+ // bare "Total issues: 0" under a printed warning read as a contradiction.
212
+ output.raw(` Total issues: ${totalIssues}`
213
+ + (totalAdvisory > 0 ? ` (+${totalAdvisory} advisory, listed above — they don't fail the check)` : ''));
214
+ if (tmEvicted > 0 && isTMDirty(tm)) {
215
+ saveTM(cwd, tm);
216
+ output.raw(` [TM] Evicted ${tmEvicted} cached translation(s) that produced damaged values — `
217
+ + '`champollion sync --force-keys <key>` now re-translates them instead of re-serving them.');
218
+ }
133
219
 
134
220
  const exitCode = (totalIssues > 0 && !args['warn-only']) ? 1 : 0;
135
221
  if (json) {
136
222
  console.log(JSON.stringify({
137
223
  command: 'integrity',
138
224
  source: config.inputLocale,
139
- sourceKeys: Object.keys(sourceFlat).length,
225
+ sourceKeys: sourceKeyCount,
140
226
  locales: localeReports,
141
227
  totalIssues,
228
+ tmEvicted,
142
229
  warnOnly: !!args['warn-only'],
143
230
  }, null, 2));
144
231
  }
@@ -6,8 +6,8 @@
6
6
  *
7
7
  * Usage:
8
8
  * champollion leaderboard # Show all results
9
- * champollion leaderboard --pair "en>crk" # Filter by language pair (quote the >)
10
- * champollion leaderboard --sort composite # Sort by composite score
9
+ * champollion leaderboard --pair "eng>crk" # Filter by language pair (quote the >; eng-crk works too)
10
+ * champollion leaderboard --sort bleu # Sort by BLEU (default: chrF++)
11
11
  * champollion leaderboard --json # Machine-readable NDJSON
12
12
  * champollion leaderboard --top 5 # Show top N results
13
13
  * champollion leaderboard --install 1 # Install method config from rank 1
@@ -15,7 +15,8 @@
15
15
  */
16
16
 
17
17
  import { output } from '../output.js';
18
- import { getLanguageCard } from '../registers.js';
18
+ import { getLanguageCard, resolveCode } from '../registers.js';
19
+ import { parseLanguagePair, formatLanguagePair } from '../language-pair.js';
19
20
  // Supabase public config — shared with the dynamic card loader
20
21
  // (RLS restricts the anon role to read-only)
21
22
  import { SUPABASE_URL, SUPABASE_ANON_KEY } from '../cards/env.js';
@@ -26,31 +27,74 @@ import fs from 'node:fs';
26
27
  import path from 'node:path';
27
28
 
28
29
  /**
29
- * Sortable metric definitions.
30
- * key: CLI flag value → column: Supabase column → label: Display header
30
+ * Scoring standard/1 (founder, 2026-10-04: "we want scoring to be industry
31
+ * standard"). As in WMT, FLORES-200 and AmericasNLP: ONE headline and ranking
32
+ * metric, chrF++ with its 95% bootstrap CI; the other standard metrics (BLEU,
33
+ * spBLEU, TER, COMET) beside it, never blended; diagnostics apart, never a
34
+ * headline or a default sort; no quality labels read off automatic scores.
35
+ * A run card declares `scores.scoring_standard`; a card without it is legacy,
36
+ * and only a legacy card's stored composite is ever shown — labelled
37
+ * "legacy composite (retired)".
31
38
  */
32
- const SORT_KEYS = {
33
- composite: { column: 'composite_score', label: 'Composite', desc: true },
34
- chrf: { column: 'chrf_plus_plus', label: 'chrF++', desc: true },
35
- exact: { column: 'exact_match_rate', label: 'Exact Match', desc: true },
36
- fst: { column: 'fst_acceptance_rate', label: 'FST Accept', desc: true },
37
- equivalent: { column: 'equivalent_match_rate', label: 'Equiv Match', desc: true },
38
- semantic: { column: 'semantic_score', label: 'Semantic', desc: true },
39
- cost: { column: 'total_cost_usd', label: 'Cost (USD)', desc: false },
40
- date: { column: 'run_timestamp', label: 'Date', desc: true },
41
- };
39
+ const SCORING_STANDARD = 'standard/1';
40
+ const LEGACY_COMPOSITE_LABEL = 'legacy composite (retired)';
41
+ const DEFAULT_SORT = 'chrf';
42
42
 
43
43
  /**
44
- * Quality tier → display label with color hint.
44
+ * Sortable metric definitions.
45
+ * key: CLI flag value → column: Supabase column → label: display header →
46
+ * group: standard | diagnostic | legacy | other. Metrics sort best-first
47
+ * (desc), TER lowest-first; nulls always last.
45
48
  */
46
- const TIER_LABELS = {
47
- baseline: '○ Baseline',
48
- emerging: '◔ Emerging',
49
- functional: '◑ Functional',
50
- deployable: '◕ Deployable',
51
- fluent: '● Fluent',
49
+ const SORT_KEYS = {
50
+ chrf: { column: 'chrf_plus_plus', label: 'chrF++', desc: true, group: 'standard' },
51
+ bleu: { column: 'corpus_bleu', label: 'BLEU', desc: true, group: 'standard' },
52
+ ter: { column: 'ter', label: 'TER (lower is better)', desc: false, group: 'standard' },
53
+ comet: { column: 'comet_score', label: 'COMET', desc: true, group: 'standard' },
54
+ exact: { column: 'exact_match_rate', label: 'Exact Match (diagnostic)', desc: true, group: 'diagnostic' },
55
+ fst: { column: 'fst_acceptance_rate', label: 'FST Accept (diagnostic)', desc: true, group: 'diagnostic' },
56
+ equivalent: { column: 'equivalent_match_rate', label: 'Equiv Match (diagnostic)', desc: true, group: 'diagnostic' },
57
+ semantic: { column: 'semantic_score', label: 'Semantic (diagnostic)', desc: true, group: 'diagnostic' },
58
+ cost: { column: 'total_cost_usd', label: 'Cost (USD)', desc: false, group: 'other' },
59
+ date: { column: 'run_timestamp', label: 'Date', desc: true, group: 'other' },
60
+ // Kept so an existing `--sort composite` keeps working: it orders by the
61
+ // retired composite, which standard/1 cards do not carry (they sort last).
62
+ composite: { column: 'composite_score', label: 'Legacy composite (retired)', desc: true, group: 'legacy' },
52
63
  };
53
64
 
65
+ /** True when the run card predates standard/1 (no scores.scoring_standard). */
66
+ function isLegacyRow(row) {
67
+ return !row?.run_card?.scores?.scoring_standard;
68
+ }
69
+
70
+ /** A finite number or null (JSON-extracted values may be strings). */
71
+ function num(v) {
72
+ if (v == null || v === '') return null;
73
+ const n = Number(v);
74
+ return Number.isFinite(n) ? n : null;
75
+ }
76
+
77
+ /** [lo, hi] of the chrF++ 95% CI, or null when either bound is missing. */
78
+ function chrfCi(row) {
79
+ const lo = num(row.chrf_ci_lower);
80
+ const hi = num(row.chrf_ci_upper);
81
+ return lo != null && hi != null ? [lo, hi] : null;
82
+ }
83
+
84
+ /** "47.5 [45.9, 49.0]" — the chrF++ headline with its CI (no label). */
85
+ function fmtChrf(row) {
86
+ const v = num(row.chrf_plus_plus);
87
+ if (v == null) return '—';
88
+ const ci = chrfCi(row);
89
+ return ci ? `${v.toFixed(1)} [${ci[0].toFixed(1)}, ${ci[1].toFixed(1)}]` : v.toFixed(1);
90
+ }
91
+
92
+ /** The stored composite of a LEGACY card, else null (never a headline). */
93
+ function legacyComposite(row) {
94
+ const c = num(row.composite_score);
95
+ return c != null && isLegacyRow(row) ? c : null;
96
+ }
97
+
54
98
  // ---------------------------------------------------------------------------
55
99
  // Contamination lane — FAIL SAFE. The lane policy lives in
56
100
  // lib/contamination-lane.js (the CLI-side mirror of the SSOT in
@@ -108,16 +152,41 @@ function pad(str, width, align = 'left') {
108
152
  return align === 'right' ? padding + s : s + padding;
109
153
  }
110
154
 
155
+ /**
156
+ * The board's key for a --pair value. Published runs store their pair as
157
+ * `eng>crk`: lower-case ISO 639-3 codes joined by > (mt-eval's publish writes
158
+ * it), and the filter is an exact match. So --pair is read in any spelling
159
+ * lib/language-pair.js accepts (eng-crk used to match nothing, silently), a
160
+ * bare 2-letter code goes to its ISO 639-3 code through the card aliases
161
+ * (en → eng, as `recommend` does), and the result is lower-cased. A code with
162
+ * a subtag (pt-BR) is kept as typed: the alias bridge would drop the subtag.
163
+ *
164
+ * @returns {{ok: true, pair: string, resolved: string[]} | {ok: false, error: string}}
165
+ */
166
+ function boardPair(raw) {
167
+ const p = parseLanguagePair(raw, { label: '--pair' });
168
+ if (!p.ok) return p;
169
+ const resolved = [];
170
+ const toBoard = (code) => {
171
+ const lower = code.toLowerCase();
172
+ if (/[-_]/.test(code)) return lower;
173
+ const r = String(resolveCode(lower)).toLowerCase();
174
+ if (r !== lower) resolved.push(`${code} → ${r}`);
175
+ return r;
176
+ };
177
+ return { ok: true, pair: formatLanguagePair({ source: toBoard(p.source), target: toBoard(p.target) }), resolved };
178
+ }
179
+
111
180
  /**
112
181
  * Fetch leaderboard data from Supabase.
113
182
  */
114
183
  async function fetchLeaderboard(sortKey, pair) {
115
- const sort = SORT_KEYS[sortKey] || SORT_KEYS.composite;
184
+ const sort = SORT_KEYS[sortKey] || SORT_KEYS[DEFAULT_SORT];
116
185
  const order = `${sort.column}.${sort.desc ? 'desc' : 'asc'}.nullslast`;
117
186
 
118
187
  let url = `${SUPABASE_URL}/rest/v1/run_cards?select=*&order=${order}`;
119
188
  if (pair) {
120
- url += `&language_pair=eq.${pair}`;
189
+ url += `&language_pair=eq.${encodeURIComponent(pair)}`;
121
190
  }
122
191
 
123
192
  const resp = await fetch(url, {
@@ -140,11 +209,27 @@ async function fetchLeaderboard(sortKey, pair) {
140
209
  * @returns {Promise<number>} Exit code (0 = success, 1 = error)
141
210
  */
142
211
  async function run(args, cwd) {
143
- const sortKey = args.sort || 'composite';
144
- const pair = args.pair || null;
212
+ const sortKey = args.sort || DEFAULT_SORT;
145
213
  const topN = args.top ? parseInt(args.top, 10) : null;
146
214
  const jsonMode = args.json || false;
147
215
 
216
+ // --pair: read in any accepted spelling, filtered as the board writes it.
217
+ let pair = null;
218
+ let resolvedNote = null;
219
+ if (args.pair !== undefined && args.pair !== false) {
220
+ const bp = boardPair(args.pair);
221
+ if (!bp.ok) {
222
+ output.error(bp.error);
223
+ return 1;
224
+ }
225
+ pair = bp.pair;
226
+ if (bp.resolved.length > 0) {
227
+ resolvedNote = `(codes resolved to ISO 639-3: ${bp.resolved.join(', ')})`;
228
+ // --json: on stderr, so stdout stays one JSON object per line.
229
+ if (jsonMode) console.error(`--pair ${pair} ${resolvedNote}`);
230
+ }
231
+ }
232
+
148
233
  // Resolve --install into an integer rank. Because --install is a string flag,
149
234
  // a bare `--install` (no rank) parses as boolean `true`, and parseInt(true)
150
235
  // is NaN — which previously slipped past the range check in _handleInstall
@@ -154,7 +239,7 @@ async function run(args, cwd) {
154
239
  if (args.install != null && args.install !== false) {
155
240
  installRank = parseInt(args.install, 10);
156
241
  if (!Number.isInteger(installRank) || installRank < 1) {
157
- output.error('--install requires a positive integer rank, e.g. `champollion leaderboard --install 1`.');
242
+ output.error('--install requires a positive integer rank, e.g. `champollion network leaderboard --install 1`.');
158
243
  return 1;
159
244
  }
160
245
  }
@@ -175,7 +260,7 @@ async function run(args, cwd) {
175
260
  console.log('[]');
176
261
  } else {
177
262
  output.raw('\n No leaderboard entries found.');
178
- if (pair) output.raw(` (filtered by pair: ${pair})`);
263
+ if (pair) output.raw(` (filtered by pair: ${pair}${resolvedNote ? ` ${resolvedNote}` : ''})`);
179
264
  output.raw(' Submit results with `mt-eval publish`.\n');
180
265
  }
181
266
  return 0;
@@ -211,18 +296,32 @@ async function run(args, cwd) {
211
296
  if (jsonMode) {
212
297
  // NDJSON output for CI/CD piping
213
298
  for (const row of display) {
299
+ const legacy = isLegacyRow(row);
214
300
  console.log(JSON.stringify({
215
301
  rank: display.indexOf(row) + 1,
216
302
  model: row.model_slug,
217
303
  condition: row.condition,
218
304
  pair: row.language_pair,
219
- composite: row.composite_score,
220
- chrF: row.chrf_plus_plus,
221
- exactMatch: row.exact_match_rate,
222
- fstAcceptance: row.fst_acceptance_rate,
223
- equivalentMatch: row.equivalent_match_rate,
224
- semanticScore: row.semantic_score,
225
- tier: row.quality_tier,
305
+ // Scoring standard/1: chrF++ (with its 95% CI) is the headline and
306
+ // the ranking metric; the other standard metrics sit beside it.
307
+ scoring_standard: legacy ? null : row.run_card.scores.scoring_standard,
308
+ primary_metric: 'chrf_plus_plus',
309
+ chrF: row.chrf_plus_plus ?? null,
310
+ chrF_ci: chrfCi(row),
311
+ bleu: row.corpus_bleu ?? null,
312
+ spbleu: num(row.run_card?.scores?.spbleu),
313
+ ter: row.ter ?? null,
314
+ comet: row.comet_score ?? null,
315
+ // Diagnostics: apart, never a headline or blended.
316
+ diagnostics: {
317
+ exactMatch: row.exact_match_rate ?? null,
318
+ fstAcceptance: row.fst_acceptance_rate ?? null,
319
+ equivalentMatch: row.equivalent_match_rate ?? null,
320
+ semanticScore: row.semantic_score ?? null,
321
+ },
322
+ // Only a legacy card's stored composite, under its retired name.
323
+ // No quality tier: tiers read off automatic scores are retired.
324
+ legacy_composite: legacyComposite(row),
226
325
  cost_usd: row.total_cost_usd,
227
326
  submitter: row.submitter,
228
327
  date: row.run_timestamp?.split('T')[0],
@@ -239,57 +338,75 @@ async function run(args, cwd) {
239
338
 
240
339
  // ---- Human-readable table output ----
241
340
 
242
- const pairDisplay = pair ? formatPairDisplay(pair) : 'All Pairs';
341
+ const pairDisplay = pair ? `${formatPairDisplay(pair)} (${pair})` : 'All Pairs';
243
342
  const sortLabel = SORT_KEYS[sortKey].label;
244
343
 
245
344
  output.raw('');
246
345
  output.raw(` champollion — Method Leaderboard`);
247
346
  output.raw(` ${pairDisplay} | Sorted by: ${sortLabel}`);
347
+ if (resolvedNote) output.raw(` ${resolvedNote}`);
248
348
  if (topN) output.raw(` Showing top ${topN} of ${rows.length}`);
349
+ if (SORT_KEYS[sortKey].group === 'diagnostic') {
350
+ output.raw(` (${sortLabel} is a diagnostic: it can explain a score, it does not rank quality — the headline is chrF++.)`);
351
+ }
352
+ if (sortKey === 'composite') {
353
+ output.raw(` (The ${LEGACY_COMPOSITE_LABEL} exists only on cards scored before ${SCORING_STANDARD}; it is not a ranking of quality. Rank by chrF++.)`);
354
+ }
249
355
  output.raw('');
250
356
 
251
- // Table header
252
- const header = [
253
- pad('#', 4),
254
- pad('Model', 28),
255
- pad('Condition', 12),
256
- pad('Score', 7, 'right'),
257
- pad('chrF++', 8, 'right'),
258
- pad('EM%', 7, 'right'),
259
- pad('Tier', 14),
260
- pad('Date', 12),
261
- pad('Lane', 9),
262
- ].join(' ');
357
+ // Columns: the chrF++ headline with its CI, the other standard metrics
358
+ // beside it, then the diagnostics apart under their own label. The legacy
359
+ // composite is shown only when it is the requested sort.
360
+ const showLegacy = sortKey === 'composite';
361
+ const cols = [
362
+ { w: 4, head: '#', cell: (r, i) => String(i + 1) },
363
+ { w: 28, head: 'Model', cell: (r) => r.model_slug || '—' },
364
+ { w: 12, head: 'Condition', cell: (r) => r.condition || '—' },
365
+ { w: 18, head: 'chrF++ [95% CI]', group: 'standard', align: 'right', cell: (r) => fmtChrf(r) },
366
+ { w: 6, head: 'BLEU', group: 'standard', align: 'right', cell: (r) => fmtMetric(r.corpus_bleu, 1) },
367
+ { w: 6, head: 'TER', group: 'standard', align: 'right', cell: (r) => fmtMetric(r.ter, 1) },
368
+ { w: 6, head: 'COMET', group: 'standard', align: 'right', cell: (r) => fmtMetric(r.comet_score, 3) },
369
+ { w: 7, head: 'EM', group: 'diagnostic', align: 'right', cell: (r) => fmtMetric(r.exact_match_rate) },
370
+ { w: 7, head: 'FST', group: 'diagnostic', align: 'right', cell: (r) => fmtMetric(r.fst_acceptance_rate) },
371
+ ...(showLegacy ? [{ w: 9, head: 'composite', group: 'legacy', align: 'right', cell: (r) => fmtMetric(legacyComposite(r), 4) }] : []),
372
+ { w: 12, head: 'Date', cell: (r) => r.run_timestamp?.split('T')[0] || '—' },
373
+ { w: 9, head: 'Lane', cell: (r) => (rowIsRelativeOnly(r) ? 'rel-only' : 'abs') },
374
+ ];
375
+ const SEP = ' ';
376
+ const GROUP_TITLES = { standard: 'standard metrics', diagnostic: 'diagnostics', legacy: 'legacy (retired)' };
377
+ // Group line above the header: each group's title over its columns.
378
+ let groupLine = '';
379
+ for (let c = 0; c < cols.length; c++) {
380
+ const g = cols[c].group;
381
+ if (g && (c === 0 || cols[c - 1].group !== g)) {
382
+ let span = 0;
383
+ let k = c;
384
+ while (k < cols.length && cols[k].group === g) { span += cols[k].w + SEP.length; k++; }
385
+ groupLine += pad(`┌ ${GROUP_TITLES[g]}`, span - SEP.length) + SEP;
386
+ c = k - 1;
387
+ } else if (!g) {
388
+ groupLine += ' '.repeat(cols[c].w) + SEP;
389
+ }
390
+ }
391
+ const header = cols.map((c) => pad(c.head, c.w, c.align)).join(SEP);
263
392
 
393
+ output.raw(` ${groupLine.trimEnd()}`);
264
394
  output.raw(` ${header}`);
265
395
  output.raw(` ${'─'.repeat(header.length)}`);
266
396
 
267
397
  for (let i = 0; i < display.length; i++) {
268
398
  const row = display[i];
269
- const tier = TIER_LABELS[row.quality_tier?.toLowerCase()] || row.quality_tier || '—';
270
- const date = row.run_timestamp?.split('T')[0] || '—';
271
-
272
- const lane = rowIsRelativeOnly(row) ? 'rel-only' : 'abs';
273
- const line = [
274
- pad(String(i + 1), 4),
275
- pad(row.model_slug || '—', 28),
276
- pad(row.condition || '—', 12),
277
- pad(fmtMetric(row.composite_score, 4), 7, 'right'),
278
- pad(fmtMetric(row.chrf_plus_plus), 8, 'right'),
279
- pad(fmtMetric(row.exact_match_rate), 7, 'right'),
280
- pad(tier, 14),
281
- pad(date, 12),
282
- pad(lane, 9),
283
- ].join(' ');
284
-
399
+ const line = cols.map((c) => pad(c.cell(row, i), c.w, c.align)).join(SEP);
285
400
  output.raw(` ${line}`);
286
401
  }
287
402
 
288
403
  output.raw('');
289
404
  output.raw(` ${display.length} result${display.length !== 1 ? 's' : ''} shown.`);
405
+ output.raw(` Headline: chrF++ with its 95% bootstrap CI (${SCORING_STANDARD}); rows whose intervals overlap are not distinguishable.`);
406
+ output.raw(` Diagnostics (EM = exact match, FST = FST acceptance) explain a score; they never rank and are never blended into it.`);
290
407
  output.raw(` Lane: abs = absolute-quality · rel-only = relative-comparison-only`);
291
408
  output.raw(` (HIGH/MEDIUM contamination, FLORES, or unknown grade — compare within that corpus only)`);
292
- output.raw(` → Install a method: champollion leaderboard --install <rank>`);
409
+ output.raw(` → Install a method: champollion network leaderboard --install <rank>`);
293
410
  output.raw(` View full leaderboard: https://champollion.dev/leaderboard`);
294
411
  output.raw('');
295
412
 
@@ -404,7 +521,10 @@ async function _handleInstall(rows, rank, sortKey, cwd, args = {}) {
404
521
  output.raw('');
405
522
  output.raw(` Source: Rank #${rank} on leaderboard (${formatPairDisplay(pair)})`);
406
523
  output.raw(` Model: ${model}`);
407
- output.raw(` Score: ${fmtMetric(row.composite_score, 4)} composite (${row.quality_tier || 'unscored'})`);
524
+ output.raw(` Score: chrF++ ${fmtChrf(row)}${chrfCi(row) ? ' (95% CI)' : ''}`);
525
+ if (legacyComposite(row) != null) {
526
+ output.raw(` ${LEGACY_COMPOSITE_LABEL}: ${fmtMetric(legacyComposite(row), 4)} — scored before ${SCORING_STANDARD}; for the record only`);
527
+ }
408
528
  if (methodCard?.name) {
409
529
  output.raw(` Method: ${methodCard.name} (${methodCard.class || 'unclassified'})`);
410
530
  }
@@ -445,7 +565,7 @@ async function _handleInstall(rows, rank, sortKey, cwd, args = {}) {
445
565
  output.raw(' To use this method in your project:');
446
566
  output.raw(` "methodPlugin": "${safeName}"`);
447
567
  output.raw('');
448
- output.raw(' Or auto-wire it: champollion leaderboard --install ' + rank + ' --apply');
568
+ output.raw(' Or auto-wire it: champollion network leaderboard --install ' + rank + ' --apply');
449
569
  output.raw('');
450
570
  }
451
571
 
@@ -18,7 +18,8 @@
18
18
  */
19
19
 
20
20
  import { resolveConfig } from '../config.js';
21
- import { fetchAvailableModels, resolveProviderApiKey, getProviderLabel, isListableProvider, getListableProviders } from '../models.js';
21
+ import { fetchAvailableModels, resolveProviderApiKey, getProviderLabel, getProviderEnvVar, isListableProvider, getListableProviders } from '../models.js';
22
+ import { missingKeyAdvice } from '../missing-key.js';
22
23
  import { output } from '../output.js';
23
24
 
24
25
  /**
@@ -91,7 +92,13 @@ async function run(args, cwd) {
91
92
  }
92
93
  output.warn(`No API key found for ${label}.`);
93
94
  output.raw('');
94
- output.raw(` Set the environment variable or add it to .env.local.`);
95
+ // On a CI runner: the repository secret, not the shell advice (lib/missing-key.js).
96
+ const envVar = getProviderEnvVar(method);
97
+ for (const line of missingKeyAdvice({
98
+ reasons: [envVar ? `No API key (${envVar})` : ''],
99
+ setupHelp: [` Set ${envVar || 'the environment variable'} in the environment or add it to .env.local.`],
100
+ step: 'the step that runs it',
101
+ })) output.raw(line);
95
102
  output.raw(` Then re-run: champollion models --method ${method}`);
96
103
  output.raw('');
97
104
  return 1;
@@ -34,8 +34,13 @@ async function run(args, cwd) {
34
34
  }
35
35
  config.resolvedLanguages = languages;
36
36
 
37
- const pairs = resolvePairs(config);
38
- const report = formatProvenanceReport(pairs);
37
+ const pairs = resolvePairs(config, { cwd });
38
+ // A pair's fallback method translates too: report it as its own route.
39
+ const routes = new Map(pairs);
40
+ for (const [key, pc] of pairs) {
41
+ if (pc.fallback) routes.set(`${key} (fallback)`, pc.fallback);
42
+ }
43
+ const report = formatProvenanceReport(routes);
39
44
  output.raw('\n champollion — Provenance Report\n');
40
45
  output.raw(report);
41
46