champollion 0.3.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/README.md +41 -26
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +34 -0
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +286 -85
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +632 -125
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +15 -9
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +194 -35
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +6 -1
  100. package/lib/seal.mjs +4 -3
  101. package/lib/sealed-qualifier.mjs +1 -1
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +1 -1
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/cards-fallback.json +1 -1
  123. package/shared/curated-orthography-conventions.json +26 -8
  124. package/shared/gettext-plural-forms.json +45 -0
  125. package/shared/method-registry.json +2 -0
  126. package/shared/metric-registry.json +96 -18
  127. package/shared/schemas/champollion-plugin.schema.json +4 -0
  128. package/shared/schemas/corpora-card.schema.json +8 -2
  129. package/shared/schemas/method-index-record.schema.json +67 -0
  130. package/shared/schemas/method-registry.schema.json +4 -0
  131. package/shared/schemas/metric-registry.schema.json +55 -1
  132. package/shared/docent/corpus.json +0 -11739
package/lib/recommend.js CHANGED
@@ -20,7 +20,17 @@
20
20
  * config vars like AWS_REGION never count) reconciled with
21
21
  * lib/methods/provider-env.js so the verdict matches exactly what the
22
22
  * method loaders read (canonical name + aliases, process.env AND .env
23
- * files) — see resolveAvailability for the precedence.
23
+ * files) — see resolveAvailability for the precedence. Each method's
24
+ * COVERAGE of the pair is read from the recorded publisher lists — the
25
+ * language card's methodSupport (through the card adapter, the same
26
+ * verdict `champollion network card` prints) and
27
+ * shared/catalogue/method-coverage.json — and a method either list
28
+ * records as not covering the pair is UNSUPPORTED, never READY (see
29
+ * languageCoverage); a runnable method no record confirms for the pair
30
+ * is UNVERIFIED, so READY means "known to cover it". The card is read
31
+ * through the CLI's normal card tier: local in a checkout; in a
32
+ * packaged install a card that is not bundled is read from the
33
+ * per-user cache or fetched once, unless offline.
24
34
  * 2. Curated cited results — shared/catalogue/external-results.json
25
35
  * (bundled into the npm package via sync:shared): hand-verified
26
36
  * published datapoints (cited ≠ reproduced), direction-exact.
@@ -54,6 +64,8 @@ import { loadMethodManifest, cliNameFor } from './method-manifest.js';
54
64
  import { providerEnvNames } from './methods/provider-env.js';
55
65
  import { getEnvOrFileVar } from './api-key.js';
56
66
  import { LANE_RELATIVE_ONLY, laneForGrade, normalizeGrade } from './contamination-lane.js';
67
+ import { isMethodSupported, getLanguageCard, getCardSourceInfo } from './registers.js';
68
+ import { isAttributed, attributions } from './cards/reader.js';
57
69
 
58
70
  // ---------------------------------------------------------------------------
59
71
  // Shared-artifact resolution (mirrors method-manifest.js's loader)
@@ -146,6 +158,21 @@ function resolveAvailability(name, entry, { env = null, cwd = undefined } = {})
146
158
  if (envVars.length === 0) {
147
159
  return { status: 'ready', detail: `no credentials required${extraNote}` };
148
160
  }
161
+ // A KEYLESS method needs nothing set: `local` talks to a server on this
162
+ // machine, Apertium to its free public API; their env vars only POINT
163
+ // ELSEWHERE. Reporting them as a missing key told people the privacy-
164
+ // preserving option needed credentials it does not. The registry says so
165
+ // explicitly (`keyless`) — a hosted API with a default URL still needs a key.
166
+ if (entry.keyless) {
167
+ const get0 = env === null ? (n) => getEnvOrFileVar(n, cwd) : (n) => env[n];
168
+ const set = envVars.find((n) => get0(n));
169
+ return {
170
+ status: 'ready',
171
+ detail: set
172
+ ? `${set} is set${extraNote}`
173
+ : `no key needed — uses ${entry.default_base_url} unless ${envVars[0]} points elsewhere${extraNote}`,
174
+ };
175
+ }
149
176
  const get = env === null ? (n) => getEnvOrFileVar(n, cwd) : (n) => env[n];
150
177
  const present = envVars.filter((n) => get(n));
151
178
  if (needAll) {
@@ -164,6 +191,86 @@ function resolveAvailability(name, entry, { env = null, cwd = undefined } = {})
164
191
  return { status: 'needs-key', detail: `set ${envVars.join(' or ')}${extraNote}` };
165
192
  }
166
193
 
194
+ /**
195
+ * What the RECORDED publisher lists say about one language for one method.
196
+ *
197
+ * Two records exist, and both are read — never one picked over the other:
198
+ * - the language card's methodSupport, through the card adapter
199
+ * (registers.js isMethodSupported — the verdict `champollion network card`
200
+ * prints). The atlas projects it from the vendors' own fetched lists
201
+ * (Apertium, Microsoft, LibreTranslate) and the curated ones (Google,
202
+ * DeepL); a card answers true/false only for the services it indexes;
203
+ * - shared/catalogue/method-coverage.json's iso6393 list.
204
+ * The card and recommend used to read different records: the card printed
205
+ * "apertium ✗ unsupported" for Plains Cree while recommend, which read only
206
+ * method-coverage.json (no Apertium list there), printed "READY … coverage
207
+ * not indexed".
208
+ *
209
+ * Every record agrees it is listed → 'listed'; every record says it is not →
210
+ * 'not-listed'; the records disagree → 'disputed' (both are named; this is an
211
+ * index, not an arbiter). "Not indexed" is said ONLY when nothing is recorded.
212
+ *
213
+ * @param {string} name - Registry method name
214
+ * @param {string} code - ISO 639-3 code (or upstream code with a script suffix)
215
+ * @param {object|null} coverageCatalogue - method-coverage.json contents
216
+ * @param {((code: string, method: string) => boolean|null)|null} cardSupport
217
+ * @returns {{coverage: 'listed'|'not-listed'|'disputed'|'unknown', note: string}}
218
+ */
219
+ function languageCoverage(name, code, coverageCatalogue, cardSupport) {
220
+ const records = [];
221
+ const onCard = cardSupport ? cardSupport(code, name) : null;
222
+ if (typeof onCard === 'boolean') records.push({ source: 'language card', listed: onCard });
223
+ const rec = (coverageCatalogue?.methods || []).find((x) => x.key === name);
224
+ const list = rec && Array.isArray(rec.iso6393) ? rec.iso6393 : [];
225
+ if (list.length > 0) {
226
+ records.push({
227
+ source: 'method-coverage.json',
228
+ listed: list.includes(code) || list.includes(baseCode(code)),
229
+ });
230
+ }
231
+ if (records.length === 0) {
232
+ return { coverage: 'unknown',
233
+ note: rec ? 'the publisher states a language count, not a list' : 'language coverage not indexed — check the service' };
234
+ }
235
+ const per = (rs) => rs.map((r) => r.source).join(' + ');
236
+ const yes = records.filter((r) => r.listed);
237
+ const no = records.filter((r) => !r.listed);
238
+ if (no.length === 0) {
239
+ return { coverage: 'listed', note: `${code} is in its published language list (${per(yes)})` };
240
+ }
241
+ if (yes.length === 0) {
242
+ return { coverage: 'not-listed', note: `${code} is NOT in its published language list (${per(no)})` };
243
+ }
244
+ return { coverage: 'disputed',
245
+ note: `the records disagree on ${code}: listed per ${per(yes)}, NOT listed per ${per(no)} — check the service` };
246
+ }
247
+
248
+ /**
249
+ * Does this method cover the PAIR? Answered from the recorded publisher lists
250
+ * (languageCoverage), never guessed. "READY" only ever meant "no key missing";
251
+ * without this a keyless public API read as ready for a language it does not
252
+ * translate (Round 1, hospital persona: Apertium "READY" for English→Ayta).
253
+ * A language a list-based service does not list cannot be translated from
254
+ * either, so the SOURCE side is read too: either side recorded as not listed
255
+ * makes the pair unsupported.
256
+ *
257
+ * @returns {{target: {coverage: string, note: string}, source: ({coverage: string, note: string}|null)}}
258
+ */
259
+ function pairCoverage(name, entry, src, tgt, coverageCatalogue, cardSupport) {
260
+ if (entry.kind === 'llm-provider' || entry.paradigm === 'llm') {
261
+ const any = { coverage: 'any', note: 'takes text in any language — quality for this one is unmeasured' };
262
+ return { target: any, source: src ? any : null };
263
+ }
264
+ if (entry.kind === 'local-model') {
265
+ const dep = { coverage: 'unknown', note: 'depends on the model you load' };
266
+ return { target: dep, source: src ? dep : null };
267
+ }
268
+ return {
269
+ target: languageCoverage(name, tgt, coverageCatalogue, cardSupport),
270
+ source: src ? languageCoverage(name, src, coverageCatalogue, cardSupport) : null,
271
+ };
272
+ }
273
+
167
274
  /**
168
275
  * Tier-1 rows: every registry method with availability + lane verdicts.
169
276
  *
@@ -173,19 +280,47 @@ function resolveAvailability(name, entry, { env = null, cwd = undefined } = {})
173
280
  * dropped. Entries whose runtimes exclude 'cli' are flagged harness-only
174
281
  * and point at mt-eval (mirrors translate.js's findHarnessOnlyEntry).
175
282
  *
283
+ * With a target, `availability` is the verdict for the PAIR: a method a
284
+ * recorded list says does not cover either side is 'unsupported' — never
285
+ * 'ready' — and the credential verdict stays readable as `key_availability`.
286
+ * A method that is runnable on keys alone but whose coverage of either side
287
+ * is not confirmed (nothing recorded, a count instead of a list, or records
288
+ * that disagree) is 'unverified': 'ready' means known to cover the pair
289
+ * (Round 5, hospital persona: Apertium READY for English→Ayta with coverage
290
+ * "not indexed"). It sorts right after 'ready' — runnable, but check first.
291
+ *
176
292
  * @param {string} useContext
177
293
  * @param {object} [opts]
178
294
  * @param {object|null} [opts.manifest] - Fixture manifest for tests
179
295
  * @param {Object<string,string>|null} [opts.env] - Fixture env for tests
180
296
  * @param {string} [opts.cwd]
297
+ * @param {string|null} [opts.src] - Source language (pair coverage)
298
+ * @param {string|null} [opts.tgt] - Target language (pair coverage)
299
+ * @param {object|null} [opts.coverage] - Fixture method-coverage.json
300
+ * @param {Function|null} [opts.cardSupport] - Fixture card lookup
301
+ * (code, method) → true/false/null; omitted = the card adapter
302
+ * (isMethodSupported); null = no card records
181
303
  * @returns {object[]}
182
304
  */
183
- function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd = undefined } = {}) {
305
+ function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd = undefined,
306
+ src = null, tgt = null, coverage = undefined, cardSupport = undefined } = {}) {
184
307
  const m = manifest !== undefined ? manifest : loadMethodManifest();
185
308
  if (!m || !m.entries) return [];
309
+ const coverageCatalogue = tgt ? (coverage !== undefined ? coverage : loadCatalogueJson('method-coverage.json')) : null;
310
+ const cardLookup = cardSupport !== undefined ? cardSupport : isMethodSupported;
186
311
  const rows = [];
187
312
  for (const [name, entry] of Object.entries(m.entries)) {
188
313
  const avail = resolveAvailability(name, entry, { env, cwd });
314
+ const cov = tgt ? pairCoverage(name, entry, src, tgt, coverageCatalogue, cardLookup) : null;
315
+ const notCovered = cov
316
+ ? [cov.source, cov.target].filter((c) => c && c.coverage === 'not-listed')
317
+ : [];
318
+ const unconfirmed = cov
319
+ ? [cov.source, cov.target].filter((c) => c && (c.coverage === 'unknown' || c.coverage === 'disputed'))
320
+ : [];
321
+ const availability = notCovered.length > 0 ? 'unsupported'
322
+ : avail.status === 'ready' && unconfirmed.length > 0 ? 'unverified'
323
+ : avail.status;
189
324
  const runtimes = entry.runtimes || ['harness', 'cli'];
190
325
  const harnessOnly = Array.isArray(entry.runtimes) && !entry.runtimes.includes('cli');
191
326
  let laneOk = true;
@@ -206,14 +341,22 @@ function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd
206
341
  runtime_note: harnessOnly
207
342
  ? `harness-only — no CLI adapter; run it with: mt-eval run --method ${name}`
208
343
  : null,
209
- availability: avail.status,
210
- availability_detail: avail.detail,
344
+ availability,
345
+ availability_detail: notCovered.length > 0
346
+ ? notCovered.map((c) => c.note).join('; ')
347
+ : avail.detail,
348
+ key_availability: avail.status,
349
+ key_availability_detail: avail.detail,
211
350
  lane_ok: laneOk,
212
351
  lane_note: laneNote,
213
352
  cost_note: entry.cost_note ?? null,
353
+ ...(cov ? { target_coverage: cov.target.coverage, target_coverage_note: cov.target.note } : {}),
354
+ ...(cov && cov.source
355
+ ? { source_coverage: cov.source.coverage, source_coverage_note: cov.source.note }
356
+ : {}),
214
357
  });
215
358
  }
216
- const order = { 'ready': 0, 'needs-key': 1, 'local-setup': 2 };
359
+ const order = { 'ready': 0, 'unverified': 1, 'needs-key': 2, 'local-setup': 3, 'unsupported': 4 };
217
360
  rows.sort((a, b) =>
218
361
  (a.lane_ok === b.lane_ok ? 0 : a.lane_ok ? -1 : 1)
219
362
  || ((order[a.availability] ?? 9) - (order[b.availability] ?? 9))
@@ -221,6 +364,120 @@ function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd
221
364
  return rows;
222
365
  }
223
366
 
367
+ // ---------------------------------------------------------------------------
368
+ // Tier 1b — open models a model card declares for the target
369
+ // ---------------------------------------------------------------------------
370
+
371
+ /**
372
+ * Weight formats the harness's local-model engine does not load from a
373
+ * Hugging Face id: it runs a transformers checkpoint by id, or a CTranslate2
374
+ * conversion only from a DIRECTORY on this machine (it picks that backend by
375
+ * the directory's model.bin) — so a quantized or ONNX export (GGUF, AWQ,
376
+ * GPTQ, MLX, ONNX), an adapter (LoRA) or a CTranslate2 conversion on the Hub
377
+ * (`…-ct2`, `…-ct2-int8`: transformers cannot read its model.bin) named on
378
+ * the card is listed as declared (`not_loadable`) but never offered as
379
+ * something to run by id. Round 11: NLLB/MADLAD cards name many `-ct2` repos.
380
+ */
381
+ const NOT_LOADABLE_BY_LOCAL_MODEL = /gguf|lora|awq|gptq|mlx|onnx|(?:^|[-_./])ct2(?:[-_.]|$)/i;
382
+
383
+ /** How many declared models are offered to try, in the card's own order. */
384
+ const DECLARED_CANDIDATE_LIMIT = 3;
385
+
386
+ /**
387
+ * Open models whose OWN model card declares the target language, that the
388
+ * harness's local-model engine can load — what to try when nothing is
389
+ * measured. ONE selection for every surface: `champollion network recommend`
390
+ * renders it, and the MCP language_overview reads it from this payload (Round
391
+ * 11: the overview suggested OmniTranslate for Plains Cree from its own copy
392
+ * of this rule while `recommend eng crk` named no model at all).
393
+ *
394
+ * Read from the language card's methodSupportEvidence (the atlas's
395
+ * methodSupport claims, through the CLI's card tier — getLanguageCard runs
396
+ * normalizeCard at the load site): every named claim that is not a service
397
+ * listing. A claim is the model publisher's statement, never a measurement;
398
+ * which of them is any good is the leaderboard's question. Candidates are the
399
+ * Hugging Face ids among them in a format local-model loads, the first
400
+ * DECLARED_CANDIDATE_LIMIT in the card's order (alphabetical by id — not a
401
+ * ranking). The card names a capped sample when there are many claims
402
+ * (`declared_total` is the full count).
403
+ *
404
+ * @param {string} code - Target ISO 639-3 code
405
+ * @param {object} [opts]
406
+ * @param {(code: string) => object|null} [opts.getCard] - Card lookup
407
+ * (default: the CLI card tier). Injectable for tests.
408
+ * @param {number} [opts.limit]
409
+ * @returns {{declared_total: number, listed: number, loadable: number,
410
+ * candidates: object[], not_loadable: string[], problem: string|null}}
411
+ */
412
+ function declaredModelCandidates(code, { getCard = getLanguageCard, limit = DECLARED_CANDIDATE_LIMIT } = {}) {
413
+ const none = { declared_total: 0, listed: 0, loadable: 0, candidates: [], not_loadable: [] };
414
+ const card = getCard(code);
415
+ if (!card) {
416
+ const packaged = getCard === getLanguageCard && getCardSourceInfo().mode === 'packaged';
417
+ return {
418
+ ...none,
419
+ problem: packaged
420
+ ? `no language card for '${code}' in this install (not bundled or cached, and not fetchable now)`
421
+ : `no language card for '${code}'`,
422
+ };
423
+ }
424
+ const ev = card.methodSupportEvidence && typeof card.methodSupportEvidence === 'object'
425
+ ? card.methodSupportEvidence : null;
426
+ const claims = (Array.isArray(ev?.named) ? ev.named : [])
427
+ .filter((n) => n && n.value !== 'service' && typeof n.variant === 'string' && n.variant);
428
+ if (claims.length === 0) {
429
+ return { ...none, problem: `the language card for '${code}' records no model that declares it` };
430
+ }
431
+ const hf = claims.filter((n) => n.variant.startsWith('hf:'));
432
+ const loadable = hf.filter((n) => !NOT_LOADABLE_BY_LOCAL_MODEL.test(n.variant));
433
+ return {
434
+ declared_total: Number.isInteger(ev.total) ? ev.total : claims.length,
435
+ listed: claims.length,
436
+ // Hugging Face ids named on the card in a format local-model loads (the
437
+ // candidates are the first `limit` of them).
438
+ loadable: loadable.length,
439
+ candidates: loadable.slice(0, limit).map((n) => ({
440
+ id: n.variant.slice('hf:'.length),
441
+ claim: n.confidence ?? n.value ?? null,
442
+ source: n.source ?? null,
443
+ })),
444
+ not_loadable: hf.filter((n) => NOT_LOADABLE_BY_LOCAL_MODEL.test(n.variant)).map((n) => n.variant.slice('hf:'.length)),
445
+ problem: null,
446
+ };
447
+ }
448
+
449
+ /**
450
+ * The declared-model section of a recommendation: the candidates, each joined
451
+ * to what can run it (the registry's local-model engine — its availability
452
+ * and lane, from the same rows the method list shows) and to the pair's
453
+ * published evidence (an exact model-id match in the curated or bulk rows:
454
+ * "runnable, no published evidence" is said only when no row names it).
455
+ *
456
+ * @returns {object}
457
+ */
458
+ function declaredModelsSection(tgt, methods, curatedRows, bulkRows, { declared = undefined } = {}) {
459
+ const found = declared !== undefined ? declared : declaredModelCandidates(tgt);
460
+ const engine = methods.find((m) => m.kind === 'local-model') || null;
461
+ const evidenced = new Set([...curatedRows, ...bulkRows]
462
+ .flatMap((r) => [r.model, r.method_ref]).filter(Boolean).map((s) => String(s).toLowerCase()));
463
+ return {
464
+ ...found,
465
+ // The engine that loads them (null: this registry has none — then they
466
+ // are listed as declared, never as runnable).
467
+ engine: engine
468
+ ? { method: engine.method, availability: engine.availability, availability_detail: engine.availability_detail,
469
+ harness_only: engine.harness_only, lane_ok: engine.lane_ok, lane_note: engine.lane_note }
470
+ : null,
471
+ candidates: found.candidates.map((c) => ({
472
+ ...c,
473
+ method: engine ? engine.method : null,
474
+ runnable: Boolean(engine) && engine.lane_ok,
475
+ published_evidence: evidenced.has(c.id.toLowerCase()),
476
+ ...(engine ? { run: `mt-eval run --method ${engine.method} --model ${c.id} --corpus <your test file>` } : {}),
477
+ })),
478
+ };
479
+ }
480
+
224
481
  // ---------------------------------------------------------------------------
225
482
  // Tier 2 — curated cited results (direction-exact)
226
483
  // ---------------------------------------------------------------------------
@@ -340,6 +597,79 @@ function bulkEvidence(src, tgt, { index = undefined, maxRows = 8 } = {}) {
340
597
  // Tier 4 — metric-reliability evidence (which metric to BELIEVE for the target)
341
598
  // ---------------------------------------------------------------------------
342
599
 
600
+ /**
601
+ * Every cited claim the target's language card makes about its family.
602
+ *
603
+ * Twin of card_family_claims in arena/mt_eval_harness/recommend.py — same
604
+ * shapes, same order of preference, same messages. Read through the CLI's
605
+ * card tier (registers.js getLanguageCard: normalizeCard at the load site,
606
+ * the per-user cache / one-time fetch in a packaged install) and the
607
+ * reader's isAttributed()/attributions() — never a bare card read:
608
+ * `classification.family` is an attribution envelope wherever Glottolog and
609
+ * WALS disagree, and a disagreement stays a disagreement here — nothing is
610
+ * elected. Three shapes, in this order:
611
+ * 1. the envelope (atlas card) → every {value, source} it carries;
612
+ * 2. the published projection's flat family + `familyAttributions` list;
613
+ * 3. a flat family, its source read from `_fieldSources`.
614
+ *
615
+ * @param {string} code
616
+ * @param {object} [opts]
617
+ * @param {(code: string) => object|null} [opts.getCard] - Card lookup
618
+ * (default: the CLI card tier). Injectable for tests.
619
+ * @returns {{claims: {value: string, source: string|null}[], problem: string|null}}
620
+ * `problem` is a plain reason when no claim could be read (no card, no
621
+ * family recorded), else null.
622
+ */
623
+ function cardFamilyClaims(code, { getCard = getLanguageCard } = {}) {
624
+ const card = getCard(code);
625
+ if (!card) {
626
+ // A packaged install holds a core set; a card that is neither bundled,
627
+ // cached nor fetchable right now is "not here", not "does not exist".
628
+ const packaged = getCard === getLanguageCard && getCardSourceInfo().mode === 'packaged';
629
+ return {
630
+ claims: [],
631
+ problem: packaged
632
+ ? `no language card for '${code}' in this install (not bundled or cached, and not fetchable now)`
633
+ : `no language card for '${code}'`,
634
+ };
635
+ }
636
+ const cls = card.classification && typeof card.classification === 'object'
637
+ && !Array.isArray(card.classification) ? card.classification : {};
638
+ const fam = cls.family;
639
+ let claims;
640
+ if (isAttributed(fam)) {
641
+ claims = attributions(fam);
642
+ } else if (Array.isArray(cls.familyAttributions) && cls.familyAttributions.length > 0) {
643
+ // The published projection: a flat family plus its attribution list.
644
+ claims = cls.familyAttributions.filter((c) => c && typeof c === 'object');
645
+ } else if (fam) {
646
+ const stamped = (card._fieldSources || {})['classification.family'];
647
+ const source = Array.isArray(stamped)
648
+ ? stamped.filter((s) => typeof s === 'string').join(', ')
649
+ : stamped;
650
+ claims = [{ value: fam, source: source || null }];
651
+ } else {
652
+ claims = [];
653
+ }
654
+ claims = claims
655
+ .filter((c) => typeof c?.value === 'string' && c.value)
656
+ .map((c) => ({ value: c.value, source: c.source ?? null }));
657
+ if (claims.length === 0) {
658
+ return { claims: [], problem: `the language card for '${code}' records no family` };
659
+ }
660
+ return { claims, problem: null };
661
+ }
662
+
663
+ /** 'Uralic (glottolog-v5.3, wals-v2020.5)' — every claim, grouped by value. */
664
+ function claimsText(claims) {
665
+ const byValue = new Map();
666
+ for (const c of claims) {
667
+ if (!byValue.has(c.value)) byValue.set(c.value, []);
668
+ byValue.get(c.value).push(c.source || 'source not recorded');
669
+ }
670
+ return [...byValue].map(([v, srcs]) => `${v} (${srcs.join(', ')})`).join('; ');
671
+ }
672
+
343
673
  /**
344
674
  * Per-target-family metric↔human correlation evidence (workstream B3).
345
675
  *
@@ -350,14 +680,31 @@ function bulkEvidence(src, tgt, { index = undefined, maxRows = 8 } = {}) {
350
680
  * Metrics-task human judgments (wmt19–wmt25); methodology spec:
351
681
  * https://champollion.dev/docs/network/specifications/metric-reliability
352
682
  *
353
- * Fail-honest: an absent index, or a target no WMT campaign ever judged,
354
- * yields an explicit note — never a silent skip, never borrowed numbers.
683
+ * A target WMT never judged is looked up by FAMILY: its language card's
684
+ * family claims (`familyClaims`, default cardFamilyClaims) are matched
685
+ * EXACTLY against the index's family roll-up names — no renaming between
686
+ * classifications, so a Glottolog "Atlantic-Congo" claim does not match a
687
+ * "Niger-Congo" roll-up, while WALS's "Niger-Congo" claim on the same card
688
+ * does (the Python twin's handling, verbatim). Exactly one evidenced family
689
+ * → that roll-up, with the claims that put the target there, any source
690
+ * disagreement, and the transfer caveat. Sources naming two different
691
+ * evidenced families → no pick, UNMEASURED. (This lookup used to stop at
692
+ * the judged languages while its note claimed "directly or via its
693
+ * family", so Northern Sami was told no family evidence covered it while
694
+ * Uralic — its own family — had evidence.)
695
+ *
696
+ * Fail-honest: an absent index, or a target neither judged nor in an
697
+ * evidenced family, yields an explicit note that says what was consulted —
698
+ * never a silent skip, never borrowed numbers.
355
699
  *
356
700
  * @param {string} tgt
357
- * @param {object|null} [reliability] - Fixture for tests
701
+ * @param {object|null} [reliability] - Fixture for tests (null = absent)
702
+ * @param {object} [opts]
703
+ * @param {(code: string) => {claims: object[], problem: string|null}} [opts.familyClaims]
704
+ * Family lookup (default: cardFamilyClaims). Injectable for tests.
358
705
  * @returns {{section: object|null, notes: string[]}}
359
706
  */
360
- function metricReliabilityEvidence(tgt, reliability = undefined) {
707
+ function metricReliabilityEvidence(tgt, reliability = undefined, { familyClaims = null } = {}) {
361
708
  const rel = reliability !== undefined
362
709
  ? reliability
363
710
  : loadCatalogueJson('metric-reliability.json');
@@ -372,6 +719,7 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
372
719
  };
373
720
  }
374
721
  const languages = rel.languages || {};
722
+ const families = rel.families || {};
375
723
  let code = null;
376
724
  let info = null;
377
725
  for (const key of Object.keys(languages).sort()) {
@@ -382,19 +730,59 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
382
730
  break;
383
731
  }
384
732
  }
733
+ const notes = [];
734
+ let familyBasis = null;
735
+ let family;
385
736
  if (info === null) {
386
- return {
387
- section: null,
388
- notes: [`No WMT human-judgment meta-evaluation covers target '${tgt}' `
389
- + '(directly or via its family) — metric choice for this language is '
390
- + 'UNMEASURED. Treat every metric as unvalidated there; prefer '
391
- + 'metrics with morphology-robust behaviour and validate locally '
392
- + 'where possible.'],
737
+ const { claims = [], problem = null } = (familyClaims || cardFamilyClaims)(tgt) || {};
738
+ const evidenced = [...new Set(claims.map((c) => c.value)
739
+ .filter((v) => Object.hasOwn(families, v)))].sort();
740
+ const unmeasuredTail = ' — metric choice for this language is UNMEASURED. Treat '
741
+ + 'every metric as unvalidated there; prefer metrics with '
742
+ + 'morphology-robust behaviour and validate locally where possible.';
743
+ if (problem) {
744
+ return {
745
+ section: null,
746
+ notes: [`No WMT human-judgment meta-evaluation covers target '${tgt}' `
747
+ + `directly, and its family could not be checked (${problem})${unmeasuredTail}`],
748
+ };
749
+ }
750
+ if (evidenced.length === 0) {
751
+ return {
752
+ section: null,
753
+ notes: [`No WMT human-judgment meta-evaluation covers target '${tgt}' `
754
+ + `directly or via its family (per its language card: ${claimsText(claims)} `
755
+ + `— no WMT-judged target language in that family)${unmeasuredTail}`],
756
+ };
757
+ }
758
+ if (evidenced.length > 1) {
759
+ return {
760
+ section: null,
761
+ notes: [`No WMT human-judgment meta-evaluation covers target '${tgt}' `
762
+ + 'directly, and its language card\'s family sources disagree '
763
+ + 'between families that each have evidence '
764
+ + `(${claimsText(claims)}) — Champollion does not pick between `
765
+ + `sources${unmeasuredTail}`],
766
+ };
767
+ }
768
+ family = evidenced[0];
769
+ familyBasis = {
770
+ via: 'language-card',
771
+ claims,
772
+ matched_sources: claims.filter((c) => c.value === family).map((c) => c.source),
393
773
  };
774
+ notes.push(`'${tgt}' was never a WMT-judged target; its language card `
775
+ + `classifies it as ${claimsText(claims.filter((c) => c.value === family))}, `
776
+ + `so the ${family} family roll-up is the closest evidence.`);
777
+ if (claims.some((c) => c.value !== family)) {
778
+ notes.push(`The card's family sources disagree (${claimsText(claims)}); `
779
+ + `only '${family}' names a family this index rolls up, so the `
780
+ + 'evidence shown rests on that classification alone.');
781
+ }
782
+ } else {
783
+ family = info.family || 'Unclassified';
394
784
  }
395
- const notes = [];
396
- const family = info.family || 'Unclassified';
397
- const famBlock = (rel.families || {})[family] || {};
785
+ const famBlock = families[family] || {};
398
786
  const metricsOut = [];
399
787
  for (const [metricId, levels] of Object.entries(famBlock.metrics || {}).sort()) {
400
788
  const sysE = levels.sys || {};
@@ -412,7 +800,7 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
412
800
  (b.sys_pearson ?? -2) - (a.sys_pearson ?? -2)
413
801
  || a.metric.localeCompare(b.metric));
414
802
  const exactPairs = [...new Set((rel.cells || [])
415
- .filter((c) => c.tgt === code && c.preferred)
803
+ .filter((c) => code !== null && c.tgt === code && c.preferred)
416
804
  .map((c) => c.pair))].sort();
417
805
  if (exactPairs.length === 0) {
418
806
  notes.push(`Family-level metric evidence only: the '${family}' `
@@ -421,18 +809,20 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
421
809
  + 'measurement.');
422
810
  }
423
811
  if ((rel.license_lane || {}).commercial_ok === false) {
424
- notes.push('Metric-reliability evidence rides a non-commercial hold (the '
425
- + 'upstream WMT judgment data license is unstated, founder review '
426
- + 'pending) — cite it in research lanes only.');
812
+ notes.push('Metric-reliability evidence rides a non-commercial hold: the '
813
+ + 'upstream WMT human-judgment data states no license, and its use '
814
+ + 'beyond research has not yet been reviewed — cite it in research '
815
+ + 'lanes only.');
427
816
  }
428
817
  return {
429
818
  section: {
430
819
  target_code: code,
431
- target_iso639_3: info.iso639_3 ?? null,
820
+ target_iso639_3: info?.iso639_3 ?? null,
432
821
  target_family: family,
433
822
  exact_pairs_measured: exactPairs,
434
823
  family_metrics: metricsOut,
435
824
  provenance: rel.provenance ?? null,
825
+ ...(familyBasis ? { family_basis: familyBasis } : {}),
436
826
  },
437
827
  notes,
438
828
  };
@@ -458,22 +848,35 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
458
848
  * @param {object|null} [opts.reliability] - Fixture for tests
459
849
  * @param {Object<string,string>|null} [opts.env] - Fixture env for tests
460
850
  * @param {string} [opts.cwd]
851
+ * @param {object|null} [opts.coverage] - Fixture method-coverage.json
852
+ * @param {Function|null} [opts.cardSupport] - Fixture card lookup
853
+ * (code, method) → true/false/null; omitted = the card adapter
854
+ * @param {Function|null} [opts.familyClaims] - Fixture family lookup for the
855
+ * metric-trust tier (code → {claims, problem}); omitted = cardFamilyClaims
856
+ * @param {object} [opts.declared] - Fixture declaredModelCandidates() result;
857
+ * omitted = read from the target's language card
461
858
  * @returns {object}
462
859
  */
463
860
  function recommend(src, tgt, {
464
861
  useContext = 'non-commercial',
465
862
  manifest = undefined,
863
+ coverage = undefined,
864
+ cardSupport = undefined,
466
865
  curated = undefined,
467
866
  bulk = undefined,
468
867
  reliability = undefined,
868
+ familyClaims = null,
869
+ declared = undefined,
469
870
  env = null,
470
871
  cwd = undefined,
471
872
  } = {}) {
472
- const methods = dispatchableMethods(useContext, { manifest, env, cwd });
873
+ const methods = dispatchableMethods(useContext,
874
+ { manifest, env, cwd, src, tgt, coverage, cardSupport });
473
875
  const { rows: curatedRows, methodsIndex: citedMethods } = curatedEvidence(src, tgt, curated);
474
876
  const { rows: bulkRows, meta: bulkMeta } = bulkEvidence(src, tgt, { index: bulk });
877
+ const declaredModels = declaredModelsSection(tgt, methods, curatedRows, bulkRows, { declared });
475
878
  const { section: reliabilitySection, notes: reliabilityNotes } =
476
- metricReliabilityEvidence(tgt, reliability);
879
+ metricReliabilityEvidence(tgt, reliability, { familyClaims });
477
880
 
478
881
  // Join: which cited models are dispatchable / commercially deployable?
479
882
  const evidencedModels = [];
@@ -523,6 +926,10 @@ function recommend(src, tgt, {
523
926
  pair: { source: src, target: tgt },
524
927
  use_context: useContext,
525
928
  runnable_methods: methods,
929
+ // Open models whose model card declares the target, loadable by the
930
+ // local-model engine: what to try when nothing is measured. A claim,
931
+ // never evidence (declaredModelCandidates).
932
+ declared_models: declaredModels,
526
933
  curated_evidence: curatedRows,
527
934
  bulk_evidence: bulkRows,
528
935
  bulk_meta: bulkMeta,
@@ -536,6 +943,63 @@ function recommend(src, tgt, {
536
943
  // Rendering
537
944
  // ---------------------------------------------------------------------------
538
945
 
946
+ /**
947
+ * The declared-model section: open models whose own model card names the
948
+ * target, labelled by what is true of each — runnable here with no published
949
+ * evidence (the usual case), runnable with evidence above, excluded from the
950
+ * lane, or declared with nothing here that loads it. Absent data is said,
951
+ * never left out (a payload without the section — an older one — renders
952
+ * nothing).
953
+ *
954
+ * @param {object} payload
955
+ * @returns {string[]}
956
+ */
957
+ function renderDeclaredModels(payload) {
958
+ const d = payload.declared_models;
959
+ if (!d) return [];
960
+ const tgt = payload.pair.target;
961
+ const out = [''];
962
+ if (d.problem) {
963
+ out.push(`Open models whose model card declares ${tgt}: none to suggest (${d.problem}).`);
964
+ return out;
965
+ }
966
+ out.push(`Open models whose model card declares ${tgt} (${d.declared_total} declared`
967
+ + `${d.listed < d.declared_total ? `, ${d.listed} named on the card` : ''}`
968
+ + ' — the publisher\'s claim, not a measurement):');
969
+ const e = d.engine;
970
+ if (d.candidates.length === 0) {
971
+ out.push(` none in a format ${e ? e.method : 'local-model'} loads (a transformers checkpoint on Hugging Face, or a CTranslate2 directory).`);
972
+ } else if (!e) {
973
+ out.push(' DECLARED — no method in this registry loads them:');
974
+ } else if (!e.lane_ok) {
975
+ out.push(` EXCLUDED — ${e.method}: ${e.lane_note}:`);
976
+ } else {
977
+ const label = (c) => (c.published_evidence ? 'RUNNABLE, PUBLISHED EVIDENCE ABOVE' : 'RUNNABLE, NO PUBLISHED EVIDENCE');
978
+ const kinds = [...new Set(d.candidates.map(label))];
979
+ out.push(` ${kinds.join(' / ')} — via ${e.method} (${e.harness_only ? 'harness-only; ' : ''}${e.availability_detail}),`
980
+ + ' in the card\'s order, not a ranking:');
981
+ }
982
+ for (const c of d.candidates) {
983
+ const tag = e && e.lane_ok && c.published_evidence ? ' (published evidence above)' : '';
984
+ out.push(` ${c.id} [${c.claim || 'claim'}; ${c.source || 'source not recorded'}]${tag}`);
985
+ }
986
+ const more = (d.loadable ?? d.candidates.length) - d.candidates.length;
987
+ if (more > 0) {
988
+ out.push(` (+${more} more on the card in a format ${e ? e.method : 'local-model'} loads — `
989
+ + `\`champollion network card ${tgt} --json\` lists every claim it names, under methodSupportEvidence)`);
990
+ }
991
+ if (d.candidates.length > 0 && e && e.lane_ok) {
992
+ out.push(` try one: ${d.candidates[0].run}`);
993
+ out.push(` (${e.method} runs seq2seq translation checkpoints — check the model card first)`);
994
+ }
995
+ if (d.not_loadable.length > 0) {
996
+ const shown = d.not_loadable.slice(0, 3).join(', ');
997
+ out.push(` not loadable by ${e ? e.method : 'local-model'} by id (an adapter, a quantized or ONNX export, or a CTranslate2 conversion): ${shown}`
998
+ + `${d.not_loadable.length > 3 ? `, … (${d.not_loadable.length} in all)` : ''}`);
999
+ }
1000
+ return out;
1001
+ }
1002
+
539
1003
  /**
540
1004
  * Human-readable rendering of a recommend() payload (mirrors the harness's
541
1005
  * render_text, plus CLI-name and harness-only annotations).
@@ -555,10 +1019,18 @@ function renderText(payload) {
555
1019
  out.push(' (method registry not available in this install)');
556
1020
  }
557
1021
  const badges = {
558
- 'ready': 'READY ',
559
- 'needs-key': 'NEEDS KEY ',
560
- 'local-setup': 'LOCAL ',
1022
+ 'ready': 'READY ',
1023
+ // Runnable on keys alone, but no record confirms the pair: the "?" line
1024
+ // under it says why (not indexed, a count only, or records disagree).
1025
+ 'unverified': 'UNVERIFIED ',
1026
+ 'needs-key': 'NEEDS KEY ',
1027
+ 'local-setup': 'LOCAL ',
1028
+ // A recorded publisher list says the pair is not covered: whatever the
1029
+ // key state, this is not runnable for this pair (the reason is the detail).
1030
+ 'unsupported': 'UNSUPPORTED ',
561
1031
  };
1032
+ const mark = (c) => (c === 'not-listed' ? '✗' : c === 'unknown' || c === 'disputed' ? '?' : '↳');
1033
+ const indent = ' '.repeat(14);
562
1034
  for (const m of payload.runnable_methods) {
563
1035
  // Show the name a champollion.config.json actually uses when it differs
564
1036
  // from the canonical registry name (e.g. openrouter → llm).
@@ -567,16 +1039,31 @@ function renderText(payload) {
567
1039
  : m.method;
568
1040
  let line;
569
1041
  if (!m.lane_ok) {
570
- line = ` EXCLUDED ${shown.padEnd(22)}${m.lane_note}`;
1042
+ line = ` EXCLUDED ${shown.padEnd(22)}${m.lane_note}`;
571
1043
  } else {
572
- const badge = badges[m.availability] || '? ';
1044
+ const badge = badges[m.availability] || '? ';
573
1045
  line = ` ${badge}${shown.padEnd(22)}${m.availability_detail}`;
574
1046
  if (m.license) line += ` [${m.license}]`;
575
1047
  }
576
1048
  out.push(line);
577
- if (m.runtime_note) out.push(` ↳ ${m.runtime_note}`);
1049
+ // An UNSUPPORTED row already states why on its own line.
1050
+ if (m.lane_ok && m.availability !== 'unsupported') {
1051
+ if (m.target_coverage_note) {
1052
+ out.push(`${indent}${mark(m.target_coverage)} ${m.target_coverage_note}`);
1053
+ }
1054
+ // The source side is shown only when it adds something the target line
1055
+ // does not: a disagreement, or no record for the source alone.
1056
+ const sc = m.source_coverage;
1057
+ if (m.source_coverage_note && (sc === 'disputed'
1058
+ || (sc === 'unknown' && m.target_coverage !== 'unknown'))) {
1059
+ out.push(`${indent}${mark(sc)} source ${payload.pair.source}: ${m.source_coverage_note}`);
1060
+ }
1061
+ }
1062
+ if (m.runtime_note) out.push(`${indent}↳ ${m.runtime_note}`);
578
1063
  }
579
1064
 
1065
+ out.push(...renderDeclaredModels(payload));
1066
+
580
1067
  out.push('');
581
1068
  out.push('Published evidence for this pair (cited — never reproduced '
582
1069
  + 'by us; relative ordering only):');
@@ -627,6 +1114,12 @@ function renderText(payload) {
627
1114
  ? rel.exact_pairs_measured.join(', ')
628
1115
  : 'none — family-level only';
629
1116
  out.push(` directly measured pairs for this target: ${measured}`);
1117
+ // How the target got its family when WMT never judged it: every claim
1118
+ // its card makes, disagreements included (the notes say which one the
1119
+ // roll-up rests on).
1120
+ if (rel.family_basis) {
1121
+ out.push(` family per the target's language card: ${claimsText(rel.family_basis.claims)}`);
1122
+ }
630
1123
  }
631
1124
 
632
1125
  out.push('');
@@ -642,6 +1135,8 @@ export {
642
1135
  dispatchableMethods,
643
1136
  curatedEvidence,
644
1137
  bulkEvidence,
1138
+ cardFamilyClaims,
1139
+ declaredModelCandidates,
645
1140
  metricReliabilityEvidence,
646
1141
  recommend,
647
1142
  renderText,