champollion 0.3.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -26
- package/bin/cli.js +53 -5
- package/index.js +63 -2
- package/lib/api-key.js +17 -4
- package/lib/autofix.js +83 -36
- package/lib/bridge/method_bridge.py +15 -3
- package/lib/cards/reader.js +34 -0
- package/lib/cards/remote.js +15 -0
- package/lib/cards/search-names.js +178 -0
- package/lib/command-help.js +286 -85
- package/lib/commands/audit.js +10 -3
- package/lib/commands/card.js +583 -226
- package/lib/commands/doctor.js +54 -18
- package/lib/commands/help.js +37 -32
- package/lib/commands/init.js +1689 -87
- package/lib/commands/integrity.js +127 -40
- package/lib/commands/leaderboard.js +187 -67
- package/lib/commands/models.js +9 -2
- package/lib/commands/provenance.js +7 -2
- package/lib/commands/recommend.js +43 -14
- package/lib/commands/register-corpus.js +632 -125
- package/lib/commands/seal-corpus.js +1 -1
- package/lib/commands/status.js +564 -27
- package/lib/commands/submit.js +17 -12
- package/lib/commands/sync.js +31 -7
- package/lib/commands/tm.js +15 -9
- package/lib/commands/verify.js +27 -3
- package/lib/commands/wrap.js +63 -5
- package/lib/commands/xliff.js +135 -64
- package/lib/commercial-eligibility.js +1 -1
- package/lib/config.js +196 -14
- package/lib/content-estimate.js +96 -0
- package/lib/content-refusals.js +270 -0
- package/lib/content-review.js +372 -0
- package/lib/content-sync.js +1127 -344
- package/lib/content.js +94 -7
- package/lib/corpus-registration.mjs +194 -35
- package/lib/cost-label.js +29 -0
- package/lib/cost-report.js +726 -78
- package/lib/diff.js +38 -4
- package/lib/docusaurus-sync.js +965 -253
- package/lib/edit-distance.js +31 -0
- package/lib/fallback.js +964 -0
- package/lib/file-scope.js +106 -0
- package/lib/flatten.js +80 -3
- package/lib/flutter-locales.js +124 -0
- package/lib/format.js +266 -12
- package/lib/hash.js +146 -21
- package/lib/icu-structure.js +929 -0
- package/lib/integrity.js +223 -75
- package/lib/language-pair.js +157 -0
- package/lib/lint.js +78 -16
- package/lib/local-only-marks.js +106 -0
- package/lib/locale-layout.js +1103 -0
- package/lib/locale-state.js +571 -0
- package/lib/methods/anthropic.js +5 -0
- package/lib/methods/apertium.js +6 -3
- package/lib/methods/api.js +138 -25
- package/lib/methods/base.js +17 -0
- package/lib/methods/coaching-data.js +153 -0
- package/lib/methods/content-separator.js +43 -0
- package/lib/methods/deepl.js +1 -1
- package/lib/methods/direct-llm.js +252 -103
- package/lib/methods/external.js +146 -63
- package/lib/methods/gemini.js +1 -0
- package/lib/methods/google-translate.js +1 -0
- package/lib/methods/http-utils.js +41 -0
- package/lib/methods/libretranslate.js +7 -2
- package/lib/methods/llm-coached.js +68 -128
- package/lib/methods/llm.js +80 -31
- package/lib/methods/local.js +93 -10
- package/lib/methods/microsoft-translator.js +1 -2
- package/lib/methods/openai.js +4 -2
- package/lib/methods/openrouter-client.js +20 -19
- package/lib/methods/openrouter-pricing.js +150 -13
- package/lib/methods/prompt-methods.js +20 -0
- package/lib/methods/provider-pricing.js +42 -1
- package/lib/methods/request-capture.js +104 -0
- package/lib/methods/tilde.js +1 -1
- package/lib/methods/translated.js +1 -2
- package/lib/missing-key.js +93 -0
- package/lib/models.js +11 -0
- package/lib/name-rules.js +32 -0
- package/lib/named-keys.js +172 -0
- package/lib/no-translate.js +4 -3
- package/lib/output.js +160 -19
- package/lib/pairs.js +586 -30
- package/lib/placeholders.js +394 -0
- package/lib/plugins.js +8 -0
- package/lib/plural-gap-redo.js +109 -0
- package/lib/plurals.js +323 -0
- package/lib/po.js +1187 -0
- package/lib/public-catalogue.js +74 -0
- package/lib/recommend.js +527 -32
- package/lib/redo.js +95 -0
- package/lib/refusal-category.js +44 -0
- package/lib/registers.js +255 -11
- package/lib/repair-script.js +20 -13
- package/lib/scripts.js +6 -1
- package/lib/seal.mjs +4 -3
- package/lib/sealed-qualifier.mjs +1 -1
- package/lib/segment.js +2 -1
- package/lib/seo.js +19 -9
- package/lib/serve.js +43 -6
- package/lib/shared-output-seed.js +164 -0
- package/lib/source-contexts.js +39 -0
- package/lib/submit.mjs +57 -5
- package/lib/sync.js +2923 -474
- package/lib/terminology.js +13 -4
- package/lib/tm-evict.js +179 -0
- package/lib/tm-seed.js +5 -2
- package/lib/tm.js +818 -36
- package/lib/translate-pair.js +639 -34
- package/lib/translate.js +78 -5
- package/lib/types.js +22 -3
- package/lib/validate.js +880 -17
- package/lib/verify.js +1296 -104
- package/lib/watch.js +32 -13
- package/lib/xliff.js +44 -3
- package/package.json +1 -1
- package/shared/CORPORA-CARDS.md +2 -0
- package/shared/cards-fallback.json +1 -1
- package/shared/curated-orthography-conventions.json +26 -8
- package/shared/gettext-plural-forms.json +45 -0
- package/shared/method-registry.json +2 -0
- package/shared/metric-registry.json +96 -18
- package/shared/schemas/champollion-plugin.schema.json +4 -0
- package/shared/schemas/corpora-card.schema.json +8 -2
- package/shared/schemas/method-index-record.schema.json +67 -0
- package/shared/schemas/method-registry.schema.json +4 -0
- package/shared/schemas/metric-registry.schema.json +55 -1
- package/shared/docent/corpus.json +0 -11739
package/lib/recommend.js
CHANGED
|
@@ -20,7 +20,17 @@
|
|
|
20
20
|
* config vars like AWS_REGION never count) reconciled with
|
|
21
21
|
* lib/methods/provider-env.js so the verdict matches exactly what the
|
|
22
22
|
* method loaders read (canonical name + aliases, process.env AND .env
|
|
23
|
-
* files) — see resolveAvailability for the precedence.
|
|
23
|
+
* files) — see resolveAvailability for the precedence. Each method's
|
|
24
|
+
* COVERAGE of the pair is read from the recorded publisher lists — the
|
|
25
|
+
* language card's methodSupport (through the card adapter, the same
|
|
26
|
+
* verdict `champollion network card` prints) and
|
|
27
|
+
* shared/catalogue/method-coverage.json — and a method either list
|
|
28
|
+
* records as not covering the pair is UNSUPPORTED, never READY (see
|
|
29
|
+
* languageCoverage); a runnable method no record confirms for the pair
|
|
30
|
+
* is UNVERIFIED, so READY means "known to cover it". The card is read
|
|
31
|
+
* through the CLI's normal card tier: local in a checkout; in a
|
|
32
|
+
* packaged install a card that is not bundled is read from the
|
|
33
|
+
* per-user cache or fetched once, unless offline.
|
|
24
34
|
* 2. Curated cited results — shared/catalogue/external-results.json
|
|
25
35
|
* (bundled into the npm package via sync:shared): hand-verified
|
|
26
36
|
* published datapoints (cited ≠ reproduced), direction-exact.
|
|
@@ -54,6 +64,8 @@ import { loadMethodManifest, cliNameFor } from './method-manifest.js';
|
|
|
54
64
|
import { providerEnvNames } from './methods/provider-env.js';
|
|
55
65
|
import { getEnvOrFileVar } from './api-key.js';
|
|
56
66
|
import { LANE_RELATIVE_ONLY, laneForGrade, normalizeGrade } from './contamination-lane.js';
|
|
67
|
+
import { isMethodSupported, getLanguageCard, getCardSourceInfo } from './registers.js';
|
|
68
|
+
import { isAttributed, attributions } from './cards/reader.js';
|
|
57
69
|
|
|
58
70
|
// ---------------------------------------------------------------------------
|
|
59
71
|
// Shared-artifact resolution (mirrors method-manifest.js's loader)
|
|
@@ -146,6 +158,21 @@ function resolveAvailability(name, entry, { env = null, cwd = undefined } = {})
|
|
|
146
158
|
if (envVars.length === 0) {
|
|
147
159
|
return { status: 'ready', detail: `no credentials required${extraNote}` };
|
|
148
160
|
}
|
|
161
|
+
// A KEYLESS method needs nothing set: `local` talks to a server on this
|
|
162
|
+
// machine, Apertium to its free public API; their env vars only POINT
|
|
163
|
+
// ELSEWHERE. Reporting them as a missing key told people the privacy-
|
|
164
|
+
// preserving option needed credentials it does not. The registry says so
|
|
165
|
+
// explicitly (`keyless`) — a hosted API with a default URL still needs a key.
|
|
166
|
+
if (entry.keyless) {
|
|
167
|
+
const get0 = env === null ? (n) => getEnvOrFileVar(n, cwd) : (n) => env[n];
|
|
168
|
+
const set = envVars.find((n) => get0(n));
|
|
169
|
+
return {
|
|
170
|
+
status: 'ready',
|
|
171
|
+
detail: set
|
|
172
|
+
? `${set} is set${extraNote}`
|
|
173
|
+
: `no key needed — uses ${entry.default_base_url} unless ${envVars[0]} points elsewhere${extraNote}`,
|
|
174
|
+
};
|
|
175
|
+
}
|
|
149
176
|
const get = env === null ? (n) => getEnvOrFileVar(n, cwd) : (n) => env[n];
|
|
150
177
|
const present = envVars.filter((n) => get(n));
|
|
151
178
|
if (needAll) {
|
|
@@ -164,6 +191,86 @@ function resolveAvailability(name, entry, { env = null, cwd = undefined } = {})
|
|
|
164
191
|
return { status: 'needs-key', detail: `set ${envVars.join(' or ')}${extraNote}` };
|
|
165
192
|
}
|
|
166
193
|
|
|
194
|
+
/**
|
|
195
|
+
* What the RECORDED publisher lists say about one language for one method.
|
|
196
|
+
*
|
|
197
|
+
* Two records exist, and both are read — never one picked over the other:
|
|
198
|
+
* - the language card's methodSupport, through the card adapter
|
|
199
|
+
* (registers.js isMethodSupported — the verdict `champollion network card`
|
|
200
|
+
* prints). The atlas projects it from the vendors' own fetched lists
|
|
201
|
+
* (Apertium, Microsoft, LibreTranslate) and the curated ones (Google,
|
|
202
|
+
* DeepL); a card answers true/false only for the services it indexes;
|
|
203
|
+
* - shared/catalogue/method-coverage.json's iso6393 list.
|
|
204
|
+
* The card and recommend used to read different records: the card printed
|
|
205
|
+
* "apertium ✗ unsupported" for Plains Cree while recommend, which read only
|
|
206
|
+
* method-coverage.json (no Apertium list there), printed "READY … coverage
|
|
207
|
+
* not indexed".
|
|
208
|
+
*
|
|
209
|
+
* Every record agrees it is listed → 'listed'; every record says it is not →
|
|
210
|
+
* 'not-listed'; the records disagree → 'disputed' (both are named; this is an
|
|
211
|
+
* index, not an arbiter). "Not indexed" is said ONLY when nothing is recorded.
|
|
212
|
+
*
|
|
213
|
+
* @param {string} name - Registry method name
|
|
214
|
+
* @param {string} code - ISO 639-3 code (or upstream code with a script suffix)
|
|
215
|
+
* @param {object|null} coverageCatalogue - method-coverage.json contents
|
|
216
|
+
* @param {((code: string, method: string) => boolean|null)|null} cardSupport
|
|
217
|
+
* @returns {{coverage: 'listed'|'not-listed'|'disputed'|'unknown', note: string}}
|
|
218
|
+
*/
|
|
219
|
+
function languageCoverage(name, code, coverageCatalogue, cardSupport) {
|
|
220
|
+
const records = [];
|
|
221
|
+
const onCard = cardSupport ? cardSupport(code, name) : null;
|
|
222
|
+
if (typeof onCard === 'boolean') records.push({ source: 'language card', listed: onCard });
|
|
223
|
+
const rec = (coverageCatalogue?.methods || []).find((x) => x.key === name);
|
|
224
|
+
const list = rec && Array.isArray(rec.iso6393) ? rec.iso6393 : [];
|
|
225
|
+
if (list.length > 0) {
|
|
226
|
+
records.push({
|
|
227
|
+
source: 'method-coverage.json',
|
|
228
|
+
listed: list.includes(code) || list.includes(baseCode(code)),
|
|
229
|
+
});
|
|
230
|
+
}
|
|
231
|
+
if (records.length === 0) {
|
|
232
|
+
return { coverage: 'unknown',
|
|
233
|
+
note: rec ? 'the publisher states a language count, not a list' : 'language coverage not indexed — check the service' };
|
|
234
|
+
}
|
|
235
|
+
const per = (rs) => rs.map((r) => r.source).join(' + ');
|
|
236
|
+
const yes = records.filter((r) => r.listed);
|
|
237
|
+
const no = records.filter((r) => !r.listed);
|
|
238
|
+
if (no.length === 0) {
|
|
239
|
+
return { coverage: 'listed', note: `${code} is in its published language list (${per(yes)})` };
|
|
240
|
+
}
|
|
241
|
+
if (yes.length === 0) {
|
|
242
|
+
return { coverage: 'not-listed', note: `${code} is NOT in its published language list (${per(no)})` };
|
|
243
|
+
}
|
|
244
|
+
return { coverage: 'disputed',
|
|
245
|
+
note: `the records disagree on ${code}: listed per ${per(yes)}, NOT listed per ${per(no)} — check the service` };
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Does this method cover the PAIR? Answered from the recorded publisher lists
|
|
250
|
+
* (languageCoverage), never guessed. "READY" only ever meant "no key missing";
|
|
251
|
+
* without this a keyless public API read as ready for a language it does not
|
|
252
|
+
* translate (Round 1, hospital persona: Apertium "READY" for English→Ayta).
|
|
253
|
+
* A language a list-based service does not list cannot be translated from
|
|
254
|
+
* either, so the SOURCE side is read too: either side recorded as not listed
|
|
255
|
+
* makes the pair unsupported.
|
|
256
|
+
*
|
|
257
|
+
* @returns {{target: {coverage: string, note: string}, source: ({coverage: string, note: string}|null)}}
|
|
258
|
+
*/
|
|
259
|
+
function pairCoverage(name, entry, src, tgt, coverageCatalogue, cardSupport) {
|
|
260
|
+
if (entry.kind === 'llm-provider' || entry.paradigm === 'llm') {
|
|
261
|
+
const any = { coverage: 'any', note: 'takes text in any language — quality for this one is unmeasured' };
|
|
262
|
+
return { target: any, source: src ? any : null };
|
|
263
|
+
}
|
|
264
|
+
if (entry.kind === 'local-model') {
|
|
265
|
+
const dep = { coverage: 'unknown', note: 'depends on the model you load' };
|
|
266
|
+
return { target: dep, source: src ? dep : null };
|
|
267
|
+
}
|
|
268
|
+
return {
|
|
269
|
+
target: languageCoverage(name, tgt, coverageCatalogue, cardSupport),
|
|
270
|
+
source: src ? languageCoverage(name, src, coverageCatalogue, cardSupport) : null,
|
|
271
|
+
};
|
|
272
|
+
}
|
|
273
|
+
|
|
167
274
|
/**
|
|
168
275
|
* Tier-1 rows: every registry method with availability + lane verdicts.
|
|
169
276
|
*
|
|
@@ -173,19 +280,47 @@ function resolveAvailability(name, entry, { env = null, cwd = undefined } = {})
|
|
|
173
280
|
* dropped. Entries whose runtimes exclude 'cli' are flagged harness-only
|
|
174
281
|
* and point at mt-eval (mirrors translate.js's findHarnessOnlyEntry).
|
|
175
282
|
*
|
|
283
|
+
* With a target, `availability` is the verdict for the PAIR: a method a
|
|
284
|
+
* recorded list says does not cover either side is 'unsupported' — never
|
|
285
|
+
* 'ready' — and the credential verdict stays readable as `key_availability`.
|
|
286
|
+
* A method that is runnable on keys alone but whose coverage of either side
|
|
287
|
+
* is not confirmed (nothing recorded, a count instead of a list, or records
|
|
288
|
+
* that disagree) is 'unverified': 'ready' means known to cover the pair
|
|
289
|
+
* (Round 5, hospital persona: Apertium READY for English→Ayta with coverage
|
|
290
|
+
* "not indexed"). It sorts right after 'ready' — runnable, but check first.
|
|
291
|
+
*
|
|
176
292
|
* @param {string} useContext
|
|
177
293
|
* @param {object} [opts]
|
|
178
294
|
* @param {object|null} [opts.manifest] - Fixture manifest for tests
|
|
179
295
|
* @param {Object<string,string>|null} [opts.env] - Fixture env for tests
|
|
180
296
|
* @param {string} [opts.cwd]
|
|
297
|
+
* @param {string|null} [opts.src] - Source language (pair coverage)
|
|
298
|
+
* @param {string|null} [opts.tgt] - Target language (pair coverage)
|
|
299
|
+
* @param {object|null} [opts.coverage] - Fixture method-coverage.json
|
|
300
|
+
* @param {Function|null} [opts.cardSupport] - Fixture card lookup
|
|
301
|
+
* (code, method) → true/false/null; omitted = the card adapter
|
|
302
|
+
* (isMethodSupported); null = no card records
|
|
181
303
|
* @returns {object[]}
|
|
182
304
|
*/
|
|
183
|
-
function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd = undefined
|
|
305
|
+
function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd = undefined,
|
|
306
|
+
src = null, tgt = null, coverage = undefined, cardSupport = undefined } = {}) {
|
|
184
307
|
const m = manifest !== undefined ? manifest : loadMethodManifest();
|
|
185
308
|
if (!m || !m.entries) return [];
|
|
309
|
+
const coverageCatalogue = tgt ? (coverage !== undefined ? coverage : loadCatalogueJson('method-coverage.json')) : null;
|
|
310
|
+
const cardLookup = cardSupport !== undefined ? cardSupport : isMethodSupported;
|
|
186
311
|
const rows = [];
|
|
187
312
|
for (const [name, entry] of Object.entries(m.entries)) {
|
|
188
313
|
const avail = resolveAvailability(name, entry, { env, cwd });
|
|
314
|
+
const cov = tgt ? pairCoverage(name, entry, src, tgt, coverageCatalogue, cardLookup) : null;
|
|
315
|
+
const notCovered = cov
|
|
316
|
+
? [cov.source, cov.target].filter((c) => c && c.coverage === 'not-listed')
|
|
317
|
+
: [];
|
|
318
|
+
const unconfirmed = cov
|
|
319
|
+
? [cov.source, cov.target].filter((c) => c && (c.coverage === 'unknown' || c.coverage === 'disputed'))
|
|
320
|
+
: [];
|
|
321
|
+
const availability = notCovered.length > 0 ? 'unsupported'
|
|
322
|
+
: avail.status === 'ready' && unconfirmed.length > 0 ? 'unverified'
|
|
323
|
+
: avail.status;
|
|
189
324
|
const runtimes = entry.runtimes || ['harness', 'cli'];
|
|
190
325
|
const harnessOnly = Array.isArray(entry.runtimes) && !entry.runtimes.includes('cli');
|
|
191
326
|
let laneOk = true;
|
|
@@ -206,14 +341,22 @@ function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd
|
|
|
206
341
|
runtime_note: harnessOnly
|
|
207
342
|
? `harness-only — no CLI adapter; run it with: mt-eval run --method ${name}`
|
|
208
343
|
: null,
|
|
209
|
-
availability
|
|
210
|
-
availability_detail:
|
|
344
|
+
availability,
|
|
345
|
+
availability_detail: notCovered.length > 0
|
|
346
|
+
? notCovered.map((c) => c.note).join('; ')
|
|
347
|
+
: avail.detail,
|
|
348
|
+
key_availability: avail.status,
|
|
349
|
+
key_availability_detail: avail.detail,
|
|
211
350
|
lane_ok: laneOk,
|
|
212
351
|
lane_note: laneNote,
|
|
213
352
|
cost_note: entry.cost_note ?? null,
|
|
353
|
+
...(cov ? { target_coverage: cov.target.coverage, target_coverage_note: cov.target.note } : {}),
|
|
354
|
+
...(cov && cov.source
|
|
355
|
+
? { source_coverage: cov.source.coverage, source_coverage_note: cov.source.note }
|
|
356
|
+
: {}),
|
|
214
357
|
});
|
|
215
358
|
}
|
|
216
|
-
const order = { 'ready': 0, 'needs-key':
|
|
359
|
+
const order = { 'ready': 0, 'unverified': 1, 'needs-key': 2, 'local-setup': 3, 'unsupported': 4 };
|
|
217
360
|
rows.sort((a, b) =>
|
|
218
361
|
(a.lane_ok === b.lane_ok ? 0 : a.lane_ok ? -1 : 1)
|
|
219
362
|
|| ((order[a.availability] ?? 9) - (order[b.availability] ?? 9))
|
|
@@ -221,6 +364,120 @@ function dispatchableMethods(useContext, { manifest = undefined, env = null, cwd
|
|
|
221
364
|
return rows;
|
|
222
365
|
}
|
|
223
366
|
|
|
367
|
+
// ---------------------------------------------------------------------------
|
|
368
|
+
// Tier 1b — open models a model card declares for the target
|
|
369
|
+
// ---------------------------------------------------------------------------
|
|
370
|
+
|
|
371
|
+
/**
|
|
372
|
+
* Weight formats the harness's local-model engine does not load from a
|
|
373
|
+
* Hugging Face id: it runs a transformers checkpoint by id, or a CTranslate2
|
|
374
|
+
* conversion only from a DIRECTORY on this machine (it picks that backend by
|
|
375
|
+
* the directory's model.bin) — so a quantized or ONNX export (GGUF, AWQ,
|
|
376
|
+
* GPTQ, MLX, ONNX), an adapter (LoRA) or a CTranslate2 conversion on the Hub
|
|
377
|
+
* (`…-ct2`, `…-ct2-int8`: transformers cannot read its model.bin) named on
|
|
378
|
+
* the card is listed as declared (`not_loadable`) but never offered as
|
|
379
|
+
* something to run by id. Round 11: NLLB/MADLAD cards name many `-ct2` repos.
|
|
380
|
+
*/
|
|
381
|
+
const NOT_LOADABLE_BY_LOCAL_MODEL = /gguf|lora|awq|gptq|mlx|onnx|(?:^|[-_./])ct2(?:[-_.]|$)/i;
|
|
382
|
+
|
|
383
|
+
/** How many declared models are offered to try, in the card's own order. */
|
|
384
|
+
const DECLARED_CANDIDATE_LIMIT = 3;
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* Open models whose OWN model card declares the target language, that the
|
|
388
|
+
* harness's local-model engine can load — what to try when nothing is
|
|
389
|
+
* measured. ONE selection for every surface: `champollion network recommend`
|
|
390
|
+
* renders it, and the MCP language_overview reads it from this payload (Round
|
|
391
|
+
* 11: the overview suggested OmniTranslate for Plains Cree from its own copy
|
|
392
|
+
* of this rule while `recommend eng crk` named no model at all).
|
|
393
|
+
*
|
|
394
|
+
* Read from the language card's methodSupportEvidence (the atlas's
|
|
395
|
+
* methodSupport claims, through the CLI's card tier — getLanguageCard runs
|
|
396
|
+
* normalizeCard at the load site): every named claim that is not a service
|
|
397
|
+
* listing. A claim is the model publisher's statement, never a measurement;
|
|
398
|
+
* which of them is any good is the leaderboard's question. Candidates are the
|
|
399
|
+
* Hugging Face ids among them in a format local-model loads, the first
|
|
400
|
+
* DECLARED_CANDIDATE_LIMIT in the card's order (alphabetical by id — not a
|
|
401
|
+
* ranking). The card names a capped sample when there are many claims
|
|
402
|
+
* (`declared_total` is the full count).
|
|
403
|
+
*
|
|
404
|
+
* @param {string} code - Target ISO 639-3 code
|
|
405
|
+
* @param {object} [opts]
|
|
406
|
+
* @param {(code: string) => object|null} [opts.getCard] - Card lookup
|
|
407
|
+
* (default: the CLI card tier). Injectable for tests.
|
|
408
|
+
* @param {number} [opts.limit]
|
|
409
|
+
* @returns {{declared_total: number, listed: number, loadable: number,
|
|
410
|
+
* candidates: object[], not_loadable: string[], problem: string|null}}
|
|
411
|
+
*/
|
|
412
|
+
function declaredModelCandidates(code, { getCard = getLanguageCard, limit = DECLARED_CANDIDATE_LIMIT } = {}) {
|
|
413
|
+
const none = { declared_total: 0, listed: 0, loadable: 0, candidates: [], not_loadable: [] };
|
|
414
|
+
const card = getCard(code);
|
|
415
|
+
if (!card) {
|
|
416
|
+
const packaged = getCard === getLanguageCard && getCardSourceInfo().mode === 'packaged';
|
|
417
|
+
return {
|
|
418
|
+
...none,
|
|
419
|
+
problem: packaged
|
|
420
|
+
? `no language card for '${code}' in this install (not bundled or cached, and not fetchable now)`
|
|
421
|
+
: `no language card for '${code}'`,
|
|
422
|
+
};
|
|
423
|
+
}
|
|
424
|
+
const ev = card.methodSupportEvidence && typeof card.methodSupportEvidence === 'object'
|
|
425
|
+
? card.methodSupportEvidence : null;
|
|
426
|
+
const claims = (Array.isArray(ev?.named) ? ev.named : [])
|
|
427
|
+
.filter((n) => n && n.value !== 'service' && typeof n.variant === 'string' && n.variant);
|
|
428
|
+
if (claims.length === 0) {
|
|
429
|
+
return { ...none, problem: `the language card for '${code}' records no model that declares it` };
|
|
430
|
+
}
|
|
431
|
+
const hf = claims.filter((n) => n.variant.startsWith('hf:'));
|
|
432
|
+
const loadable = hf.filter((n) => !NOT_LOADABLE_BY_LOCAL_MODEL.test(n.variant));
|
|
433
|
+
return {
|
|
434
|
+
declared_total: Number.isInteger(ev.total) ? ev.total : claims.length,
|
|
435
|
+
listed: claims.length,
|
|
436
|
+
// Hugging Face ids named on the card in a format local-model loads (the
|
|
437
|
+
// candidates are the first `limit` of them).
|
|
438
|
+
loadable: loadable.length,
|
|
439
|
+
candidates: loadable.slice(0, limit).map((n) => ({
|
|
440
|
+
id: n.variant.slice('hf:'.length),
|
|
441
|
+
claim: n.confidence ?? n.value ?? null,
|
|
442
|
+
source: n.source ?? null,
|
|
443
|
+
})),
|
|
444
|
+
not_loadable: hf.filter((n) => NOT_LOADABLE_BY_LOCAL_MODEL.test(n.variant)).map((n) => n.variant.slice('hf:'.length)),
|
|
445
|
+
problem: null,
|
|
446
|
+
};
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
/**
|
|
450
|
+
* The declared-model section of a recommendation: the candidates, each joined
|
|
451
|
+
* to what can run it (the registry's local-model engine — its availability
|
|
452
|
+
* and lane, from the same rows the method list shows) and to the pair's
|
|
453
|
+
* published evidence (an exact model-id match in the curated or bulk rows:
|
|
454
|
+
* "runnable, no published evidence" is said only when no row names it).
|
|
455
|
+
*
|
|
456
|
+
* @returns {object}
|
|
457
|
+
*/
|
|
458
|
+
function declaredModelsSection(tgt, methods, curatedRows, bulkRows, { declared = undefined } = {}) {
|
|
459
|
+
const found = declared !== undefined ? declared : declaredModelCandidates(tgt);
|
|
460
|
+
const engine = methods.find((m) => m.kind === 'local-model') || null;
|
|
461
|
+
const evidenced = new Set([...curatedRows, ...bulkRows]
|
|
462
|
+
.flatMap((r) => [r.model, r.method_ref]).filter(Boolean).map((s) => String(s).toLowerCase()));
|
|
463
|
+
return {
|
|
464
|
+
...found,
|
|
465
|
+
// The engine that loads them (null: this registry has none — then they
|
|
466
|
+
// are listed as declared, never as runnable).
|
|
467
|
+
engine: engine
|
|
468
|
+
? { method: engine.method, availability: engine.availability, availability_detail: engine.availability_detail,
|
|
469
|
+
harness_only: engine.harness_only, lane_ok: engine.lane_ok, lane_note: engine.lane_note }
|
|
470
|
+
: null,
|
|
471
|
+
candidates: found.candidates.map((c) => ({
|
|
472
|
+
...c,
|
|
473
|
+
method: engine ? engine.method : null,
|
|
474
|
+
runnable: Boolean(engine) && engine.lane_ok,
|
|
475
|
+
published_evidence: evidenced.has(c.id.toLowerCase()),
|
|
476
|
+
...(engine ? { run: `mt-eval run --method ${engine.method} --model ${c.id} --corpus <your test file>` } : {}),
|
|
477
|
+
})),
|
|
478
|
+
};
|
|
479
|
+
}
|
|
480
|
+
|
|
224
481
|
// ---------------------------------------------------------------------------
|
|
225
482
|
// Tier 2 — curated cited results (direction-exact)
|
|
226
483
|
// ---------------------------------------------------------------------------
|
|
@@ -340,6 +597,79 @@ function bulkEvidence(src, tgt, { index = undefined, maxRows = 8 } = {}) {
|
|
|
340
597
|
// Tier 4 — metric-reliability evidence (which metric to BELIEVE for the target)
|
|
341
598
|
// ---------------------------------------------------------------------------
|
|
342
599
|
|
|
600
|
+
/**
|
|
601
|
+
* Every cited claim the target's language card makes about its family.
|
|
602
|
+
*
|
|
603
|
+
* Twin of card_family_claims in arena/mt_eval_harness/recommend.py — same
|
|
604
|
+
* shapes, same order of preference, same messages. Read through the CLI's
|
|
605
|
+
* card tier (registers.js getLanguageCard: normalizeCard at the load site,
|
|
606
|
+
* the per-user cache / one-time fetch in a packaged install) and the
|
|
607
|
+
* reader's isAttributed()/attributions() — never a bare card read:
|
|
608
|
+
* `classification.family` is an attribution envelope wherever Glottolog and
|
|
609
|
+
* WALS disagree, and a disagreement stays a disagreement here — nothing is
|
|
610
|
+
* elected. Three shapes, in this order:
|
|
611
|
+
* 1. the envelope (atlas card) → every {value, source} it carries;
|
|
612
|
+
* 2. the published projection's flat family + `familyAttributions` list;
|
|
613
|
+
* 3. a flat family, its source read from `_fieldSources`.
|
|
614
|
+
*
|
|
615
|
+
* @param {string} code
|
|
616
|
+
* @param {object} [opts]
|
|
617
|
+
* @param {(code: string) => object|null} [opts.getCard] - Card lookup
|
|
618
|
+
* (default: the CLI card tier). Injectable for tests.
|
|
619
|
+
* @returns {{claims: {value: string, source: string|null}[], problem: string|null}}
|
|
620
|
+
* `problem` is a plain reason when no claim could be read (no card, no
|
|
621
|
+
* family recorded), else null.
|
|
622
|
+
*/
|
|
623
|
+
function cardFamilyClaims(code, { getCard = getLanguageCard } = {}) {
|
|
624
|
+
const card = getCard(code);
|
|
625
|
+
if (!card) {
|
|
626
|
+
// A packaged install holds a core set; a card that is neither bundled,
|
|
627
|
+
// cached nor fetchable right now is "not here", not "does not exist".
|
|
628
|
+
const packaged = getCard === getLanguageCard && getCardSourceInfo().mode === 'packaged';
|
|
629
|
+
return {
|
|
630
|
+
claims: [],
|
|
631
|
+
problem: packaged
|
|
632
|
+
? `no language card for '${code}' in this install (not bundled or cached, and not fetchable now)`
|
|
633
|
+
: `no language card for '${code}'`,
|
|
634
|
+
};
|
|
635
|
+
}
|
|
636
|
+
const cls = card.classification && typeof card.classification === 'object'
|
|
637
|
+
&& !Array.isArray(card.classification) ? card.classification : {};
|
|
638
|
+
const fam = cls.family;
|
|
639
|
+
let claims;
|
|
640
|
+
if (isAttributed(fam)) {
|
|
641
|
+
claims = attributions(fam);
|
|
642
|
+
} else if (Array.isArray(cls.familyAttributions) && cls.familyAttributions.length > 0) {
|
|
643
|
+
// The published projection: a flat family plus its attribution list.
|
|
644
|
+
claims = cls.familyAttributions.filter((c) => c && typeof c === 'object');
|
|
645
|
+
} else if (fam) {
|
|
646
|
+
const stamped = (card._fieldSources || {})['classification.family'];
|
|
647
|
+
const source = Array.isArray(stamped)
|
|
648
|
+
? stamped.filter((s) => typeof s === 'string').join(', ')
|
|
649
|
+
: stamped;
|
|
650
|
+
claims = [{ value: fam, source: source || null }];
|
|
651
|
+
} else {
|
|
652
|
+
claims = [];
|
|
653
|
+
}
|
|
654
|
+
claims = claims
|
|
655
|
+
.filter((c) => typeof c?.value === 'string' && c.value)
|
|
656
|
+
.map((c) => ({ value: c.value, source: c.source ?? null }));
|
|
657
|
+
if (claims.length === 0) {
|
|
658
|
+
return { claims: [], problem: `the language card for '${code}' records no family` };
|
|
659
|
+
}
|
|
660
|
+
return { claims, problem: null };
|
|
661
|
+
}
|
|
662
|
+
|
|
663
|
+
/** 'Uralic (glottolog-v5.3, wals-v2020.5)' — every claim, grouped by value. */
|
|
664
|
+
function claimsText(claims) {
|
|
665
|
+
const byValue = new Map();
|
|
666
|
+
for (const c of claims) {
|
|
667
|
+
if (!byValue.has(c.value)) byValue.set(c.value, []);
|
|
668
|
+
byValue.get(c.value).push(c.source || 'source not recorded');
|
|
669
|
+
}
|
|
670
|
+
return [...byValue].map(([v, srcs]) => `${v} (${srcs.join(', ')})`).join('; ');
|
|
671
|
+
}
|
|
672
|
+
|
|
343
673
|
/**
|
|
344
674
|
* Per-target-family metric↔human correlation evidence (workstream B3).
|
|
345
675
|
*
|
|
@@ -350,14 +680,31 @@ function bulkEvidence(src, tgt, { index = undefined, maxRows = 8 } = {}) {
|
|
|
350
680
|
* Metrics-task human judgments (wmt19–wmt25); methodology spec:
|
|
351
681
|
* https://champollion.dev/docs/network/specifications/metric-reliability
|
|
352
682
|
*
|
|
353
|
-
*
|
|
354
|
-
*
|
|
683
|
+
* A target WMT never judged is looked up by FAMILY: its language card's
|
|
684
|
+
* family claims (`familyClaims`, default cardFamilyClaims) are matched
|
|
685
|
+
* EXACTLY against the index's family roll-up names — no renaming between
|
|
686
|
+
* classifications, so a Glottolog "Atlantic-Congo" claim does not match a
|
|
687
|
+
* "Niger-Congo" roll-up, while WALS's "Niger-Congo" claim on the same card
|
|
688
|
+
* does (the Python twin's handling, verbatim). Exactly one evidenced family
|
|
689
|
+
* → that roll-up, with the claims that put the target there, any source
|
|
690
|
+
* disagreement, and the transfer caveat. Sources naming two different
|
|
691
|
+
* evidenced families → no pick, UNMEASURED. (This lookup used to stop at
|
|
692
|
+
* the judged languages while its note claimed "directly or via its
|
|
693
|
+
* family", so Northern Sami was told no family evidence covered it while
|
|
694
|
+
* Uralic — its own family — had evidence.)
|
|
695
|
+
*
|
|
696
|
+
* Fail-honest: an absent index, or a target neither judged nor in an
|
|
697
|
+
* evidenced family, yields an explicit note that says what was consulted —
|
|
698
|
+
* never a silent skip, never borrowed numbers.
|
|
355
699
|
*
|
|
356
700
|
* @param {string} tgt
|
|
357
|
-
* @param {object|null} [reliability] - Fixture for tests
|
|
701
|
+
* @param {object|null} [reliability] - Fixture for tests (null = absent)
|
|
702
|
+
* @param {object} [opts]
|
|
703
|
+
* @param {(code: string) => {claims: object[], problem: string|null}} [opts.familyClaims]
|
|
704
|
+
* Family lookup (default: cardFamilyClaims). Injectable for tests.
|
|
358
705
|
* @returns {{section: object|null, notes: string[]}}
|
|
359
706
|
*/
|
|
360
|
-
function metricReliabilityEvidence(tgt, reliability = undefined) {
|
|
707
|
+
function metricReliabilityEvidence(tgt, reliability = undefined, { familyClaims = null } = {}) {
|
|
361
708
|
const rel = reliability !== undefined
|
|
362
709
|
? reliability
|
|
363
710
|
: loadCatalogueJson('metric-reliability.json');
|
|
@@ -372,6 +719,7 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
|
|
|
372
719
|
};
|
|
373
720
|
}
|
|
374
721
|
const languages = rel.languages || {};
|
|
722
|
+
const families = rel.families || {};
|
|
375
723
|
let code = null;
|
|
376
724
|
let info = null;
|
|
377
725
|
for (const key of Object.keys(languages).sort()) {
|
|
@@ -382,19 +730,59 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
|
|
|
382
730
|
break;
|
|
383
731
|
}
|
|
384
732
|
}
|
|
733
|
+
const notes = [];
|
|
734
|
+
let familyBasis = null;
|
|
735
|
+
let family;
|
|
385
736
|
if (info === null) {
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
737
|
+
const { claims = [], problem = null } = (familyClaims || cardFamilyClaims)(tgt) || {};
|
|
738
|
+
const evidenced = [...new Set(claims.map((c) => c.value)
|
|
739
|
+
.filter((v) => Object.hasOwn(families, v)))].sort();
|
|
740
|
+
const unmeasuredTail = ' — metric choice for this language is UNMEASURED. Treat '
|
|
741
|
+
+ 'every metric as unvalidated there; prefer metrics with '
|
|
742
|
+
+ 'morphology-robust behaviour and validate locally where possible.';
|
|
743
|
+
if (problem) {
|
|
744
|
+
return {
|
|
745
|
+
section: null,
|
|
746
|
+
notes: [`No WMT human-judgment meta-evaluation covers target '${tgt}' `
|
|
747
|
+
+ `directly, and its family could not be checked (${problem})${unmeasuredTail}`],
|
|
748
|
+
};
|
|
749
|
+
}
|
|
750
|
+
if (evidenced.length === 0) {
|
|
751
|
+
return {
|
|
752
|
+
section: null,
|
|
753
|
+
notes: [`No WMT human-judgment meta-evaluation covers target '${tgt}' `
|
|
754
|
+
+ `directly or via its family (per its language card: ${claimsText(claims)} `
|
|
755
|
+
+ `— no WMT-judged target language in that family)${unmeasuredTail}`],
|
|
756
|
+
};
|
|
757
|
+
}
|
|
758
|
+
if (evidenced.length > 1) {
|
|
759
|
+
return {
|
|
760
|
+
section: null,
|
|
761
|
+
notes: [`No WMT human-judgment meta-evaluation covers target '${tgt}' `
|
|
762
|
+
+ 'directly, and its language card\'s family sources disagree '
|
|
763
|
+
+ 'between families that each have evidence '
|
|
764
|
+
+ `(${claimsText(claims)}) — Champollion does not pick between `
|
|
765
|
+
+ `sources${unmeasuredTail}`],
|
|
766
|
+
};
|
|
767
|
+
}
|
|
768
|
+
family = evidenced[0];
|
|
769
|
+
familyBasis = {
|
|
770
|
+
via: 'language-card',
|
|
771
|
+
claims,
|
|
772
|
+
matched_sources: claims.filter((c) => c.value === family).map((c) => c.source),
|
|
393
773
|
};
|
|
774
|
+
notes.push(`'${tgt}' was never a WMT-judged target; its language card `
|
|
775
|
+
+ `classifies it as ${claimsText(claims.filter((c) => c.value === family))}, `
|
|
776
|
+
+ `so the ${family} family roll-up is the closest evidence.`);
|
|
777
|
+
if (claims.some((c) => c.value !== family)) {
|
|
778
|
+
notes.push(`The card's family sources disagree (${claimsText(claims)}); `
|
|
779
|
+
+ `only '${family}' names a family this index rolls up, so the `
|
|
780
|
+
+ 'evidence shown rests on that classification alone.');
|
|
781
|
+
}
|
|
782
|
+
} else {
|
|
783
|
+
family = info.family || 'Unclassified';
|
|
394
784
|
}
|
|
395
|
-
const
|
|
396
|
-
const family = info.family || 'Unclassified';
|
|
397
|
-
const famBlock = (rel.families || {})[family] || {};
|
|
785
|
+
const famBlock = families[family] || {};
|
|
398
786
|
const metricsOut = [];
|
|
399
787
|
for (const [metricId, levels] of Object.entries(famBlock.metrics || {}).sort()) {
|
|
400
788
|
const sysE = levels.sys || {};
|
|
@@ -412,7 +800,7 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
|
|
|
412
800
|
(b.sys_pearson ?? -2) - (a.sys_pearson ?? -2)
|
|
413
801
|
|| a.metric.localeCompare(b.metric));
|
|
414
802
|
const exactPairs = [...new Set((rel.cells || [])
|
|
415
|
-
.filter((c) => c.tgt === code && c.preferred)
|
|
803
|
+
.filter((c) => code !== null && c.tgt === code && c.preferred)
|
|
416
804
|
.map((c) => c.pair))].sort();
|
|
417
805
|
if (exactPairs.length === 0) {
|
|
418
806
|
notes.push(`Family-level metric evidence only: the '${family}' `
|
|
@@ -421,18 +809,20 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
|
|
|
421
809
|
+ 'measurement.');
|
|
422
810
|
}
|
|
423
811
|
if ((rel.license_lane || {}).commercial_ok === false) {
|
|
424
|
-
notes.push('Metric-reliability evidence rides a non-commercial hold
|
|
425
|
-
+ 'upstream WMT judgment data
|
|
426
|
-
+ '
|
|
812
|
+
notes.push('Metric-reliability evidence rides a non-commercial hold: the '
|
|
813
|
+
+ 'upstream WMT human-judgment data states no license, and its use '
|
|
814
|
+
+ 'beyond research has not yet been reviewed — cite it in research '
|
|
815
|
+
+ 'lanes only.');
|
|
427
816
|
}
|
|
428
817
|
return {
|
|
429
818
|
section: {
|
|
430
819
|
target_code: code,
|
|
431
|
-
target_iso639_3: info
|
|
820
|
+
target_iso639_3: info?.iso639_3 ?? null,
|
|
432
821
|
target_family: family,
|
|
433
822
|
exact_pairs_measured: exactPairs,
|
|
434
823
|
family_metrics: metricsOut,
|
|
435
824
|
provenance: rel.provenance ?? null,
|
|
825
|
+
...(familyBasis ? { family_basis: familyBasis } : {}),
|
|
436
826
|
},
|
|
437
827
|
notes,
|
|
438
828
|
};
|
|
@@ -458,22 +848,35 @@ function metricReliabilityEvidence(tgt, reliability = undefined) {
|
|
|
458
848
|
* @param {object|null} [opts.reliability] - Fixture for tests
|
|
459
849
|
* @param {Object<string,string>|null} [opts.env] - Fixture env for tests
|
|
460
850
|
* @param {string} [opts.cwd]
|
|
851
|
+
* @param {object|null} [opts.coverage] - Fixture method-coverage.json
|
|
852
|
+
* @param {Function|null} [opts.cardSupport] - Fixture card lookup
|
|
853
|
+
* (code, method) → true/false/null; omitted = the card adapter
|
|
854
|
+
* @param {Function|null} [opts.familyClaims] - Fixture family lookup for the
|
|
855
|
+
* metric-trust tier (code → {claims, problem}); omitted = cardFamilyClaims
|
|
856
|
+
* @param {object} [opts.declared] - Fixture declaredModelCandidates() result;
|
|
857
|
+
* omitted = read from the target's language card
|
|
461
858
|
* @returns {object}
|
|
462
859
|
*/
|
|
463
860
|
function recommend(src, tgt, {
|
|
464
861
|
useContext = 'non-commercial',
|
|
465
862
|
manifest = undefined,
|
|
863
|
+
coverage = undefined,
|
|
864
|
+
cardSupport = undefined,
|
|
466
865
|
curated = undefined,
|
|
467
866
|
bulk = undefined,
|
|
468
867
|
reliability = undefined,
|
|
868
|
+
familyClaims = null,
|
|
869
|
+
declared = undefined,
|
|
469
870
|
env = null,
|
|
470
871
|
cwd = undefined,
|
|
471
872
|
} = {}) {
|
|
472
|
-
const methods = dispatchableMethods(useContext,
|
|
873
|
+
const methods = dispatchableMethods(useContext,
|
|
874
|
+
{ manifest, env, cwd, src, tgt, coverage, cardSupport });
|
|
473
875
|
const { rows: curatedRows, methodsIndex: citedMethods } = curatedEvidence(src, tgt, curated);
|
|
474
876
|
const { rows: bulkRows, meta: bulkMeta } = bulkEvidence(src, tgt, { index: bulk });
|
|
877
|
+
const declaredModels = declaredModelsSection(tgt, methods, curatedRows, bulkRows, { declared });
|
|
475
878
|
const { section: reliabilitySection, notes: reliabilityNotes } =
|
|
476
|
-
metricReliabilityEvidence(tgt, reliability);
|
|
879
|
+
metricReliabilityEvidence(tgt, reliability, { familyClaims });
|
|
477
880
|
|
|
478
881
|
// Join: which cited models are dispatchable / commercially deployable?
|
|
479
882
|
const evidencedModels = [];
|
|
@@ -523,6 +926,10 @@ function recommend(src, tgt, {
|
|
|
523
926
|
pair: { source: src, target: tgt },
|
|
524
927
|
use_context: useContext,
|
|
525
928
|
runnable_methods: methods,
|
|
929
|
+
// Open models whose model card declares the target, loadable by the
|
|
930
|
+
// local-model engine: what to try when nothing is measured. A claim,
|
|
931
|
+
// never evidence (declaredModelCandidates).
|
|
932
|
+
declared_models: declaredModels,
|
|
526
933
|
curated_evidence: curatedRows,
|
|
527
934
|
bulk_evidence: bulkRows,
|
|
528
935
|
bulk_meta: bulkMeta,
|
|
@@ -536,6 +943,63 @@ function recommend(src, tgt, {
|
|
|
536
943
|
// Rendering
|
|
537
944
|
// ---------------------------------------------------------------------------
|
|
538
945
|
|
|
946
|
+
/**
|
|
947
|
+
* The declared-model section: open models whose own model card names the
|
|
948
|
+
* target, labelled by what is true of each — runnable here with no published
|
|
949
|
+
* evidence (the usual case), runnable with evidence above, excluded from the
|
|
950
|
+
* lane, or declared with nothing here that loads it. Absent data is said,
|
|
951
|
+
* never left out (a payload without the section — an older one — renders
|
|
952
|
+
* nothing).
|
|
953
|
+
*
|
|
954
|
+
* @param {object} payload
|
|
955
|
+
* @returns {string[]}
|
|
956
|
+
*/
|
|
957
|
+
function renderDeclaredModels(payload) {
|
|
958
|
+
const d = payload.declared_models;
|
|
959
|
+
if (!d) return [];
|
|
960
|
+
const tgt = payload.pair.target;
|
|
961
|
+
const out = [''];
|
|
962
|
+
if (d.problem) {
|
|
963
|
+
out.push(`Open models whose model card declares ${tgt}: none to suggest (${d.problem}).`);
|
|
964
|
+
return out;
|
|
965
|
+
}
|
|
966
|
+
out.push(`Open models whose model card declares ${tgt} (${d.declared_total} declared`
|
|
967
|
+
+ `${d.listed < d.declared_total ? `, ${d.listed} named on the card` : ''}`
|
|
968
|
+
+ ' — the publisher\'s claim, not a measurement):');
|
|
969
|
+
const e = d.engine;
|
|
970
|
+
if (d.candidates.length === 0) {
|
|
971
|
+
out.push(` none in a format ${e ? e.method : 'local-model'} loads (a transformers checkpoint on Hugging Face, or a CTranslate2 directory).`);
|
|
972
|
+
} else if (!e) {
|
|
973
|
+
out.push(' DECLARED — no method in this registry loads them:');
|
|
974
|
+
} else if (!e.lane_ok) {
|
|
975
|
+
out.push(` EXCLUDED — ${e.method}: ${e.lane_note}:`);
|
|
976
|
+
} else {
|
|
977
|
+
const label = (c) => (c.published_evidence ? 'RUNNABLE, PUBLISHED EVIDENCE ABOVE' : 'RUNNABLE, NO PUBLISHED EVIDENCE');
|
|
978
|
+
const kinds = [...new Set(d.candidates.map(label))];
|
|
979
|
+
out.push(` ${kinds.join(' / ')} — via ${e.method} (${e.harness_only ? 'harness-only; ' : ''}${e.availability_detail}),`
|
|
980
|
+
+ ' in the card\'s order, not a ranking:');
|
|
981
|
+
}
|
|
982
|
+
for (const c of d.candidates) {
|
|
983
|
+
const tag = e && e.lane_ok && c.published_evidence ? ' (published evidence above)' : '';
|
|
984
|
+
out.push(` ${c.id} [${c.claim || 'claim'}; ${c.source || 'source not recorded'}]${tag}`);
|
|
985
|
+
}
|
|
986
|
+
const more = (d.loadable ?? d.candidates.length) - d.candidates.length;
|
|
987
|
+
if (more > 0) {
|
|
988
|
+
out.push(` (+${more} more on the card in a format ${e ? e.method : 'local-model'} loads — `
|
|
989
|
+
+ `\`champollion network card ${tgt} --json\` lists every claim it names, under methodSupportEvidence)`);
|
|
990
|
+
}
|
|
991
|
+
if (d.candidates.length > 0 && e && e.lane_ok) {
|
|
992
|
+
out.push(` try one: ${d.candidates[0].run}`);
|
|
993
|
+
out.push(` (${e.method} runs seq2seq translation checkpoints — check the model card first)`);
|
|
994
|
+
}
|
|
995
|
+
if (d.not_loadable.length > 0) {
|
|
996
|
+
const shown = d.not_loadable.slice(0, 3).join(', ');
|
|
997
|
+
out.push(` not loadable by ${e ? e.method : 'local-model'} by id (an adapter, a quantized or ONNX export, or a CTranslate2 conversion): ${shown}`
|
|
998
|
+
+ `${d.not_loadable.length > 3 ? `, … (${d.not_loadable.length} in all)` : ''}`);
|
|
999
|
+
}
|
|
1000
|
+
return out;
|
|
1001
|
+
}
|
|
1002
|
+
|
|
539
1003
|
/**
|
|
540
1004
|
* Human-readable rendering of a recommend() payload (mirrors the harness's
|
|
541
1005
|
* render_text, plus CLI-name and harness-only annotations).
|
|
@@ -555,10 +1019,18 @@ function renderText(payload) {
|
|
|
555
1019
|
out.push(' (method registry not available in this install)');
|
|
556
1020
|
}
|
|
557
1021
|
const badges = {
|
|
558
|
-
'ready': 'READY
|
|
559
|
-
|
|
560
|
-
|
|
1022
|
+
'ready': 'READY ',
|
|
1023
|
+
// Runnable on keys alone, but no record confirms the pair: the "?" line
|
|
1024
|
+
// under it says why (not indexed, a count only, or records disagree).
|
|
1025
|
+
'unverified': 'UNVERIFIED ',
|
|
1026
|
+
'needs-key': 'NEEDS KEY ',
|
|
1027
|
+
'local-setup': 'LOCAL ',
|
|
1028
|
+
// A recorded publisher list says the pair is not covered: whatever the
|
|
1029
|
+
// key state, this is not runnable for this pair (the reason is the detail).
|
|
1030
|
+
'unsupported': 'UNSUPPORTED ',
|
|
561
1031
|
};
|
|
1032
|
+
const mark = (c) => (c === 'not-listed' ? '✗' : c === 'unknown' || c === 'disputed' ? '?' : '↳');
|
|
1033
|
+
const indent = ' '.repeat(14);
|
|
562
1034
|
for (const m of payload.runnable_methods) {
|
|
563
1035
|
// Show the name a champollion.config.json actually uses when it differs
|
|
564
1036
|
// from the canonical registry name (e.g. openrouter → llm).
|
|
@@ -567,16 +1039,31 @@ function renderText(payload) {
|
|
|
567
1039
|
: m.method;
|
|
568
1040
|
let line;
|
|
569
1041
|
if (!m.lane_ok) {
|
|
570
|
-
line = ` EXCLUDED
|
|
1042
|
+
line = ` EXCLUDED ${shown.padEnd(22)}${m.lane_note}`;
|
|
571
1043
|
} else {
|
|
572
|
-
const badge = badges[m.availability] || '?
|
|
1044
|
+
const badge = badges[m.availability] || '? ';
|
|
573
1045
|
line = ` ${badge}${shown.padEnd(22)}${m.availability_detail}`;
|
|
574
1046
|
if (m.license) line += ` [${m.license}]`;
|
|
575
1047
|
}
|
|
576
1048
|
out.push(line);
|
|
577
|
-
|
|
1049
|
+
// An UNSUPPORTED row already states why on its own line.
|
|
1050
|
+
if (m.lane_ok && m.availability !== 'unsupported') {
|
|
1051
|
+
if (m.target_coverage_note) {
|
|
1052
|
+
out.push(`${indent}${mark(m.target_coverage)} ${m.target_coverage_note}`);
|
|
1053
|
+
}
|
|
1054
|
+
// The source side is shown only when it adds something the target line
|
|
1055
|
+
// does not: a disagreement, or no record for the source alone.
|
|
1056
|
+
const sc = m.source_coverage;
|
|
1057
|
+
if (m.source_coverage_note && (sc === 'disputed'
|
|
1058
|
+
|| (sc === 'unknown' && m.target_coverage !== 'unknown'))) {
|
|
1059
|
+
out.push(`${indent}${mark(sc)} source ${payload.pair.source}: ${m.source_coverage_note}`);
|
|
1060
|
+
}
|
|
1061
|
+
}
|
|
1062
|
+
if (m.runtime_note) out.push(`${indent}↳ ${m.runtime_note}`);
|
|
578
1063
|
}
|
|
579
1064
|
|
|
1065
|
+
out.push(...renderDeclaredModels(payload));
|
|
1066
|
+
|
|
580
1067
|
out.push('');
|
|
581
1068
|
out.push('Published evidence for this pair (cited — never reproduced '
|
|
582
1069
|
+ 'by us; relative ordering only):');
|
|
@@ -627,6 +1114,12 @@ function renderText(payload) {
|
|
|
627
1114
|
? rel.exact_pairs_measured.join(', ')
|
|
628
1115
|
: 'none — family-level only';
|
|
629
1116
|
out.push(` directly measured pairs for this target: ${measured}`);
|
|
1117
|
+
// How the target got its family when WMT never judged it: every claim
|
|
1118
|
+
// its card makes, disagreements included (the notes say which one the
|
|
1119
|
+
// roll-up rests on).
|
|
1120
|
+
if (rel.family_basis) {
|
|
1121
|
+
out.push(` family per the target's language card: ${claimsText(rel.family_basis.claims)}`);
|
|
1122
|
+
}
|
|
630
1123
|
}
|
|
631
1124
|
|
|
632
1125
|
out.push('');
|
|
@@ -642,6 +1135,8 @@ export {
|
|
|
642
1135
|
dispatchableMethods,
|
|
643
1136
|
curatedEvidence,
|
|
644
1137
|
bulkEvidence,
|
|
1138
|
+
cardFamilyClaims,
|
|
1139
|
+
declaredModelCandidates,
|
|
645
1140
|
metricReliabilityEvidence,
|
|
646
1141
|
recommend,
|
|
647
1142
|
renderText,
|