champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,620 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "SSOT for METRIC IDENTITY, consumed by the Python arena harness (arena/mt_eval_harness) and available to the JS CLI/website. One metric has up to four coordinated names: the canonical_id (this file's key = the run-card scores key), the Python MetricPlugin 'name' attribute that computes it (plugin_name), the language-card evalMetrics key that declares it (card_key), and the denormalized run_cards DB/leaderboard column (db_column). This file maps them so no consumer ever guesses. Definitions come from the scoring spec (cli/website/docs/network/specifications/scoring.md §2) and scoring.py; arena/tests/test_metric_registry_ssot.py FAILS if scoring.py weight tables or publish.py run-card score keys drift from this file. verifier_reproducible means: deterministically re-derivable by the verifier from the sha-pinned corpus + stored entries (+ card-pinned tools like the FST, or the pinned neural model under the fail-closed contract). in_composite means: currently enters at least one composite profile in scoring.PROFILE_REGISTRY when available (declared-but-INACTIVE metrics are false). Confidence-interval columns (chrf_ci_lower/...), boolean flags (has_references, morph_in_composite), model-id fields (comet_model, qe_model, metricx_model) and counts (corpus_size, total, evaluated, errors) are metadata ABOUT metrics, not metrics — they are deliberately not entries here.",
|
|
3
|
+
"version": 1,
|
|
4
|
+
"entries": {
|
|
5
|
+
"exact_match_rate": {
|
|
6
|
+
"category": "surface",
|
|
7
|
+
"status": "implemented",
|
|
8
|
+
"display_name": "Exact Match",
|
|
9
|
+
"plugin_name": null,
|
|
10
|
+
"card_key": null,
|
|
11
|
+
"db_column": "exact_match_rate",
|
|
12
|
+
"scale": "0.0-1.0",
|
|
13
|
+
"direction": "higher",
|
|
14
|
+
"level": "both",
|
|
15
|
+
"in_composite": true,
|
|
16
|
+
"verifier_reproducible": true,
|
|
17
|
+
"notes": "Harness core (tester.py), no plugin. Binary predicted == reference; corpus rate = matches/total."
|
|
18
|
+
},
|
|
19
|
+
"equivalent_match_rate": {
|
|
20
|
+
"category": "surface",
|
|
21
|
+
"status": "partial",
|
|
22
|
+
"display_name": "Equivalent Match",
|
|
23
|
+
"plugin_name": "crk_linter",
|
|
24
|
+
"card_key": "lyss-eq",
|
|
25
|
+
"db_column": "equivalent_match_rate",
|
|
26
|
+
"scale": "0.0-1.0",
|
|
27
|
+
"direction": "higher",
|
|
28
|
+
"level": "both",
|
|
29
|
+
"in_composite": true,
|
|
30
|
+
"verifier_reproducible": true,
|
|
31
|
+
"notes": "CRK proxy today (champollion_lyss.crk.metrics.CrkLinterMetric, declared on the crk card as 'lyss-eq'). publish.py discovers it by the is_equivalence_linter aggregate flag, not by name — any language's linter plugin feeds this same canonical metric."
|
|
32
|
+
},
|
|
33
|
+
"chrf_plus_plus": {
|
|
34
|
+
"category": "surface",
|
|
35
|
+
"status": "implemented",
|
|
36
|
+
"display_name": "chrF++",
|
|
37
|
+
"plugin_name": null,
|
|
38
|
+
"card_key": null,
|
|
39
|
+
"db_column": "chrf_plus_plus",
|
|
40
|
+
"scale": "0-100",
|
|
41
|
+
"direction": "higher",
|
|
42
|
+
"level": "both",
|
|
43
|
+
"in_composite": true,
|
|
44
|
+
"verifier_reproducible": true,
|
|
45
|
+
"notes": "sacrebleu (word_order=2), harness core. Normalized /100 for the composite (scoring.NORMALIZATIONS)."
|
|
46
|
+
},
|
|
47
|
+
"bleu": {
|
|
48
|
+
"category": "surface",
|
|
49
|
+
"status": "implemented",
|
|
50
|
+
"display_name": "BLEU",
|
|
51
|
+
"plugin_name": null,
|
|
52
|
+
"card_key": null,
|
|
53
|
+
"db_column": "corpus_bleu",
|
|
54
|
+
"scale": "0-100",
|
|
55
|
+
"direction": "higher",
|
|
56
|
+
"level": "corpus",
|
|
57
|
+
"in_composite": false,
|
|
58
|
+
"verifier_reproducible": true,
|
|
59
|
+
"notes": "sacrebleu, harness core. Reported for MT-literature compatibility, never composited. Run-card key is top-level 'corpus_bleu' (not scores.bleu) — see _proposed_renames."
|
|
60
|
+
},
|
|
61
|
+
"ter": {
|
|
62
|
+
"category": "surface",
|
|
63
|
+
"status": "implemented",
|
|
64
|
+
"display_name": "Translation Edit Rate",
|
|
65
|
+
"plugin_name": null,
|
|
66
|
+
"card_key": null,
|
|
67
|
+
"db_column": "ter",
|
|
68
|
+
"scale": "0-inf",
|
|
69
|
+
"direction": "lower",
|
|
70
|
+
"level": "both",
|
|
71
|
+
"in_composite": false,
|
|
72
|
+
"verifier_reproducible": true,
|
|
73
|
+
"notes": "sacrebleu corpus_ter. Excluded from composite (correlates with chrF++)."
|
|
74
|
+
},
|
|
75
|
+
"length_ratio": {
|
|
76
|
+
"category": "surface",
|
|
77
|
+
"status": "implemented",
|
|
78
|
+
"display_name": "Length Ratio",
|
|
79
|
+
"plugin_name": null,
|
|
80
|
+
"card_key": null,
|
|
81
|
+
"db_column": "length_ratio",
|
|
82
|
+
"scale": "0-inf (ideal 1.0)",
|
|
83
|
+
"direction": "neutral",
|
|
84
|
+
"level": "both",
|
|
85
|
+
"in_composite": false,
|
|
86
|
+
"verifier_reproducible": true,
|
|
87
|
+
"notes": "Diagnostic: len(predicted)/len(reference); <0.5 truncation, >2.0 inflation."
|
|
88
|
+
},
|
|
89
|
+
"fst_acceptance_rate": {
|
|
90
|
+
"category": "structural",
|
|
91
|
+
"status": "implemented",
|
|
92
|
+
"display_name": "FST Acceptance",
|
|
93
|
+
"plugin_name": "giellalt_fst_validity",
|
|
94
|
+
"card_key": null,
|
|
95
|
+
"db_column": "fst_acceptance_rate",
|
|
96
|
+
"scale": "0.0-1.0",
|
|
97
|
+
"direction": "higher",
|
|
98
|
+
"level": "both",
|
|
99
|
+
"in_composite": true,
|
|
100
|
+
"verifier_reproducible": true,
|
|
101
|
+
"notes": "GiellaLTFSTMetric (plugins/giellalt_fst.py). Declared per-language via the card's resources.fsts install metadata, not evalMetrics. Plugin aggregate key is 'avg_fst_validity' (legacy standalone reports used 'acceptance_rate')."
|
|
102
|
+
},
|
|
103
|
+
"morphological_accuracy": {
|
|
104
|
+
"category": "structural",
|
|
105
|
+
"status": "implemented",
|
|
106
|
+
"display_name": "Morphological Accuracy",
|
|
107
|
+
"plugin_name": "giellalt_fst_validity",
|
|
108
|
+
"card_key": null,
|
|
109
|
+
"db_column": "morphological_accuracy",
|
|
110
|
+
"scale": "0.0-1.0",
|
|
111
|
+
"direction": "higher",
|
|
112
|
+
"level": "both",
|
|
113
|
+
"in_composite": true,
|
|
114
|
+
"verifier_reproducible": true,
|
|
115
|
+
"notes": "FST-derived, lemma-matched (same plugin as fst_acceptance_rate). Enters the fst-coverage composite only when morph_coverage >= MORPH_COVERAGE_FLOOR (0.25); verifier re-derives via recompute_corpus_morph."
|
|
116
|
+
},
|
|
117
|
+
"morph_coverage": {
|
|
118
|
+
"category": "structural",
|
|
119
|
+
"status": "implemented",
|
|
120
|
+
"display_name": "Morphological Coverage",
|
|
121
|
+
"plugin_name": "giellalt_fst_validity",
|
|
122
|
+
"card_key": null,
|
|
123
|
+
"db_column": "morph_coverage",
|
|
124
|
+
"scale": "0.0-1.0",
|
|
125
|
+
"direction": "neutral",
|
|
126
|
+
"level": "corpus",
|
|
127
|
+
"in_composite": false,
|
|
128
|
+
"verifier_reproducible": true,
|
|
129
|
+
"notes": "Companion disclosure for morphological_accuracy (fraction of analyzable predicted words lemma-matched); a coverage figure, not a quality score."
|
|
130
|
+
},
|
|
131
|
+
"orthographic_accuracy": {
|
|
132
|
+
"category": "structural",
|
|
133
|
+
"status": "planned",
|
|
134
|
+
"display_name": "Orthographic Accuracy",
|
|
135
|
+
"plugin_name": null,
|
|
136
|
+
"card_key": null,
|
|
137
|
+
"db_column": null,
|
|
138
|
+
"scale": "0.0-1.0",
|
|
139
|
+
"direction": "higher",
|
|
140
|
+
"level": "both",
|
|
141
|
+
"in_composite": false,
|
|
142
|
+
"verifier_reproducible": false,
|
|
143
|
+
"notes": "Carries a DECLARED 0.05 weight in the surface-only profile but is in scoring.INACTIVE_METRICS (needs per-language orthographic rule sets) — never scores yet."
|
|
144
|
+
},
|
|
145
|
+
"gloss_word_accuracy": {
|
|
146
|
+
"category": "structural",
|
|
147
|
+
"status": "implemented",
|
|
148
|
+
"display_name": "IGT Gloss Word Accuracy",
|
|
149
|
+
"plugin_name": "igt_gloss",
|
|
150
|
+
"card_key": null,
|
|
151
|
+
"db_column": null,
|
|
152
|
+
"scale": "0.0-1.0",
|
|
153
|
+
"direction": "higher",
|
|
154
|
+
"level": "both",
|
|
155
|
+
"in_composite": false,
|
|
156
|
+
"verifier_reproducible": true,
|
|
157
|
+
"notes": "SIGMORPHON-2023 glossing shared-task accuracy (plugins/igt_gloss.py, IGTGlossMetric). LANGUAGE-NEUTRAL Layer-1 lane metric for the interlinear-gloss TASK — not MT scoring; shipped 2026-07-04 (COMPUTEL). Headline is word-level positional accuracy; the plugin also emits companion keys gloss_morpheme_accuracy (+ _average_/_overall_), and stem/gram precision/recall/F1 (gloss_classes). ACTIVATION-PENDING: any language activates it by declaring module mt_eval_harness.plugins.igt_gloss on its card's evalMetrics — no card does yet, so it lives in plugin_metrics JSONB with no run_cards column. Deterministic (positional gold comparison), verifier-re-derivable."
|
|
158
|
+
},
|
|
159
|
+
"seg_micro_f1": {
|
|
160
|
+
"category": "structural",
|
|
161
|
+
"status": "implemented",
|
|
162
|
+
"display_name": "Morpheme Segmentation F1 (micro)",
|
|
163
|
+
"plugin_name": "morph_segmentation",
|
|
164
|
+
"card_key": null,
|
|
165
|
+
"db_column": null,
|
|
166
|
+
"scale": "0.0-1.0",
|
|
167
|
+
"direction": "higher",
|
|
168
|
+
"level": "both",
|
|
169
|
+
"in_composite": false,
|
|
170
|
+
"verifier_reproducible": true,
|
|
171
|
+
"notes": "SIGMORPHON-style morpheme-segmentation F1 (plugins/morph_segmentation.py, MorphSegmentationMetric). LANGUAGE-NEUTRAL Layer-1 lane metric for the segmentation TASK — not MT scoring; shipped 2026-07-04 (COMPUTEL). Headline is corpus micro F1; the plugin also emits seg_micro_precision/recall, seg_macro_f1, and per-entry seg_precision/recall/f1. ACTIVATION-PENDING: declared per-card via evalMetrics module mt_eval_harness.plugins.morph_segmentation — no card does yet, so it lives in plugin_metrics JSONB with no run_cards column. Deterministic multiset match, verifier-re-derivable."
|
|
172
|
+
},
|
|
173
|
+
"semantic_score": {
|
|
174
|
+
"category": "semantic",
|
|
175
|
+
"status": "partial",
|
|
176
|
+
"display_name": "Semantic Similarity",
|
|
177
|
+
"plugin_name": "crk_semantic",
|
|
178
|
+
"card_key": "lyss-sem",
|
|
179
|
+
"db_column": "semantic_score",
|
|
180
|
+
"scale": "0.0-1.0",
|
|
181
|
+
"direction": "higher",
|
|
182
|
+
"level": "both",
|
|
183
|
+
"in_composite": true,
|
|
184
|
+
"verifier_reproducible": true,
|
|
185
|
+
"notes": "CRK proxy today (champollion_lyss.crk.metrics.CrkSemanticMetric, declared on the crk card as 'lyss-sem'). publish.py discovers it by the semantic_verdict_counts aggregate field and applies the verdict weights defined there."
|
|
186
|
+
},
|
|
187
|
+
"comet_score": {
|
|
188
|
+
"category": "neural",
|
|
189
|
+
"status": "implemented",
|
|
190
|
+
"display_name": "COMET / AfriCOMET",
|
|
191
|
+
"plugin_name": null,
|
|
192
|
+
"card_key": null,
|
|
193
|
+
"db_column": "comet_score",
|
|
194
|
+
"scale": "~0.0-1.0",
|
|
195
|
+
"direction": "higher",
|
|
196
|
+
"level": "both",
|
|
197
|
+
"in_composite": false,
|
|
198
|
+
"verifier_reproducible": true,
|
|
199
|
+
"notes": "Harness core (metrics_comet.py); model auto-selected via the card's metricModelSupport. NEURAL — reported in the separate lane, never composited; verifier re-derives fail-closed (recompute_corpus_comet)."
|
|
200
|
+
},
|
|
201
|
+
"qe_score": {
|
|
202
|
+
"category": "neural",
|
|
203
|
+
"status": "implemented",
|
|
204
|
+
"display_name": "Reference-Free QE (AfriCOMET-QE)",
|
|
205
|
+
"plugin_name": null,
|
|
206
|
+
"card_key": null,
|
|
207
|
+
"db_column": "qe_score",
|
|
208
|
+
"scale": "0.0-1.0",
|
|
209
|
+
"direction": "higher",
|
|
210
|
+
"level": "both",
|
|
211
|
+
"in_composite": false,
|
|
212
|
+
"verifier_reproducible": true,
|
|
213
|
+
"notes": "Neural reference-free lane for no-reference corpora; verifier re-derives (recompute_corpus_qe)."
|
|
214
|
+
},
|
|
215
|
+
"metricx_score": {
|
|
216
|
+
"category": "neural",
|
|
217
|
+
"status": "implemented",
|
|
218
|
+
"display_name": "MetricX-24",
|
|
219
|
+
"plugin_name": null,
|
|
220
|
+
"card_key": null,
|
|
221
|
+
"db_column": null,
|
|
222
|
+
"scale": "0-25",
|
|
223
|
+
"direction": "lower",
|
|
224
|
+
"level": "both",
|
|
225
|
+
"in_composite": false,
|
|
226
|
+
"verifier_reproducible": false,
|
|
227
|
+
"notes": "LOWER-IS-BETTER error score (the WMT24++/TranslateGemma reference metric). Lives in run_card JSONB only (dedicated column is a follow-up); not yet re-derived by the verifier."
|
|
228
|
+
},
|
|
229
|
+
"code_switching_rate": {
|
|
230
|
+
"category": "behavioral",
|
|
231
|
+
"status": "implemented",
|
|
232
|
+
"display_name": "Code-Switching Rate",
|
|
233
|
+
"plugin_name": "code_switching",
|
|
234
|
+
"card_key": null,
|
|
235
|
+
"db_column": "code_switching_rate",
|
|
236
|
+
"scale": "0.0-1.0",
|
|
237
|
+
"direction": "lower",
|
|
238
|
+
"level": "both",
|
|
239
|
+
"in_composite": true,
|
|
240
|
+
"verifier_reproducible": true,
|
|
241
|
+
"notes": "CodeSwitchingPlugin (language-agnostic, always loaded). Inverted (1-x) for the composite. Plugin aggregate key: avg_code_switching_rate."
|
|
242
|
+
},
|
|
243
|
+
"hallucination_rate": {
|
|
244
|
+
"category": "behavioral",
|
|
245
|
+
"status": "implemented",
|
|
246
|
+
"display_name": "Hallucination Rate",
|
|
247
|
+
"plugin_name": "hallucination",
|
|
248
|
+
"card_key": null,
|
|
249
|
+
"db_column": "hallucination_rate",
|
|
250
|
+
"scale": "0.0-1.0",
|
|
251
|
+
"direction": "lower",
|
|
252
|
+
"level": "both",
|
|
253
|
+
"in_composite": true,
|
|
254
|
+
"verifier_reproducible": true,
|
|
255
|
+
"notes": "HallucinationPlugin (language-agnostic, always loaded). Inverted (1-x) for the composite. Plugin aggregate key: avg_hallucination_rate."
|
|
256
|
+
},
|
|
257
|
+
"terminology_adherence": {
|
|
258
|
+
"category": "behavioral",
|
|
259
|
+
"status": "implemented",
|
|
260
|
+
"display_name": "Terminology Adherence",
|
|
261
|
+
"plugin_name": "terminology",
|
|
262
|
+
"card_key": null,
|
|
263
|
+
"db_column": "terminology_adherence",
|
|
264
|
+
"scale": "0.0-1.0",
|
|
265
|
+
"direction": "higher",
|
|
266
|
+
"level": "both",
|
|
267
|
+
"in_composite": true,
|
|
268
|
+
"verifier_reproducible": true,
|
|
269
|
+
"notes": "TerminologyPlugin; null (metric inactive) when the run supplies no coaching glossary. Reproducible given the run's glossary provenance. Plugin aggregate key: avg_terminology_adherence."
|
|
270
|
+
},
|
|
271
|
+
"consistency_score": {
|
|
272
|
+
"category": "behavioral",
|
|
273
|
+
"status": "planned",
|
|
274
|
+
"display_name": "Cross-Entry Consistency",
|
|
275
|
+
"plugin_name": null,
|
|
276
|
+
"card_key": null,
|
|
277
|
+
"db_column": null,
|
|
278
|
+
"scale": "0.0-1.0",
|
|
279
|
+
"direction": "higher",
|
|
280
|
+
"level": "corpus",
|
|
281
|
+
"in_composite": false,
|
|
282
|
+
"verifier_reproducible": false,
|
|
283
|
+
"notes": "Specified in scoring spec §2.4, not yet implemented."
|
|
284
|
+
},
|
|
285
|
+
"compliance_index": {
|
|
286
|
+
"category": "compliance",
|
|
287
|
+
"status": "implemented",
|
|
288
|
+
"display_name": "Double-Pass Compliance",
|
|
289
|
+
"plugin_name": "double_pass_compliance",
|
|
290
|
+
"card_key": null,
|
|
291
|
+
"db_column": null,
|
|
292
|
+
"scale": "0.0-1.0",
|
|
293
|
+
"direction": "higher",
|
|
294
|
+
"level": "both",
|
|
295
|
+
"in_composite": false,
|
|
296
|
+
"verifier_reproducible": true,
|
|
297
|
+
"notes": "DoublePassCompliancePlugin. Quality GATE (placeholder/quote/casing integrity), not a quality score; lives in plugin_metrics/report only, no run_cards column."
|
|
298
|
+
},
|
|
299
|
+
"repair_effectiveness": {
|
|
300
|
+
"category": "compliance",
|
|
301
|
+
"status": "implemented",
|
|
302
|
+
"display_name": "Repair Effectiveness",
|
|
303
|
+
"plugin_name": "double_pass_compliance",
|
|
304
|
+
"card_key": null,
|
|
305
|
+
"db_column": null,
|
|
306
|
+
"scale": "0.0-1.0",
|
|
307
|
+
"direction": "higher",
|
|
308
|
+
"level": "corpus",
|
|
309
|
+
"in_composite": false,
|
|
310
|
+
"verifier_reproducible": true,
|
|
311
|
+
"notes": "Fraction of compliance violations auto-repaired by post-translation hooks; same plugin as compliance_index."
|
|
312
|
+
},
|
|
313
|
+
"spbleu": {
|
|
314
|
+
"category": "comparator",
|
|
315
|
+
"status": "implemented",
|
|
316
|
+
"display_name": "spBLEU (FLORES-200)",
|
|
317
|
+
"plugin_name": null,
|
|
318
|
+
"card_key": null,
|
|
319
|
+
"db_column": null,
|
|
320
|
+
"scale": "0-100",
|
|
321
|
+
"direction": "higher",
|
|
322
|
+
"level": "corpus",
|
|
323
|
+
"in_composite": false,
|
|
324
|
+
"verifier_reproducible": true,
|
|
325
|
+
"notes": "Comparability sidecar (FLORES/NLLB lingua-franca tokenizer). JSONB only."
|
|
326
|
+
},
|
|
327
|
+
"chrf_plain": {
|
|
328
|
+
"category": "comparator",
|
|
329
|
+
"status": "implemented",
|
|
330
|
+
"display_name": "Plain chrF (word_order=0)",
|
|
331
|
+
"plugin_name": null,
|
|
332
|
+
"card_key": null,
|
|
333
|
+
"db_column": null,
|
|
334
|
+
"scale": "0-100",
|
|
335
|
+
"direction": "higher",
|
|
336
|
+
"level": "corpus",
|
|
337
|
+
"in_composite": false,
|
|
338
|
+
"verifier_reproducible": true,
|
|
339
|
+
"notes": "The chrF figure FLORES/WMT tables report. JSONB only."
|
|
340
|
+
},
|
|
341
|
+
"fuse_score": {
|
|
342
|
+
"category": "comparator",
|
|
343
|
+
"status": "partial",
|
|
344
|
+
"display_name": "FUSE-style Comparator (untrained)",
|
|
345
|
+
"plugin_name": null,
|
|
346
|
+
"card_key": null,
|
|
347
|
+
"db_column": null,
|
|
348
|
+
"scale": "0.0-1.0",
|
|
349
|
+
"direction": "higher",
|
|
350
|
+
"level": "corpus",
|
|
351
|
+
"in_composite": false,
|
|
352
|
+
"verifier_reproducible": false,
|
|
353
|
+
"notes": "Opt-in (--fuse); UNTRAINED reimplementation of AmericasNLP-2025 FUSE (Raja & Vats), flagged fuse_untrained=true; needs the 'fuse' extra (LaBSE). Not re-derived by the verifier."
|
|
354
|
+
},
|
|
355
|
+
"linted_chrf": {
|
|
356
|
+
"category": "comparator",
|
|
357
|
+
"status": "partial",
|
|
358
|
+
"display_name": "LYSS-normalized chrF++",
|
|
359
|
+
"plugin_name": "crk_linted_chrf",
|
|
360
|
+
"card_key": "lyss-chrf",
|
|
361
|
+
"db_column": null,
|
|
362
|
+
"scale": "0-100",
|
|
363
|
+
"direction": "higher",
|
|
364
|
+
"level": "both",
|
|
365
|
+
"in_composite": false,
|
|
366
|
+
"verifier_reproducible": true,
|
|
367
|
+
"notes": "CRK proxy today: the third LYSS plugin (champollion_lyss.crk.metrics.CrkLintedChrF), sibling to crk_linter/crk_semantic. chrF++ where lint-verdict EXACT/EQUIVALENT pairs score 100 and MISS pairs fall back to canonical chrF++ — NOT comparable to published chrF++ numbers (disclosed in the class docstring). Declared on crk.json as 'lyss-chrf' (activated 2026-07-07; runs before that date never computed it). No run_cards column — comparator, plugin_metrics JSONB only, never composited. Deterministic (LYSS lint + sacrebleu chrF), re-derivable given the pinned LYSS release."
|
|
368
|
+
},
|
|
369
|
+
"style_consistency_rate": {
|
|
370
|
+
"category": "behavioral",
|
|
371
|
+
"status": "implemented",
|
|
372
|
+
"display_name": "Writing Style Consistency",
|
|
373
|
+
"plugin_name": "writing_style",
|
|
374
|
+
"card_key": null,
|
|
375
|
+
"db_column": "style_consistency_rate",
|
|
376
|
+
"scale": "0.0-1.0",
|
|
377
|
+
"direction": "higher",
|
|
378
|
+
"level": "both",
|
|
379
|
+
"in_composite": false,
|
|
380
|
+
"verifier_reproducible": false,
|
|
381
|
+
"notes": "WritingStyleMetric — informational only (register/length/formality alignment); profile may be auto-detected from the corpus, so not verifier-contracted."
|
|
382
|
+
},
|
|
383
|
+
"composite": {
|
|
384
|
+
"category": "composite",
|
|
385
|
+
"status": "implemented",
|
|
386
|
+
"display_name": "Composite Score (experimental)",
|
|
387
|
+
"plugin_name": null,
|
|
388
|
+
"card_key": null,
|
|
389
|
+
"db_column": "composite_score",
|
|
390
|
+
"scale": "0.0-1.0",
|
|
391
|
+
"direction": "higher",
|
|
392
|
+
"level": "corpus",
|
|
393
|
+
"in_composite": false,
|
|
394
|
+
"verifier_reproducible": true,
|
|
395
|
+
"notes": "Derived by scoring.compute_composite_score from the profile weight tables (SSOT: scoring spec §4.3). A convenience sort key, NOT a validated quality measurement."
|
|
396
|
+
},
|
|
397
|
+
"cost_adjusted": {
|
|
398
|
+
"category": "composite",
|
|
399
|
+
"status": "implemented",
|
|
400
|
+
"display_name": "Cost-Adjusted Score",
|
|
401
|
+
"plugin_name": null,
|
|
402
|
+
"card_key": null,
|
|
403
|
+
"db_column": null,
|
|
404
|
+
"scale": "0.0-1.0",
|
|
405
|
+
"direction": "higher",
|
|
406
|
+
"level": "corpus",
|
|
407
|
+
"in_composite": false,
|
|
408
|
+
"verifier_reproducible": true,
|
|
409
|
+
"notes": "scoring.cost_adjusted_score (composite / log2(1 + cost*1000), penalty-only). JSONB only."
|
|
410
|
+
},
|
|
411
|
+
"quality_tier": {
|
|
412
|
+
"category": "composite",
|
|
413
|
+
"status": "implemented",
|
|
414
|
+
"display_name": "Quality Tier",
|
|
415
|
+
"plugin_name": null,
|
|
416
|
+
"card_key": null,
|
|
417
|
+
"db_column": "quality_tier",
|
|
418
|
+
"scale": "label (baseline/emerging/functional/deployable/fluent/unscored)",
|
|
419
|
+
"direction": "neutral",
|
|
420
|
+
"level": "corpus",
|
|
421
|
+
"in_composite": false,
|
|
422
|
+
"verifier_reproducible": true,
|
|
423
|
+
"notes": "Heuristic label on the composite (scoring.QUALITY_TIERS); only human review confirms usability."
|
|
424
|
+
},
|
|
425
|
+
"tokens_per_second": {
|
|
426
|
+
"category": "efficiency",
|
|
427
|
+
"status": "implemented",
|
|
428
|
+
"display_name": "Tokens / Second",
|
|
429
|
+
"plugin_name": null,
|
|
430
|
+
"card_key": null,
|
|
431
|
+
"db_column": "tokens_per_second",
|
|
432
|
+
"scale": "0-inf",
|
|
433
|
+
"direction": "higher",
|
|
434
|
+
"level": "corpus",
|
|
435
|
+
"in_composite": false,
|
|
436
|
+
"verifier_reproducible": false,
|
|
437
|
+
"notes": "Speed metric (spec §7) — derived from RunLog timing; environment-dependent, never re-derived."
|
|
438
|
+
},
|
|
439
|
+
"entries_per_minute": {
|
|
440
|
+
"category": "efficiency",
|
|
441
|
+
"status": "implemented",
|
|
442
|
+
"display_name": "Entries / Minute",
|
|
443
|
+
"plugin_name": null,
|
|
444
|
+
"card_key": null,
|
|
445
|
+
"db_column": "entries_per_minute",
|
|
446
|
+
"scale": "0-inf",
|
|
447
|
+
"direction": "higher",
|
|
448
|
+
"level": "corpus",
|
|
449
|
+
"in_composite": false,
|
|
450
|
+
"verifier_reproducible": false,
|
|
451
|
+
"notes": "Speed metric (spec §7)."
|
|
452
|
+
},
|
|
453
|
+
"avg_latency_seconds": {
|
|
454
|
+
"category": "efficiency",
|
|
455
|
+
"status": "implemented",
|
|
456
|
+
"display_name": "Avg Latency (s)",
|
|
457
|
+
"plugin_name": null,
|
|
458
|
+
"card_key": null,
|
|
459
|
+
"db_column": "avg_latency_seconds",
|
|
460
|
+
"scale": "0-inf",
|
|
461
|
+
"direction": "lower",
|
|
462
|
+
"level": "corpus",
|
|
463
|
+
"in_composite": false,
|
|
464
|
+
"verifier_reproducible": false,
|
|
465
|
+
"notes": "Mean per-entry latency across non-error entries."
|
|
466
|
+
},
|
|
467
|
+
"median_latency_seconds": {
|
|
468
|
+
"category": "efficiency",
|
|
469
|
+
"status": "implemented",
|
|
470
|
+
"display_name": "Median Latency (s)",
|
|
471
|
+
"plugin_name": null,
|
|
472
|
+
"card_key": null,
|
|
473
|
+
"db_column": "median_latency_seconds",
|
|
474
|
+
"scale": "0-inf",
|
|
475
|
+
"direction": "lower",
|
|
476
|
+
"level": "corpus",
|
|
477
|
+
"in_composite": false,
|
|
478
|
+
"verifier_reproducible": false,
|
|
479
|
+
"notes": ""
|
|
480
|
+
},
|
|
481
|
+
"p95_latency_seconds": {
|
|
482
|
+
"category": "efficiency",
|
|
483
|
+
"status": "implemented",
|
|
484
|
+
"display_name": "P95 Latency (s)",
|
|
485
|
+
"plugin_name": null,
|
|
486
|
+
"card_key": null,
|
|
487
|
+
"db_column": "p95_latency_seconds",
|
|
488
|
+
"scale": "0-inf",
|
|
489
|
+
"direction": "lower",
|
|
490
|
+
"level": "corpus",
|
|
491
|
+
"in_composite": false,
|
|
492
|
+
"verifier_reproducible": false,
|
|
493
|
+
"notes": ""
|
|
494
|
+
},
|
|
495
|
+
"elapsed_seconds": {
|
|
496
|
+
"category": "efficiency",
|
|
497
|
+
"status": "implemented",
|
|
498
|
+
"display_name": "Elapsed (s)",
|
|
499
|
+
"plugin_name": null,
|
|
500
|
+
"card_key": null,
|
|
501
|
+
"db_column": "elapsed_seconds",
|
|
502
|
+
"scale": "0-inf",
|
|
503
|
+
"direction": "lower",
|
|
504
|
+
"level": "corpus",
|
|
505
|
+
"in_composite": false,
|
|
506
|
+
"verifier_reproducible": false,
|
|
507
|
+
"notes": "Total wall time (top-level run-card field, not in scores)."
|
|
508
|
+
},
|
|
509
|
+
"total_cost_usd": {
|
|
510
|
+
"category": "efficiency",
|
|
511
|
+
"status": "implemented",
|
|
512
|
+
"display_name": "Total Cost (USD)",
|
|
513
|
+
"plugin_name": null,
|
|
514
|
+
"card_key": null,
|
|
515
|
+
"db_column": "total_cost_usd",
|
|
516
|
+
"scale": "USD",
|
|
517
|
+
"direction": "lower",
|
|
518
|
+
"level": "corpus",
|
|
519
|
+
"in_composite": false,
|
|
520
|
+
"verifier_reproducible": false,
|
|
521
|
+
"notes": "Actual spend of THIS run; null = cost UNKNOWN (never coerced to 0). Lives in totals, not scores."
|
|
522
|
+
},
|
|
523
|
+
"cost_per_entry_usd": {
|
|
524
|
+
"category": "efficiency",
|
|
525
|
+
"status": "implemented",
|
|
526
|
+
"display_name": "Cost / Entry (USD)",
|
|
527
|
+
"plugin_name": null,
|
|
528
|
+
"card_key": null,
|
|
529
|
+
"db_column": "cost_per_entry_usd",
|
|
530
|
+
"scale": "USD",
|
|
531
|
+
"direction": "lower",
|
|
532
|
+
"level": "corpus",
|
|
533
|
+
"in_composite": false,
|
|
534
|
+
"verifier_reproducible": false,
|
|
535
|
+
"notes": "Feeds cost_adjusted (spec §6.3). Lives in totals."
|
|
536
|
+
},
|
|
537
|
+
"cost_per_source_char": {
|
|
538
|
+
"category": "efficiency",
|
|
539
|
+
"status": "implemented",
|
|
540
|
+
"display_name": "Cost / Source Char (USD)",
|
|
541
|
+
"plugin_name": null,
|
|
542
|
+
"card_key": null,
|
|
543
|
+
"db_column": "cost_per_source_char",
|
|
544
|
+
"scale": "USD",
|
|
545
|
+
"direction": "lower",
|
|
546
|
+
"level": "corpus",
|
|
547
|
+
"in_composite": false,
|
|
548
|
+
"verifier_reproducible": false,
|
|
549
|
+
"notes": "Tokenization-independent cost normalization. Lives in totals."
|
|
550
|
+
},
|
|
551
|
+
"tokens_per_entry": {
|
|
552
|
+
"category": "efficiency",
|
|
553
|
+
"status": "implemented",
|
|
554
|
+
"display_name": "Tokens / Entry",
|
|
555
|
+
"plugin_name": null,
|
|
556
|
+
"card_key": null,
|
|
557
|
+
"db_column": "tokens_per_entry",
|
|
558
|
+
"scale": "0-inf",
|
|
559
|
+
"direction": "neutral",
|
|
560
|
+
"level": "corpus",
|
|
561
|
+
"in_composite": false,
|
|
562
|
+
"verifier_reproducible": false,
|
|
563
|
+
"notes": "Verbosity diagnostic. Lives in totals."
|
|
564
|
+
},
|
|
565
|
+
"cost_per_1k_tokens": {
|
|
566
|
+
"category": "efficiency",
|
|
567
|
+
"status": "implemented",
|
|
568
|
+
"display_name": "Cost / 1K Tokens (USD)",
|
|
569
|
+
"plugin_name": null,
|
|
570
|
+
"card_key": null,
|
|
571
|
+
"db_column": "cost_per_1k_tokens",
|
|
572
|
+
"scale": "USD",
|
|
573
|
+
"direction": "lower",
|
|
574
|
+
"level": "corpus",
|
|
575
|
+
"in_composite": false,
|
|
576
|
+
"verifier_reproducible": false,
|
|
577
|
+
"notes": "Provider-pricing comparability. Lives in totals."
|
|
578
|
+
},
|
|
579
|
+
"cchrf": {
|
|
580
|
+
"category": "comparator",
|
|
581
|
+
"status": "partial",
|
|
582
|
+
"display_name": "Chance-corrected chrF++ (cchrF++)",
|
|
583
|
+
"plugin_name": null,
|
|
584
|
+
"card_key": null,
|
|
585
|
+
"db_column": null,
|
|
586
|
+
"scale": "0.0-1.0",
|
|
587
|
+
"direction": "higher",
|
|
588
|
+
"level": "corpus",
|
|
589
|
+
"in_composite": false,
|
|
590
|
+
"verifier_reproducible": true,
|
|
591
|
+
"notes": "clamp0((chrF++ - floor)/(100 - floor)) — removes the orthography-specific chance floor so scores are cross-language comparable; the clamp at 0 doubles as the noise rail (at-or-below-floor = indistinguishable from chance). PARTIAL: implemented today only in the connection-quality lane (arena/mt_eval_harness/connection_quality.py cchrf(); JS twins cli/website/src/utils/connectionQuality.mjs + arcStrength.mjs; constants SSOT shared/connection-quality.json cq-v1) — NOT computed by tester.py and never a leaderboard column. Floors: cli/website/src/data/cchrf-floors.json (196 languages, champollion-derived), regenerated from research/cchrf/results/atlas.json (Monte-Carlo N1-unigram floors over FLORES-200 dev monolingual text; the study + paper live in research/cchrf). Known caveat before any RANKING consumer wires this: N1 undershoots fluent-output chance by ~2.3 chrF++ measured on 24/204 languages (research/cchrf REVIEW_2026-07-11 M1); the N1-vs-N_w estimator choice is an open founder decision recorded in docs/METRICS_RESEARCH_PROGRAM_2026-07-10.md. Forbidden on language cards (card-integrity R3)."
|
|
592
|
+
}
|
|
593
|
+
},
|
|
594
|
+
"_proposed_renames": [
|
|
595
|
+
{
|
|
596
|
+
"current": "run-card key 'corpus_bleu' (top-level) vs canonical 'bleu' vs db_column 'corpus_bleu'",
|
|
597
|
+
"proposal": "Move BLEU into scores as scores.bleu (spec §9 already shows it there) and keep db_column corpus_bleu; OR rename nothing and let this registry carry the mapping.",
|
|
598
|
+
"impact": "Published run cards embed the top-level corpus_bleu key; leaderboard row-expand readers would need a fallback read.",
|
|
599
|
+
"decision": "KEEP (2026-07-07) — do not rename. corpus_bleu stays the top-level run-card key and the DB column; this registry carries the bleu↔corpus_bleu mapping. Renaming a published key breaks archived-card readers for no functional gain."
|
|
600
|
+
},
|
|
601
|
+
{
|
|
602
|
+
"current": "plugin 'crk_linter' vs card key 'lyss-eq' vs scores/db 'equivalent_match_rate'",
|
|
603
|
+
"proposal": "Rename the plugin to 'lyss_eq_crk' (LYSS branding + language qualifier) when a second language ships a linter; publish.py already discovers by the is_equivalence_linter flag, so the rename is display-only.",
|
|
604
|
+
"impact": "Published run cards embed plugin_metrics['crk_linter']; archived-report rescoring tools that key on the name would need the alias.",
|
|
605
|
+
"decision": "DEFER to LYSS 0.2 / language #2 (2026-07-07) — keep 'crk_linter' for launch. Discovery is by the is_equivalence_linter envelope flag, so the name is display-only; the alias-first rename to lyss_eq_crk lands with the champollion_lyss.roles spine when a second language forces the generalization, not as pre-launch churn."
|
|
606
|
+
},
|
|
607
|
+
{
|
|
608
|
+
"current": "plugin 'crk_semantic' vs card key 'lyss-sem' vs scores/db 'semantic_score'",
|
|
609
|
+
"proposal": "Rename the plugin to 'lyss_sem_crk' at the same time as crk_linter, same rationale (discovery is by semantic_verdict_counts, display-only).",
|
|
610
|
+
"impact": "Same as crk_linter: plugin_metrics keys in published cards.",
|
|
611
|
+
"decision": "DEFER to LYSS 0.2 / language #2 (2026-07-07) — keep 'crk_semantic' for launch, alias-first rename to lyss_sem_crk alongside crk_linter. Discovery is by the semantic_verdict_counts envelope, so the name is display-only."
|
|
612
|
+
},
|
|
613
|
+
{
|
|
614
|
+
"current": "plugin 'giellalt_fst_validity' (aggregate key 'avg_fst_validity') vs scores/db 'fst_acceptance_rate'",
|
|
615
|
+
"proposal": "None — keep as-is. The plugin name states the tool (GiellaLT FST), the metric name states the measurement; this registry is the bridge.",
|
|
616
|
+
"impact": "n/a",
|
|
617
|
+
"decision": "KEEP (2026-07-07) — no rename; the tool-vs-measurement split is intentional and the registry bridges it."
|
|
618
|
+
}
|
|
619
|
+
]
|
|
620
|
+
}
|