champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,3888 @@
1
+ {
2
+ "_comment": "CITE-ONLY catalogue of externally-published MT results. Each record is ONE concrete, verified datapoint — a single (model, benchmark, NAMED metric, value) tuple with an explicit lower_is_better direction and a source_url. We NEVER re-host the underlying corpus; we record the citation, the metric, and the published value. TWO integrity axes are tracked per record and must NOT be conflated: (1) verified:true — the (model, benchmark, metric, value) was checked against the cited source (citation accuracy); a record may not be added unverified. (2) badge 'unverified — not reproduced by Champollion' — the number is someone else's measurement, run by them on their data, NOT re-run in the Champollion harness. So verified:true does NOT mean we reproduced the result. Metric VARIANTS (legacy-BLEU vs spBLEU vs chrF vs chrF++ vs MetricX-24 vs a specific COMET version) are NOT directly comparable — metric_variant_flag on each record says so. Aggregate/leaderboard pointers (WMT, Papers with Code, OPUS-MT leaderboard) and model reports we have NOT pinned to a single verified number live in manifest.cite, NOT here. This file is an SSOT in the same family as shared/method-registry.json and shared/human-services.json; it is read at build time by the public Index page (cli/website/src/pages/catalogue.js). DO add new verified datapoints; do NOT paste corpus content, gold references, or per-sentence outputs here. Validate against shared/schemas/external-results.schema.json. v3 ADDS three things WITHOUT changing the integrity rules above: (a) optional `pair` {source,target ISO-639-3} on records that are a single directed language pair, so the datapoint surfaces as a FLAGGED EXTERNAL EDGE on /mesh (cite-only, never a chain bridge, never in routing); (b) optional `signal_strength` — the OBJECTIVE A–F grade of the corpus the number was measured on (size · example length · domain breadth · contamination), SEPARATE from provenance: an external result is flagged not-reproduced but is NOT penalized on grade for being external; a high model score on a tiny or HIGH-contamination test set is still a WEAK edge (interim rubric — the coordinator reconciles with the canonical mesh-routing rubric; each grade carries a plain-English rationale); (c) a top-level `methods` availability INDEX cataloguing every method seen in results (availability / weight openness / access / license), a SUPERSET of method-registry.json — entries with runnable_in_champollion:false are cited-only (point users to the method now, implement later).",
3
+ "version": 3,
4
+ "results": [
5
+ {
6
+ "id": "translategemma-27b-metricx",
7
+ "model": "TranslateGemma 27B",
8
+ "source": "TranslateGemma Technical Report",
9
+ "org": "Google",
10
+ "year": 2026,
11
+ "category": "model-report",
12
+ "benchmark": "WMT24++",
13
+ "metric": "MetricX-24",
14
+ "value": 3.09,
15
+ "lower_is_better": true,
16
+ "metric_variant_flag": "MetricX-24 — a learned MQM-style error metric on a ~0–25 scale where LOWER is better; correlates with human MQM error counts. NOT comparable to BLEU/chrF/chrF++ (higher-better surface metrics), to COMET-22, or across MetricX versions.",
17
+ "headline": "Google's open Gemma-3-based MT suite (55 languages); the 27B reaches the report's best MetricX on WMT24++.",
18
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
19
+ "source_url": "https://arxiv.org/abs/2601.09012",
20
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
21
+ "verified": true,
22
+ "badge": "unverified — not reproduced by Champollion",
23
+ "notes": "TranslateGemma is a DEDICATED MT model fine-tuned from Gemma 3 — not to be confused with the general-purpose Gemma 2 LLM (see manifest.cite).",
24
+ "signal_strength": {
25
+ "grade": "B",
26
+ "corpus_size": null,
27
+ "example_length": "sentence",
28
+ "domain_breadth": "multi-domain",
29
+ "contamination": "MEDIUM",
30
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
31
+ },
32
+ "method_ref": "translategemma"
33
+ },
34
+ {
35
+ "id": "translategemma-27b-comet22",
36
+ "model": "TranslateGemma 27B",
37
+ "source": "TranslateGemma Technical Report",
38
+ "org": "Google",
39
+ "year": 2026,
40
+ "category": "model-report",
41
+ "benchmark": "WMT24++",
42
+ "metric": "COMET-22",
43
+ "value": 84.4,
44
+ "lower_is_better": false,
45
+ "metric_variant_flag": "COMET-22 — the Unbabel wmt22-comet-da learned metric, reported on a 0–100 scale where HIGHER is better. NOT comparable to BLEU/chrF/chrF++, to MetricX-24, or across COMET model versions.",
46
+ "headline": "Google's open Gemma-3-based MT suite (55 languages); the 27B reaches the report's best COMET-22 on WMT24++.",
47
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
48
+ "source_url": "https://arxiv.org/abs/2601.09012",
49
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
50
+ "verified": true,
51
+ "badge": "unverified — not reproduced by Champollion",
52
+ "notes": null,
53
+ "signal_strength": {
54
+ "grade": "B",
55
+ "corpus_size": null,
56
+ "example_length": "sentence",
57
+ "domain_breadth": "multi-domain",
58
+ "contamination": "MEDIUM",
59
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
60
+ },
61
+ "method_ref": "translategemma"
62
+ },
63
+ {
64
+ "id": "translategemma-12b-metricx",
65
+ "model": "TranslateGemma 12B",
66
+ "source": "TranslateGemma Technical Report",
67
+ "org": "Google",
68
+ "year": 2026,
69
+ "category": "model-report",
70
+ "benchmark": "WMT24++",
71
+ "metric": "MetricX-24",
72
+ "value": 3.6,
73
+ "lower_is_better": true,
74
+ "metric_variant_flag": "MetricX-24 — a learned MQM-style error metric on a ~0–25 scale where LOWER is better; correlates with human MQM error counts. NOT comparable to BLEU/chrF/chrF++ (higher-better surface metrics), to COMET-22, or across MetricX versions.",
75
+ "headline": "On WMT24++ the 12B TranslateGemma surpasses the 27B Gemma 3 baseline (4.04 MetricX) — fine-tuning beats scale.",
76
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
77
+ "source_url": "https://arxiv.org/abs/2601.09012",
78
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
79
+ "verified": true,
80
+ "badge": "unverified — not reproduced by Champollion",
81
+ "notes": null,
82
+ "signal_strength": {
83
+ "grade": "B",
84
+ "corpus_size": null,
85
+ "example_length": "sentence",
86
+ "domain_breadth": "multi-domain",
87
+ "contamination": "MEDIUM",
88
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
89
+ },
90
+ "method_ref": "translategemma"
91
+ },
92
+ {
93
+ "id": "translategemma-12b-comet22",
94
+ "model": "TranslateGemma 12B",
95
+ "source": "TranslateGemma Technical Report",
96
+ "org": "Google",
97
+ "year": 2026,
98
+ "category": "model-report",
99
+ "benchmark": "WMT24++",
100
+ "metric": "COMET-22",
101
+ "value": 83.5,
102
+ "lower_is_better": false,
103
+ "metric_variant_flag": "COMET-22 — the Unbabel wmt22-comet-da learned metric, reported on a 0–100 scale where HIGHER is better. NOT comparable to BLEU/chrF/chrF++, to MetricX-24, or across COMET model versions.",
104
+ "headline": "On WMT24++ the 12B TranslateGemma surpasses the 27B Gemma 3 baseline (83.1 COMET-22).",
105
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
106
+ "source_url": "https://arxiv.org/abs/2601.09012",
107
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
108
+ "verified": true,
109
+ "badge": "unverified — not reproduced by Champollion",
110
+ "notes": null,
111
+ "signal_strength": {
112
+ "grade": "B",
113
+ "corpus_size": null,
114
+ "example_length": "sentence",
115
+ "domain_breadth": "multi-domain",
116
+ "contamination": "MEDIUM",
117
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
118
+ },
119
+ "method_ref": "translategemma"
120
+ },
121
+ {
122
+ "id": "translategemma-4b-metricx",
123
+ "model": "TranslateGemma 4B",
124
+ "source": "TranslateGemma Technical Report",
125
+ "org": "Google",
126
+ "year": 2026,
127
+ "category": "model-report",
128
+ "benchmark": "WMT24++",
129
+ "metric": "MetricX-24",
130
+ "value": 5.32,
131
+ "lower_is_better": true,
132
+ "metric_variant_flag": "MetricX-24 — a learned MQM-style error metric on a ~0–25 scale where LOWER is better; correlates with human MQM error counts. NOT comparable to BLEU/chrF/chrF++ (higher-better surface metrics), to COMET-22, or across MetricX versions.",
133
+ "headline": "Smallest TranslateGemma size; still a large MetricX gain over the 4B Gemma 3 baseline (6.97).",
134
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
135
+ "source_url": "https://arxiv.org/abs/2601.09012",
136
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
137
+ "verified": true,
138
+ "badge": "unverified — not reproduced by Champollion",
139
+ "notes": null,
140
+ "signal_strength": {
141
+ "grade": "B",
142
+ "corpus_size": null,
143
+ "example_length": "sentence",
144
+ "domain_breadth": "multi-domain",
145
+ "contamination": "MEDIUM",
146
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
147
+ },
148
+ "method_ref": "translategemma"
149
+ },
150
+ {
151
+ "id": "translategemma-4b-comet22",
152
+ "model": "TranslateGemma 4B",
153
+ "source": "TranslateGemma Technical Report",
154
+ "org": "Google",
155
+ "year": 2026,
156
+ "category": "model-report",
157
+ "benchmark": "WMT24++",
158
+ "metric": "COMET-22",
159
+ "value": 80.1,
160
+ "lower_is_better": false,
161
+ "metric_variant_flag": "COMET-22 — the Unbabel wmt22-comet-da learned metric, reported on a 0–100 scale where HIGHER is better. NOT comparable to BLEU/chrF/chrF++, to MetricX-24, or across COMET model versions.",
162
+ "headline": "Smallest TranslateGemma size; COMET-22 gain over the 4B Gemma 3 baseline (77.2).",
163
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
164
+ "source_url": "https://arxiv.org/abs/2601.09012",
165
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
166
+ "verified": true,
167
+ "badge": "unverified — not reproduced by Champollion",
168
+ "notes": null,
169
+ "signal_strength": {
170
+ "grade": "B",
171
+ "corpus_size": null,
172
+ "example_length": "sentence",
173
+ "domain_breadth": "multi-domain",
174
+ "contamination": "MEDIUM",
175
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
176
+ },
177
+ "method_ref": "translategemma"
178
+ },
179
+ {
180
+ "id": "gemma3-27b-metricx",
181
+ "model": "Gemma 3 27B (baseline)",
182
+ "source": "TranslateGemma Technical Report",
183
+ "org": "Google",
184
+ "year": 2026,
185
+ "category": "model-report",
186
+ "benchmark": "WMT24++",
187
+ "metric": "MetricX-24",
188
+ "value": 4.04,
189
+ "lower_is_better": true,
190
+ "metric_variant_flag": "MetricX-24 — a learned MQM-style error metric on a ~0–25 scale where LOWER is better; correlates with human MQM error counts. NOT comparable to BLEU/chrF/chrF++ (higher-better surface metrics), to COMET-22, or across MetricX versions.",
191
+ "headline": "The pre-fine-tuning Gemma 3 27B MT baseline reported in the TranslateGemma study (fine-tuning takes it from 4.04 to 3.09).",
192
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
193
+ "source_url": "https://arxiv.org/abs/2601.09012",
194
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
195
+ "verified": true,
196
+ "badge": "unverified — not reproduced by Champollion",
197
+ "notes": "Gemma 3 is a general-purpose open LLM; this is its baseline MT score in the TranslateGemma report, included to show the fine-tuning delta.",
198
+ "signal_strength": {
199
+ "grade": "B",
200
+ "corpus_size": null,
201
+ "example_length": "sentence",
202
+ "domain_breadth": "multi-domain",
203
+ "contamination": "MEDIUM",
204
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
205
+ },
206
+ "method_ref": "gemma-3"
207
+ },
208
+ {
209
+ "id": "gemma3-27b-comet22",
210
+ "model": "Gemma 3 27B (baseline)",
211
+ "source": "TranslateGemma Technical Report",
212
+ "org": "Google",
213
+ "year": 2026,
214
+ "category": "model-report",
215
+ "benchmark": "WMT24++",
216
+ "metric": "COMET-22",
217
+ "value": 83.1,
218
+ "lower_is_better": false,
219
+ "metric_variant_flag": "COMET-22 — the Unbabel wmt22-comet-da learned metric, reported on a 0–100 scale where HIGHER is better. NOT comparable to BLEU/chrF/chrF++, to MetricX-24, or across COMET model versions.",
220
+ "headline": "The pre-fine-tuning Gemma 3 27B MT baseline reported in the TranslateGemma study.",
221
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
222
+ "source_url": "https://arxiv.org/abs/2601.09012",
223
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
224
+ "verified": true,
225
+ "badge": "unverified — not reproduced by Champollion",
226
+ "notes": "Gemma 3 is a general-purpose open LLM; this is its baseline MT score in the TranslateGemma report, included to show the fine-tuning delta.",
227
+ "signal_strength": {
228
+ "grade": "B",
229
+ "corpus_size": null,
230
+ "example_length": "sentence",
231
+ "domain_breadth": "multi-domain",
232
+ "contamination": "MEDIUM",
233
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
234
+ },
235
+ "method_ref": "gemma-3"
236
+ },
237
+ {
238
+ "id": "gemma3-12b-metricx",
239
+ "model": "Gemma 3 12B (baseline)",
240
+ "source": "TranslateGemma Technical Report",
241
+ "org": "Google",
242
+ "year": 2026,
243
+ "category": "model-report",
244
+ "benchmark": "WMT24++",
245
+ "metric": "MetricX-24",
246
+ "value": 4.86,
247
+ "lower_is_better": true,
248
+ "metric_variant_flag": "MetricX-24 — a learned MQM-style error metric on a ~0–25 scale where LOWER is better; correlates with human MQM error counts. NOT comparable to BLEU/chrF/chrF++ (higher-better surface metrics), to COMET-22, or across MetricX versions.",
249
+ "headline": "The pre-fine-tuning Gemma 3 12B MT baseline reported in the TranslateGemma study.",
250
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
251
+ "source_url": "https://arxiv.org/abs/2601.09012",
252
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
253
+ "verified": true,
254
+ "badge": "unverified — not reproduced by Champollion",
255
+ "notes": "Gemma 3 is a general-purpose open LLM; this is its baseline MT score in the TranslateGemma report, included to show the fine-tuning delta.",
256
+ "signal_strength": {
257
+ "grade": "B",
258
+ "corpus_size": null,
259
+ "example_length": "sentence",
260
+ "domain_breadth": "multi-domain",
261
+ "contamination": "MEDIUM",
262
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
263
+ },
264
+ "method_ref": "gemma-3"
265
+ },
266
+ {
267
+ "id": "gemma3-12b-comet22",
268
+ "model": "Gemma 3 12B (baseline)",
269
+ "source": "TranslateGemma Technical Report",
270
+ "org": "Google",
271
+ "year": 2026,
272
+ "category": "model-report",
273
+ "benchmark": "WMT24++",
274
+ "metric": "COMET-22",
275
+ "value": 81.6,
276
+ "lower_is_better": false,
277
+ "metric_variant_flag": "COMET-22 — the Unbabel wmt22-comet-da learned metric, reported on a 0–100 scale where HIGHER is better. NOT comparable to BLEU/chrF/chrF++, to MetricX-24, or across COMET model versions.",
278
+ "headline": "The pre-fine-tuning Gemma 3 12B MT baseline reported in the TranslateGemma study.",
279
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
280
+ "source_url": "https://arxiv.org/abs/2601.09012",
281
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
282
+ "verified": true,
283
+ "badge": "unverified — not reproduced by Champollion",
284
+ "notes": "Gemma 3 is a general-purpose open LLM; this is its baseline MT score in the TranslateGemma report, included to show the fine-tuning delta.",
285
+ "signal_strength": {
286
+ "grade": "B",
287
+ "corpus_size": null,
288
+ "example_length": "sentence",
289
+ "domain_breadth": "multi-domain",
290
+ "contamination": "MEDIUM",
291
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
292
+ },
293
+ "method_ref": "gemma-3"
294
+ },
295
+ {
296
+ "id": "gemma3-4b-metricx",
297
+ "model": "Gemma 3 4B (baseline)",
298
+ "source": "TranslateGemma Technical Report",
299
+ "org": "Google",
300
+ "year": 2026,
301
+ "category": "model-report",
302
+ "benchmark": "WMT24++",
303
+ "metric": "MetricX-24",
304
+ "value": 6.97,
305
+ "lower_is_better": true,
306
+ "metric_variant_flag": "MetricX-24 — a learned MQM-style error metric on a ~0–25 scale where LOWER is better; correlates with human MQM error counts. NOT comparable to BLEU/chrF/chrF++ (higher-better surface metrics), to COMET-22, or across MetricX versions.",
307
+ "headline": "The pre-fine-tuning Gemma 3 4B MT baseline reported in the TranslateGemma study.",
308
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
309
+ "source_url": "https://arxiv.org/abs/2601.09012",
310
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
311
+ "verified": true,
312
+ "badge": "unverified — not reproduced by Champollion",
313
+ "notes": "Gemma 3 is a general-purpose open LLM; this is its baseline MT score in the TranslateGemma report, included to show the fine-tuning delta.",
314
+ "signal_strength": {
315
+ "grade": "B",
316
+ "corpus_size": null,
317
+ "example_length": "sentence",
318
+ "domain_breadth": "multi-domain",
319
+ "contamination": "MEDIUM",
320
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
321
+ },
322
+ "method_ref": "gemma-3"
323
+ },
324
+ {
325
+ "id": "gemma3-4b-comet22",
326
+ "model": "Gemma 3 4B (baseline)",
327
+ "source": "TranslateGemma Technical Report",
328
+ "org": "Google",
329
+ "year": 2026,
330
+ "category": "model-report",
331
+ "benchmark": "WMT24++",
332
+ "metric": "COMET-22",
333
+ "value": 77.2,
334
+ "lower_is_better": false,
335
+ "metric_variant_flag": "COMET-22 — the Unbabel wmt22-comet-da learned metric, reported on a 0–100 scale where HIGHER is better. NOT comparable to BLEU/chrF/chrF++, to MetricX-24, or across COMET model versions.",
336
+ "headline": "The pre-fine-tuning Gemma 3 4B MT baseline reported in the TranslateGemma study.",
337
+ "langs_or_pairs": "WMT24++ — 55 language pairs, automatic eval averaged across directions.",
338
+ "source_url": "https://arxiv.org/abs/2601.09012",
339
+ "citation": "Finkelstein, Caswell, Domhan, Peter, Juraska, Riley, Deutsch, et al. (2026), 'TranslateGemma Technical Report', arXiv:2601.09012.",
340
+ "verified": true,
341
+ "badge": "unverified — not reproduced by Champollion",
342
+ "notes": "Gemma 3 is a general-purpose open LLM; this is its baseline MT score in the TranslateGemma report, included to show the fine-tuning delta.",
343
+ "signal_strength": {
344
+ "grade": "B",
345
+ "corpus_size": null,
346
+ "example_length": "sentence",
347
+ "domain_breadth": "multi-domain",
348
+ "contamination": "MEDIUM",
349
+ "rationale": "Average over 55 WMT24++ out-of-English directions (general-MT, multi-domain, sentence-level). WMT24++ is a recent held-out-style set (MEDIUM contamination). Solid multi-pair breadth, but an AGGREGATE average rather than a single edge → B."
350
+ },
351
+ "method_ref": "gemma-3"
352
+ },
353
+ {
354
+ "id": "opus-en-fr-tatoeba-bleu",
355
+ "model": "OPUS-MT en-fr",
356
+ "source": "OPUS-MT model card (Helsinki-NLP)",
357
+ "org": "Helsinki-NLP",
358
+ "year": 2020,
359
+ "category": "model-report",
360
+ "benchmark": "Tatoeba-test (eng-fra)",
361
+ "metric": "BLEU",
362
+ "value": 50.5,
363
+ "lower_is_better": false,
364
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
365
+ "headline": "OPUS-MT en→fr scores 50.5 BLEU on the Tatoeba test set.",
366
+ "langs_or_pairs": "eng→fra (single directed pair).",
367
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-fr",
368
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
369
+ "verified": true,
370
+ "badge": "unverified — not reproduced by Champollion",
371
+ "notes": null,
372
+ "pair": {
373
+ "source": "eng",
374
+ "target": "fra"
375
+ },
376
+ "signal_strength": {
377
+ "grade": "B",
378
+ "corpus_size": null,
379
+ "example_length": "sentence",
380
+ "domain_breadth": "broad",
381
+ "contamination": "MEDIUM",
382
+ "rationale": "Large Tatoeba crowd-sentence test set (thousands of short, mixed-register sentences); Tatoeba is broadly trained on → MEDIUM contamination. Strong size, informal short examples → B."
383
+ },
384
+ "method_ref": "opus-mt"
385
+ },
386
+ {
387
+ "id": "opus-en-fr-tatoeba-chrf",
388
+ "model": "OPUS-MT en-fr",
389
+ "source": "OPUS-MT model card (Helsinki-NLP)",
390
+ "org": "Helsinki-NLP",
391
+ "year": 2020,
392
+ "category": "model-report",
393
+ "benchmark": "Tatoeba-test (eng-fra)",
394
+ "metric": "chrF",
395
+ "value": 0.672,
396
+ "lower_is_better": false,
397
+ "metric_variant_flag": "chrF (Helsinki chrF2, 0–1 scale) — character-n-gram F-score on a 0–1 scale where HIGHER is better (e.g. 0.672). NOT the same as chrF++ (word+char, 0–100), and NOT comparable to BLEU/spBLEU/COMET/MetricX.",
398
+ "headline": "OPUS-MT en→fr: chrF 0.672 on Tatoeba.",
399
+ "langs_or_pairs": "eng→fra (single directed pair).",
400
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-fr",
401
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
402
+ "verified": true,
403
+ "badge": "unverified — not reproduced by Champollion",
404
+ "notes": null,
405
+ "pair": {
406
+ "source": "eng",
407
+ "target": "fra"
408
+ },
409
+ "signal_strength": {
410
+ "grade": "B",
411
+ "corpus_size": null,
412
+ "example_length": "sentence",
413
+ "domain_breadth": "broad",
414
+ "contamination": "MEDIUM",
415
+ "rationale": "Same Tatoeba eng-fra test; large but short crowd sentences, MEDIUM contamination → B."
416
+ },
417
+ "method_ref": "opus-mt"
418
+ },
419
+ {
420
+ "id": "opus-en-fr-newstest2013-bleu",
421
+ "model": "OPUS-MT en-fr",
422
+ "source": "OPUS-MT model card (Helsinki-NLP)",
423
+ "org": "Helsinki-NLP",
424
+ "year": 2020,
425
+ "category": "model-report",
426
+ "benchmark": "newstest2013 (eng-fra)",
427
+ "metric": "BLEU",
428
+ "value": 33.2,
429
+ "lower_is_better": false,
430
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
431
+ "headline": "OPUS-MT en→fr: 33.2 BLEU on WMT newstest2013 (news).",
432
+ "langs_or_pairs": "eng→fra (single directed pair).",
433
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-fr",
434
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
435
+ "verified": true,
436
+ "badge": "unverified — not reproduced by Champollion",
437
+ "notes": null,
438
+ "pair": {
439
+ "source": "eng",
440
+ "target": "fra"
441
+ },
442
+ "signal_strength": {
443
+ "grade": "C",
444
+ "corpus_size": 3000,
445
+ "example_length": "sentence",
446
+ "domain_breadth": "single-domain",
447
+ "contamination": "MEDIUM",
448
+ "rationale": "WMT newstest2013: ~3000 professional NEWS sentences — single-domain, and an older WMT set that is broadly trained on (MEDIUM-HIGH contamination) → C."
449
+ },
450
+ "method_ref": "opus-mt"
451
+ },
452
+ {
453
+ "id": "opus-en-es-tatoeba-bleu",
454
+ "model": "OPUS-MT en-es",
455
+ "source": "OPUS-MT model card (Helsinki-NLP)",
456
+ "org": "Helsinki-NLP",
457
+ "year": 2020,
458
+ "category": "model-report",
459
+ "benchmark": "Tatoeba-test (eng-spa)",
460
+ "metric": "BLEU",
461
+ "value": 54.9,
462
+ "lower_is_better": false,
463
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
464
+ "headline": "OPUS-MT en→es: 54.9 BLEU on Tatoeba.",
465
+ "langs_or_pairs": "eng→spa (single directed pair).",
466
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-es",
467
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
468
+ "verified": true,
469
+ "badge": "unverified — not reproduced by Champollion",
470
+ "notes": null,
471
+ "pair": {
472
+ "source": "eng",
473
+ "target": "spa"
474
+ },
475
+ "signal_strength": {
476
+ "grade": "B",
477
+ "corpus_size": null,
478
+ "example_length": "sentence",
479
+ "domain_breadth": "broad",
480
+ "contamination": "MEDIUM",
481
+ "rationale": "Large Tatoeba eng-spa test; MEDIUM contamination → B."
482
+ },
483
+ "method_ref": "opus-mt"
484
+ },
485
+ {
486
+ "id": "opus-en-es-tatoeba-chrf",
487
+ "model": "OPUS-MT en-es",
488
+ "source": "OPUS-MT model card (Helsinki-NLP)",
489
+ "org": "Helsinki-NLP",
490
+ "year": 2020,
491
+ "category": "model-report",
492
+ "benchmark": "Tatoeba-test (eng-spa)",
493
+ "metric": "chrF",
494
+ "value": 0.721,
495
+ "lower_is_better": false,
496
+ "metric_variant_flag": "chrF (Helsinki chrF2, 0–1 scale) — character-n-gram F-score on a 0–1 scale where HIGHER is better (e.g. 0.672). NOT the same as chrF++ (word+char, 0–100), and NOT comparable to BLEU/spBLEU/COMET/MetricX.",
497
+ "headline": "OPUS-MT en→es: chrF 0.721 on Tatoeba.",
498
+ "langs_or_pairs": "eng→spa (single directed pair).",
499
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-es",
500
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
501
+ "verified": true,
502
+ "badge": "unverified — not reproduced by Champollion",
503
+ "notes": null,
504
+ "pair": {
505
+ "source": "eng",
506
+ "target": "spa"
507
+ },
508
+ "signal_strength": {
509
+ "grade": "B",
510
+ "corpus_size": null,
511
+ "example_length": "sentence",
512
+ "domain_breadth": "broad",
513
+ "contamination": "MEDIUM",
514
+ "rationale": "Same eng-spa Tatoeba test → B."
515
+ },
516
+ "method_ref": "opus-mt"
517
+ },
518
+ {
519
+ "id": "opus-en-de-tatoeba-bleu",
520
+ "model": "OPUS-MT en-de",
521
+ "source": "OPUS-MT model card (Helsinki-NLP)",
522
+ "org": "Helsinki-NLP",
523
+ "year": 2020,
524
+ "category": "model-report",
525
+ "benchmark": "Tatoeba-test (eng-deu)",
526
+ "metric": "BLEU",
527
+ "value": 47.3,
528
+ "lower_is_better": false,
529
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
530
+ "headline": "OPUS-MT en→de: 47.3 BLEU on Tatoeba.",
531
+ "langs_or_pairs": "eng→deu (single directed pair).",
532
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-de",
533
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
534
+ "verified": true,
535
+ "badge": "unverified — not reproduced by Champollion",
536
+ "notes": null,
537
+ "pair": {
538
+ "source": "eng",
539
+ "target": "deu"
540
+ },
541
+ "signal_strength": {
542
+ "grade": "B",
543
+ "corpus_size": null,
544
+ "example_length": "sentence",
545
+ "domain_breadth": "broad",
546
+ "contamination": "MEDIUM",
547
+ "rationale": "Large Tatoeba eng-deu test; MEDIUM contamination → B."
548
+ },
549
+ "method_ref": "opus-mt"
550
+ },
551
+ {
552
+ "id": "opus-en-de-tatoeba-chrf",
553
+ "model": "OPUS-MT en-de",
554
+ "source": "OPUS-MT model card (Helsinki-NLP)",
555
+ "org": "Helsinki-NLP",
556
+ "year": 2020,
557
+ "category": "model-report",
558
+ "benchmark": "Tatoeba-test (eng-deu)",
559
+ "metric": "chrF",
560
+ "value": 0.664,
561
+ "lower_is_better": false,
562
+ "metric_variant_flag": "chrF (Helsinki chrF2, 0–1 scale) — character-n-gram F-score on a 0–1 scale where HIGHER is better (e.g. 0.672). NOT the same as chrF++ (word+char, 0–100), and NOT comparable to BLEU/spBLEU/COMET/MetricX.",
563
+ "headline": "OPUS-MT en→de: chrF 0.664 on Tatoeba.",
564
+ "langs_or_pairs": "eng→deu (single directed pair).",
565
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-de",
566
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
567
+ "verified": true,
568
+ "badge": "unverified — not reproduced by Champollion",
569
+ "notes": null,
570
+ "pair": {
571
+ "source": "eng",
572
+ "target": "deu"
573
+ },
574
+ "signal_strength": {
575
+ "grade": "B",
576
+ "corpus_size": null,
577
+ "example_length": "sentence",
578
+ "domain_breadth": "broad",
579
+ "contamination": "MEDIUM",
580
+ "rationale": "Same eng-deu Tatoeba test → B."
581
+ },
582
+ "method_ref": "opus-mt"
583
+ },
584
+ {
585
+ "id": "opus-en-de-newstest2019-bleu",
586
+ "model": "OPUS-MT en-de",
587
+ "source": "OPUS-MT model card (Helsinki-NLP)",
588
+ "org": "Helsinki-NLP",
589
+ "year": 2020,
590
+ "category": "model-report",
591
+ "benchmark": "newstest2019 (eng-deu)",
592
+ "metric": "BLEU",
593
+ "value": 40.9,
594
+ "lower_is_better": false,
595
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
596
+ "headline": "OPUS-MT en→de: 40.9 BLEU on WMT newstest2019.",
597
+ "langs_or_pairs": "eng→deu (single directed pair).",
598
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-de",
599
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
600
+ "verified": true,
601
+ "badge": "unverified — not reproduced by Champollion",
602
+ "notes": null,
603
+ "pair": {
604
+ "source": "eng",
605
+ "target": "deu"
606
+ },
607
+ "signal_strength": {
608
+ "grade": "C",
609
+ "corpus_size": 2000,
610
+ "example_length": "sentence",
611
+ "domain_breadth": "single-domain",
612
+ "contamination": "MEDIUM",
613
+ "rationale": "WMT newstest2019: news single-domain, broadly trained on → C."
614
+ },
615
+ "method_ref": "opus-mt"
616
+ },
617
+ {
618
+ "id": "opus-en-ru-tatoeba-bleu",
619
+ "model": "OPUS-MT en-ru",
620
+ "source": "OPUS-MT model card (Helsinki-NLP)",
621
+ "org": "Helsinki-NLP",
622
+ "year": 2020,
623
+ "category": "model-report",
624
+ "benchmark": "Tatoeba-test (eng-rus)",
625
+ "metric": "BLEU",
626
+ "value": 48.4,
627
+ "lower_is_better": false,
628
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
629
+ "headline": "OPUS-MT en→ru: 48.4 BLEU on Tatoeba.",
630
+ "langs_or_pairs": "eng→rus (single directed pair).",
631
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-ru",
632
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
633
+ "verified": true,
634
+ "badge": "unverified — not reproduced by Champollion",
635
+ "notes": null,
636
+ "pair": {
637
+ "source": "eng",
638
+ "target": "rus"
639
+ },
640
+ "signal_strength": {
641
+ "grade": "B",
642
+ "corpus_size": null,
643
+ "example_length": "sentence",
644
+ "domain_breadth": "broad",
645
+ "contamination": "MEDIUM",
646
+ "rationale": "Large Tatoeba eng-rus test; MEDIUM contamination → B."
647
+ },
648
+ "method_ref": "opus-mt"
649
+ },
650
+ {
651
+ "id": "opus-en-ru-tatoeba-chrf",
652
+ "model": "OPUS-MT en-ru",
653
+ "source": "OPUS-MT model card (Helsinki-NLP)",
654
+ "org": "Helsinki-NLP",
655
+ "year": 2020,
656
+ "category": "model-report",
657
+ "benchmark": "Tatoeba-test (eng-rus)",
658
+ "metric": "chrF",
659
+ "value": 0.669,
660
+ "lower_is_better": false,
661
+ "metric_variant_flag": "chrF (Helsinki chrF2, 0–1 scale) — character-n-gram F-score on a 0–1 scale where HIGHER is better (e.g. 0.672). NOT the same as chrF++ (word+char, 0–100), and NOT comparable to BLEU/spBLEU/COMET/MetricX.",
662
+ "headline": "OPUS-MT en→ru: chrF 0.669 on Tatoeba.",
663
+ "langs_or_pairs": "eng→rus (single directed pair).",
664
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-ru",
665
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
666
+ "verified": true,
667
+ "badge": "unverified — not reproduced by Champollion",
668
+ "notes": null,
669
+ "pair": {
670
+ "source": "eng",
671
+ "target": "rus"
672
+ },
673
+ "signal_strength": {
674
+ "grade": "B",
675
+ "corpus_size": null,
676
+ "example_length": "sentence",
677
+ "domain_breadth": "broad",
678
+ "contamination": "MEDIUM",
679
+ "rationale": "Same eng-rus Tatoeba test → B."
680
+ },
681
+ "method_ref": "opus-mt"
682
+ },
683
+ {
684
+ "id": "opus-tcbig-en-tr-tatoeba-bleu",
685
+ "model": "OPUS-MT tc-big en-tr",
686
+ "source": "OPUS-MT model card (Helsinki-NLP)",
687
+ "org": "Helsinki-NLP",
688
+ "year": 2020,
689
+ "category": "model-report",
690
+ "benchmark": "tatoeba-test-v2021-08-07 (eng-tur)",
691
+ "metric": "BLEU",
692
+ "value": 42.3,
693
+ "lower_is_better": false,
694
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
695
+ "headline": "OPUS-MT tc-big en→tr: 42.3 BLEU on 13,907-sentence Tatoeba test.",
696
+ "langs_or_pairs": "eng→tur (single directed pair).",
697
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-tc-big-en-tr",
698
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
699
+ "verified": true,
700
+ "badge": "unverified — not reproduced by Champollion",
701
+ "notes": null,
702
+ "pair": {
703
+ "source": "eng",
704
+ "target": "tur"
705
+ },
706
+ "signal_strength": {
707
+ "grade": "B",
708
+ "corpus_size": 13907,
709
+ "example_length": "sentence",
710
+ "domain_breadth": "broad",
711
+ "contamination": "MEDIUM",
712
+ "rationale": "Large stated test set (13,907 sentences) of mixed crowd sentences; MEDIUM contamination → B. Size is explicit on the card."
713
+ },
714
+ "method_ref": "opus-mt"
715
+ },
716
+ {
717
+ "id": "opus-tcbig-en-tr-tatoeba-chrf",
718
+ "model": "OPUS-MT tc-big en-tr",
719
+ "source": "OPUS-MT model card (Helsinki-NLP)",
720
+ "org": "Helsinki-NLP",
721
+ "year": 2020,
722
+ "category": "model-report",
723
+ "benchmark": "tatoeba-test-v2021-08-07 (eng-tur)",
724
+ "metric": "chrF",
725
+ "value": 0.68726,
726
+ "lower_is_better": false,
727
+ "metric_variant_flag": "chrF (Helsinki chrF2, 0–1 scale) — character-n-gram F-score on a 0–1 scale where HIGHER is better (e.g. 0.672). NOT the same as chrF++ (word+char, 0–100), and NOT comparable to BLEU/spBLEU/COMET/MetricX.",
728
+ "headline": "OPUS-MT tc-big en→tr: chrF 0.687 on Tatoeba.",
729
+ "langs_or_pairs": "eng→tur (single directed pair).",
730
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-tc-big-en-tr",
731
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
732
+ "verified": true,
733
+ "badge": "unverified — not reproduced by Champollion",
734
+ "notes": null,
735
+ "pair": {
736
+ "source": "eng",
737
+ "target": "tur"
738
+ },
739
+ "signal_strength": {
740
+ "grade": "B",
741
+ "corpus_size": 13907,
742
+ "example_length": "sentence",
743
+ "domain_breadth": "broad",
744
+ "contamination": "MEDIUM",
745
+ "rationale": "Same 13,907-sentence eng-tur test → B."
746
+ },
747
+ "method_ref": "opus-mt"
748
+ },
749
+ {
750
+ "id": "opus-tcbig-en-tr-flores-bleu",
751
+ "model": "OPUS-MT tc-big en-tr",
752
+ "source": "OPUS-MT model card (Helsinki-NLP)",
753
+ "org": "Helsinki-NLP",
754
+ "year": 2020,
755
+ "category": "model-report",
756
+ "benchmark": "FLORES-101 devtest (eng-tur)",
757
+ "metric": "BLEU",
758
+ "value": 31.4,
759
+ "lower_is_better": false,
760
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
761
+ "headline": "OPUS-MT tc-big en→tr: 31.4 BLEU on FLORES-101 (relative-only).",
762
+ "langs_or_pairs": "eng→tur (single directed pair).",
763
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-tc-big-en-tr",
764
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
765
+ "verified": true,
766
+ "badge": "unverified — not reproduced by Champollion",
767
+ "notes": null,
768
+ "pair": {
769
+ "source": "eng",
770
+ "target": "tur"
771
+ },
772
+ "signal_strength": {
773
+ "grade": "C",
774
+ "corpus_size": 1012,
775
+ "example_length": "sentence",
776
+ "domain_breadth": "multi-domain",
777
+ "contamination": "HIGH",
778
+ "rationale": "FLORES-101 devtest (1012 multi-domain Wikipedia sentences) — strong size and breadth, BUT FLORES is broadly memorized (HIGH contamination) so it is RELATIVE-ONLY / illustration; contamination caps the signal at C."
779
+ },
780
+ "method_ref": "opus-mt"
781
+ },
782
+ {
783
+ "id": "opus-tcbig-en-tr-flores-chrf",
784
+ "model": "OPUS-MT tc-big en-tr",
785
+ "source": "OPUS-MT model card (Helsinki-NLP)",
786
+ "org": "Helsinki-NLP",
787
+ "year": 2020,
788
+ "category": "model-report",
789
+ "benchmark": "FLORES-101 devtest (eng-tur)",
790
+ "metric": "chrF",
791
+ "value": 0.62829,
792
+ "lower_is_better": false,
793
+ "metric_variant_flag": "chrF (Helsinki chrF2, 0–1 scale) — character-n-gram F-score on a 0–1 scale where HIGHER is better (e.g. 0.672). NOT the same as chrF++ (word+char, 0–100), and NOT comparable to BLEU/spBLEU/COMET/MetricX.",
794
+ "headline": "OPUS-MT tc-big en→tr: chrF 0.628 on FLORES-101 (relative-only).",
795
+ "langs_or_pairs": "eng→tur (single directed pair).",
796
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-tc-big-en-tr",
797
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
798
+ "verified": true,
799
+ "badge": "unverified — not reproduced by Champollion",
800
+ "notes": null,
801
+ "pair": {
802
+ "source": "eng",
803
+ "target": "tur"
804
+ },
805
+ "signal_strength": {
806
+ "grade": "C",
807
+ "corpus_size": 1012,
808
+ "example_length": "sentence",
809
+ "domain_breadth": "multi-domain",
810
+ "contamination": "HIGH",
811
+ "rationale": "Same eng-tur FLORES-101 devtest; HIGH contamination → relative-only → C."
812
+ },
813
+ "method_ref": "opus-mt"
814
+ },
815
+ {
816
+ "id": "opus-en-ha-tatoeba-bleu",
817
+ "model": "OPUS-MT en-ha",
818
+ "source": "OPUS-MT model card (Helsinki-NLP)",
819
+ "org": "Helsinki-NLP",
820
+ "year": 2020,
821
+ "category": "model-report",
822
+ "benchmark": "Tatoeba-test (eng-hau)",
823
+ "metric": "BLEU",
824
+ "value": 17.6,
825
+ "lower_is_better": false,
826
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
827
+ "headline": "OPUS-MT en→ha: 17.6 BLEU on a small Tatoeba test.",
828
+ "langs_or_pairs": "eng→hau (single directed pair).",
829
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-ha",
830
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
831
+ "verified": true,
832
+ "badge": "unverified — not reproduced by Champollion",
833
+ "notes": null,
834
+ "pair": {
835
+ "source": "eng",
836
+ "target": "hau"
837
+ },
838
+ "signal_strength": {
839
+ "grade": "D",
840
+ "corpus_size": null,
841
+ "example_length": "sentence",
842
+ "domain_breadth": "narrow",
843
+ "contamination": "LOW",
844
+ "rationale": "Low-resource Tatoeba eng-hau test (small; size not stated on the card). Little contamination (LOW) but the SMALL low-resource test set is the dominant weakness → D."
845
+ },
846
+ "method_ref": "opus-mt"
847
+ },
848
+ {
849
+ "id": "opus-ha-en-tatoeba-bleu",
850
+ "model": "OPUS-MT ha-en",
851
+ "source": "OPUS-MT model card (Helsinki-NLP)",
852
+ "org": "Helsinki-NLP",
853
+ "year": 2020,
854
+ "category": "model-report",
855
+ "benchmark": "Tatoeba-test (hau-eng)",
856
+ "metric": "BLEU",
857
+ "value": 39.0,
858
+ "lower_is_better": false,
859
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
860
+ "headline": "OPUS-MT ha→en: 39.0 BLEU — high score, small test → still a weak edge.",
861
+ "langs_or_pairs": "hau→eng (single directed pair).",
862
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-ha-en",
863
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
864
+ "verified": true,
865
+ "badge": "unverified — not reproduced by Champollion",
866
+ "notes": null,
867
+ "pair": {
868
+ "source": "hau",
869
+ "target": "eng"
870
+ },
871
+ "signal_strength": {
872
+ "grade": "D",
873
+ "corpus_size": null,
874
+ "example_length": "sentence",
875
+ "domain_breadth": "narrow",
876
+ "contamination": "LOW",
877
+ "rationale": "Reverse direction scores higher (39.0) but is measured on the SAME small low-resource Tatoeba set — a high score on a weak test is still a WEAK edge → D."
878
+ },
879
+ "method_ref": "opus-mt"
880
+ },
881
+ {
882
+ "id": "opus-en-mul-amh-bleu",
883
+ "model": "OPUS-MT en-mul",
884
+ "source": "OPUS-MT model card (Helsinki-NLP)",
885
+ "org": "Helsinki-NLP",
886
+ "year": 2020,
887
+ "category": "model-report",
888
+ "benchmark": "Tatoeba-test (eng-amh)",
889
+ "metric": "BLEU",
890
+ "value": 8.5,
891
+ "lower_is_better": false,
892
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
893
+ "headline": "OPUS-MT (multilingual) en→am: 8.5 BLEU.",
894
+ "langs_or_pairs": "eng→amh (single directed pair).",
895
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-mul",
896
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
897
+ "verified": true,
898
+ "badge": "unverified — not reproduced by Champollion",
899
+ "notes": null,
900
+ "pair": {
901
+ "source": "eng",
902
+ "target": "amh"
903
+ },
904
+ "signal_strength": {
905
+ "grade": "D",
906
+ "corpus_size": null,
907
+ "example_length": "sentence",
908
+ "domain_breadth": "narrow",
909
+ "contamination": "LOW",
910
+ "rationale": "Low-resource pair via the multilingual en-mul model; small Tatoeba eng-amh test → weak signal D."
911
+ },
912
+ "method_ref": "opus-mt"
913
+ },
914
+ {
915
+ "id": "opus-mul-en-amh-bleu",
916
+ "model": "OPUS-MT mul-en",
917
+ "source": "OPUS-MT model card (Helsinki-NLP)",
918
+ "org": "Helsinki-NLP",
919
+ "year": 2020,
920
+ "category": "model-report",
921
+ "benchmark": "Tatoeba-test (amh-eng)",
922
+ "metric": "BLEU",
923
+ "value": 25.6,
924
+ "lower_is_better": false,
925
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
926
+ "headline": "OPUS-MT (multilingual) am→en: 25.6 BLEU.",
927
+ "langs_or_pairs": "amh→eng (single directed pair).",
928
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-mul-en",
929
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
930
+ "verified": true,
931
+ "badge": "unverified — not reproduced by Champollion",
932
+ "notes": null,
933
+ "pair": {
934
+ "source": "amh",
935
+ "target": "eng"
936
+ },
937
+ "signal_strength": {
938
+ "grade": "D",
939
+ "corpus_size": null,
940
+ "example_length": "sentence",
941
+ "domain_breadth": "narrow",
942
+ "contamination": "LOW",
943
+ "rationale": "amh→en via multilingual mul-en; small low-resource test → D."
944
+ },
945
+ "method_ref": "opus-mt"
946
+ },
947
+ {
948
+ "id": "opus-en-mul-uig-bleu",
949
+ "model": "OPUS-MT en-mul",
950
+ "source": "OPUS-MT model card (Helsinki-NLP)",
951
+ "org": "Helsinki-NLP",
952
+ "year": 2020,
953
+ "category": "model-report",
954
+ "benchmark": "Tatoeba-test (eng-uig)",
955
+ "metric": "BLEU",
956
+ "value": 2.1,
957
+ "lower_is_better": false,
958
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
959
+ "headline": "OPUS-MT (multilingual) en→ug: 2.1 BLEU — essentially no signal.",
960
+ "langs_or_pairs": "eng→uig (single directed pair).",
961
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-mul",
962
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
963
+ "verified": true,
964
+ "badge": "unverified — not reproduced by Champollion",
965
+ "notes": null,
966
+ "pair": {
967
+ "source": "eng",
968
+ "target": "uig"
969
+ },
970
+ "signal_strength": {
971
+ "grade": "F",
972
+ "corpus_size": null,
973
+ "example_length": "sentence",
974
+ "domain_breadth": "narrow",
975
+ "contamination": "LOW",
976
+ "rationale": "Near-zero BLEU (2.1) on a tiny low-resource eng-uig test — essentially no usable signal → F."
977
+ },
978
+ "method_ref": "opus-mt"
979
+ },
980
+ {
981
+ "id": "opus-mul-en-uig-bleu",
982
+ "model": "OPUS-MT mul-en",
983
+ "source": "OPUS-MT model card (Helsinki-NLP)",
984
+ "org": "Helsinki-NLP",
985
+ "year": 2020,
986
+ "category": "model-report",
987
+ "benchmark": "Tatoeba-test (uig-eng)",
988
+ "metric": "BLEU",
989
+ "value": 11.4,
990
+ "lower_is_better": false,
991
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
992
+ "headline": "OPUS-MT (multilingual) ug→en: 11.4 BLEU.",
993
+ "langs_or_pairs": "uig→eng (single directed pair).",
994
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-mul-en",
995
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
996
+ "verified": true,
997
+ "badge": "unverified — not reproduced by Champollion",
998
+ "notes": null,
999
+ "pair": {
1000
+ "source": "uig",
1001
+ "target": "eng"
1002
+ },
1003
+ "signal_strength": {
1004
+ "grade": "D",
1005
+ "corpus_size": null,
1006
+ "example_length": "sentence",
1007
+ "domain_breadth": "narrow",
1008
+ "contamination": "LOW",
1009
+ "rationale": "ug→en via multilingual mul-en; small low-resource test → D."
1010
+ },
1011
+ "method_ref": "opus-mt"
1012
+ },
1013
+ {
1014
+ "id": "opus-en-mul-sah-bleu",
1015
+ "model": "OPUS-MT en-mul",
1016
+ "source": "OPUS-MT model card (Helsinki-NLP)",
1017
+ "org": "Helsinki-NLP",
1018
+ "year": 2020,
1019
+ "category": "model-report",
1020
+ "benchmark": "Tatoeba-test (eng-sah)",
1021
+ "metric": "BLEU",
1022
+ "value": 0.2,
1023
+ "lower_is_better": false,
1024
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
1025
+ "headline": "OPUS-MT (multilingual) en→sah: 0.2 BLEU — the model essentially fails this pair.",
1026
+ "langs_or_pairs": "eng→sah (single directed pair).",
1027
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-mul",
1028
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
1029
+ "verified": true,
1030
+ "badge": "unverified — not reproduced by Champollion",
1031
+ "notes": null,
1032
+ "pair": {
1033
+ "source": "eng",
1034
+ "target": "sah"
1035
+ },
1036
+ "signal_strength": {
1037
+ "grade": "F",
1038
+ "corpus_size": null,
1039
+ "example_length": "sentence",
1040
+ "domain_breadth": "narrow",
1041
+ "contamination": "LOW",
1042
+ "rationale": "BLEU 0.2 / chrF 0.007 — the model essentially fails this direction; a near-empty-signal measurement → F. The starkest weak edge."
1043
+ },
1044
+ "method_ref": "opus-mt"
1045
+ },
1046
+ {
1047
+ "id": "opus-mul-en-sah-bleu",
1048
+ "model": "OPUS-MT mul-en",
1049
+ "source": "OPUS-MT model card (Helsinki-NLP)",
1050
+ "org": "Helsinki-NLP",
1051
+ "year": 2020,
1052
+ "category": "model-report",
1053
+ "benchmark": "Tatoeba-test (sah-eng)",
1054
+ "metric": "BLEU",
1055
+ "value": 5.0,
1056
+ "lower_is_better": false,
1057
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
1058
+ "headline": "OPUS-MT (multilingual) sah→en: 5.0 BLEU.",
1059
+ "langs_or_pairs": "sah→eng (single directed pair).",
1060
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-mul-en",
1061
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
1062
+ "verified": true,
1063
+ "badge": "unverified — not reproduced by Champollion",
1064
+ "notes": null,
1065
+ "pair": {
1066
+ "source": "sah",
1067
+ "target": "eng"
1068
+ },
1069
+ "signal_strength": {
1070
+ "grade": "F",
1071
+ "corpus_size": null,
1072
+ "example_length": "sentence",
1073
+ "domain_breadth": "narrow",
1074
+ "contamination": "LOW",
1075
+ "rationale": "sah→en 5.0 BLEU on a tiny low-resource test → very weak signal F."
1076
+ },
1077
+ "method_ref": "opus-mt"
1078
+ },
1079
+ {
1080
+ "id": "opus-en-trk-kaz-bleu",
1081
+ "model": "OPUS-MT en-trk (Turkic)",
1082
+ "source": "OPUS-MT model card (Helsinki-NLP)",
1083
+ "org": "Helsinki-NLP",
1084
+ "year": 2020,
1085
+ "category": "model-report",
1086
+ "benchmark": "Tatoeba-test (eng-kaz)",
1087
+ "metric": "BLEU",
1088
+ "value": 11.1,
1089
+ "lower_is_better": false,
1090
+ "metric_variant_flag": "BLEU (OPUS-MT model-card, tokenized) — a tokenized n-gram precision score on a 0–100 scale where HIGHER is better. The exact tokenization is the OPUS-MT card's, NOT a pinned sacreBLEU signature, so it is NOT directly comparable to spBLEU, sacreBLEU, chrF/chrF++, or COMET/MetricX.",
1091
+ "headline": "OPUS-MT (Turkic) en→kk: 11.1 BLEU.",
1092
+ "langs_or_pairs": "eng→kaz (single directed pair).",
1093
+ "source_url": "https://huggingface.co/Helsinki-NLP/opus-mt-en-trk",
1094
+ "citation": "Tiedemann & Thottingal (2020), 'OPUS-MT — Building open translation services for the World', EAMT 2020; benchmark table from the model's Hugging Face card.",
1095
+ "verified": true,
1096
+ "badge": "unverified — not reproduced by Champollion",
1097
+ "notes": null,
1098
+ "pair": {
1099
+ "source": "eng",
1100
+ "target": "kaz"
1101
+ },
1102
+ "signal_strength": {
1103
+ "grade": "D",
1104
+ "corpus_size": null,
1105
+ "example_length": "sentence",
1106
+ "domain_breadth": "narrow",
1107
+ "contamination": "LOW",
1108
+ "rationale": "eng→kaz via the Turkic-family en-trk model; small low-resource Tatoeba test → D."
1109
+ },
1110
+ "method_ref": "opus-mt"
1111
+ },
1112
+ {
1113
+ "id": "nllb-eng-hau-flores-spbleu",
1114
+ "model": "NLLB-200",
1115
+ "source": "No Language Left Behind (NLLB-200)",
1116
+ "org": "Meta AI",
1117
+ "year": 2022,
1118
+ "category": "paper",
1119
+ "benchmark": "FLORES-101 devtest (eng-hau)",
1120
+ "metric": "spBLEU",
1121
+ "value": 33.6,
1122
+ "lower_is_better": false,
1123
+ "metric_variant_flag": "spBLEU (FLORES SentencePiece-tokenized BLEU) — 0–100, HIGHER better; uses a fixed SPM tokenizer so cross-language sums are comparable WITHIN FLORES but NOT comparable to tokenized BLEU, chrF/chrF++, or COMET/MetricX.",
1124
+ "headline": "NLLB-200 en→ha: 33.6 spBLEU on FLORES-101 (relative-only).",
1125
+ "langs_or_pairs": "eng→hau (single directed pair).",
1126
+ "source_url": "https://arxiv.org/abs/2207.04672",
1127
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1128
+ "verified": true,
1129
+ "badge": "unverified — not reproduced by Champollion",
1130
+ "notes": "FLORES is HIGH-contamination illustration-only data (the data-boundaries doctrine): cited as a relative reading, never an absolute quality claim.",
1131
+ "signal_strength": {
1132
+ "grade": "C",
1133
+ "corpus_size": 1012,
1134
+ "example_length": "sentence",
1135
+ "domain_breadth": "multi-domain",
1136
+ "contamination": "HIGH",
1137
+ "rationale": "FLORES-101 devtest (1012 multi-domain sentences): strong size/breadth, but FLORES is broadly memorized (HIGH contamination) so it is RELATIVE-ONLY — contamination caps the signal at C."
1138
+ },
1139
+ "method_ref": "nllb-200",
1140
+ "pair": {
1141
+ "source": "eng",
1142
+ "target": "hau"
1143
+ }
1144
+ },
1145
+ {
1146
+ "id": "nllb-eng-hau-flores-chrfpp",
1147
+ "model": "NLLB-200",
1148
+ "source": "No Language Left Behind (NLLB-200)",
1149
+ "org": "Meta AI",
1150
+ "year": 2022,
1151
+ "category": "paper",
1152
+ "benchmark": "FLORES-101 devtest (eng-hau)",
1153
+ "metric": "chrF++",
1154
+ "value": 53.5,
1155
+ "lower_is_better": false,
1156
+ "metric_variant_flag": "chrF++ (word+character n-gram F-score, 0–100) — HIGHER better. NOT comparable to chrF (0–1), to BLEU/spBLEU, or to COMET/MetricX.",
1157
+ "headline": "NLLB-200 en→ha: 53.5 chrF++ on FLORES-101 (relative-only).",
1158
+ "langs_or_pairs": "eng→hau (single directed pair).",
1159
+ "source_url": "https://arxiv.org/abs/2207.04672",
1160
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1161
+ "verified": true,
1162
+ "badge": "unverified — not reproduced by Champollion",
1163
+ "notes": "FLORES relative-only; chrF++ (0–100) ≠ OPUS-MT chrF (0–1).",
1164
+ "signal_strength": {
1165
+ "grade": "C",
1166
+ "corpus_size": 1012,
1167
+ "example_length": "sentence",
1168
+ "domain_breadth": "multi-domain",
1169
+ "contamination": "HIGH",
1170
+ "rationale": "Same eng-hau FLORES-101 devtest; HIGH contamination → relative-only → C."
1171
+ },
1172
+ "method_ref": "nllb-200",
1173
+ "pair": {
1174
+ "source": "eng",
1175
+ "target": "hau"
1176
+ }
1177
+ },
1178
+ {
1179
+ "id": "nllb-hau-eng-flores-chrfpp",
1180
+ "model": "NLLB-200",
1181
+ "source": "No Language Left Behind (NLLB-200)",
1182
+ "org": "Meta AI",
1183
+ "year": 2022,
1184
+ "category": "paper",
1185
+ "benchmark": "FLORES-101 devtest (hau-eng)",
1186
+ "metric": "chrF++",
1187
+ "value": 57.3,
1188
+ "lower_is_better": false,
1189
+ "metric_variant_flag": "chrF++ (word+character n-gram F-score, 0–100) — HIGHER better. NOT comparable to chrF (0–1), to BLEU/spBLEU, or to COMET/MetricX.",
1190
+ "headline": "NLLB-200 ha→en: 57.3 chrF++ on FLORES-101 (relative-only).",
1191
+ "langs_or_pairs": "hau→eng (single directed pair).",
1192
+ "source_url": "https://arxiv.org/abs/2207.04672",
1193
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1194
+ "verified": true,
1195
+ "badge": "unverified — not reproduced by Champollion",
1196
+ "notes": "FLORES relative-only.",
1197
+ "signal_strength": {
1198
+ "grade": "C",
1199
+ "corpus_size": 1012,
1200
+ "example_length": "sentence",
1201
+ "domain_breadth": "multi-domain",
1202
+ "contamination": "HIGH",
1203
+ "rationale": "ha→en on FLORES-101; HIGH contamination → relative-only → C."
1204
+ },
1205
+ "method_ref": "nllb-200",
1206
+ "pair": {
1207
+ "source": "hau",
1208
+ "target": "eng"
1209
+ }
1210
+ },
1211
+ {
1212
+ "id": "nllb-eng-amh-tico-bleu",
1213
+ "model": "NLLB-200",
1214
+ "source": "No Language Left Behind (NLLB-200)",
1215
+ "org": "Meta AI",
1216
+ "year": 2022,
1217
+ "category": "paper",
1218
+ "benchmark": "TICO (eng-amh)",
1219
+ "metric": "BLEU",
1220
+ "value": 13.7,
1221
+ "lower_is_better": false,
1222
+ "metric_variant_flag": "BLEU (detokenized, as reported in the NLLB paper Tico table) — 0–100, HIGHER better; tokenization is the paper's, NOT a pinned sacreBLEU signature, so not comparable to spBLEU/chrF/chrF++/COMET.",
1223
+ "headline": "NLLB-200 en→am: 13.7 BLEU on the TICO (health) set.",
1224
+ "langs_or_pairs": "eng→amh (single directed pair).",
1225
+ "source_url": "https://arxiv.org/abs/2207.04672",
1226
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1227
+ "verified": true,
1228
+ "badge": "unverified — not reproduced by Champollion",
1229
+ "notes": "Reported on the TICO public-health set in the NLLB paper (Table 58).",
1230
+ "signal_strength": {
1231
+ "grade": "C",
1232
+ "corpus_size": null,
1233
+ "example_length": "sentence",
1234
+ "domain_breadth": "single-domain",
1235
+ "contamination": "MEDIUM",
1236
+ "rationale": "NLLB paper Tico table (public-health domain): single-domain but a genuine held-out set; MEDIUM contamination → C."
1237
+ },
1238
+ "method_ref": "nllb-200",
1239
+ "pair": {
1240
+ "source": "eng",
1241
+ "target": "amh"
1242
+ }
1243
+ },
1244
+ {
1245
+ "id": "nllb-eng-amh-tico-chrfpp",
1246
+ "model": "NLLB-200",
1247
+ "source": "No Language Left Behind (NLLB-200)",
1248
+ "org": "Meta AI",
1249
+ "year": 2022,
1250
+ "category": "paper",
1251
+ "benchmark": "TICO (eng-amh)",
1252
+ "metric": "chrF++",
1253
+ "value": 36.7,
1254
+ "lower_is_better": false,
1255
+ "metric_variant_flag": "chrF++ (word+character n-gram F-score, 0–100) — HIGHER better. NOT comparable to chrF (0–1), to BLEU/spBLEU, or to COMET/MetricX.",
1256
+ "headline": "NLLB-200 en→am: 36.7 chrF++ on TICO.",
1257
+ "langs_or_pairs": "eng→amh (single directed pair).",
1258
+ "source_url": "https://arxiv.org/abs/2207.04672",
1259
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1260
+ "verified": true,
1261
+ "badge": "unverified — not reproduced by Champollion",
1262
+ "notes": "TICO health domain (NLLB Table 58).",
1263
+ "signal_strength": {
1264
+ "grade": "C",
1265
+ "corpus_size": null,
1266
+ "example_length": "sentence",
1267
+ "domain_breadth": "single-domain",
1268
+ "contamination": "MEDIUM",
1269
+ "rationale": "Same eng-amh TICO set; single-domain health → C."
1270
+ },
1271
+ "method_ref": "nllb-200",
1272
+ "pair": {
1273
+ "source": "eng",
1274
+ "target": "amh"
1275
+ }
1276
+ },
1277
+ {
1278
+ "id": "nllb-amh-eng-tico-chrfpp",
1279
+ "model": "NLLB-200",
1280
+ "source": "No Language Left Behind (NLLB-200)",
1281
+ "org": "Meta AI",
1282
+ "year": 2022,
1283
+ "category": "paper",
1284
+ "benchmark": "TICO (amh-eng)",
1285
+ "metric": "chrF++",
1286
+ "value": 60.2,
1287
+ "lower_is_better": false,
1288
+ "metric_variant_flag": "chrF++ (word+character n-gram F-score, 0–100) — HIGHER better. NOT comparable to chrF (0–1), to BLEU/spBLEU, or to COMET/MetricX.",
1289
+ "headline": "NLLB-200 am→en: 60.2 chrF++ on TICO.",
1290
+ "langs_or_pairs": "amh→eng (single directed pair).",
1291
+ "source_url": "https://arxiv.org/abs/2207.04672",
1292
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1293
+ "verified": true,
1294
+ "badge": "unverified — not reproduced by Champollion",
1295
+ "notes": "TICO health domain (NLLB Table 58).",
1296
+ "signal_strength": {
1297
+ "grade": "C",
1298
+ "corpus_size": null,
1299
+ "example_length": "sentence",
1300
+ "domain_breadth": "single-domain",
1301
+ "contamination": "MEDIUM",
1302
+ "rationale": "am→en on TICO; single-domain health, MEDIUM contamination → C."
1303
+ },
1304
+ "method_ref": "nllb-200",
1305
+ "pair": {
1306
+ "source": "amh",
1307
+ "target": "eng"
1308
+ }
1309
+ },
1310
+ {
1311
+ "id": "nllb-flores101-avg-spbleu",
1312
+ "model": "NLLB-200",
1313
+ "source": "No Language Left Behind (NLLB-200)",
1314
+ "org": "Meta AI",
1315
+ "year": 2022,
1316
+ "category": "paper",
1317
+ "benchmark": "FLORES-101 devtest (101-language average)",
1318
+ "metric": "spBLEU",
1319
+ "value": 24.0,
1320
+ "lower_is_better": false,
1321
+ "metric_variant_flag": "spBLEU (FLORES SentencePiece-tokenized BLEU) — 0–100, HIGHER better; uses a fixed SPM tokenizer so cross-language sums are comparable WITHIN FLORES but NOT comparable to tokenized BLEU, chrF/chrF++, or COMET/MetricX.",
1322
+ "headline": "NLLB-200 averages 24.0 spBLEU across all FLORES-101 directions (relative-only headline).",
1323
+ "langs_or_pairs": "Average across all FLORES-101 devtest directions.",
1324
+ "source_url": "https://arxiv.org/abs/2207.04672",
1325
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1326
+ "verified": true,
1327
+ "badge": "unverified — not reproduced by Champollion",
1328
+ "notes": "Table 30 average (101 languages). FLORES relative-only; aggregate, not a single edge.",
1329
+ "signal_strength": {
1330
+ "grade": "C",
1331
+ "corpus_size": null,
1332
+ "example_length": "sentence",
1333
+ "domain_breadth": "multi-domain",
1334
+ "contamination": "HIGH",
1335
+ "rationale": "Average over all FLORES-101 devtest directions; FLORES HIGH contamination → relative-only → C. Aggregate, not a single pair."
1336
+ },
1337
+ "method_ref": "nllb-200"
1338
+ },
1339
+ {
1340
+ "id": "nllb-flores101-avg-chrfpp",
1341
+ "model": "NLLB-200",
1342
+ "source": "No Language Left Behind (NLLB-200)",
1343
+ "org": "Meta AI",
1344
+ "year": 2022,
1345
+ "category": "paper",
1346
+ "benchmark": "FLORES-101 devtest (101-language average)",
1347
+ "metric": "chrF++",
1348
+ "value": 41.7,
1349
+ "lower_is_better": false,
1350
+ "metric_variant_flag": "chrF++ (word+character n-gram F-score, 0–100) — HIGHER better. NOT comparable to chrF (0–1), to BLEU/spBLEU, or to COMET/MetricX.",
1351
+ "headline": "NLLB-200 averages 41.7 chrF++ across all FLORES-101 directions (relative-only headline).",
1352
+ "langs_or_pairs": "Average across all FLORES-101 devtest directions.",
1353
+ "source_url": "https://arxiv.org/abs/2207.04672",
1354
+ "citation": "NLLB Team et al. (2022), 'No Language Left Behind: Scaling Human-Centered Machine Translation', arXiv:2207.04672.",
1355
+ "verified": true,
1356
+ "badge": "unverified — not reproduced by Champollion",
1357
+ "notes": "Table 30 average (101 languages). FLORES relative-only; aggregate.",
1358
+ "signal_strength": {
1359
+ "grade": "C",
1360
+ "corpus_size": null,
1361
+ "example_length": "sentence",
1362
+ "domain_breadth": "multi-domain",
1363
+ "contamination": "HIGH",
1364
+ "rationale": "FLORES-101 devtest 101-language average chrF++; HIGH contamination → relative-only → C."
1365
+ },
1366
+ "method_ref": "nllb-200"
1367
+ },
1368
+ {
1369
+ "id": "opus-afr-deu-tatoeba-chrf2",
1370
+ "model": "OPUS-MT afr-deu",
1371
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1372
+ "org": "Helsinki-NLP",
1373
+ "year": 2021,
1374
+ "category": "model-report",
1375
+ "benchmark": "Tatoeba-test (afr-deu)",
1376
+ "metric": "chrF2",
1377
+ "value": 67.1,
1378
+ "lower_is_better": false,
1379
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1380
+ "headline": "OPUS-MT afr→deu: chrF2 67.1 on Tatoeba-test.",
1381
+ "langs_or_pairs": "afr→deu (single directed pair).",
1382
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1383
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1384
+ "verified": true,
1385
+ "badge": "unverified — not reproduced by Champollion",
1386
+ "notes": null,
1387
+ "pair": {
1388
+ "source": "afr",
1389
+ "target": "deu"
1390
+ },
1391
+ "signal_strength": {
1392
+ "grade": "C",
1393
+ "corpus_size": null,
1394
+ "example_length": "sentence",
1395
+ "domain_breadth": "broad",
1396
+ "contamination": "MEDIUM",
1397
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1398
+ }
1399
+ },
1400
+ {
1401
+ "id": "opus-afr-eng-tatoeba-chrf2",
1402
+ "model": "OPUS-MT afr-eng",
1403
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1404
+ "org": "Helsinki-NLP",
1405
+ "year": 2021,
1406
+ "category": "model-report",
1407
+ "benchmark": "Tatoeba-test (afr-eng)",
1408
+ "metric": "chrF2",
1409
+ "value": 73.8,
1410
+ "lower_is_better": false,
1411
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1412
+ "headline": "OPUS-MT afr→eng: chrF2 73.8 on Tatoeba-test.",
1413
+ "langs_or_pairs": "afr→eng (single directed pair).",
1414
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1415
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1416
+ "verified": true,
1417
+ "badge": "unverified — not reproduced by Champollion",
1418
+ "notes": null,
1419
+ "pair": {
1420
+ "source": "afr",
1421
+ "target": "eng"
1422
+ },
1423
+ "signal_strength": {
1424
+ "grade": "C",
1425
+ "corpus_size": null,
1426
+ "example_length": "sentence",
1427
+ "domain_breadth": "broad",
1428
+ "contamination": "MEDIUM",
1429
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1430
+ }
1431
+ },
1432
+ {
1433
+ "id": "opus-afr-epo-tatoeba-chrf2",
1434
+ "model": "OPUS-MT afr-epo",
1435
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1436
+ "org": "Helsinki-NLP",
1437
+ "year": 2021,
1438
+ "category": "model-report",
1439
+ "benchmark": "Tatoeba-test (afr-epo)",
1440
+ "metric": "chrF2",
1441
+ "value": 41.1,
1442
+ "lower_is_better": false,
1443
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1444
+ "headline": "OPUS-MT afr→epo: chrF2 41.1 on Tatoeba-test.",
1445
+ "langs_or_pairs": "afr→epo (single directed pair).",
1446
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1447
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1448
+ "verified": true,
1449
+ "badge": "unverified — not reproduced by Champollion",
1450
+ "notes": null,
1451
+ "pair": {
1452
+ "source": "afr",
1453
+ "target": "epo"
1454
+ },
1455
+ "signal_strength": {
1456
+ "grade": "C",
1457
+ "corpus_size": null,
1458
+ "example_length": "sentence",
1459
+ "domain_breadth": "broad",
1460
+ "contamination": "MEDIUM",
1461
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1462
+ }
1463
+ },
1464
+ {
1465
+ "id": "opus-afr-rus-tatoeba-chrf2",
1466
+ "model": "OPUS-MT afr-rus",
1467
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1468
+ "org": "Helsinki-NLP",
1469
+ "year": 2021,
1470
+ "category": "model-report",
1471
+ "benchmark": "Tatoeba-test (afr-rus)",
1472
+ "metric": "chrF2",
1473
+ "value": 58,
1474
+ "lower_is_better": false,
1475
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1476
+ "headline": "OPUS-MT afr→rus: chrF2 58 on Tatoeba-test.",
1477
+ "langs_or_pairs": "afr→rus (single directed pair).",
1478
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1479
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1480
+ "verified": true,
1481
+ "badge": "unverified — not reproduced by Champollion",
1482
+ "notes": null,
1483
+ "pair": {
1484
+ "source": "afr",
1485
+ "target": "rus"
1486
+ },
1487
+ "signal_strength": {
1488
+ "grade": "C",
1489
+ "corpus_size": null,
1490
+ "example_length": "sentence",
1491
+ "domain_breadth": "broad",
1492
+ "contamination": "MEDIUM",
1493
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1494
+ }
1495
+ },
1496
+ {
1497
+ "id": "opus-afr-spa-tatoeba-chrf2",
1498
+ "model": "OPUS-MT afr-spa",
1499
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1500
+ "org": "Helsinki-NLP",
1501
+ "year": 2021,
1502
+ "category": "model-report",
1503
+ "benchmark": "Tatoeba-test (afr-spa)",
1504
+ "metric": "chrF2",
1505
+ "value": 68,
1506
+ "lower_is_better": false,
1507
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1508
+ "headline": "OPUS-MT afr→spa: chrF2 68 on Tatoeba-test.",
1509
+ "langs_or_pairs": "afr→spa (single directed pair).",
1510
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1511
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1512
+ "verified": true,
1513
+ "badge": "unverified — not reproduced by Champollion",
1514
+ "notes": null,
1515
+ "pair": {
1516
+ "source": "afr",
1517
+ "target": "spa"
1518
+ },
1519
+ "signal_strength": {
1520
+ "grade": "C",
1521
+ "corpus_size": null,
1522
+ "example_length": "sentence",
1523
+ "domain_breadth": "broad",
1524
+ "contamination": "MEDIUM",
1525
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1526
+ }
1527
+ },
1528
+ {
1529
+ "id": "opus-ara-deu-tatoeba-chrf2",
1530
+ "model": "OPUS-MT ara-deu",
1531
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1532
+ "org": "Helsinki-NLP",
1533
+ "year": 2021,
1534
+ "category": "model-report",
1535
+ "benchmark": "Tatoeba-test (ara-deu)",
1536
+ "metric": "chrF2",
1537
+ "value": 63,
1538
+ "lower_is_better": false,
1539
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1540
+ "headline": "OPUS-MT ara→deu: chrF2 63 on Tatoeba-test.",
1541
+ "langs_or_pairs": "ara→deu (single directed pair).",
1542
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1543
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1544
+ "verified": true,
1545
+ "badge": "unverified — not reproduced by Champollion",
1546
+ "notes": null,
1547
+ "pair": {
1548
+ "source": "ara",
1549
+ "target": "deu"
1550
+ },
1551
+ "signal_strength": {
1552
+ "grade": "C",
1553
+ "corpus_size": null,
1554
+ "example_length": "sentence",
1555
+ "domain_breadth": "broad",
1556
+ "contamination": "MEDIUM",
1557
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1558
+ }
1559
+ },
1560
+ {
1561
+ "id": "opus-ara-ell-tatoeba-chrf2",
1562
+ "model": "OPUS-MT ara-ell",
1563
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1564
+ "org": "Helsinki-NLP",
1565
+ "year": 2021,
1566
+ "category": "model-report",
1567
+ "benchmark": "Tatoeba-test (ara-ell)",
1568
+ "metric": "chrF2",
1569
+ "value": 64.1,
1570
+ "lower_is_better": false,
1571
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1572
+ "headline": "OPUS-MT ara→ell: chrF2 64.1 on Tatoeba-test.",
1573
+ "langs_or_pairs": "ara→ell (single directed pair).",
1574
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1575
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1576
+ "verified": true,
1577
+ "badge": "unverified — not reproduced by Champollion",
1578
+ "notes": null,
1579
+ "pair": {
1580
+ "source": "ara",
1581
+ "target": "ell"
1582
+ },
1583
+ "signal_strength": {
1584
+ "grade": "C",
1585
+ "corpus_size": null,
1586
+ "example_length": "sentence",
1587
+ "domain_breadth": "broad",
1588
+ "contamination": "MEDIUM",
1589
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1590
+ }
1591
+ },
1592
+ {
1593
+ "id": "opus-ara-eng-tatoeba-chrf2",
1594
+ "model": "OPUS-MT ara-eng",
1595
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1596
+ "org": "Helsinki-NLP",
1597
+ "year": 2021,
1598
+ "category": "model-report",
1599
+ "benchmark": "Tatoeba-test (ara-eng)",
1600
+ "metric": "chrF2",
1601
+ "value": 61.7,
1602
+ "lower_is_better": false,
1603
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1604
+ "headline": "OPUS-MT ara→eng: chrF2 61.7 on Tatoeba-test.",
1605
+ "langs_or_pairs": "ara→eng (single directed pair).",
1606
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1607
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1608
+ "verified": true,
1609
+ "badge": "unverified — not reproduced by Champollion",
1610
+ "notes": null,
1611
+ "pair": {
1612
+ "source": "ara",
1613
+ "target": "eng"
1614
+ },
1615
+ "signal_strength": {
1616
+ "grade": "C",
1617
+ "corpus_size": null,
1618
+ "example_length": "sentence",
1619
+ "domain_breadth": "broad",
1620
+ "contamination": "MEDIUM",
1621
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1622
+ }
1623
+ },
1624
+ {
1625
+ "id": "opus-ara-epo-tatoeba-chrf2",
1626
+ "model": "OPUS-MT ara-epo",
1627
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1628
+ "org": "Helsinki-NLP",
1629
+ "year": 2021,
1630
+ "category": "model-report",
1631
+ "benchmark": "Tatoeba-test (ara-epo)",
1632
+ "metric": "chrF2",
1633
+ "value": 37.6,
1634
+ "lower_is_better": false,
1635
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1636
+ "headline": "OPUS-MT ara→epo: chrF2 37.6 on Tatoeba-test.",
1637
+ "langs_or_pairs": "ara→epo (single directed pair).",
1638
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1639
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1640
+ "verified": true,
1641
+ "badge": "unverified — not reproduced by Champollion",
1642
+ "notes": null,
1643
+ "pair": {
1644
+ "source": "ara",
1645
+ "target": "epo"
1646
+ },
1647
+ "signal_strength": {
1648
+ "grade": "C",
1649
+ "corpus_size": null,
1650
+ "example_length": "sentence",
1651
+ "domain_breadth": "broad",
1652
+ "contamination": "MEDIUM",
1653
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1654
+ }
1655
+ },
1656
+ {
1657
+ "id": "opus-ara-fra-tatoeba-chrf2",
1658
+ "model": "OPUS-MT ara-fra",
1659
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1660
+ "org": "Helsinki-NLP",
1661
+ "year": 2021,
1662
+ "category": "model-report",
1663
+ "benchmark": "Tatoeba-test (ara-fra)",
1664
+ "metric": "chrF2",
1665
+ "value": 56.2,
1666
+ "lower_is_better": false,
1667
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1668
+ "headline": "OPUS-MT ara→fra: chrF2 56.2 on Tatoeba-test.",
1669
+ "langs_or_pairs": "ara→fra (single directed pair).",
1670
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1671
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1672
+ "verified": true,
1673
+ "badge": "unverified — not reproduced by Champollion",
1674
+ "notes": null,
1675
+ "pair": {
1676
+ "source": "ara",
1677
+ "target": "fra"
1678
+ },
1679
+ "signal_strength": {
1680
+ "grade": "C",
1681
+ "corpus_size": null,
1682
+ "example_length": "sentence",
1683
+ "domain_breadth": "broad",
1684
+ "contamination": "MEDIUM",
1685
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1686
+ }
1687
+ },
1688
+ {
1689
+ "id": "opus-ara-heb-tatoeba-chrf2",
1690
+ "model": "OPUS-MT ara-heb",
1691
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1692
+ "org": "Helsinki-NLP",
1693
+ "year": 2021,
1694
+ "category": "model-report",
1695
+ "benchmark": "Tatoeba-test (ara-heb)",
1696
+ "metric": "chrF2",
1697
+ "value": 60.5,
1698
+ "lower_is_better": false,
1699
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1700
+ "headline": "OPUS-MT ara→heb: chrF2 60.5 on Tatoeba-test.",
1701
+ "langs_or_pairs": "ara→heb (single directed pair).",
1702
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1703
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1704
+ "verified": true,
1705
+ "badge": "unverified — not reproduced by Champollion",
1706
+ "notes": null,
1707
+ "pair": {
1708
+ "source": "ara",
1709
+ "target": "heb"
1710
+ },
1711
+ "signal_strength": {
1712
+ "grade": "C",
1713
+ "corpus_size": null,
1714
+ "example_length": "sentence",
1715
+ "domain_breadth": "broad",
1716
+ "contamination": "MEDIUM",
1717
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1718
+ }
1719
+ },
1720
+ {
1721
+ "id": "opus-ara-ita-tatoeba-chrf2",
1722
+ "model": "OPUS-MT ara-ita",
1723
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1724
+ "org": "Helsinki-NLP",
1725
+ "year": 2021,
1726
+ "category": "model-report",
1727
+ "benchmark": "Tatoeba-test (ara-ita)",
1728
+ "metric": "chrF2",
1729
+ "value": 63.9,
1730
+ "lower_is_better": false,
1731
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1732
+ "headline": "OPUS-MT ara→ita: chrF2 63.9 on Tatoeba-test.",
1733
+ "langs_or_pairs": "ara→ita (single directed pair).",
1734
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1735
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1736
+ "verified": true,
1737
+ "badge": "unverified — not reproduced by Champollion",
1738
+ "notes": null,
1739
+ "pair": {
1740
+ "source": "ara",
1741
+ "target": "ita"
1742
+ },
1743
+ "signal_strength": {
1744
+ "grade": "C",
1745
+ "corpus_size": null,
1746
+ "example_length": "sentence",
1747
+ "domain_breadth": "broad",
1748
+ "contamination": "MEDIUM",
1749
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1750
+ }
1751
+ },
1752
+ {
1753
+ "id": "opus-ara-jpn-tatoeba-chrf2",
1754
+ "model": "OPUS-MT ara-jpn",
1755
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1756
+ "org": "Helsinki-NLP",
1757
+ "year": 2021,
1758
+ "category": "model-report",
1759
+ "benchmark": "Tatoeba-test (ara-jpn)",
1760
+ "metric": "chrF2",
1761
+ "value": 19.6,
1762
+ "lower_is_better": false,
1763
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1764
+ "headline": "OPUS-MT ara→jpn: chrF2 19.6 on Tatoeba-test.",
1765
+ "langs_or_pairs": "ara→jpn (single directed pair).",
1766
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1767
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1768
+ "verified": true,
1769
+ "badge": "unverified — not reproduced by Champollion",
1770
+ "notes": null,
1771
+ "pair": {
1772
+ "source": "ara",
1773
+ "target": "jpn"
1774
+ },
1775
+ "signal_strength": {
1776
+ "grade": "C",
1777
+ "corpus_size": null,
1778
+ "example_length": "sentence",
1779
+ "domain_breadth": "broad",
1780
+ "contamination": "MEDIUM",
1781
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1782
+ }
1783
+ },
1784
+ {
1785
+ "id": "opus-ara-rus-tatoeba-chrf2",
1786
+ "model": "OPUS-MT ara-rus",
1787
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1788
+ "org": "Helsinki-NLP",
1789
+ "year": 2021,
1790
+ "category": "model-report",
1791
+ "benchmark": "Tatoeba-test (ara-rus)",
1792
+ "metric": "chrF2",
1793
+ "value": 60.6,
1794
+ "lower_is_better": false,
1795
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1796
+ "headline": "OPUS-MT ara→rus: chrF2 60.6 on Tatoeba-test.",
1797
+ "langs_or_pairs": "ara→rus (single directed pair).",
1798
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1799
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1800
+ "verified": true,
1801
+ "badge": "unverified — not reproduced by Champollion",
1802
+ "notes": null,
1803
+ "pair": {
1804
+ "source": "ara",
1805
+ "target": "rus"
1806
+ },
1807
+ "signal_strength": {
1808
+ "grade": "C",
1809
+ "corpus_size": null,
1810
+ "example_length": "sentence",
1811
+ "domain_breadth": "broad",
1812
+ "contamination": "MEDIUM",
1813
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1814
+ }
1815
+ },
1816
+ {
1817
+ "id": "opus-ara-tur-tatoeba-chrf2",
1818
+ "model": "OPUS-MT ara-tur",
1819
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1820
+ "org": "Helsinki-NLP",
1821
+ "year": 2021,
1822
+ "category": "model-report",
1823
+ "benchmark": "Tatoeba-test (ara-tur)",
1824
+ "metric": "chrF2",
1825
+ "value": 61.4,
1826
+ "lower_is_better": false,
1827
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1828
+ "headline": "OPUS-MT ara→tur: chrF2 61.4 on Tatoeba-test.",
1829
+ "langs_or_pairs": "ara→tur (single directed pair).",
1830
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1831
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1832
+ "verified": true,
1833
+ "badge": "unverified — not reproduced by Champollion",
1834
+ "notes": null,
1835
+ "pair": {
1836
+ "source": "ara",
1837
+ "target": "tur"
1838
+ },
1839
+ "signal_strength": {
1840
+ "grade": "C",
1841
+ "corpus_size": null,
1842
+ "example_length": "sentence",
1843
+ "domain_breadth": "broad",
1844
+ "contamination": "MEDIUM",
1845
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1846
+ }
1847
+ },
1848
+ {
1849
+ "id": "opus-ben-eng-tatoeba-chrf2",
1850
+ "model": "OPUS-MT ben-eng",
1851
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1852
+ "org": "Helsinki-NLP",
1853
+ "year": 2021,
1854
+ "category": "model-report",
1855
+ "benchmark": "Tatoeba-test (ben-eng)",
1856
+ "metric": "chrF2",
1857
+ "value": 63.9,
1858
+ "lower_is_better": false,
1859
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1860
+ "headline": "OPUS-MT ben→eng: chrF2 63.9 on Tatoeba-test.",
1861
+ "langs_or_pairs": "ben→eng (single directed pair).",
1862
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1863
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1864
+ "verified": true,
1865
+ "badge": "unverified — not reproduced by Champollion",
1866
+ "notes": null,
1867
+ "pair": {
1868
+ "source": "ben",
1869
+ "target": "eng"
1870
+ },
1871
+ "signal_strength": {
1872
+ "grade": "C",
1873
+ "corpus_size": null,
1874
+ "example_length": "sentence",
1875
+ "domain_breadth": "broad",
1876
+ "contamination": "MEDIUM",
1877
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1878
+ }
1879
+ },
1880
+ {
1881
+ "id": "opus-ber-eng-tatoeba-chrf2",
1882
+ "model": "OPUS-MT ber-eng",
1883
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1884
+ "org": "Helsinki-NLP",
1885
+ "year": 2021,
1886
+ "category": "model-report",
1887
+ "benchmark": "Tatoeba-test (ber-eng)",
1888
+ "metric": "chrF2",
1889
+ "value": 9.5,
1890
+ "lower_is_better": false,
1891
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1892
+ "headline": "OPUS-MT ber→eng: chrF2 9.5 on Tatoeba-test.",
1893
+ "langs_or_pairs": "ber→eng (single directed pair).",
1894
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1895
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1896
+ "verified": true,
1897
+ "badge": "unverified — not reproduced by Champollion",
1898
+ "notes": null,
1899
+ "pair": {
1900
+ "source": "ber",
1901
+ "target": "eng"
1902
+ },
1903
+ "signal_strength": {
1904
+ "grade": "C",
1905
+ "corpus_size": null,
1906
+ "example_length": "sentence",
1907
+ "domain_breadth": "broad",
1908
+ "contamination": "MEDIUM",
1909
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1910
+ }
1911
+ },
1912
+ {
1913
+ "id": "opus-bre-eng-tatoeba-chrf2",
1914
+ "model": "OPUS-MT bre-eng",
1915
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1916
+ "org": "Helsinki-NLP",
1917
+ "year": 2021,
1918
+ "category": "model-report",
1919
+ "benchmark": "Tatoeba-test (bre-eng)",
1920
+ "metric": "chrF2",
1921
+ "value": 25.6,
1922
+ "lower_is_better": false,
1923
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1924
+ "headline": "OPUS-MT bre→eng: chrF2 25.6 on Tatoeba-test.",
1925
+ "langs_or_pairs": "bre→eng (single directed pair).",
1926
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1927
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1928
+ "verified": true,
1929
+ "badge": "unverified — not reproduced by Champollion",
1930
+ "notes": null,
1931
+ "pair": {
1932
+ "source": "bre",
1933
+ "target": "eng"
1934
+ },
1935
+ "signal_strength": {
1936
+ "grade": "C",
1937
+ "corpus_size": null,
1938
+ "example_length": "sentence",
1939
+ "domain_breadth": "broad",
1940
+ "contamination": "MEDIUM",
1941
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1942
+ }
1943
+ },
1944
+ {
1945
+ "id": "opus-bre-fra-tatoeba-chrf2",
1946
+ "model": "OPUS-MT bre-fra",
1947
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1948
+ "org": "Helsinki-NLP",
1949
+ "year": 2021,
1950
+ "category": "model-report",
1951
+ "benchmark": "Tatoeba-test (bre-fra)",
1952
+ "metric": "chrF2",
1953
+ "value": 23.3,
1954
+ "lower_is_better": false,
1955
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1956
+ "headline": "OPUS-MT bre→fra: chrF2 23.3 on Tatoeba-test.",
1957
+ "langs_or_pairs": "bre→fra (single directed pair).",
1958
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1959
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1960
+ "verified": true,
1961
+ "badge": "unverified — not reproduced by Champollion",
1962
+ "notes": null,
1963
+ "pair": {
1964
+ "source": "bre",
1965
+ "target": "fra"
1966
+ },
1967
+ "signal_strength": {
1968
+ "grade": "C",
1969
+ "corpus_size": null,
1970
+ "example_length": "sentence",
1971
+ "domain_breadth": "broad",
1972
+ "contamination": "MEDIUM",
1973
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
1974
+ }
1975
+ },
1976
+ {
1977
+ "id": "opus-bul-deu-tatoeba-chrf2",
1978
+ "model": "OPUS-MT bul-deu",
1979
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
1980
+ "org": "Helsinki-NLP",
1981
+ "year": 2021,
1982
+ "category": "model-report",
1983
+ "benchmark": "Tatoeba-test (bul-deu)",
1984
+ "metric": "chrF2",
1985
+ "value": 67.7,
1986
+ "lower_is_better": false,
1987
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
1988
+ "headline": "OPUS-MT bul→deu: chrF2 67.7 on Tatoeba-test.",
1989
+ "langs_or_pairs": "bul→deu (single directed pair).",
1990
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
1991
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
1992
+ "verified": true,
1993
+ "badge": "unverified — not reproduced by Champollion",
1994
+ "notes": null,
1995
+ "pair": {
1996
+ "source": "bul",
1997
+ "target": "deu"
1998
+ },
1999
+ "signal_strength": {
2000
+ "grade": "C",
2001
+ "corpus_size": null,
2002
+ "example_length": "sentence",
2003
+ "domain_breadth": "broad",
2004
+ "contamination": "MEDIUM",
2005
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2006
+ }
2007
+ },
2008
+ {
2009
+ "id": "opus-bul-eng-tatoeba-chrf2",
2010
+ "model": "OPUS-MT bul-eng",
2011
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2012
+ "org": "Helsinki-NLP",
2013
+ "year": 2021,
2014
+ "category": "model-report",
2015
+ "benchmark": "Tatoeba-test (bul-eng)",
2016
+ "metric": "chrF2",
2017
+ "value": 72,
2018
+ "lower_is_better": false,
2019
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2020
+ "headline": "OPUS-MT bul→eng: chrF2 72 on Tatoeba-test.",
2021
+ "langs_or_pairs": "bul→eng (single directed pair).",
2022
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2023
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2024
+ "verified": true,
2025
+ "badge": "unverified — not reproduced by Champollion",
2026
+ "notes": null,
2027
+ "pair": {
2028
+ "source": "bul",
2029
+ "target": "eng"
2030
+ },
2031
+ "signal_strength": {
2032
+ "grade": "C",
2033
+ "corpus_size": null,
2034
+ "example_length": "sentence",
2035
+ "domain_breadth": "broad",
2036
+ "contamination": "MEDIUM",
2037
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2038
+ }
2039
+ },
2040
+ {
2041
+ "id": "opus-bul-epo-tatoeba-chrf2",
2042
+ "model": "OPUS-MT bul-epo",
2043
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2044
+ "org": "Helsinki-NLP",
2045
+ "year": 2021,
2046
+ "category": "model-report",
2047
+ "benchmark": "Tatoeba-test (bul-epo)",
2048
+ "metric": "chrF2",
2049
+ "value": 43.8,
2050
+ "lower_is_better": false,
2051
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2052
+ "headline": "OPUS-MT bul→epo: chrF2 43.8 on Tatoeba-test.",
2053
+ "langs_or_pairs": "bul→epo (single directed pair).",
2054
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2055
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2056
+ "verified": true,
2057
+ "badge": "unverified — not reproduced by Champollion",
2058
+ "notes": null,
2059
+ "pair": {
2060
+ "source": "bul",
2061
+ "target": "epo"
2062
+ },
2063
+ "signal_strength": {
2064
+ "grade": "C",
2065
+ "corpus_size": null,
2066
+ "example_length": "sentence",
2067
+ "domain_breadth": "broad",
2068
+ "contamination": "MEDIUM",
2069
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2070
+ }
2071
+ },
2072
+ {
2073
+ "id": "opus-bul-fra-tatoeba-chrf2",
2074
+ "model": "OPUS-MT bul-fra",
2075
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2076
+ "org": "Helsinki-NLP",
2077
+ "year": 2021,
2078
+ "category": "model-report",
2079
+ "benchmark": "Tatoeba-test (bul-fra)",
2080
+ "metric": "chrF2",
2081
+ "value": 69.3,
2082
+ "lower_is_better": false,
2083
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2084
+ "headline": "OPUS-MT bul→fra: chrF2 69.3 on Tatoeba-test.",
2085
+ "langs_or_pairs": "bul→fra (single directed pair).",
2086
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2087
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2088
+ "verified": true,
2089
+ "badge": "unverified — not reproduced by Champollion",
2090
+ "notes": null,
2091
+ "pair": {
2092
+ "source": "bul",
2093
+ "target": "fra"
2094
+ },
2095
+ "signal_strength": {
2096
+ "grade": "C",
2097
+ "corpus_size": null,
2098
+ "example_length": "sentence",
2099
+ "domain_breadth": "broad",
2100
+ "contamination": "MEDIUM",
2101
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2102
+ }
2103
+ },
2104
+ {
2105
+ "id": "opus-bul-ita-tatoeba-chrf2",
2106
+ "model": "OPUS-MT bul-ita",
2107
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2108
+ "org": "Helsinki-NLP",
2109
+ "year": 2021,
2110
+ "category": "model-report",
2111
+ "benchmark": "Tatoeba-test (bul-ita)",
2112
+ "metric": "chrF2",
2113
+ "value": 65.3,
2114
+ "lower_is_better": false,
2115
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2116
+ "headline": "OPUS-MT bul→ita: chrF2 65.3 on Tatoeba-test.",
2117
+ "langs_or_pairs": "bul→ita (single directed pair).",
2118
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2119
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2120
+ "verified": true,
2121
+ "badge": "unverified — not reproduced by Champollion",
2122
+ "notes": null,
2123
+ "pair": {
2124
+ "source": "bul",
2125
+ "target": "ita"
2126
+ },
2127
+ "signal_strength": {
2128
+ "grade": "C",
2129
+ "corpus_size": null,
2130
+ "example_length": "sentence",
2131
+ "domain_breadth": "broad",
2132
+ "contamination": "MEDIUM",
2133
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2134
+ }
2135
+ },
2136
+ {
2137
+ "id": "opus-bul-jpn-tatoeba-chrf2",
2138
+ "model": "OPUS-MT bul-jpn",
2139
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2140
+ "org": "Helsinki-NLP",
2141
+ "year": 2021,
2142
+ "category": "model-report",
2143
+ "benchmark": "Tatoeba-test (bul-jpn)",
2144
+ "metric": "chrF2",
2145
+ "value": 24.1,
2146
+ "lower_is_better": false,
2147
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2148
+ "headline": "OPUS-MT bul→jpn: chrF2 24.1 on Tatoeba-test.",
2149
+ "langs_or_pairs": "bul→jpn (single directed pair).",
2150
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2151
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2152
+ "verified": true,
2153
+ "badge": "unverified — not reproduced by Champollion",
2154
+ "notes": null,
2155
+ "pair": {
2156
+ "source": "bul",
2157
+ "target": "jpn"
2158
+ },
2159
+ "signal_strength": {
2160
+ "grade": "C",
2161
+ "corpus_size": null,
2162
+ "example_length": "sentence",
2163
+ "domain_breadth": "broad",
2164
+ "contamination": "MEDIUM",
2165
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2166
+ }
2167
+ },
2168
+ {
2169
+ "id": "opus-bul-spa-tatoeba-chrf2",
2170
+ "model": "OPUS-MT bul-spa",
2171
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2172
+ "org": "Helsinki-NLP",
2173
+ "year": 2021,
2174
+ "category": "model-report",
2175
+ "benchmark": "Tatoeba-test (bul-spa)",
2176
+ "metric": "chrF2",
2177
+ "value": 66.1,
2178
+ "lower_is_better": false,
2179
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2180
+ "headline": "OPUS-MT bul→spa: chrF2 66.1 on Tatoeba-test.",
2181
+ "langs_or_pairs": "bul→spa (single directed pair).",
2182
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2183
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2184
+ "verified": true,
2185
+ "badge": "unverified — not reproduced by Champollion",
2186
+ "notes": null,
2187
+ "pair": {
2188
+ "source": "bul",
2189
+ "target": "spa"
2190
+ },
2191
+ "signal_strength": {
2192
+ "grade": "C",
2193
+ "corpus_size": null,
2194
+ "example_length": "sentence",
2195
+ "domain_breadth": "broad",
2196
+ "contamination": "MEDIUM",
2197
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2198
+ }
2199
+ },
2200
+ {
2201
+ "id": "opus-bul-tur-tatoeba-chrf2",
2202
+ "model": "OPUS-MT bul-tur",
2203
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2204
+ "org": "Helsinki-NLP",
2205
+ "year": 2021,
2206
+ "category": "model-report",
2207
+ "benchmark": "Tatoeba-test (bul-tur)",
2208
+ "metric": "chrF2",
2209
+ "value": 68.6,
2210
+ "lower_is_better": false,
2211
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2212
+ "headline": "OPUS-MT bul→tur: chrF2 68.6 on Tatoeba-test.",
2213
+ "langs_or_pairs": "bul→tur (single directed pair).",
2214
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2215
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2216
+ "verified": true,
2217
+ "badge": "unverified — not reproduced by Champollion",
2218
+ "notes": null,
2219
+ "pair": {
2220
+ "source": "bul",
2221
+ "target": "tur"
2222
+ },
2223
+ "signal_strength": {
2224
+ "grade": "C",
2225
+ "corpus_size": null,
2226
+ "example_length": "sentence",
2227
+ "domain_breadth": "broad",
2228
+ "contamination": "MEDIUM",
2229
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2230
+ }
2231
+ },
2232
+ {
2233
+ "id": "opus-bul-ukr-tatoeba-chrf2",
2234
+ "model": "OPUS-MT bul-ukr",
2235
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2236
+ "org": "Helsinki-NLP",
2237
+ "year": 2021,
2238
+ "category": "model-report",
2239
+ "benchmark": "Tatoeba-test (bul-ukr)",
2240
+ "metric": "chrF2",
2241
+ "value": 68.3,
2242
+ "lower_is_better": false,
2243
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2244
+ "headline": "OPUS-MT bul→ukr: chrF2 68.3 on Tatoeba-test.",
2245
+ "langs_or_pairs": "bul→ukr (single directed pair).",
2246
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2247
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2248
+ "verified": true,
2249
+ "badge": "unverified — not reproduced by Champollion",
2250
+ "notes": null,
2251
+ "pair": {
2252
+ "source": "bul",
2253
+ "target": "ukr"
2254
+ },
2255
+ "signal_strength": {
2256
+ "grade": "C",
2257
+ "corpus_size": null,
2258
+ "example_length": "sentence",
2259
+ "domain_breadth": "broad",
2260
+ "contamination": "MEDIUM",
2261
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2262
+ }
2263
+ },
2264
+ {
2265
+ "id": "opus-cat-deu-tatoeba-chrf2",
2266
+ "model": "OPUS-MT cat-deu",
2267
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2268
+ "org": "Helsinki-NLP",
2269
+ "year": 2021,
2270
+ "category": "model-report",
2271
+ "benchmark": "Tatoeba-test (cat-deu)",
2272
+ "metric": "chrF2",
2273
+ "value": 59.3,
2274
+ "lower_is_better": false,
2275
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2276
+ "headline": "OPUS-MT cat→deu: chrF2 59.3 on Tatoeba-test.",
2277
+ "langs_or_pairs": "cat→deu (single directed pair).",
2278
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2279
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2280
+ "verified": true,
2281
+ "badge": "unverified — not reproduced by Champollion",
2282
+ "notes": null,
2283
+ "pair": {
2284
+ "source": "cat",
2285
+ "target": "deu"
2286
+ },
2287
+ "signal_strength": {
2288
+ "grade": "C",
2289
+ "corpus_size": null,
2290
+ "example_length": "sentence",
2291
+ "domain_breadth": "broad",
2292
+ "contamination": "MEDIUM",
2293
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2294
+ }
2295
+ },
2296
+ {
2297
+ "id": "opus-cat-eng-tatoeba-chrf2",
2298
+ "model": "OPUS-MT cat-eng",
2299
+ "source": "Tatoeba-Challenge results (Helsinki-NLP)",
2300
+ "org": "Helsinki-NLP",
2301
+ "year": 2021,
2302
+ "category": "model-report",
2303
+ "benchmark": "Tatoeba-test (cat-eng)",
2304
+ "metric": "chrF2",
2305
+ "value": 66.8,
2306
+ "lower_is_better": false,
2307
+ "metric_variant_flag": "chrF2 (chrF, beta=2) on the Tatoeba-test set — 0–100, higher better, read from the Tatoeba-Challenge results tables. NOT comparable to BLEU/spBLEU, COMET/MetricX, or chrF++ from other setups.",
2308
+ "headline": "OPUS-MT cat→eng: chrF2 66.8 on Tatoeba-test.",
2309
+ "langs_or_pairs": "cat→eng (single directed pair).",
2310
+ "source_url": "https://raw.githubusercontent.com/Helsinki-NLP/Tatoeba-Challenge/master/results/tatoeba-results-chrF2-sorted-langpair.md",
2311
+ "citation": "Tiedemann (2020), The Tatoeba Translation Challenge; per-pair chrF from the repo results tables.",
2312
+ "verified": true,
2313
+ "badge": "unverified — not reproduced by Champollion",
2314
+ "notes": null,
2315
+ "pair": {
2316
+ "source": "cat",
2317
+ "target": "eng"
2318
+ },
2319
+ "signal_strength": {
2320
+ "grade": "C",
2321
+ "corpus_size": null,
2322
+ "example_length": "sentence",
2323
+ "domain_breadth": "broad",
2324
+ "contamination": "MEDIUM",
2325
+ "rationale": "Tatoeba-test, read from the Tatoeba-Challenge results tables. Test-set size varies by pair (often small for low-resource); general crowd-sourced sentences widely available online (MEDIUM contamination). Conservative C default for this cite-only tier."
2326
+ }
2327
+ },
2328
+ {
2329
+ "id": "nllb-amh-eng-flores-chrfpp",
2330
+ "model": "NLLB-200 3.3B",
2331
+ "source": "NLLB-200 published FLORES-200 metrics",
2332
+ "org": "Meta AI",
2333
+ "year": 2022,
2334
+ "category": "model-report",
2335
+ "benchmark": "FLORES-200 devtest (amh-eng)",
2336
+ "metric": "chrF++",
2337
+ "value": 59,
2338
+ "lower_is_better": false,
2339
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2340
+ "headline": "NLLB-200 3.3B amh→eng: chrF++ 59 on FLORES-200.",
2341
+ "langs_or_pairs": "amh→eng (single directed pair).",
2342
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2343
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2344
+ "verified": true,
2345
+ "badge": "unverified — not reproduced by Champollion",
2346
+ "notes": null,
2347
+ "pair": {
2348
+ "source": "amh",
2349
+ "target": "eng"
2350
+ },
2351
+ "signal_strength": {
2352
+ "grade": "B",
2353
+ "corpus_size": 1012,
2354
+ "example_length": "sentence",
2355
+ "domain_breadth": "multi-domain",
2356
+ "contamination": "MEDIUM",
2357
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2358
+ }
2359
+ },
2360
+ {
2361
+ "id": "nllb-arb-eng-flores-chrfpp",
2362
+ "model": "NLLB-200 3.3B",
2363
+ "source": "NLLB-200 published FLORES-200 metrics",
2364
+ "org": "Meta AI",
2365
+ "year": 2022,
2366
+ "category": "model-report",
2367
+ "benchmark": "FLORES-200 devtest (arb-eng)",
2368
+ "metric": "chrF++",
2369
+ "value": 65.8,
2370
+ "lower_is_better": false,
2371
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2372
+ "headline": "NLLB-200 3.3B arb→eng: chrF++ 65.8 on FLORES-200.",
2373
+ "langs_or_pairs": "arb→eng (single directed pair).",
2374
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2375
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2376
+ "verified": true,
2377
+ "badge": "unverified — not reproduced by Champollion",
2378
+ "notes": null,
2379
+ "pair": {
2380
+ "source": "arb",
2381
+ "target": "eng"
2382
+ },
2383
+ "signal_strength": {
2384
+ "grade": "B",
2385
+ "corpus_size": 1012,
2386
+ "example_length": "sentence",
2387
+ "domain_breadth": "multi-domain",
2388
+ "contamination": "MEDIUM",
2389
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2390
+ }
2391
+ },
2392
+ {
2393
+ "id": "nllb-ben-eng-flores-chrfpp",
2394
+ "model": "NLLB-200 3.3B",
2395
+ "source": "NLLB-200 published FLORES-200 metrics",
2396
+ "org": "Meta AI",
2397
+ "year": 2022,
2398
+ "category": "model-report",
2399
+ "benchmark": "FLORES-200 devtest (ben-eng)",
2400
+ "metric": "chrF++",
2401
+ "value": 61.1,
2402
+ "lower_is_better": false,
2403
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2404
+ "headline": "NLLB-200 3.3B ben→eng: chrF++ 61.1 on FLORES-200.",
2405
+ "langs_or_pairs": "ben→eng (single directed pair).",
2406
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2407
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2408
+ "verified": true,
2409
+ "badge": "unverified — not reproduced by Champollion",
2410
+ "notes": null,
2411
+ "pair": {
2412
+ "source": "ben",
2413
+ "target": "eng"
2414
+ },
2415
+ "signal_strength": {
2416
+ "grade": "B",
2417
+ "corpus_size": 1012,
2418
+ "example_length": "sentence",
2419
+ "domain_breadth": "multi-domain",
2420
+ "contamination": "MEDIUM",
2421
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2422
+ }
2423
+ },
2424
+ {
2425
+ "id": "nllb-deu-eng-flores-chrfpp",
2426
+ "model": "NLLB-200 3.3B",
2427
+ "source": "NLLB-200 published FLORES-200 metrics",
2428
+ "org": "Meta AI",
2429
+ "year": 2022,
2430
+ "category": "model-report",
2431
+ "benchmark": "FLORES-200 devtest (deu-eng)",
2432
+ "metric": "chrF++",
2433
+ "value": 67.4,
2434
+ "lower_is_better": false,
2435
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2436
+ "headline": "NLLB-200 3.3B deu→eng: chrF++ 67.4 on FLORES-200.",
2437
+ "langs_or_pairs": "deu→eng (single directed pair).",
2438
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2439
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2440
+ "verified": true,
2441
+ "badge": "unverified — not reproduced by Champollion",
2442
+ "notes": null,
2443
+ "pair": {
2444
+ "source": "deu",
2445
+ "target": "eng"
2446
+ },
2447
+ "signal_strength": {
2448
+ "grade": "B",
2449
+ "corpus_size": 1012,
2450
+ "example_length": "sentence",
2451
+ "domain_breadth": "multi-domain",
2452
+ "contamination": "MEDIUM",
2453
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2454
+ }
2455
+ },
2456
+ {
2457
+ "id": "nllb-ell-eng-flores-chrfpp",
2458
+ "model": "NLLB-200 3.3B",
2459
+ "source": "NLLB-200 published FLORES-200 metrics",
2460
+ "org": "Meta AI",
2461
+ "year": 2022,
2462
+ "category": "model-report",
2463
+ "benchmark": "FLORES-200 devtest (ell-eng)",
2464
+ "metric": "chrF++",
2465
+ "value": 62,
2466
+ "lower_is_better": false,
2467
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2468
+ "headline": "NLLB-200 3.3B ell→eng: chrF++ 62 on FLORES-200.",
2469
+ "langs_or_pairs": "ell→eng (single directed pair).",
2470
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2471
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2472
+ "verified": true,
2473
+ "badge": "unverified — not reproduced by Champollion",
2474
+ "notes": null,
2475
+ "pair": {
2476
+ "source": "ell",
2477
+ "target": "eng"
2478
+ },
2479
+ "signal_strength": {
2480
+ "grade": "B",
2481
+ "corpus_size": 1012,
2482
+ "example_length": "sentence",
2483
+ "domain_breadth": "multi-domain",
2484
+ "contamination": "MEDIUM",
2485
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2486
+ }
2487
+ },
2488
+ {
2489
+ "id": "nllb-fra-eng-flores-chrfpp",
2490
+ "model": "NLLB-200 3.3B",
2491
+ "source": "NLLB-200 published FLORES-200 metrics",
2492
+ "org": "Meta AI",
2493
+ "year": 2022,
2494
+ "category": "model-report",
2495
+ "benchmark": "FLORES-200 devtest (fra-eng)",
2496
+ "metric": "chrF++",
2497
+ "value": 68.1,
2498
+ "lower_is_better": false,
2499
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2500
+ "headline": "NLLB-200 3.3B fra→eng: chrF++ 68.1 on FLORES-200.",
2501
+ "langs_or_pairs": "fra→eng (single directed pair).",
2502
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2503
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2504
+ "verified": true,
2505
+ "badge": "unverified — not reproduced by Champollion",
2506
+ "notes": null,
2507
+ "pair": {
2508
+ "source": "fra",
2509
+ "target": "eng"
2510
+ },
2511
+ "signal_strength": {
2512
+ "grade": "B",
2513
+ "corpus_size": 1012,
2514
+ "example_length": "sentence",
2515
+ "domain_breadth": "multi-domain",
2516
+ "contamination": "MEDIUM",
2517
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2518
+ }
2519
+ },
2520
+ {
2521
+ "id": "nllb-hin-eng-flores-chrfpp",
2522
+ "model": "NLLB-200 3.3B",
2523
+ "source": "NLLB-200 published FLORES-200 metrics",
2524
+ "org": "Meta AI",
2525
+ "year": 2022,
2526
+ "category": "model-report",
2527
+ "benchmark": "FLORES-200 devtest (hin-eng)",
2528
+ "metric": "chrF++",
2529
+ "value": 65.9,
2530
+ "lower_is_better": false,
2531
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2532
+ "headline": "NLLB-200 3.3B hin→eng: chrF++ 65.9 on FLORES-200.",
2533
+ "langs_or_pairs": "hin→eng (single directed pair).",
2534
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2535
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2536
+ "verified": true,
2537
+ "badge": "unverified — not reproduced by Champollion",
2538
+ "notes": null,
2539
+ "pair": {
2540
+ "source": "hin",
2541
+ "target": "eng"
2542
+ },
2543
+ "signal_strength": {
2544
+ "grade": "B",
2545
+ "corpus_size": 1012,
2546
+ "example_length": "sentence",
2547
+ "domain_breadth": "multi-domain",
2548
+ "contamination": "MEDIUM",
2549
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2550
+ }
2551
+ },
2552
+ {
2553
+ "id": "nllb-isl-eng-flores-chrfpp",
2554
+ "model": "NLLB-200 3.3B",
2555
+ "source": "NLLB-200 published FLORES-200 metrics",
2556
+ "org": "Meta AI",
2557
+ "year": 2022,
2558
+ "category": "model-report",
2559
+ "benchmark": "FLORES-200 devtest (isl-eng)",
2560
+ "metric": "chrF++",
2561
+ "value": 56.6,
2562
+ "lower_is_better": false,
2563
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2564
+ "headline": "NLLB-200 3.3B isl→eng: chrF++ 56.6 on FLORES-200.",
2565
+ "langs_or_pairs": "isl→eng (single directed pair).",
2566
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2567
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2568
+ "verified": true,
2569
+ "badge": "unverified — not reproduced by Champollion",
2570
+ "notes": null,
2571
+ "pair": {
2572
+ "source": "isl",
2573
+ "target": "eng"
2574
+ },
2575
+ "signal_strength": {
2576
+ "grade": "B",
2577
+ "corpus_size": 1012,
2578
+ "example_length": "sentence",
2579
+ "domain_breadth": "multi-domain",
2580
+ "contamination": "MEDIUM",
2581
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2582
+ }
2583
+ },
2584
+ {
2585
+ "id": "nllb-ita-eng-flores-chrfpp",
2586
+ "model": "NLLB-200 3.3B",
2587
+ "source": "NLLB-200 published FLORES-200 metrics",
2588
+ "org": "Meta AI",
2589
+ "year": 2022,
2590
+ "category": "model-report",
2591
+ "benchmark": "FLORES-200 devtest (ita-eng)",
2592
+ "metric": "chrF++",
2593
+ "value": 61.2,
2594
+ "lower_is_better": false,
2595
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2596
+ "headline": "NLLB-200 3.3B ita→eng: chrF++ 61.2 on FLORES-200.",
2597
+ "langs_or_pairs": "ita→eng (single directed pair).",
2598
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2599
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2600
+ "verified": true,
2601
+ "badge": "unverified — not reproduced by Champollion",
2602
+ "notes": null,
2603
+ "pair": {
2604
+ "source": "ita",
2605
+ "target": "eng"
2606
+ },
2607
+ "signal_strength": {
2608
+ "grade": "B",
2609
+ "corpus_size": 1012,
2610
+ "example_length": "sentence",
2611
+ "domain_breadth": "multi-domain",
2612
+ "contamination": "MEDIUM",
2613
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2614
+ }
2615
+ },
2616
+ {
2617
+ "id": "nllb-jpn-eng-flores-chrfpp",
2618
+ "model": "NLLB-200 3.3B",
2619
+ "source": "NLLB-200 published FLORES-200 metrics",
2620
+ "org": "Meta AI",
2621
+ "year": 2022,
2622
+ "category": "model-report",
2623
+ "benchmark": "FLORES-200 devtest (jpn-eng)",
2624
+ "metric": "chrF++",
2625
+ "value": 55.1,
2626
+ "lower_is_better": false,
2627
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2628
+ "headline": "NLLB-200 3.3B jpn→eng: chrF++ 55.1 on FLORES-200.",
2629
+ "langs_or_pairs": "jpn→eng (single directed pair).",
2630
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2631
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2632
+ "verified": true,
2633
+ "badge": "unverified — not reproduced by Champollion",
2634
+ "notes": null,
2635
+ "pair": {
2636
+ "source": "jpn",
2637
+ "target": "eng"
2638
+ },
2639
+ "signal_strength": {
2640
+ "grade": "B",
2641
+ "corpus_size": 1012,
2642
+ "example_length": "sentence",
2643
+ "domain_breadth": "multi-domain",
2644
+ "contamination": "MEDIUM",
2645
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2646
+ }
2647
+ },
2648
+ {
2649
+ "id": "nllb-kor-eng-flores-chrfpp",
2650
+ "model": "NLLB-200 3.3B",
2651
+ "source": "NLLB-200 published FLORES-200 metrics",
2652
+ "org": "Meta AI",
2653
+ "year": 2022,
2654
+ "category": "model-report",
2655
+ "benchmark": "FLORES-200 devtest (kor-eng)",
2656
+ "metric": "chrF++",
2657
+ "value": 56.1,
2658
+ "lower_is_better": false,
2659
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2660
+ "headline": "NLLB-200 3.3B kor→eng: chrF++ 56.1 on FLORES-200.",
2661
+ "langs_or_pairs": "kor→eng (single directed pair).",
2662
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2663
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2664
+ "verified": true,
2665
+ "badge": "unverified — not reproduced by Champollion",
2666
+ "notes": null,
2667
+ "pair": {
2668
+ "source": "kor",
2669
+ "target": "eng"
2670
+ },
2671
+ "signal_strength": {
2672
+ "grade": "B",
2673
+ "corpus_size": 1012,
2674
+ "example_length": "sentence",
2675
+ "domain_breadth": "multi-domain",
2676
+ "contamination": "MEDIUM",
2677
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2678
+ }
2679
+ },
2680
+ {
2681
+ "id": "nllb-nld-eng-flores-chrfpp",
2682
+ "model": "NLLB-200 3.3B",
2683
+ "source": "NLLB-200 published FLORES-200 metrics",
2684
+ "org": "Meta AI",
2685
+ "year": 2022,
2686
+ "category": "model-report",
2687
+ "benchmark": "FLORES-200 devtest (nld-eng)",
2688
+ "metric": "chrF++",
2689
+ "value": 59,
2690
+ "lower_is_better": false,
2691
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2692
+ "headline": "NLLB-200 3.3B nld→eng: chrF++ 59 on FLORES-200.",
2693
+ "langs_or_pairs": "nld→eng (single directed pair).",
2694
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2695
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2696
+ "verified": true,
2697
+ "badge": "unverified — not reproduced by Champollion",
2698
+ "notes": null,
2699
+ "pair": {
2700
+ "source": "nld",
2701
+ "target": "eng"
2702
+ },
2703
+ "signal_strength": {
2704
+ "grade": "B",
2705
+ "corpus_size": 1012,
2706
+ "example_length": "sentence",
2707
+ "domain_breadth": "multi-domain",
2708
+ "contamination": "MEDIUM",
2709
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2710
+ }
2711
+ },
2712
+ {
2713
+ "id": "nllb-pol-eng-flores-chrfpp",
2714
+ "model": "NLLB-200 3.3B",
2715
+ "source": "NLLB-200 published FLORES-200 metrics",
2716
+ "org": "Meta AI",
2717
+ "year": 2022,
2718
+ "category": "model-report",
2719
+ "benchmark": "FLORES-200 devtest (pol-eng)",
2720
+ "metric": "chrF++",
2721
+ "value": 57,
2722
+ "lower_is_better": false,
2723
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2724
+ "headline": "NLLB-200 3.3B pol→eng: chrF++ 57 on FLORES-200.",
2725
+ "langs_or_pairs": "pol→eng (single directed pair).",
2726
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2727
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2728
+ "verified": true,
2729
+ "badge": "unverified — not reproduced by Champollion",
2730
+ "notes": null,
2731
+ "pair": {
2732
+ "source": "pol",
2733
+ "target": "eng"
2734
+ },
2735
+ "signal_strength": {
2736
+ "grade": "B",
2737
+ "corpus_size": 1012,
2738
+ "example_length": "sentence",
2739
+ "domain_breadth": "multi-domain",
2740
+ "contamination": "MEDIUM",
2741
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2742
+ }
2743
+ },
2744
+ {
2745
+ "id": "nllb-por-eng-flores-chrfpp",
2746
+ "model": "NLLB-200 3.3B",
2747
+ "source": "NLLB-200 published FLORES-200 metrics",
2748
+ "org": "Meta AI",
2749
+ "year": 2022,
2750
+ "category": "model-report",
2751
+ "benchmark": "FLORES-200 devtest (por-eng)",
2752
+ "metric": "chrF++",
2753
+ "value": 71.3,
2754
+ "lower_is_better": false,
2755
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2756
+ "headline": "NLLB-200 3.3B por→eng: chrF++ 71.3 on FLORES-200.",
2757
+ "langs_or_pairs": "por→eng (single directed pair).",
2758
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2759
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2760
+ "verified": true,
2761
+ "badge": "unverified — not reproduced by Champollion",
2762
+ "notes": null,
2763
+ "pair": {
2764
+ "source": "por",
2765
+ "target": "eng"
2766
+ },
2767
+ "signal_strength": {
2768
+ "grade": "B",
2769
+ "corpus_size": 1012,
2770
+ "example_length": "sentence",
2771
+ "domain_breadth": "multi-domain",
2772
+ "contamination": "MEDIUM",
2773
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2774
+ }
2775
+ },
2776
+ {
2777
+ "id": "nllb-ron-eng-flores-chrfpp",
2778
+ "model": "NLLB-200 3.3B",
2779
+ "source": "NLLB-200 published FLORES-200 metrics",
2780
+ "org": "Meta AI",
2781
+ "year": 2022,
2782
+ "category": "model-report",
2783
+ "benchmark": "FLORES-200 devtest (ron-eng)",
2784
+ "metric": "chrF++",
2785
+ "value": 68.1,
2786
+ "lower_is_better": false,
2787
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2788
+ "headline": "NLLB-200 3.3B ron→eng: chrF++ 68.1 on FLORES-200.",
2789
+ "langs_or_pairs": "ron→eng (single directed pair).",
2790
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2791
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2792
+ "verified": true,
2793
+ "badge": "unverified — not reproduced by Champollion",
2794
+ "notes": null,
2795
+ "pair": {
2796
+ "source": "ron",
2797
+ "target": "eng"
2798
+ },
2799
+ "signal_strength": {
2800
+ "grade": "B",
2801
+ "corpus_size": 1012,
2802
+ "example_length": "sentence",
2803
+ "domain_breadth": "multi-domain",
2804
+ "contamination": "MEDIUM",
2805
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2806
+ }
2807
+ },
2808
+ {
2809
+ "id": "nllb-rus-eng-flores-chrfpp",
2810
+ "model": "NLLB-200 3.3B",
2811
+ "source": "NLLB-200 published FLORES-200 metrics",
2812
+ "org": "Meta AI",
2813
+ "year": 2022,
2814
+ "category": "model-report",
2815
+ "benchmark": "FLORES-200 devtest (rus-eng)",
2816
+ "metric": "chrF++",
2817
+ "value": 61.3,
2818
+ "lower_is_better": false,
2819
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2820
+ "headline": "NLLB-200 3.3B rus→eng: chrF++ 61.3 on FLORES-200.",
2821
+ "langs_or_pairs": "rus→eng (single directed pair).",
2822
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2823
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2824
+ "verified": true,
2825
+ "badge": "unverified — not reproduced by Champollion",
2826
+ "notes": null,
2827
+ "pair": {
2828
+ "source": "rus",
2829
+ "target": "eng"
2830
+ },
2831
+ "signal_strength": {
2832
+ "grade": "B",
2833
+ "corpus_size": 1012,
2834
+ "example_length": "sentence",
2835
+ "domain_breadth": "multi-domain",
2836
+ "contamination": "MEDIUM",
2837
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2838
+ }
2839
+ },
2840
+ {
2841
+ "id": "nllb-spa-eng-flores-chrfpp",
2842
+ "model": "NLLB-200 3.3B",
2843
+ "source": "NLLB-200 published FLORES-200 metrics",
2844
+ "org": "Meta AI",
2845
+ "year": 2022,
2846
+ "category": "model-report",
2847
+ "benchmark": "FLORES-200 devtest (spa-eng)",
2848
+ "metric": "chrF++",
2849
+ "value": 59.1,
2850
+ "lower_is_better": false,
2851
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2852
+ "headline": "NLLB-200 3.3B spa→eng: chrF++ 59.1 on FLORES-200.",
2853
+ "langs_or_pairs": "spa→eng (single directed pair).",
2854
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2855
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2856
+ "verified": true,
2857
+ "badge": "unverified — not reproduced by Champollion",
2858
+ "notes": null,
2859
+ "pair": {
2860
+ "source": "spa",
2861
+ "target": "eng"
2862
+ },
2863
+ "signal_strength": {
2864
+ "grade": "B",
2865
+ "corpus_size": 1012,
2866
+ "example_length": "sentence",
2867
+ "domain_breadth": "multi-domain",
2868
+ "contamination": "MEDIUM",
2869
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2870
+ }
2871
+ },
2872
+ {
2873
+ "id": "nllb-swh-eng-flores-chrfpp",
2874
+ "model": "NLLB-200 3.3B",
2875
+ "source": "NLLB-200 published FLORES-200 metrics",
2876
+ "org": "Meta AI",
2877
+ "year": 2022,
2878
+ "category": "model-report",
2879
+ "benchmark": "FLORES-200 devtest (swh-eng)",
2880
+ "metric": "chrF++",
2881
+ "value": 65,
2882
+ "lower_is_better": false,
2883
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2884
+ "headline": "NLLB-200 3.3B swh→eng: chrF++ 65 on FLORES-200.",
2885
+ "langs_or_pairs": "swh→eng (single directed pair).",
2886
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2887
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2888
+ "verified": true,
2889
+ "badge": "unverified — not reproduced by Champollion",
2890
+ "notes": null,
2891
+ "pair": {
2892
+ "source": "swh",
2893
+ "target": "eng"
2894
+ },
2895
+ "signal_strength": {
2896
+ "grade": "B",
2897
+ "corpus_size": 1012,
2898
+ "example_length": "sentence",
2899
+ "domain_breadth": "multi-domain",
2900
+ "contamination": "MEDIUM",
2901
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2902
+ }
2903
+ },
2904
+ {
2905
+ "id": "nllb-tur-eng-flores-chrfpp",
2906
+ "model": "NLLB-200 3.3B",
2907
+ "source": "NLLB-200 published FLORES-200 metrics",
2908
+ "org": "Meta AI",
2909
+ "year": 2022,
2910
+ "category": "model-report",
2911
+ "benchmark": "FLORES-200 devtest (tur-eng)",
2912
+ "metric": "chrF++",
2913
+ "value": 63.9,
2914
+ "lower_is_better": false,
2915
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2916
+ "headline": "NLLB-200 3.3B tur→eng: chrF++ 63.9 on FLORES-200.",
2917
+ "langs_or_pairs": "tur→eng (single directed pair).",
2918
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2919
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2920
+ "verified": true,
2921
+ "badge": "unverified — not reproduced by Champollion",
2922
+ "notes": null,
2923
+ "pair": {
2924
+ "source": "tur",
2925
+ "target": "eng"
2926
+ },
2927
+ "signal_strength": {
2928
+ "grade": "B",
2929
+ "corpus_size": 1012,
2930
+ "example_length": "sentence",
2931
+ "domain_breadth": "multi-domain",
2932
+ "contamination": "MEDIUM",
2933
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2934
+ }
2935
+ },
2936
+ {
2937
+ "id": "nllb-ukr-eng-flores-chrfpp",
2938
+ "model": "NLLB-200 3.3B",
2939
+ "source": "NLLB-200 published FLORES-200 metrics",
2940
+ "org": "Meta AI",
2941
+ "year": 2022,
2942
+ "category": "model-report",
2943
+ "benchmark": "FLORES-200 devtest (ukr-eng)",
2944
+ "metric": "chrF++",
2945
+ "value": 64.2,
2946
+ "lower_is_better": false,
2947
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2948
+ "headline": "NLLB-200 3.3B ukr→eng: chrF++ 64.2 on FLORES-200.",
2949
+ "langs_or_pairs": "ukr→eng (single directed pair).",
2950
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2951
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2952
+ "verified": true,
2953
+ "badge": "unverified — not reproduced by Champollion",
2954
+ "notes": null,
2955
+ "pair": {
2956
+ "source": "ukr",
2957
+ "target": "eng"
2958
+ },
2959
+ "signal_strength": {
2960
+ "grade": "B",
2961
+ "corpus_size": 1012,
2962
+ "example_length": "sentence",
2963
+ "domain_breadth": "multi-domain",
2964
+ "contamination": "MEDIUM",
2965
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2966
+ }
2967
+ },
2968
+ {
2969
+ "id": "nllb-vie-eng-flores-chrfpp",
2970
+ "model": "NLLB-200 3.3B",
2971
+ "source": "NLLB-200 published FLORES-200 metrics",
2972
+ "org": "Meta AI",
2973
+ "year": 2022,
2974
+ "category": "model-report",
2975
+ "benchmark": "FLORES-200 devtest (vie-eng)",
2976
+ "metric": "chrF++",
2977
+ "value": 61.5,
2978
+ "lower_is_better": false,
2979
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
2980
+ "headline": "NLLB-200 3.3B vie→eng: chrF++ 61.5 on FLORES-200.",
2981
+ "langs_or_pairs": "vie→eng (single directed pair).",
2982
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
2983
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
2984
+ "verified": true,
2985
+ "badge": "unverified — not reproduced by Champollion",
2986
+ "notes": null,
2987
+ "pair": {
2988
+ "source": "vie",
2989
+ "target": "eng"
2990
+ },
2991
+ "signal_strength": {
2992
+ "grade": "B",
2993
+ "corpus_size": 1012,
2994
+ "example_length": "sentence",
2995
+ "domain_breadth": "multi-domain",
2996
+ "contamination": "MEDIUM",
2997
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
2998
+ }
2999
+ },
3000
+ {
3001
+ "id": "nllb-yor-eng-flores-chrfpp",
3002
+ "model": "NLLB-200 3.3B",
3003
+ "source": "NLLB-200 published FLORES-200 metrics",
3004
+ "org": "Meta AI",
3005
+ "year": 2022,
3006
+ "category": "model-report",
3007
+ "benchmark": "FLORES-200 devtest (yor-eng)",
3008
+ "metric": "chrF++",
3009
+ "value": 44.5,
3010
+ "lower_is_better": false,
3011
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3012
+ "headline": "NLLB-200 3.3B yor→eng: chrF++ 44.5 on FLORES-200.",
3013
+ "langs_or_pairs": "yor→eng (single directed pair).",
3014
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3015
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3016
+ "verified": true,
3017
+ "badge": "unverified — not reproduced by Champollion",
3018
+ "notes": null,
3019
+ "pair": {
3020
+ "source": "yor",
3021
+ "target": "eng"
3022
+ },
3023
+ "signal_strength": {
3024
+ "grade": "B",
3025
+ "corpus_size": 1012,
3026
+ "example_length": "sentence",
3027
+ "domain_breadth": "multi-domain",
3028
+ "contamination": "MEDIUM",
3029
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3030
+ }
3031
+ },
3032
+ {
3033
+ "id": "nllb-zho-eng-flores-chrfpp",
3034
+ "model": "NLLB-200 3.3B",
3035
+ "source": "NLLB-200 published FLORES-200 metrics",
3036
+ "org": "Meta AI",
3037
+ "year": 2022,
3038
+ "category": "model-report",
3039
+ "benchmark": "FLORES-200 devtest (zho-eng)",
3040
+ "metric": "chrF++",
3041
+ "value": 56.2,
3042
+ "lower_is_better": false,
3043
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3044
+ "headline": "NLLB-200 3.3B zho→eng: chrF++ 56.2 on FLORES-200.",
3045
+ "langs_or_pairs": "zho→eng (single directed pair).",
3046
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3047
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3048
+ "verified": true,
3049
+ "badge": "unverified — not reproduced by Champollion",
3050
+ "notes": null,
3051
+ "pair": {
3052
+ "source": "zho",
3053
+ "target": "eng"
3054
+ },
3055
+ "signal_strength": {
3056
+ "grade": "B",
3057
+ "corpus_size": 1012,
3058
+ "example_length": "sentence",
3059
+ "domain_breadth": "multi-domain",
3060
+ "contamination": "MEDIUM",
3061
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3062
+ }
3063
+ },
3064
+ {
3065
+ "id": "nllb-zul-eng-flores-chrfpp",
3066
+ "model": "NLLB-200 3.3B",
3067
+ "source": "NLLB-200 published FLORES-200 metrics",
3068
+ "org": "Meta AI",
3069
+ "year": 2022,
3070
+ "category": "model-report",
3071
+ "benchmark": "FLORES-200 devtest (zul-eng)",
3072
+ "metric": "chrF++",
3073
+ "value": 60.9,
3074
+ "lower_is_better": false,
3075
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3076
+ "headline": "NLLB-200 3.3B zul→eng: chrF++ 60.9 on FLORES-200.",
3077
+ "langs_or_pairs": "zul→eng (single directed pair).",
3078
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3079
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3080
+ "verified": true,
3081
+ "badge": "unverified — not reproduced by Champollion",
3082
+ "notes": null,
3083
+ "pair": {
3084
+ "source": "zul",
3085
+ "target": "eng"
3086
+ },
3087
+ "signal_strength": {
3088
+ "grade": "B",
3089
+ "corpus_size": 1012,
3090
+ "example_length": "sentence",
3091
+ "domain_breadth": "multi-domain",
3092
+ "contamination": "MEDIUM",
3093
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3094
+ }
3095
+ },
3096
+ {
3097
+ "id": "nllb-eng-arb-flores-chrfpp",
3098
+ "model": "NLLB-200 3.3B",
3099
+ "source": "NLLB-200 published FLORES-200 metrics",
3100
+ "org": "Meta AI",
3101
+ "year": 2022,
3102
+ "category": "model-report",
3103
+ "benchmark": "FLORES-200 devtest (eng-arb)",
3104
+ "metric": "chrF++",
3105
+ "value": 55,
3106
+ "lower_is_better": false,
3107
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3108
+ "headline": "NLLB-200 3.3B eng→arb: chrF++ 55 on FLORES-200.",
3109
+ "langs_or_pairs": "eng→arb (single directed pair).",
3110
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3111
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3112
+ "verified": true,
3113
+ "badge": "unverified — not reproduced by Champollion",
3114
+ "notes": null,
3115
+ "pair": {
3116
+ "source": "eng",
3117
+ "target": "arb"
3118
+ },
3119
+ "signal_strength": {
3120
+ "grade": "B",
3121
+ "corpus_size": 1012,
3122
+ "example_length": "sentence",
3123
+ "domain_breadth": "multi-domain",
3124
+ "contamination": "MEDIUM",
3125
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3126
+ }
3127
+ },
3128
+ {
3129
+ "id": "nllb-eng-ben-flores-chrfpp",
3130
+ "model": "NLLB-200 3.3B",
3131
+ "source": "NLLB-200 published FLORES-200 metrics",
3132
+ "org": "Meta AI",
3133
+ "year": 2022,
3134
+ "category": "model-report",
3135
+ "benchmark": "FLORES-200 devtest (eng-ben)",
3136
+ "metric": "chrF++",
3137
+ "value": 48.7,
3138
+ "lower_is_better": false,
3139
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3140
+ "headline": "NLLB-200 3.3B eng→ben: chrF++ 48.7 on FLORES-200.",
3141
+ "langs_or_pairs": "eng→ben (single directed pair).",
3142
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3143
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3144
+ "verified": true,
3145
+ "badge": "unverified — not reproduced by Champollion",
3146
+ "notes": null,
3147
+ "pair": {
3148
+ "source": "eng",
3149
+ "target": "ben"
3150
+ },
3151
+ "signal_strength": {
3152
+ "grade": "B",
3153
+ "corpus_size": 1012,
3154
+ "example_length": "sentence",
3155
+ "domain_breadth": "multi-domain",
3156
+ "contamination": "MEDIUM",
3157
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3158
+ }
3159
+ },
3160
+ {
3161
+ "id": "nllb-eng-deu-flores-chrfpp",
3162
+ "model": "NLLB-200 3.3B",
3163
+ "source": "NLLB-200 published FLORES-200 metrics",
3164
+ "org": "Meta AI",
3165
+ "year": 2022,
3166
+ "category": "model-report",
3167
+ "benchmark": "FLORES-200 devtest (eng-deu)",
3168
+ "metric": "chrF++",
3169
+ "value": 62.8,
3170
+ "lower_is_better": false,
3171
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3172
+ "headline": "NLLB-200 3.3B eng→deu: chrF++ 62.8 on FLORES-200.",
3173
+ "langs_or_pairs": "eng→deu (single directed pair).",
3174
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3175
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3176
+ "verified": true,
3177
+ "badge": "unverified — not reproduced by Champollion",
3178
+ "notes": null,
3179
+ "pair": {
3180
+ "source": "eng",
3181
+ "target": "deu"
3182
+ },
3183
+ "signal_strength": {
3184
+ "grade": "B",
3185
+ "corpus_size": 1012,
3186
+ "example_length": "sentence",
3187
+ "domain_breadth": "multi-domain",
3188
+ "contamination": "MEDIUM",
3189
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3190
+ }
3191
+ },
3192
+ {
3193
+ "id": "nllb-eng-fra-flores-chrfpp",
3194
+ "model": "NLLB-200 3.3B",
3195
+ "source": "NLLB-200 published FLORES-200 metrics",
3196
+ "org": "Meta AI",
3197
+ "year": 2022,
3198
+ "category": "model-report",
3199
+ "benchmark": "FLORES-200 devtest (eng-fra)",
3200
+ "metric": "chrF++",
3201
+ "value": 69.6,
3202
+ "lower_is_better": false,
3203
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3204
+ "headline": "NLLB-200 3.3B eng→fra: chrF++ 69.6 on FLORES-200.",
3205
+ "langs_or_pairs": "eng→fra (single directed pair).",
3206
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3207
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3208
+ "verified": true,
3209
+ "badge": "unverified — not reproduced by Champollion",
3210
+ "notes": null,
3211
+ "pair": {
3212
+ "source": "eng",
3213
+ "target": "fra"
3214
+ },
3215
+ "signal_strength": {
3216
+ "grade": "B",
3217
+ "corpus_size": 1012,
3218
+ "example_length": "sentence",
3219
+ "domain_breadth": "multi-domain",
3220
+ "contamination": "MEDIUM",
3221
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3222
+ }
3223
+ },
3224
+ {
3225
+ "id": "nllb-eng-hin-flores-chrfpp",
3226
+ "model": "NLLB-200 3.3B",
3227
+ "source": "NLLB-200 published FLORES-200 metrics",
3228
+ "org": "Meta AI",
3229
+ "year": 2022,
3230
+ "category": "model-report",
3231
+ "benchmark": "FLORES-200 devtest (eng-hin)",
3232
+ "metric": "chrF++",
3233
+ "value": 57,
3234
+ "lower_is_better": false,
3235
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3236
+ "headline": "NLLB-200 3.3B eng→hin: chrF++ 57 on FLORES-200.",
3237
+ "langs_or_pairs": "eng→hin (single directed pair).",
3238
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3239
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3240
+ "verified": true,
3241
+ "badge": "unverified — not reproduced by Champollion",
3242
+ "notes": null,
3243
+ "pair": {
3244
+ "source": "eng",
3245
+ "target": "hin"
3246
+ },
3247
+ "signal_strength": {
3248
+ "grade": "B",
3249
+ "corpus_size": 1012,
3250
+ "example_length": "sentence",
3251
+ "domain_breadth": "multi-domain",
3252
+ "contamination": "MEDIUM",
3253
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3254
+ }
3255
+ },
3256
+ {
3257
+ "id": "nllb-eng-ita-flores-chrfpp",
3258
+ "model": "NLLB-200 3.3B",
3259
+ "source": "NLLB-200 published FLORES-200 metrics",
3260
+ "org": "Meta AI",
3261
+ "year": 2022,
3262
+ "category": "model-report",
3263
+ "benchmark": "FLORES-200 devtest (eng-ita)",
3264
+ "metric": "chrF++",
3265
+ "value": 57.1,
3266
+ "lower_is_better": false,
3267
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3268
+ "headline": "NLLB-200 3.3B eng→ita: chrF++ 57.1 on FLORES-200.",
3269
+ "langs_or_pairs": "eng→ita (single directed pair).",
3270
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3271
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3272
+ "verified": true,
3273
+ "badge": "unverified — not reproduced by Champollion",
3274
+ "notes": null,
3275
+ "pair": {
3276
+ "source": "eng",
3277
+ "target": "ita"
3278
+ },
3279
+ "signal_strength": {
3280
+ "grade": "B",
3281
+ "corpus_size": 1012,
3282
+ "example_length": "sentence",
3283
+ "domain_breadth": "multi-domain",
3284
+ "contamination": "MEDIUM",
3285
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3286
+ }
3287
+ },
3288
+ {
3289
+ "id": "nllb-eng-jpn-flores-chrfpp",
3290
+ "model": "NLLB-200 3.3B",
3291
+ "source": "NLLB-200 published FLORES-200 metrics",
3292
+ "org": "Meta AI",
3293
+ "year": 2022,
3294
+ "category": "model-report",
3295
+ "benchmark": "FLORES-200 devtest (eng-jpn)",
3296
+ "metric": "chrF++",
3297
+ "value": 25.2,
3298
+ "lower_is_better": false,
3299
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3300
+ "headline": "NLLB-200 3.3B eng→jpn: chrF++ 25.2 on FLORES-200.",
3301
+ "langs_or_pairs": "eng→jpn (single directed pair).",
3302
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3303
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3304
+ "verified": true,
3305
+ "badge": "unverified — not reproduced by Champollion",
3306
+ "notes": null,
3307
+ "pair": {
3308
+ "source": "eng",
3309
+ "target": "jpn"
3310
+ },
3311
+ "signal_strength": {
3312
+ "grade": "B",
3313
+ "corpus_size": 1012,
3314
+ "example_length": "sentence",
3315
+ "domain_breadth": "multi-domain",
3316
+ "contamination": "MEDIUM",
3317
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3318
+ }
3319
+ },
3320
+ {
3321
+ "id": "nllb-eng-kor-flores-chrfpp",
3322
+ "model": "NLLB-200 3.3B",
3323
+ "source": "NLLB-200 published FLORES-200 metrics",
3324
+ "org": "Meta AI",
3325
+ "year": 2022,
3326
+ "category": "model-report",
3327
+ "benchmark": "FLORES-200 devtest (eng-kor)",
3328
+ "metric": "chrF++",
3329
+ "value": 34.3,
3330
+ "lower_is_better": false,
3331
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3332
+ "headline": "NLLB-200 3.3B eng→kor: chrF++ 34.3 on FLORES-200.",
3333
+ "langs_or_pairs": "eng→kor (single directed pair).",
3334
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3335
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3336
+ "verified": true,
3337
+ "badge": "unverified — not reproduced by Champollion",
3338
+ "notes": null,
3339
+ "pair": {
3340
+ "source": "eng",
3341
+ "target": "kor"
3342
+ },
3343
+ "signal_strength": {
3344
+ "grade": "B",
3345
+ "corpus_size": 1012,
3346
+ "example_length": "sentence",
3347
+ "domain_breadth": "multi-domain",
3348
+ "contamination": "MEDIUM",
3349
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3350
+ }
3351
+ },
3352
+ {
3353
+ "id": "nllb-eng-nld-flores-chrfpp",
3354
+ "model": "NLLB-200 3.3B",
3355
+ "source": "NLLB-200 published FLORES-200 metrics",
3356
+ "org": "Meta AI",
3357
+ "year": 2022,
3358
+ "category": "model-report",
3359
+ "benchmark": "FLORES-200 devtest (eng-nld)",
3360
+ "metric": "chrF++",
3361
+ "value": 54.9,
3362
+ "lower_is_better": false,
3363
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3364
+ "headline": "NLLB-200 3.3B eng→nld: chrF++ 54.9 on FLORES-200.",
3365
+ "langs_or_pairs": "eng→nld (single directed pair).",
3366
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3367
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3368
+ "verified": true,
3369
+ "badge": "unverified — not reproduced by Champollion",
3370
+ "notes": null,
3371
+ "pair": {
3372
+ "source": "eng",
3373
+ "target": "nld"
3374
+ },
3375
+ "signal_strength": {
3376
+ "grade": "B",
3377
+ "corpus_size": 1012,
3378
+ "example_length": "sentence",
3379
+ "domain_breadth": "multi-domain",
3380
+ "contamination": "MEDIUM",
3381
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3382
+ }
3383
+ },
3384
+ {
3385
+ "id": "nllb-eng-por-flores-chrfpp",
3386
+ "model": "NLLB-200 3.3B",
3387
+ "source": "NLLB-200 published FLORES-200 metrics",
3388
+ "org": "Meta AI",
3389
+ "year": 2022,
3390
+ "category": "model-report",
3391
+ "benchmark": "FLORES-200 devtest (eng-por)",
3392
+ "metric": "chrF++",
3393
+ "value": 69.4,
3394
+ "lower_is_better": false,
3395
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3396
+ "headline": "NLLB-200 3.3B eng→por: chrF++ 69.4 on FLORES-200.",
3397
+ "langs_or_pairs": "eng→por (single directed pair).",
3398
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3399
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3400
+ "verified": true,
3401
+ "badge": "unverified — not reproduced by Champollion",
3402
+ "notes": null,
3403
+ "pair": {
3404
+ "source": "eng",
3405
+ "target": "por"
3406
+ },
3407
+ "signal_strength": {
3408
+ "grade": "B",
3409
+ "corpus_size": 1012,
3410
+ "example_length": "sentence",
3411
+ "domain_breadth": "multi-domain",
3412
+ "contamination": "MEDIUM",
3413
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3414
+ }
3415
+ },
3416
+ {
3417
+ "id": "nllb-eng-rus-flores-chrfpp",
3418
+ "model": "NLLB-200 3.3B",
3419
+ "source": "NLLB-200 published FLORES-200 metrics",
3420
+ "org": "Meta AI",
3421
+ "year": 2022,
3422
+ "category": "model-report",
3423
+ "benchmark": "FLORES-200 devtest (eng-rus)",
3424
+ "metric": "chrF++",
3425
+ "value": 56.1,
3426
+ "lower_is_better": false,
3427
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3428
+ "headline": "NLLB-200 3.3B eng→rus: chrF++ 56.1 on FLORES-200.",
3429
+ "langs_or_pairs": "eng→rus (single directed pair).",
3430
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3431
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3432
+ "verified": true,
3433
+ "badge": "unverified — not reproduced by Champollion",
3434
+ "notes": null,
3435
+ "pair": {
3436
+ "source": "eng",
3437
+ "target": "rus"
3438
+ },
3439
+ "signal_strength": {
3440
+ "grade": "B",
3441
+ "corpus_size": 1012,
3442
+ "example_length": "sentence",
3443
+ "domain_breadth": "multi-domain",
3444
+ "contamination": "MEDIUM",
3445
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3446
+ }
3447
+ },
3448
+ {
3449
+ "id": "nllb-eng-spa-flores-chrfpp",
3450
+ "model": "NLLB-200 3.3B",
3451
+ "source": "NLLB-200 published FLORES-200 metrics",
3452
+ "org": "Meta AI",
3453
+ "year": 2022,
3454
+ "category": "model-report",
3455
+ "benchmark": "FLORES-200 devtest (eng-spa)",
3456
+ "metric": "chrF++",
3457
+ "value": 54.2,
3458
+ "lower_is_better": false,
3459
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3460
+ "headline": "NLLB-200 3.3B eng→spa: chrF++ 54.2 on FLORES-200.",
3461
+ "langs_or_pairs": "eng→spa (single directed pair).",
3462
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3463
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3464
+ "verified": true,
3465
+ "badge": "unverified — not reproduced by Champollion",
3466
+ "notes": null,
3467
+ "pair": {
3468
+ "source": "eng",
3469
+ "target": "spa"
3470
+ },
3471
+ "signal_strength": {
3472
+ "grade": "B",
3473
+ "corpus_size": 1012,
3474
+ "example_length": "sentence",
3475
+ "domain_breadth": "multi-domain",
3476
+ "contamination": "MEDIUM",
3477
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3478
+ }
3479
+ },
3480
+ {
3481
+ "id": "nllb-eng-swh-flores-chrfpp",
3482
+ "model": "NLLB-200 3.3B",
3483
+ "source": "NLLB-200 published FLORES-200 metrics",
3484
+ "org": "Meta AI",
3485
+ "year": 2022,
3486
+ "category": "model-report",
3487
+ "benchmark": "FLORES-200 devtest (eng-swh)",
3488
+ "metric": "chrF++",
3489
+ "value": 60,
3490
+ "lower_is_better": false,
3491
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3492
+ "headline": "NLLB-200 3.3B eng→swh: chrF++ 60 on FLORES-200.",
3493
+ "langs_or_pairs": "eng→swh (single directed pair).",
3494
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3495
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3496
+ "verified": true,
3497
+ "badge": "unverified — not reproduced by Champollion",
3498
+ "notes": null,
3499
+ "pair": {
3500
+ "source": "eng",
3501
+ "target": "swh"
3502
+ },
3503
+ "signal_strength": {
3504
+ "grade": "B",
3505
+ "corpus_size": 1012,
3506
+ "example_length": "sentence",
3507
+ "domain_breadth": "multi-domain",
3508
+ "contamination": "MEDIUM",
3509
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3510
+ }
3511
+ },
3512
+ {
3513
+ "id": "nllb-eng-tur-flores-chrfpp",
3514
+ "model": "NLLB-200 3.3B",
3515
+ "source": "NLLB-200 published FLORES-200 metrics",
3516
+ "org": "Meta AI",
3517
+ "year": 2022,
3518
+ "category": "model-report",
3519
+ "benchmark": "FLORES-200 devtest (eng-tur)",
3520
+ "metric": "chrF++",
3521
+ "value": 57.8,
3522
+ "lower_is_better": false,
3523
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3524
+ "headline": "NLLB-200 3.3B eng→tur: chrF++ 57.8 on FLORES-200.",
3525
+ "langs_or_pairs": "eng→tur (single directed pair).",
3526
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3527
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3528
+ "verified": true,
3529
+ "badge": "unverified — not reproduced by Champollion",
3530
+ "notes": null,
3531
+ "pair": {
3532
+ "source": "eng",
3533
+ "target": "tur"
3534
+ },
3535
+ "signal_strength": {
3536
+ "grade": "B",
3537
+ "corpus_size": 1012,
3538
+ "example_length": "sentence",
3539
+ "domain_breadth": "multi-domain",
3540
+ "contamination": "MEDIUM",
3541
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3542
+ }
3543
+ },
3544
+ {
3545
+ "id": "nllb-eng-vie-flores-chrfpp",
3546
+ "model": "NLLB-200 3.3B",
3547
+ "source": "NLLB-200 published FLORES-200 metrics",
3548
+ "org": "Meta AI",
3549
+ "year": 2022,
3550
+ "category": "model-report",
3551
+ "benchmark": "FLORES-200 devtest (eng-vie)",
3552
+ "metric": "chrF++",
3553
+ "value": 59.3,
3554
+ "lower_is_better": false,
3555
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3556
+ "headline": "NLLB-200 3.3B eng→vie: chrF++ 59.3 on FLORES-200.",
3557
+ "langs_or_pairs": "eng→vie (single directed pair).",
3558
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3559
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3560
+ "verified": true,
3561
+ "badge": "unverified — not reproduced by Champollion",
3562
+ "notes": null,
3563
+ "pair": {
3564
+ "source": "eng",
3565
+ "target": "vie"
3566
+ },
3567
+ "signal_strength": {
3568
+ "grade": "B",
3569
+ "corpus_size": 1012,
3570
+ "example_length": "sentence",
3571
+ "domain_breadth": "multi-domain",
3572
+ "contamination": "MEDIUM",
3573
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3574
+ }
3575
+ },
3576
+ {
3577
+ "id": "nllb-eng-zho-flores-chrfpp",
3578
+ "model": "NLLB-200 3.3B",
3579
+ "source": "NLLB-200 published FLORES-200 metrics",
3580
+ "org": "Meta AI",
3581
+ "year": 2022,
3582
+ "category": "model-report",
3583
+ "benchmark": "FLORES-200 devtest (eng-zho)",
3584
+ "metric": "chrF++",
3585
+ "value": 22.3,
3586
+ "lower_is_better": false,
3587
+ "metric_variant_flag": "chrF++ on FLORES-200 devtest — 0–100, higher better, from the NLLB-200 3.3B public metrics. NOT comparable to BLEU, COMET, or MetricX.",
3588
+ "headline": "NLLB-200 3.3B eng→zho: chrF++ 22.3 on FLORES-200.",
3589
+ "langs_or_pairs": "eng→zho (single directed pair).",
3590
+ "source_url": "https://dl.fbaipublicfiles.com/large_objects/nllb/models/nllb_200_dense_3b/metrics.csv",
3591
+ "citation": "NLLB Team et al. (2022), No Language Left Behind, arXiv:2207.04672; FLORES-200 devtest chrF++ from the released metrics.",
3592
+ "verified": true,
3593
+ "badge": "unverified — not reproduced by Champollion",
3594
+ "notes": null,
3595
+ "pair": {
3596
+ "source": "eng",
3597
+ "target": "zho"
3598
+ },
3599
+ "signal_strength": {
3600
+ "grade": "B",
3601
+ "corpus_size": 1012,
3602
+ "example_length": "sentence",
3603
+ "domain_breadth": "multi-domain",
3604
+ "contamination": "MEDIUM",
3605
+ "rationale": "FLORES-200 devtest = 1012 professionally-translated, multi-parallel sentences (multi-domain). Strong standard benchmark, but a widely-used public set (MEDIUM contamination) → B."
3606
+ }
3607
+ }
3608
+ ],
3609
+ "manifest": {
3610
+ "run": [
3611
+ {
3612
+ "family": "flores",
3613
+ "name": "FLORES-200 / FLORES+",
3614
+ "license": "CC-BY-SA-4.0",
3615
+ "contamination_posture": "HIGH — relative-only (illustrative, never an absolute quality claim)",
3616
+ "note": "Many-to-many devtest used for breadth; the most broadly trained-on suite, so scored on the relative-only lane and used to illustrate coverage, not to rank absolute quality."
3617
+ },
3618
+ {
3619
+ "family": "tico19",
3620
+ "name": "TICO-19 (COVID-19 translation)",
3621
+ "license": "CC0-1.0",
3622
+ "contamination_posture": "varies — eligibility-gated per dataset",
3623
+ "note": "Public-health domain parallel set fetched from source; widely available so treated cautiously by the contamination lane."
3624
+ },
3625
+ {
3626
+ "family": "in22",
3627
+ "name": "IN22 (Indian languages)",
3628
+ "license": "CC-BY-4.0",
3629
+ "contamination_posture": "varies — eligibility-gated per dataset",
3630
+ "note": "Conversational + general Indic evaluation pairs (gated download; fetch-from-source)."
3631
+ },
3632
+ {
3633
+ "family": "tatoeba",
3634
+ "name": "Tatoeba clean bridges",
3635
+ "license": "CC-BY-2.0",
3636
+ "contamination_posture": "varies — eligibility-gated per dataset",
3637
+ "note": "Per-pair clean bridges rebuilt fetch-from-source (Challenge tar + OPUS moses), including the flagship spa→que Quechua bridge."
3638
+ },
3639
+ {
3640
+ "family": "globalvoices",
3641
+ "name": "GlobalVoices news",
3642
+ "license": "CC-BY-3.0",
3643
+ "contamination_posture": "varies — eligibility-gated per dataset",
3644
+ "note": "Citizen-media news parallel corpus, fetch-from-source."
3645
+ },
3646
+ {
3647
+ "family": "smol",
3648
+ "name": "SMOL (sentence + document)",
3649
+ "license": "CC-BY-4.0",
3650
+ "contamination_posture": "varies — eligibility-gated per dataset",
3651
+ "note": "Low-resource sentence- and document-level pairs, per-pair sha-pinned."
3652
+ },
3653
+ {
3654
+ "family": "alt",
3655
+ "name": "ALT (Asian Language Treebank)",
3656
+ "license": "CC-BY-4.0",
3657
+ "contamination_posture": "varies — eligibility-gated per dataset",
3658
+ "note": "Multi-way SEA evaluation (Lao, Khmer, Burmese, Filipino, Malay…); only the bare parallel layer (CC-BY-4.0) is used, not the NC treebank annotations."
3659
+ },
3660
+ {
3661
+ "family": "turkicxwmt",
3662
+ "name": "TurkicX-WMT",
3663
+ "license": "CC-BY-SA-4.0",
3664
+ "contamination_posture": "varies — eligibility-gated per dataset",
3665
+ "note": "Turkic-language evaluation pairs with non-uniform per-pair sizes (data-driven size map)."
3666
+ },
3667
+ {
3668
+ "family": "wmt24pp",
3669
+ "name": "WMT24++ ",
3670
+ "license": "Apache-2.0",
3671
+ "contamination_posture": "varies — eligibility-gated per dataset",
3672
+ "note": "Per-pair built corpus, sha-pinned; out-of-English directions. (Same benchmark family the cited TranslateGemma WMT24++ numbers are reported on — but those numbers are the authors', not a Champollion run.)"
3673
+ },
3674
+ {
3675
+ "family": "gamayun",
3676
+ "name": "Gamayun kit (TWB)",
3677
+ "license": "LicenseRef-TWB-Gamayun",
3678
+ "contamination_posture": "non-commercial research lane",
3679
+ "note": "Translators-without-Borders humanitarian mini-kits; carved into the non-commercial research lane."
3680
+ },
3681
+ {
3682
+ "family": "prize",
3683
+ "name": "Prize / contest sets",
3684
+ "license": "varies",
3685
+ "contamination_posture": "held-out where applicable",
3686
+ "note": "Contest-specific evaluation corpora."
3687
+ }
3688
+ ],
3689
+ "cite": [
3690
+ {
3691
+ "source": "TranslateGemma Technical Report",
3692
+ "result_ids": [
3693
+ "translategemma-27b-metricx",
3694
+ "translategemma-27b-comet22",
3695
+ "translategemma-12b-metricx",
3696
+ "translategemma-12b-comet22",
3697
+ "translategemma-4b-metricx",
3698
+ "translategemma-4b-comet22",
3699
+ "gemma3-27b-metricx",
3700
+ "gemma3-27b-comet22",
3701
+ "gemma3-12b-metricx",
3702
+ "gemma3-12b-comet22",
3703
+ "gemma3-4b-metricx",
3704
+ "gemma3-4b-comet22"
3705
+ ],
3706
+ "note": "Google's open MT suite fine-tuned from Gemma 3 (55 languages). We cite the verified WMT24++ MetricX-24 / COMET-22 numbers for TranslateGemma 4B/12B/27B and their Gemma 3 baselines (see the External results tab); we have not re-run the models."
3707
+ },
3708
+ {
3709
+ "source": "Gemma 2 (general-purpose open LLM)",
3710
+ "note": "A general open-weights LLM (Gemma Team, arXiv:2403.08295 / Gemma 2, arXiv:2408.00118) whose translation ability is incidental and reported inside a broad capability suite — NOT a dedicated MT system, and NOT to be confused with TranslateGemma (a separate Google MT model fine-tuned from Gemma 3). Cited for context; no dedicated-MT single number is pinned because the source publishes none on a stated MT protocol."
3711
+ },
3712
+ {
3713
+ "source": "Frontier and open MT model reports",
3714
+ "note": "Model papers/cards whose translation results we point to but have NOT pinned to a single verified datapoint here (different metric variants and test suites, or no stated MT protocol): MADLAD-400 MT models (arXiv:2309.04662), SeamlessM4T (arXiv:2308.11596), Aya / Aya 23 (arXiv:2402.07827 / 2405.15032), GPT-4 (arXiv:2303.08774), Gemini (arXiv:2312.11805), Claude 3 (model card). Cited as the field's reference systems; numbers are not pooled and not reproduced. (NLLB-200 per-pair numbers ARE now pinned — see its own entry.)"
3715
+ },
3716
+ {
3717
+ "source": "WMT General MT shared task",
3718
+ "note": "The field's reference human-ranked event (Kocmi et al., WMT24, https://www2.statmt.org/wmt24/); cited, not re-run — the test sets are broadly trained on. Human ESA/MQM is the official ranking; automatic COMET/MetricX/chrF are version-specific and not pooled."
3719
+ },
3720
+ {
3721
+ "source": "Papers with Code — MT leaderboards",
3722
+ "note": "Community aggregator of self-reported SOTA per benchmark (https://paperswithcode.com/task/machine-translation); predominantly legacy tokenized BLEU with inconsistent tooling — a literature map, not a like-for-like ranking."
3723
+ },
3724
+ {
3725
+ "source": "OPUS-MT / Tatoeba-Challenge leaderboard",
3726
+ "note": "Open, reproducible neural-MT models + BLEU/chrF benchmark cards across a very large number of pairs incl. low-resource directions (Tiedemann 2020; https://github.com/Helsinki-NLP/OPUS-MT-leaderboard). We pin a sample of VERIFIED per-pair card numbers (see the External results tab) spanning strong high-resource edges and weak low-resource ones; Champollion has not re-run them and runs its own clean Tatoeba bridges fetch-from-source rather than re-hosting.",
3727
+ "result_ids": [
3728
+ "opus-en-fr-tatoeba-bleu",
3729
+ "opus-en-fr-tatoeba-chrf",
3730
+ "opus-en-fr-newstest2013-bleu",
3731
+ "opus-en-es-tatoeba-bleu",
3732
+ "opus-en-es-tatoeba-chrf",
3733
+ "opus-en-de-tatoeba-bleu",
3734
+ "opus-en-de-tatoeba-chrf",
3735
+ "opus-en-de-newstest2019-bleu",
3736
+ "opus-en-ru-tatoeba-bleu",
3737
+ "opus-en-ru-tatoeba-chrf",
3738
+ "opus-tcbig-en-tr-tatoeba-bleu",
3739
+ "opus-tcbig-en-tr-tatoeba-chrf",
3740
+ "opus-tcbig-en-tr-flores-bleu",
3741
+ "opus-tcbig-en-tr-flores-chrf",
3742
+ "opus-en-ha-tatoeba-bleu",
3743
+ "opus-ha-en-tatoeba-bleu",
3744
+ "opus-en-mul-amh-bleu",
3745
+ "opus-mul-en-amh-bleu",
3746
+ "opus-en-mul-uig-bleu",
3747
+ "opus-mul-en-uig-bleu",
3748
+ "opus-en-mul-sah-bleu",
3749
+ "opus-mul-en-sah-bleu",
3750
+ "opus-en-trk-kaz-bleu"
3751
+ ]
3752
+ },
3753
+ {
3754
+ "source": "NLLB-200 (No Language Left Behind)",
3755
+ "result_ids": [
3756
+ "nllb-eng-hau-flores-spbleu",
3757
+ "nllb-eng-hau-flores-chrfpp",
3758
+ "nllb-hau-eng-flores-chrfpp",
3759
+ "nllb-eng-amh-tico-bleu",
3760
+ "nllb-eng-amh-tico-chrfpp",
3761
+ "nllb-amh-eng-tico-chrfpp",
3762
+ "nllb-flores101-avg-spbleu",
3763
+ "nllb-flores101-avg-chrfpp"
3764
+ ],
3765
+ "note": "Meta's 200-language open model (arXiv:2207.04672). We pin VERIFIED per-pair numbers read from the paper: FLORES-101 eng↔hau (relative-only, FLORES is HIGH-contamination) and the TICO health-domain eng↔amh directions, plus the FLORES-101 101-language average. Weights are CC-BY-NC-4.0 (non-commercial) — see the method index. Champollion has not re-run NLLB-200; numbers are the authors'."
3766
+ }
3767
+ ],
3768
+ "exclude": [
3769
+ {
3770
+ "name": "Bible corpora / JW300",
3771
+ "reason_code": "contaminated-colonial",
3772
+ "reason": "Broadly memorized by modern models (high contamination) AND rooted in colonial/missionary translation programs; excluded on both integrity and sovereignty grounds — per the data-boundaries doctrine."
3773
+ },
3774
+ {
3775
+ "name": "OPUS-100 (wholesale)",
3776
+ "reason_code": "mixed-license",
3777
+ "reason": "A grab-bag of sub-corpora under mixed and sometimes unclear licenses; we instead pull specific, license-clean pairs fetch-from-source rather than ingesting the whole bundle."
3778
+ },
3779
+ {
3780
+ "name": "MTEB / BUCC",
3781
+ "reason_code": "embedding-not-mt",
3782
+ "reason": "Sentence-embedding / bitext-mining benchmarks, not machine-translation generation — out of scope for a translation-quality index."
3783
+ },
3784
+ {
3785
+ "name": "MADLAD-400 corpus",
3786
+ "reason_code": "monolingual-training",
3787
+ "reason": "A monolingual training corpus (419 languages), not a parallel evaluation set. We cite the MADLAD-400 MT MODEL results but do not index the corpus as a benchmark."
3788
+ },
3789
+ {
3790
+ "name": "FLEURS",
3791
+ "reason_code": "speech-out-of-scope",
3792
+ "reason": "A speech (ASR/speech-translation) benchmark; Champollion's Network indexes text machine translation, so speech corpora are out of scope."
3793
+ },
3794
+ {
3795
+ "name": "EdTeKLA easy crk subsets (62 / 90-phase1 / 124 / crk-master)",
3796
+ "reason_code": "improper-subset",
3797
+ "reason": "Small, easy slices that must never rank as if they were full held-out evaluations — enforced by the datasets.quarantined flag + DB trigger. The full EdTeKLA protocol stays in the non-commercial research lane and corpus CONTENT is never hosted."
3798
+ },
3799
+ {
3800
+ "name": "Wolvengrey Cree dictionary data",
3801
+ "reason_code": "no-redistribute",
3802
+ "reason": "Redistribution permission is pending; never copied into corpora, cards, or exports."
3803
+ }
3804
+ ]
3805
+ },
3806
+ "methods": [
3807
+ {
3808
+ "id": "opus-mt",
3809
+ "name": "OPUS-MT",
3810
+ "org": "Helsinki-NLP",
3811
+ "paradigm": "neural-mt",
3812
+ "availability": "open-weights",
3813
+ "weights": "open",
3814
+ "access": [
3815
+ "huggingface",
3816
+ "download",
3817
+ "pip",
3818
+ "github"
3819
+ ],
3820
+ "license": "CC-BY-4.0 / Apache-2.0 (per model)",
3821
+ "commercial_use": true,
3822
+ "homepage": "https://github.com/Helsinki-NLP/Opus-MT",
3823
+ "weights_url": "https://huggingface.co/Helsinki-NLP",
3824
+ "source_url": "https://github.com/Helsinki-NLP/Opus-MT",
3825
+ "runnable_in_champollion": false,
3826
+ "notes": "1000+ open Marian-NMT models (single-pair + multilingual). Weights freely downloadable via HF (MarianMTModel) or .zip from object.pouta.csc.fi. Not yet a first-class Champollion method — reachable today by self-hosting behind the local/OpenAI-compatible lane; a native adapter is implement-for-them later."
3827
+ },
3828
+ {
3829
+ "id": "nllb-200",
3830
+ "name": "NLLB-200",
3831
+ "org": "Meta AI",
3832
+ "paradigm": "neural-mt",
3833
+ "availability": "open-weights",
3834
+ "weights": "open",
3835
+ "access": [
3836
+ "huggingface",
3837
+ "download",
3838
+ "github"
3839
+ ],
3840
+ "license": "CC-BY-NC-4.0",
3841
+ "commercial_use": false,
3842
+ "homepage": "https://github.com/facebookresearch/fairseq/tree/nllb",
3843
+ "weights_url": "https://huggingface.co/facebook/nllb-200-3.3B",
3844
+ "source_url": "https://arxiv.org/abs/2207.04672",
3845
+ "runnable_in_champollion": false,
3846
+ "notes": "200-language single model; weights openly downloadable (facebook/nllb-200-3.3B, -distilled-600M) but the CC-BY-NC-4.0 license is NON-COMMERCIAL — carved out of any commercial lane. Research model, not for production deployment."
3847
+ },
3848
+ {
3849
+ "id": "translategemma",
3850
+ "name": "TranslateGemma",
3851
+ "org": "Google",
3852
+ "paradigm": "dedicated-mt",
3853
+ "availability": "open-weights",
3854
+ "weights": "open",
3855
+ "access": [
3856
+ "huggingface",
3857
+ "download"
3858
+ ],
3859
+ "license": "Gemma Terms of Use",
3860
+ "commercial_use": true,
3861
+ "homepage": "https://ai.google.dev/gemma",
3862
+ "weights_url": "https://huggingface.co/google",
3863
+ "source_url": "https://arxiv.org/abs/2601.09012",
3864
+ "runnable_in_champollion": false,
3865
+ "notes": "Dedicated MT suite fine-tuned from Gemma 3 (4B/12B/27B), 55 languages. Open weights under the Gemma license. Cited-only here; a native adapter is implement-for-them later."
3866
+ },
3867
+ {
3868
+ "id": "gemma-3",
3869
+ "name": "Gemma 3",
3870
+ "org": "Google",
3871
+ "paradigm": "llm",
3872
+ "availability": "open-weights",
3873
+ "weights": "open",
3874
+ "access": [
3875
+ "huggingface",
3876
+ "download",
3877
+ "api"
3878
+ ],
3879
+ "license": "Gemma Terms of Use",
3880
+ "commercial_use": true,
3881
+ "homepage": "https://ai.google.dev/gemma",
3882
+ "weights_url": "https://huggingface.co/google",
3883
+ "source_url": "https://arxiv.org/abs/2601.09012",
3884
+ "runnable_in_champollion": false,
3885
+ "notes": "General-purpose open LLM; its MT scores here are the pre-fine-tuning baseline reported in the TranslateGemma study. As a general LLM it is reachable via OpenRouter/local lanes, but not registered as a dedicated MT method."
3886
+ }
3887
+ ]
3888
+ }