champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,253 @@
1
+ {
2
+ "_doc": "SSOT mapping between external source codes and project ISO 639-3 codes. Each source (flores, ntrex, tatoeba) has its own namespace. Only codes that DIFFER from ISO 639-3 are listed — unlisted codes pass through unchanged. For FLORES, the pattern is {iso639-3}_{script} where the script suffix is stripped by default, except where the base code itself needs remapping (e.g., zho → cmn).",
3
+ "_script_handling": "For FLORES codes with same ISO base but different scripts (ace_Arab vs ace_Latn), the script suffix is stripped to produce the ISO 639-3 code. Only zho needs base remapping → cmn, with script preserved as BCP-47 subtag (cmn-Hans, cmn-Hant). This means ace_Arab and ace_Latn both map to 'ace' and will produce equivalent eval pairs — which is correct, as the FLORES data for both scripts covers the same language.",
4
+ "flores": {
5
+ "zho_Hans": "cmn-Hans",
6
+ "zho_Hant": "cmn-Hant"
7
+ },
8
+ "ntrex": {
9
+ "fas": "pes",
10
+ "fil": "tgl",
11
+ "msa": "zsm",
12
+ "swa": "swh",
13
+ "nep": "npi",
14
+ "ori": "ory",
15
+ "lav": "lvs",
16
+ "mon": "khk"
17
+ },
18
+ "tatoeba": {
19
+ "zho": "cmn"
20
+ },
21
+ "_reverse_doc": "Reverse mappings for NTREX: when fetching data, we need to map project ISO 639-3 codes back to the codes used in NTREX filenames. The NTREX repo uses macrolanguage codes in its filenames.",
22
+ "ntrex_reverse": {
23
+ "pes": "fas",
24
+ "tgl": "fil",
25
+ "zsm": "msa",
26
+ "swh": "swa",
27
+ "npi": "nep",
28
+ "ory": "ori",
29
+ "lvs": "lav",
30
+ "khk": "mon"
31
+ },
32
+ "_tico19_doc": "TICO-19 file codes map from short (mostly ISO 639-1) codes to project ISO 639-3. Files are named test.en-{code}.tsv. Tigrinya has three variants: ti (original), ti_ER (Eritrean), ti_ET (Ethiopian) — all map to tir. Codes that match ISO 639-3 (ckb, fuv, nus, prs) are omitted — they pass through unchanged.",
33
+ "tico19": {
34
+ "am": "amh",
35
+ "ar": "arb",
36
+ "bn": "ben",
37
+ "din": "dip",
38
+ "es-LA": "spa",
39
+ "fa": "pes",
40
+ "fr": "fra",
41
+ "ha": "hau",
42
+ "hi": "hin",
43
+ "id": "ind",
44
+ "km": "khm",
45
+ "kr": "knc",
46
+ "ku": "kmr",
47
+ "lg": "lug",
48
+ "ln": "lin",
49
+ "mr": "mar",
50
+ "ms": "zsm",
51
+ "my": "mya",
52
+ "ne": "npi",
53
+ "om": "gaz",
54
+ "ps": "pbt",
55
+ "pt-BR": "por",
56
+ "ru": "rus",
57
+ "rw": "kin",
58
+ "so": "som",
59
+ "sw": "swh",
60
+ "ta": "tam",
61
+ "ti": "tir",
62
+ "ti_ER": "tir",
63
+ "ti_ET": "tir",
64
+ "tl": "tgl",
65
+ "ur": "urd",
66
+ "zh": "cmn-Hans",
67
+ "zu": "zul"
68
+ },
69
+ "tico19_reverse": {
70
+ "amh": "am",
71
+ "arb": "ar",
72
+ "ben": "bn",
73
+ "dip": "din",
74
+ "spa": "es-LA",
75
+ "pes": "fa",
76
+ "fra": "fr",
77
+ "hau": "ha",
78
+ "hin": "hi",
79
+ "ind": "id",
80
+ "khm": "km",
81
+ "knc": "kr",
82
+ "kmr": "ku",
83
+ "lug": "lg",
84
+ "lin": "ln",
85
+ "mar": "mr",
86
+ "zsm": "ms",
87
+ "mya": "my",
88
+ "npi": "ne",
89
+ "gaz": "om",
90
+ "pbt": "ps",
91
+ "por": "pt-BR",
92
+ "rus": "ru",
93
+ "kin": "rw",
94
+ "som": "so",
95
+ "swh": "sw",
96
+ "tam": "ta",
97
+ "tir": "ti",
98
+ "tgl": "tl",
99
+ "urd": "ur",
100
+ "cmn-Hans": "zh",
101
+ "zul": "zu"
102
+ },
103
+ "_in22_doc": "IN22 (AI4Bharat IndicTrans2) uses ISO 639-1 / short codes as column names in the HuggingFace Parquet files. Codes that already match ISO 639-3 (brx, doi, kok, mai, mni, sat) are omitted — they pass through unchanged.",
104
+ "in22": {
105
+ "as": "asm",
106
+ "bn": "ben",
107
+ "gu": "guj",
108
+ "hi": "hin",
109
+ "kn": "kan",
110
+ "ks": "kas",
111
+ "ml": "mal",
112
+ "mr": "mar",
113
+ "ne": "npi",
114
+ "or": "ori",
115
+ "pa": "pan",
116
+ "sa": "san",
117
+ "sd": "snd",
118
+ "ta": "tam",
119
+ "te": "tel",
120
+ "ur": "urd",
121
+ "en": "eng"
122
+ },
123
+ "in22_reverse": {
124
+ "asm": "as",
125
+ "ben": "bn",
126
+ "brx": "brx",
127
+ "doi": "doi",
128
+ "guj": "gu",
129
+ "hin": "hi",
130
+ "kan": "kn",
131
+ "kas": "ks",
132
+ "kok": "kok",
133
+ "mai": "mai",
134
+ "mal": "ml",
135
+ "mni": "mni",
136
+ "mar": "mr",
137
+ "npi": "ne",
138
+ "ori": "or",
139
+ "pan": "pa",
140
+ "san": "sa",
141
+ "sat": "sat",
142
+ "snd": "sd",
143
+ "tam": "ta",
144
+ "tel": "te",
145
+ "urd": "ur",
146
+ "eng": "en"
147
+ },
148
+ "_globalvoices_doc": "GlobalVoices v2018q4 via OPUS uses ISO 639-1 codes in filenames. Only codes that differ from ISO 639-3 are listed — unlisted codes pass through unchanged. fil passes through as-is since OPUS uses the ISO 639-2T code.",
149
+ "globalvoices": {
150
+ "am": "amh",
151
+ "ar": "arb",
152
+ "bg": "bul",
153
+ "bn": "ben",
154
+ "ca": "cat",
155
+ "cs": "ces",
156
+ "da": "dan",
157
+ "de": "deu",
158
+ "el": "ell",
159
+ "en": "eng",
160
+ "eo": "epo",
161
+ "es": "spa",
162
+ "fa": "fas",
163
+ "fr": "fra",
164
+ "he": "heb",
165
+ "hi": "hin",
166
+ "hu": "hun",
167
+ "id": "ind",
168
+ "it": "ita",
169
+ "km": "khm",
170
+ "ko": "kor",
171
+ "mg": "mlg",
172
+ "mk": "mkd",
173
+ "ms": "zsm",
174
+ "my": "mya",
175
+ "ne": "npi",
176
+ "nl": "nld",
177
+ "pl": "pol",
178
+ "pt": "por",
179
+ "ro": "ron",
180
+ "ru": "rus",
181
+ "sq": "sqi",
182
+ "sr": "srp",
183
+ "sv": "swe",
184
+ "sw": "swh",
185
+ "tr": "tur",
186
+ "uk": "ukr",
187
+ "ur": "urd",
188
+ "vi": "vie",
189
+ "zh": "cmn"
190
+ },
191
+ "globalvoices_reverse": {
192
+ "amh": "am",
193
+ "arb": "ar",
194
+ "bul": "bg",
195
+ "ben": "bn",
196
+ "cat": "ca",
197
+ "ces": "cs",
198
+ "dan": "da",
199
+ "deu": "de",
200
+ "ell": "el",
201
+ "eng": "en",
202
+ "epo": "eo",
203
+ "spa": "es",
204
+ "fas": "fa",
205
+ "fra": "fr",
206
+ "heb": "he",
207
+ "hin": "hi",
208
+ "hun": "hu",
209
+ "ind": "id",
210
+ "ita": "it",
211
+ "khm": "km",
212
+ "kor": "ko",
213
+ "mlg": "mg",
214
+ "mkd": "mk",
215
+ "zsm": "ms",
216
+ "mya": "my",
217
+ "npi": "ne",
218
+ "nld": "nl",
219
+ "pol": "pl",
220
+ "por": "pt",
221
+ "ron": "ro",
222
+ "rus": "ru",
223
+ "sqi": "sq",
224
+ "srp": "sr",
225
+ "swe": "sv",
226
+ "swh": "sw",
227
+ "tur": "tr",
228
+ "ukr": "uk",
229
+ "urd": "ur",
230
+ "vie": "vi",
231
+ "cmn": "zh"
232
+ },
233
+ "_gamayun_doc": "Gamayun (CLEAR Global) Parallel Sentence Kits use short language codes for en/fr, and ISO 639-3 for target languages. Only en→eng and fr→fra need remapping — target codes (hau, knc, lin, rhg, tir, swh, swc, nnb) already match ISO 639-3 and pass through unchanged.",
234
+ "gamayun": {
235
+ "en": "eng",
236
+ "fr": "fra"
237
+ },
238
+ "gamayun_reverse": {
239
+ "eng": "en",
240
+ "fra": "fr"
241
+ },
242
+ "_macrolanguage_doc": "ISO 639-3 macrolanguage code -> the individual language Champollion's language cards use for it. Some corpora label a pair with a macrolanguage code (OPUS/Tatoeba convention, e.g. 'fas', 'ara', 'sqi') while our language cards are keyed by the dominant individual variant ('pes', 'arb', 'als'). This map lets coverage from macro-coded corpora be recorded on the correct individual-language card without renaming the corpora. Sourced from the ISO 639-3 macrolanguage table (dominant individual variant) and consistent with the ntrex/tico19 mappings above. Genuinely ambiguous collections (e.g. 'ber' Berber, a family; 'ajp' which is itself individual but lacks a card) are intentionally omitted and surfaced loudly by sync-eval-datasets.mjs.",
243
+ "macrolanguage": {
244
+ "fas": "pes",
245
+ "ara": "arb",
246
+ "sqi": "als",
247
+ "msa": "zsm",
248
+ "swa": "swh",
249
+ "zho": "cmn",
250
+ "cmn-Hans": "cmn",
251
+ "cmn-Hant": "cmn"
252
+ }
253
+ }
@@ -0,0 +1,281 @@
1
+ # Corpora Cards v1 — Reference Document
2
+
3
+ > **⚠️ SUPERSEDED SNAPSHOT (frozen 2026-06-09).** The cards under
4
+ > [`corpora-cards/`](./corpora-cards/) are the source of truth and have moved
5
+ > on — notably the EdTeKLA entries: the corpus has a **public** GitHub source
6
+ > (pinned ref) and its contamination rating was corrected from "NONE /
7
+ > private corpus" to **MEDIUM** on 2026-06-11. Where this document and a card
8
+ > disagree, the card wins. Kept for rebuild archaeology only.
9
+
10
+ > **Generated**: 2026-06-09
11
+ > **Cards**: 51 total (2 EDTeKLA, 46 Tatoeba eval, 3 reference)
12
+ > **Schema**: [`corpora-card.schema.json`](./schemas/corpora-card.schema.json)
13
+ > **Card directory**: [`corpora-cards/`](./corpora-cards/)
14
+
15
+ This document catalogues every v1 corpora card for use when rebuilding, validating, or extending the card set.
16
+
17
+ ---
18
+
19
+ ## Table of Contents
20
+
21
+ 1. [EDTeKLA Cards](#1-edtekla-cards-2)
22
+ 2. [Tatoeba Eval Cards](#2-tatoeba-eval-cards-46)
23
+ 3. [Reference Cards](#3-reference-cards-3)
24
+ 4. [Field Inventory](#4-field-inventory)
25
+ 5. [Schema Migration Notes](#5-schema-migration-notes-oldnew-field-mapping)
26
+ 6. [Vitality Legend](#6-vitality-legend)
27
+
28
+ ---
29
+
30
+ ## 1. EDTeKLA Cards (2)
31
+
32
+ Private Plains Cree evaluation sets from the EDTeKLA Project, University of Alberta. Not publicly available. Zero contamination risk.
33
+
34
+ | ID | Pair | Dev Size | License | Commercial | doNotTrain | secretTest | stewardship | Contam. | Domain | Target Vitality |
35
+ |----|------|----------|---------|------------|------------|------------|-------------|---------|--------|-----------------|
36
+ | `eval-eng-crk-edtekla-dev-v1` | eng→crk | 436 entries | CC BY-NC-SA 4.0 | ❌ | ✅ | null | null | NONE | educational | **severely-endangered** |
37
+ | `eval-eng-crk-edtekla-textbook` | eng→crk | 486 entries | CC BY-NC-SA 4.0 | ❌ | ✅ | null | null | NONE | educational | **severely-endangered** |
38
+
39
+ **Key details:**
40
+ - **Relationship**: `textbook` is the full corpus (486 = 436 dev + 50 held-out). `dev-v1` is the 436-entry dev split only.
41
+ - **Quality**: Human-translated by L1 Plains Cree speakers, certified educators. Multi-pass expert review. SRO orthography.
42
+ - **AI Training**: Non-commercial only.
43
+ - **Publisher**: EDTeKLA Project, University of Alberta. No public URL, paper, or citation.
44
+ - **Data file**: `curated/eng-crk-dev-v1.json` (harness-json format).
45
+ - **Submission**: null (no submission terms defined yet).
46
+
47
+ ---
48
+
49
+ ## 2. Tatoeba Eval Cards (46)
50
+
51
+ Community-curated evaluation sets built via Tatoeba API by `corpora-builder v0.1.0`. All are:
52
+ - **License**: CC-BY-2.0 (commercial ✅, redistribution ✅)
53
+ - **doNotTrain**: ✅ (all set to `true`)
54
+ - **secretTest**: null (none have secret test sets yet)
55
+ - **stewardship**: null (no steward governance established yet)
56
+ - **submission**: null (no submission terms defined yet)
57
+ - **Domain**: mixed
58
+ - **Quality**: Human-translated (Tatoeba community volunteers). No translator qualifications, review process, or orthography specified.
59
+ - **Format**: harness-json
60
+
61
+ ### Grouped by Target Language Vitality
62
+
63
+ #### Critically Endangered (1)
64
+
65
+ | ID | Pair | Dev Size | Contam. | Target Language |
66
+ |----|------|----------|---------|-----------------|
67
+ | `eval-eng-haw-tatoeba-dev-v1` | eng→haw | 194 entries | MEDIUM | Hawaiian |
68
+
69
+ #### Definitely Endangered (1)
70
+
71
+ | ID | Pair | Dev Size | Contam. | Target Language |
72
+ |----|------|----------|---------|-----------------|
73
+ | `eval-eng-sme-tatoeba-dev-v1` | eng→sme | 58 entries | LOW | Northern Sámi |
74
+
75
+ #### Vulnerable (10)
76
+
77
+ | ID | Pair | Dev Size | Contam. | Target Language |
78
+ |----|------|----------|---------|-----------------|
79
+ | `eval-eng-cym-tatoeba-dev-v1` | eng→cym | 47 entries | LOW | Welsh |
80
+ | `eval-eng-ibo-tatoeba-dev-v1` | eng→ibo | 35 entries | LOW | Igbo |
81
+ | `eval-eng-pag-tatoeba-dev-v1` | eng→pag | 60 entries | LOW | Pangasinan |
82
+ | `eval-eng-pam-tatoeba-dev-v1` | eng→pam | 48 entries | LOW | Kapampangan |
83
+ | `eval-eng-war-tatoeba-dev-v1` | eng→war | 131 entries | LOW | Waray |
84
+ | `eval-fra-cat-tatoeba-dev-v1` | fra→cat | 57 entries | LOW | Catalan |
85
+ | `eval-fra-eus-tatoeba-dev-v1` | fra→eus | 59 entries | LOW | Basque |
86
+ | `eval-nld-fry-tatoeba-dev-v1` | nld→fry | 58 entries | LOW | Western Frisian |
87
+ | `eval-por-glg-tatoeba-dev-v1` | por→glg | 102 entries | LOW | Galician |
88
+ | `eval-spa-que-tatoeba-dev-v1` | spa→que | 95 entries | LOW | Quechua |
89
+
90
+ #### Safe (33)
91
+
92
+ | ID | Pair | Dev Size | Contam. | Target Language |
93
+ |----|------|----------|---------|-----------------|
94
+ | `eval-dan-fao-tatoeba-dev-v1` | dan→fao | 168 entries | MEDIUM | Faroese |
95
+ | `eval-deu-ltz-tatoeba-dev-v1` | deu→ltz | 179 entries | MEDIUM | Luxembourgish |
96
+ | `eval-eng-amh-tatoeba-dev-v1` | eng→amh | 73 entries | LOW | Amharic |
97
+ | `eval-eng-bos-tatoeba-dev-v1` | eng→bos | 64 entries | LOW | Bosnian |
98
+ | `eval-eng-ceb-tatoeba-dev-v1` | eng→ceb | 132 entries | LOW | Cebuano |
99
+ | `eval-eng-guj-tatoeba-dev-v1` | eng→guj | 165 entries | MEDIUM | Gujarati |
100
+ | `eval-eng-hau-tatoeba-dev-v1` | eng→hau | 140 entries | LOW | Hausa |
101
+ | `eval-eng-hil-tatoeba-dev-v1` | eng→hil | 56 entries | LOW | Hiligaynon |
102
+ | `eval-eng-ilo-tatoeba-dev-v1` | eng→ilo | 105 entries | LOW | Ilocano |
103
+ | `eval-eng-kan-tatoeba-dev-v1` | eng→kan | 61 entries | LOW | Kannada |
104
+ | `eval-eng-kaz-tatoeba-dev-v1` | eng→kaz | 113 entries | LOW | Kazakh |
105
+ | `eval-eng-lao-tatoeba-dev-v1` | eng→lao | 68 entries | LOW | Lao |
106
+ | `eval-eng-lug-tatoeba-dev-v1` | eng→lug | 183 entries | MEDIUM | Ganda |
107
+ | `eval-eng-mal-tatoeba-dev-v1` | eng→mal | 59 entries | LOW | Malayalam |
108
+ | `eval-eng-mlt-tatoeba-dev-v1` | eng→mlt | 129 entries | LOW | Maltese |
109
+ | `eval-eng-mon-tatoeba-dev-v1` | eng→mon | 138 entries | LOW | Mongolian |
110
+ | `eval-eng-mya-tatoeba-dev-v1` | eng→mya | 77 entries | LOW | Burmese |
111
+ | `eval-eng-pan-tatoeba-dev-v1` | eng→pan | 68 entries | LOW | Panjabi |
112
+ | `eval-eng-sin-tatoeba-dev-v1` | eng→sin | 69 entries | LOW | Sinhala |
113
+ | `eval-eng-sna-tatoeba-dev-v1` | eng→sna | 47 entries | LOW | Shona |
114
+ | `eval-eng-tam-tatoeba-dev-v1` | eng→tam | 153 entries | MEDIUM | Tamil |
115
+ | `eval-eng-tel-tatoeba-dev-v1` | eng→tel | 71 entries | LOW | Telugu |
116
+ | `eval-eng-tir-tatoeba-dev-v1` | eng→tir | 54 entries | LOW | Tigrinya |
117
+ | `eval-eng-urd-tatoeba-dev-v1` | eng→urd | 181 entries | MEDIUM | Urdu |
118
+ | `eval-eng-uzb-tatoeba-dev-v1` | eng→uzb | 167 entries | MEDIUM | Uzbek |
119
+ | `eval-eng-xho-tatoeba-dev-v1` | eng→xho | 75 entries | LOW | Xhosa |
120
+ | `eval-eng-yor-tatoeba-dev-v1` | eng→yor | 68 entries | LOW | Yoruba |
121
+ | `eval-eng-zsm-tatoeba-dev-v1` | eng→zsm | 148 entries | LOW | Standard Malay |
122
+ | `eval-eng-zul-tatoeba-dev-v1` | eng→zul | 112 entries | LOW | Zulu |
123
+ | `eval-fra-hau-tatoeba-dev-v1` | fra→hau | 168 entries | MEDIUM | Hausa |
124
+ | `eval-fra-ltz-tatoeba-dev-v1` | fra→ltz | 196 entries | MEDIUM | Luxembourgish |
125
+ | `eval-ita-mlt-tatoeba-dev-v1` | ita→mlt | 180 entries | MEDIUM | Maltese |
126
+ | `eval-rus-uzb-tatoeba-dev-v1` | rus→uzb | 51 entries | LOW | Uzbek |
127
+
128
+ #### No Language Card (1)
129
+
130
+ | ID | Pair | Dev Size | Contam. | Target Language |
131
+ |----|------|----------|---------|-----------------|
132
+ | `eval-eng-sqi-tatoeba-dev-v1` | eng→sqi | 61 entries | LOW | Albanian (no `sqi.json` language card) |
133
+
134
+ **Note**: `sqi` uses the ISO 639-3 macrolanguage code for Albanian. The language cards directory may use a more specific code (e.g., `aln` for Gheg Albanian or `als` for Tosk Albanian). This needs resolution.
135
+
136
+ ### Tatoeba Source Pairs Summary
137
+
138
+ | Source Language | # Pairs |
139
+ |----------------|---------|
140
+ | eng (English) | 34 |
141
+ | fra (French) | 4 |
142
+ | dan (Danish) | 1 |
143
+ | deu (German) | 1 |
144
+ | ita (Italian) | 1 |
145
+ | nld (Dutch) | 1 |
146
+ | por (Portuguese) | 1 |
147
+ | rus (Russian) | 1 |
148
+ | spa (Spanish) | 1 |
149
+
150
+ ### Dev Set Size Distribution
151
+
152
+ | Range | Count | Cards |
153
+ |-------|-------|-------|
154
+ | < 50 entries | 5 | cym (47), ibo (35), pam (48), sna (47), bos (64→corrected: see below) |
155
+ | 50–99 | 18 | amh, hil, kaz, kan, lao, mal, pan, sin, sme, sqi, tel, tir, uzb-rus, yor, cat, eus, fry, mya |
156
+ | 100–149 | 11 | ceb, hau, ilo, kaz, mlt, mon, war, glg, que, zul, zsm |
157
+ | 150–199 | 9 | dan→fao, deu→ltz, guj, haw, lug, tam, urd, uzb-eng, fra→ltz |
158
+ | ≥ 200 | 1 | fra→hau (168→corrected: none ≥200) |
159
+
160
+ **Actual smallest**: eng→ibo at 35 entries.
161
+ **Actual largest**: fra→ltz at 196 entries.
162
+
163
+ ---
164
+
165
+ ## 3. Reference Cards (3)
166
+
167
+ Multi-language reference corpora catalogued for development use. These do NOT have `pair`, `dev`, `doNotTrain`, `secretTest`, or `stewardship` fields — they use `languages`, `segments`, and `download` instead.
168
+
169
+ | ID | Name | Version | # Languages | Segments | License | Commercial | Contam. |
170
+ |----|------|---------|-------------|----------|---------|------------|---------|
171
+ | `ref-flores-plus` | FLORES+ | 2.0 | 197 | dev: 997 sent, devtest: 1012 sent | CC-BY-SA-4.0 | ❌ | **HIGH** |
172
+ | `ref-ntrex-128` | NTREX-128 | 1.0 | 104 | test: 1997 sent | CC-BY-SA-4.0 | ❌ | MEDIUM |
173
+ | `ref-tatoeba-challenge` | Tatoeba Challenge | 2023-09-26 | 108 | test: varies, dev: varies | CC-BY-2.0 | ✅ | MEDIUM |
174
+
175
+ ### FLORES+
176
+ - **Publisher**: Meta AI (NLLB Team)
177
+ - **Source**: English Wikipedia + Wikinews, professionally translated
178
+ - **Known in training of**: NLLB-200
179
+ - **Download**: `git clone https://github.com/openlanguagedata/flores.git`
180
+ - **Warning**: Heavily contaminated in frontier LLM training data. Scores should be treated as relative comparisons only.
181
+
182
+ ### NTREX-128
183
+ - **Publisher**: Microsoft Research
184
+ - **Source**: WMT19 English news test set, professionally translated into 128 languages
185
+ - **Download**: `git clone https://github.com/MicrosoftTranslator/NTREX.git`
186
+ - **Note**: Less contaminated than FLORES+ but source sentences from WMT may appear in training data.
187
+
188
+ ### Tatoeba Challenge
189
+ - **Publisher**: Tatoeba community / OPUS / University of Helsinki
190
+ - **Source**: User-contributed translations, short conversational sentences
191
+ - **Known in training of**: OPUS-MT, Helsinki-NLP models
192
+ - **Download**: Per-pair packages from `https://github.com/Helsinki-NLP/Tatoeba-Challenge/tree/master/data`
193
+ - **License caveat**: Individual sentences may have different licenses. CC-BY-2.0 applies to the collection as distributed.
194
+
195
+ ---
196
+
197
+ ## 4. Field Inventory
198
+
199
+ ### Fields Present on All Cards
200
+
201
+ | Field | Type | Required | Notes |
202
+ |-------|------|----------|-------|
203
+ | `id` | string | ✅ | Pattern: `^(ref\|eval)-[a-z0-9][a-z0-9-]*$` |
204
+ | `type` | `"reference"` \| `"eval"` | ✅ | |
205
+ | `name` | string | ✅ | |
206
+ | `version` | string | ✅ | Semver recommended |
207
+ | `description` | string | ✅ | |
208
+ | `source` | object | ✅ | `{publisher, url, paper, citation, fundedBy}` |
209
+ | `license` | object | ✅ | `{spdx, commercial, redistribution, aiTraining, notes}` |
210
+ | `contamination` | object | ✅ | `{risk, reasoning, knownInTrainingOf?}` |
211
+ | `_provenance` | object | ✅ | `{addedAt, lastUpdated, populatedFrom}` |
212
+
213
+ ### Eval-Only Fields (required when `type: "eval"`)
214
+
215
+ | Field | Type | Required | Current Status |
216
+ |-------|------|----------|----------------|
217
+ | `pair` | object | ✅ | `{source, target, direction}` — all cards use `"unidirectional"` |
218
+ | `dev` | object | ✅ | `{size, sizeUnit, domain, domainDistribution, dataFile, format}` |
219
+ | `doNotTrain` | boolean | ✅ | All 48 eval cards set to `true` |
220
+ | `secretTest` | object \| null | — | All cards currently `null` |
221
+ | `stewardship` | object \| null | — | All cards currently `null` |
222
+ | `submission` | object \| null | — | All cards currently `null` |
223
+ | `quality` | object \| null | — | Populated for EDTeKLA; partially populated for Tatoeba |
224
+
225
+ ### Reference-Only Fields (required when `type: "reference"`)
226
+
227
+ | Field | Type | Required | Notes |
228
+ |-------|------|----------|-------|
229
+ | `languages` | string[] | ✅ | ISO 639-3 codes |
230
+ | `download` | object | ✅ | `{method, url, instructions, sha256}` |
231
+ | `segments` | object[] | — | `{id, name, size?, sizeUnit?, purpose}` |
232
+
233
+ ---
234
+
235
+ ## 5. Schema Migration Notes (Old→New Field Mapping)
236
+
237
+ The schema currently contains these fields inside `submission`, which represent an older naming convention that will be restructured:
238
+
239
+ | Old Field (in schema) | Current Location | Planned New Location | Purpose |
240
+ |-----------------------|------------------|---------------------|---------|
241
+ | `ipTransfer` | `submission.ipTransfer` | `submission.transfer` | Whether method authors must transfer IP rights |
242
+ | `ipTransferTerms` | `submission.ipTransferTerms` | `submission.transfer` (merged) / `submission.retained` | Full text of transfer agreement |
243
+ | `doNotTrain` | top-level `doNotTrain` | Stays top-level + also informs `usageRestrictions` | Prohibits use for ML training |
244
+
245
+ **Current state**: The `submission` field is `null` on all 51 cards. The `ipTransfer` and `ipTransferTerms` fields exist in the schema but have never been populated in any card. When these are eventually populated, the field names should be updated per the mapping above.
246
+
247
+ **`doNotTrain`**: Currently a top-level boolean on all eval cards (always `true`). In a future schema revision, this may also feed into a broader `usageRestrictions` object, but the top-level field will remain for backwards compatibility.
248
+
249
+ ---
250
+
251
+ ## 6. Vitality Legend
252
+
253
+ UNESCO vitality status for target languages, sourced from language cards:
254
+
255
+ | Status | Meaning | # Targets |
256
+ |--------|---------|-----------|
257
+ | **critically-endangered** | Most members of youngest generation are speakers. But the language is not spoken in most everyday contexts. | 1 (Hawaiian) |
258
+ | **severely-endangered** | Language is spoken by grandparents and older generations. While the parent generation may understand it, they do not speak it to children or among themselves. | 1 (Plains Cree — EDTeKLA) |
259
+ | **definitely-endangered** | Children no longer learn the language as mother tongue in the home. | 1 (Northern Sámi) |
260
+ | **vulnerable** | Most children speak the language, but it may be restricted to certain domains (e.g., home). | 10 (Welsh, Igbo, Pangasinan, Kapampangan, Waray, Catalan, Basque, Western Frisian, Galician, Quechua) |
261
+ | **safe** | Language is spoken by all generations; intergenerational transmission is uninterrupted. | 33 |
262
+ | **NO_CARD** | No language card found for this ISO 639-3 code. | 1 (sqi — Albanian macrolanguage) |
263
+
264
+ ---
265
+
266
+ ## Quick Stats
267
+
268
+ | Metric | Value |
269
+ |--------|-------|
270
+ | Total cards | 51 |
271
+ | Eval cards | 48 |
272
+ | Reference cards | 3 |
273
+ | Unique source languages | 9 (eng, fra, dan, deu, ita, nld, por, rus, spa) |
274
+ | Unique target languages | 44 |
275
+ | Endangered/vulnerable targets | 13 of 44 (30%) |
276
+ | Smallest eval set | eng→ibo: 35 entries |
277
+ | Largest eval set | eng→crk-textbook: 486 entries |
278
+ | All eval doNotTrain | ✅ (100%) |
279
+ | Cards with secretTest | 0 |
280
+ | Cards with stewardship | 0 |
281
+ | Cards with submission terms | 0 |
@@ -0,0 +1,35 @@
1
+ {
2
+ "_comment": [
3
+ "Curated per-dictionary capability/redistribution flags consumed by scripts/derive-dictionaries.mjs.",
4
+ "Keyed by language code, then by a case-insensitive substring of the dictionary name as it",
5
+ "appears on the card. Flags: machineReadable (structured/API access exists), redistributable",
6
+ "(false = pointer-only: content must never be copied into corpora, cards, exports, or public",
7
+ "repos), license (SPDX-ish string, only when actually known). Every entry MUST cite its basis",
8
+ "in the source string. Only record what is documented — omitted flags mean unknown, and the",
9
+ "deriver writes nothing it cannot cite (index, not arbiter).",
10
+ "Data-over-code (docs/AGENTS.md §1): language-specific facts live in this file, never",
11
+ "hardcoded in the deriver."
12
+ ],
13
+ "flags": {
14
+ "crk": {
15
+ "itwêwina": {
16
+ "machineReadable": true,
17
+ "redistributable": false,
18
+ "source": "manual-curation (the card's own evalStandard.ipNotice: gloss data is fetched live from the public itwêwina API, and the underlying dictionary content — Wolvengrey CW, Maskwacîs MD, AECD — is NOT openly licensed and must never be redistributed. Founder ruling 2026-07-19: redistribution consent will not be sought and is not needed — the dictionary is INDEXED as a cited resource, pointer-only, permanently; CLAUDE.md Wolvengrey boundary)"
19
+ },
20
+ "Wolvengrey": {
21
+ "machineReadable": false,
22
+ "redistributable": false,
23
+ "source": "print dictionary (University of Regina Press, uofrpress.ca/Books/C/Cree-Words); rights-holder Arok Wolvengrey (First Nations University of Canada) — no public license — indexed pointer-only, permanently (founder ruling 2026-07-19; CLAUDE.md Wolvengrey boundary; shared/licenses.json wolvengrey-itwewina UNLICENSED)"
24
+ },
25
+ "Maskwacîs": {
26
+ "redistributable": false,
27
+ "source": "Maskwachees Cultural College dictionary (1998/2009, named in giellalt/lang-crk LICENSE); served via the itwêwina API — content not openly licensed, pointer-only (lyss NOTICE; shared/licenses.json)"
28
+ },
29
+ "Alberta Elders'": {
30
+ "redistributable": false,
31
+ "source": "Alberta Elders' Cree Dictionary (Earle Waugh, ed.); served via the itwêwina API — content not openly licensed, pointer-only (lyss NOTICE; shared/licenses.json)"
32
+ }
33
+ }
34
+ }
35
+ }
@@ -0,0 +1,35 @@
1
+ {
2
+ "_meta": {
3
+ "description": "Hand-curated endonym seed for languages absent from every harvested source (docs/CARD_DATA_QUALITY_AUDIT.md section 3b: 'Curated seed for high-visibility families'). Every entry MUST cite a published, checkable source — an endonym that cannot be cited stays out of this file. Applied by cli/scripts/apply-curated-endonyms.mjs (merge-only), stamped 'manual-curation+<source_url>'. Surfaced for community review; community-preferred display names override on request (docs/LANGUAGE_TAXONOMY.md).",
4
+ "format": "latin orthography / syllabics, following the existing crk card convention",
5
+ "created": "2026-06-11"
6
+ },
7
+ "crm": {
8
+ "nativeName": "ililîmowin / ᐃᓕᓖᒧᐎᓐ",
9
+ "script": "Cans",
10
+ "source_url": "https://en.wikipedia.org/wiki/Moose_Cree_language",
11
+ "source_note": "Wikipedia 'Moose Cree language' infobox: ᐃᓕᓖᒧᐎᓐ Ililîmowin. Corroborated by aaniskohtaaw.ca (Moose Cree language class, 'ililîmowin') and the Dictionary of Moose Cree (moosecree.ca).",
12
+ "retrieved": "2026-06-11"
13
+ },
14
+ "crl": {
15
+ "nativeName": "īyiyū ayimūn / ᐄᔨᔫ ᐊᔨᒨᓐ",
16
+ "script": "Cans",
17
+ "source_url": "https://en.wikipedia.org/wiki/East_Cree",
18
+ "source_note": "Wikipedia 'East Cree' lead: Northern East Cree ᐄᔨᔫ ᐊᔨᒨᓐ Īyiyū Ayimūn.",
19
+ "retrieved": "2026-06-11"
20
+ },
21
+ "crj": {
22
+ "nativeName": "īnū ayimūn / ᐄᓅ ᐊᔨᒨᓐ",
23
+ "script": "Cans",
24
+ "source_url": "https://en.wikipedia.org/wiki/East_Cree",
25
+ "source_note": "Wikipedia 'East Cree' lead: Southern East Cree ᐄᓅ ᐊᔨᒨᓐ Īnū Ayimūn.",
26
+ "retrieved": "2026-06-11"
27
+ },
28
+ "cwd": {
29
+ "nativeName": "nīhithawīwin / ᓀᐦᐃᖬᐍᐏᐣ",
30
+ "script": "Cans",
31
+ "source_url": "https://en.wikipedia.org/wiki/Woods_Cree",
32
+ "source_note": "Wikipedia 'Woods Cree' infobox: Nīhithawīwin ᓀᐦᐃᖬᐍᐏᐣ (the 'th' dialect; ᖬ = th-series syllabic distinctive to Woods Cree). Same form in the 2020-07-27 enwiki snapshot on archive.org.",
33
+ "retrieved": "2026-06-11"
34
+ }
35
+ }
@@ -0,0 +1,51 @@
1
+ {
2
+ "_comment": [
3
+ "Curated per-language FST registrations consumed by scripts/derive-fsts.mjs",
4
+ "(merged into resources.fsts[] — resource EXISTENCE only, boundary invariant kind 2).",
5
+ "Seeded 2026-07-19 with the GiellaLT PRODUCTION-maturity languages missing from the",
6
+ "cards (the live-check lane in generate-language-card.mjs only ran for ~20 launch",
7
+ "languages). Maturity is GiellaLT's own published tier (map-maturity-prod.html), not",
8
+ "our judgment. NOTE the repo slug is not always the ISO code (Kalaallisut = lang-kl).",
9
+ "Per the fst_installer.py doctrine, entries carry NO install pins: GitHub releases",
10
+ "are vestigial for GiellaLT; distribution is their nightly channel.",
11
+ "Every entry MUST carry a source string citing its basis (index, not arbiter)."
12
+ ],
13
+ "perLanguage": {
14
+ "fao": [
15
+ {
16
+ "name": "GiellaLT Faroese FST (lang-fao)",
17
+ "url": "https://github.com/giellalt/lang-fao",
18
+ "type": "morphological-analyzer",
19
+ "maturity": "production",
20
+ "source": "GiellaLT maturity map (giellalt.github.io/map-maturity-prod.html), production tier — verified 2026-07-19"
21
+ }
22
+ ],
23
+ "fkv": [
24
+ {
25
+ "name": "GiellaLT Kven FST (lang-fkv)",
26
+ "url": "https://github.com/giellalt/lang-fkv",
27
+ "type": "morphological-analyzer",
28
+ "maturity": "production",
29
+ "source": "GiellaLT maturity map (giellalt.github.io/map-maturity-prod.html), production tier — verified 2026-07-19"
30
+ }
31
+ ],
32
+ "kal": [
33
+ {
34
+ "name": "GiellaLT Kalaallisut FST (lang-kal)",
35
+ "url": "https://github.com/giellalt/lang-kal",
36
+ "type": "morphological-analyzer",
37
+ "maturity": "production",
38
+ "source": "GiellaLT maturity map (giellalt.github.io/map-maturity-prod.html), production tier; repo verified live (github.com/giellalt/lang-kal HTTP 200) 2026-07-19"
39
+ }
40
+ ],
41
+ "nno": [
42
+ {
43
+ "name": "GiellaLT Norwegian Nynorsk FST (lang-nno)",
44
+ "url": "https://github.com/giellalt/lang-nno",
45
+ "type": "morphological-analyzer",
46
+ "maturity": "production",
47
+ "source": "GiellaLT maturity map (giellalt.github.io/map-maturity-prod.html), production tier — verified 2026-07-19"
48
+ }
49
+ ]
50
+ }
51
+ }