champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,2975 @@
1
+ {
2
+ "_meta": {
3
+ "description": "Plain-language glossary of technical terms that appear in rendered language-card text (linguisticChallenges, typology labels, formality/gender descriptions, scripts, numeral systems, eval metadata) and in tc-features.json feature names/values. Definitions are original prose written for Champollion; citations point to the standard reference for each concept.",
4
+ "curated": "hand-written 2026-06-11; harvest source: cli/shared/language-cards/*.json text fields + cli/shared/explainers/tc-features.json names/values",
5
+ "policy": "plain = 2-3 sentence original definition (no copied text); mt_relevance = 1 sentence; related entries must exist in this file; `also` lists surface variants used on cards (detection vocabulary for the linter)."
6
+ },
7
+ "terms": [
8
+ {
9
+ "term": "morpheme",
10
+ "also": [
11
+ "morphemes"
12
+ ],
13
+ "plain": "The smallest piece of a word that carries meaning. The English word 'unhappiness' contains three: un-, happy, and -ness. Languages differ enormously in how many morphemes they pack into one word.",
14
+ "mt_relevance": "MT systems work on tokens, and a token that contains many morphemes hides grammar the system needs to translate correctly.",
15
+ "citations": [
16
+ {
17
+ "source": "SIL Glossary of Linguistic Terms",
18
+ "url": "https://glossary.sil.org/term/morpheme"
19
+ }
20
+ ],
21
+ "related": [
22
+ "affix",
23
+ "root",
24
+ "tokenization"
25
+ ]
26
+ },
27
+ {
28
+ "term": "morphology",
29
+ "also": [
30
+ "morphological complexity",
31
+ "morphologically complex",
32
+ "complex morphology"
33
+ ],
34
+ "plain": "The study of how words are built from smaller meaningful parts, and the word-building system of a language itself. A morphologically complex language expresses with word endings what English expresses with separate words.",
35
+ "mt_relevance": "Complex morphology multiplies the distinct word forms an MT system must learn from limited data.",
36
+ "citations": [
37
+ {
38
+ "source": "SIL Glossary of Linguistic Terms",
39
+ "url": "https://glossary.sil.org/term/morphology"
40
+ }
41
+ ],
42
+ "related": [
43
+ "morpheme",
44
+ "inflection",
45
+ "agglutinative",
46
+ "polysynthesis"
47
+ ]
48
+ },
49
+ {
50
+ "term": "inflection",
51
+ "also": [
52
+ "inflectional",
53
+ "inflectional synthesis",
54
+ "inflected"
55
+ ],
56
+ "plain": "Changing a word's form to express grammar — tense, number, case, gender — without changing its core meaning, like English sing/sang or cat/cats. Highly inflected languages can mark six or more categories on a single word.",
57
+ "mt_relevance": "Every inflected form is a separate token for an MT system, so heavy inflection means more rare words and more agreement errors.",
58
+ "citations": [
59
+ {
60
+ "source": "SIL Glossary of Linguistic Terms",
61
+ "url": "https://glossary.sil.org/term/inflection"
62
+ }
63
+ ],
64
+ "related": [
65
+ "morphology",
66
+ "paradigm",
67
+ "agreement"
68
+ ]
69
+ },
70
+ {
71
+ "term": "derivation",
72
+ "also": [
73
+ "derivational"
74
+ ],
75
+ "plain": "Building new words from existing ones, like teach → teacher or happy → unhappy. Unlike inflection, derivation creates a new dictionary word rather than a grammatical variant of the same word.",
76
+ "mt_relevance": "Productive derivation lets speakers coin words on the fly that an MT system has never seen.",
77
+ "citations": [
78
+ {
79
+ "source": "SIL Glossary of Linguistic Terms",
80
+ "url": "https://glossary.sil.org/term/derivation"
81
+ }
82
+ ],
83
+ "related": [
84
+ "inflection",
85
+ "compounding",
86
+ "morpheme"
87
+ ]
88
+ },
89
+ {
90
+ "term": "affix",
91
+ "also": [
92
+ "affixes",
93
+ "affixation"
94
+ ],
95
+ "plain": "A morpheme attached to a word stem to modify its meaning or grammar. Prefixes attach before the stem, suffixes after, and a few languages use infixes inside the stem.",
96
+ "mt_relevance": "Where affixes attach determines where subword tokenizers should split words for the MT model.",
97
+ "citations": [
98
+ {
99
+ "source": "SIL Glossary of Linguistic Terms",
100
+ "url": "https://glossary.sil.org/term/affix"
101
+ }
102
+ ],
103
+ "related": [
104
+ "prefix",
105
+ "suffix",
106
+ "infix",
107
+ "morpheme"
108
+ ]
109
+ },
110
+ {
111
+ "term": "prefix",
112
+ "also": [
113
+ "prefixes",
114
+ "prefixing"
115
+ ],
116
+ "plain": "An affix that attaches to the front of a word stem, like re- in 'rewrite'. Some languages, including many Bantu and Athabaskan languages, carry most of their grammar in strings of prefixes.",
117
+ "mt_relevance": "Prefix-heavy languages put grammatical information at the start of words, the opposite of what suffix-trained tokenizers expect.",
118
+ "citations": [
119
+ {
120
+ "source": "WALS Online, chapter 26 (Dryer)",
121
+ "url": "https://wals.info/chapter/26"
122
+ }
123
+ ],
124
+ "related": [
125
+ "suffix",
126
+ "affix"
127
+ ]
128
+ },
129
+ {
130
+ "term": "suffix",
131
+ "also": [
132
+ "suffixes",
133
+ "suffixing"
134
+ ],
135
+ "plain": "An affix that attaches to the end of a word stem, like -ness in 'kindness'. Suffixing is the most common affixation strategy across the world's languages.",
136
+ "mt_relevance": "Stacked suffixes create long, rare word forms that MT systems must segment correctly to translate.",
137
+ "citations": [
138
+ {
139
+ "source": "WALS Online, chapter 26 (Dryer)",
140
+ "url": "https://wals.info/chapter/26"
141
+ }
142
+ ],
143
+ "related": [
144
+ "prefix",
145
+ "affix"
146
+ ]
147
+ },
148
+ {
149
+ "term": "infix",
150
+ "also": [
151
+ "infixes",
152
+ "infixation"
153
+ ],
154
+ "plain": "An affix inserted inside a word stem rather than before or after it. Tagalog, for example, turns sulat 'write' into sumulat 'wrote' by inserting -um- after the first consonant.",
155
+ "mt_relevance": "Infixes break the assumption that a word's stem is a contiguous string, which defeats simple subword segmentation.",
156
+ "citations": [
157
+ {
158
+ "source": "SIL Glossary of Linguistic Terms",
159
+ "url": "https://glossary.sil.org/term/infix"
160
+ }
161
+ ],
162
+ "related": [
163
+ "affix",
164
+ "root-and-pattern morphology"
165
+ ]
166
+ },
167
+ {
168
+ "term": "clitic",
169
+ "also": [
170
+ "clitics",
171
+ "proclitic",
172
+ "enclitic"
173
+ ],
174
+ "plain": "A small grammatical word that cannot stand on its own and leans on a neighboring word, like the 's in \"the queen of England's hat\". Clitics behave partly like words and partly like affixes.",
175
+ "mt_relevance": "Clitics blur word boundaries, so tokenizers may attach them to the wrong host and scramble the grammar.",
176
+ "citations": [
177
+ {
178
+ "source": "SIL Glossary of Linguistic Terms",
179
+ "url": "https://glossary.sil.org/term/clitic"
180
+ }
181
+ ],
182
+ "related": [
183
+ "affix",
184
+ "particle",
185
+ "tokenization"
186
+ ]
187
+ },
188
+ {
189
+ "term": "stem",
190
+ "also": [
191
+ "stems",
192
+ "word stem"
193
+ ],
194
+ "plain": "The core part of a word that affixes attach to. In 'unbelievable', the stem of -able is 'believe'. In many languages a stem never appears alone and must carry at least some inflection.",
195
+ "mt_relevance": "Identifying shared stems across word forms is how MT systems generalize from limited training data.",
196
+ "citations": [
197
+ {
198
+ "source": "SIL Glossary of Linguistic Terms",
199
+ "url": "https://glossary.sil.org/term/stem-1"
200
+ }
201
+ ],
202
+ "related": [
203
+ "root",
204
+ "affix",
205
+ "inflection"
206
+ ]
207
+ },
208
+ {
209
+ "term": "root",
210
+ "also": [
211
+ "roots",
212
+ "word root"
213
+ ],
214
+ "plain": "The irreducible core of a word once all affixes are stripped away. In Semitic languages a root is often just three consonants (like k-t-b 'write' in Arabic) that vowel patterns turn into words.",
215
+ "mt_relevance": "Languages whose roots are discontinuous (consonant skeletons) need specialized segmentation for MT to see word relationships.",
216
+ "citations": [
217
+ {
218
+ "source": "SIL Glossary of Linguistic Terms",
219
+ "url": "https://glossary.sil.org/term/root-1"
220
+ }
221
+ ],
222
+ "related": [
223
+ "stem",
224
+ "root-and-pattern morphology",
225
+ "morpheme"
226
+ ]
227
+ },
228
+ {
229
+ "term": "paradigm",
230
+ "also": [
231
+ "inflectional paradigm",
232
+ "paradigms"
233
+ ],
234
+ "plain": "The full set of inflected forms a word can take — like a verb conjugation table. Paradigms range from two forms (English 'must') to thousands in polysynthetic languages.",
235
+ "mt_relevance": "Large paradigms guarantee that most word forms are rare or unseen in training data.",
236
+ "citations": [
237
+ {
238
+ "source": "SIL Glossary of Linguistic Terms",
239
+ "url": "https://glossary.sil.org/term/paradigm"
240
+ }
241
+ ],
242
+ "related": [
243
+ "inflection",
244
+ "suppletion"
245
+ ]
246
+ },
247
+ {
248
+ "term": "agglutinative",
249
+ "also": [
250
+ "agglutination",
251
+ "agglutinating",
252
+ "agglutinative morphology"
253
+ ],
254
+ "plain": "A word-building style where each grammatical meaning gets its own clearly separable affix, stacked one after another — as in Turkish or Swahili. Words can get long, but each piece has one job.",
255
+ "mt_relevance": "Agglutinative words segment cleanly into subwords, but only if the tokenizer learns the language's affix order.",
256
+ "citations": [
257
+ {
258
+ "source": "SIL Glossary of Linguistic Terms",
259
+ "url": "https://glossary.sil.org/term/agglutinative-language"
260
+ }
261
+ ],
262
+ "related": [
263
+ "fusional",
264
+ "isolating",
265
+ "polysynthesis",
266
+ "morphology"
267
+ ]
268
+ },
269
+ {
270
+ "term": "fusional",
271
+ "also": [
272
+ "fusional morphology",
273
+ "inflecting language"
274
+ ],
275
+ "plain": "A word-building style where one affix fuses several grammatical meanings at once. The Spanish ending -ó in habló marks past tense, third person, and singular simultaneously — no part of it can be assigned to just one meaning.",
276
+ "mt_relevance": "Fused endings cannot be decomposed by tokenizers, so each combination must be learned as a unit.",
277
+ "citations": [
278
+ {
279
+ "source": "SIL Glossary of Linguistic Terms",
280
+ "url": "https://glossary.sil.org/term/fusional-language"
281
+ }
282
+ ],
283
+ "related": [
284
+ "agglutinative",
285
+ "isolating",
286
+ "inflection"
287
+ ]
288
+ },
289
+ {
290
+ "term": "isolating",
291
+ "also": [
292
+ "analytic language",
293
+ "isolating morphology"
294
+ ],
295
+ "plain": "A word-building style where words are mostly single morphemes and grammar is expressed by word order and helper words instead of endings — as in Vietnamese or Mandarin. The opposite extreme from polysynthesis.",
296
+ "mt_relevance": "Isolating languages shift the MT problem from word segmentation to word order and function-word choice.",
297
+ "citations": [
298
+ {
299
+ "source": "SIL Glossary of Linguistic Terms",
300
+ "url": "https://glossary.sil.org/term/isolating-language"
301
+ }
302
+ ],
303
+ "related": [
304
+ "agglutinative",
305
+ "fusional",
306
+ "word order"
307
+ ]
308
+ },
309
+ {
310
+ "term": "polysynthesis",
311
+ "also": [
312
+ "polysynthetic",
313
+ "polysynthetic language"
314
+ ],
315
+ "plain": "A word-building style where a single verb can contain what other languages express as a whole sentence — subject, object, location, instrument and more, all as parts of one word. Many Indigenous American languages work this way.",
316
+ "mt_relevance": "Polysynthetic words rarely repeat exactly, so word-level MT sees an endless stream of unknown tokens.",
317
+ "example": "Plains Cree (crk): a single verb can incorporate subject/object pronouns, instrumentals, locations, and actions (card field linguisticChallenges.polysynthesis).",
318
+ "citations": [
319
+ {
320
+ "source": "SIL Glossary of Linguistic Terms",
321
+ "url": "https://glossary.sil.org/term/polysynthetic-language"
322
+ }
323
+ ],
324
+ "related": [
325
+ "noun incorporation",
326
+ "agglutinative",
327
+ "morphology"
328
+ ]
329
+ },
330
+ {
331
+ "term": "noun incorporation",
332
+ "also": [
333
+ "incorporation",
334
+ "incorporated noun"
335
+ ],
336
+ "plain": "Folding a noun into the verb to make one word, roughly like turning 'hunt seals' into 'seal-hunt' as a verb. Common in polysynthetic languages, where it changes the sentence's emphasis and grammar.",
337
+ "mt_relevance": "An incorporated noun disappears from the sentence as a separate word, so alignment-based translation loses it.",
338
+ "example": "Surfaced on cards as linguisticChallenges.nounIncorporation (Grambank-based), e.g. Plains Cree (crk).",
339
+ "citations": [
340
+ {
341
+ "source": "SIL Glossary of Linguistic Terms",
342
+ "url": "https://glossary.sil.org/term/incorporation"
343
+ }
344
+ ],
345
+ "related": [
346
+ "polysynthesis",
347
+ "compounding"
348
+ ]
349
+ },
350
+ {
351
+ "term": "compounding",
352
+ "also": [
353
+ "compound nouns",
354
+ "compounds",
355
+ "compound words"
356
+ ],
357
+ "plain": "Joining two or more independent words into one new word, like 'tooth' + 'brush' → 'toothbrush'. German-style compounding can chain many words into very long single tokens.",
358
+ "mt_relevance": "Novel compounds are unseen tokens that MT must split into known parts to translate.",
359
+ "citations": [
360
+ {
361
+ "source": "SIL Glossary of Linguistic Terms",
362
+ "url": "https://glossary.sil.org/term/compound-word"
363
+ }
364
+ ],
365
+ "related": [
366
+ "derivation",
367
+ "noun incorporation",
368
+ "tokenization"
369
+ ]
370
+ },
371
+ {
372
+ "term": "reduplication",
373
+ "also": [
374
+ "reduplicated",
375
+ "reduplicative"
376
+ ],
377
+ "plain": "Repeating all or part of a word to change its meaning — to mark plurals, intensity, or ongoing action. Indonesian orang 'person' becomes orang-orang 'people'.",
378
+ "mt_relevance": "MT must recognize that a doubled word is grammar, not an accidental repetition to be deleted.",
379
+ "citations": [
380
+ {
381
+ "source": "WALS Online, chapter 27 (Rubino)",
382
+ "url": "https://wals.info/chapter/27"
383
+ }
384
+ ],
385
+ "related": [
386
+ "morphology",
387
+ "grammatical number"
388
+ ]
389
+ },
390
+ {
391
+ "term": "ablaut",
392
+ "also": [
393
+ "apophony",
394
+ "vowel gradation"
395
+ ],
396
+ "plain": "Changing a vowel inside a word to change its grammar, like English sing/sang/sung. The word's consonant frame stays put while the vowel does the grammatical work.",
397
+ "mt_relevance": "Ablaut hides inflection inside the stem where subword tokenizers cannot isolate it.",
398
+ "citations": [
399
+ {
400
+ "source": "SIL Glossary of Linguistic Terms",
401
+ "url": "https://glossary.sil.org/term/ablaut"
402
+ }
403
+ ],
404
+ "related": [
405
+ "root-and-pattern morphology",
406
+ "umlaut",
407
+ "inflection"
408
+ ]
409
+ },
410
+ {
411
+ "term": "suppletion",
412
+ "also": [
413
+ "suppletive"
414
+ ],
415
+ "plain": "When a word's inflected forms come from completely different roots, like go/went or good/better. The grammar treats them as one word even though they share no sounds.",
416
+ "mt_relevance": "Suppletive forms cannot be derived by rule, so MT must have seen each one in training data.",
417
+ "citations": [
418
+ {
419
+ "source": "WALS Online, chapter 79 (Veselinova)",
420
+ "url": "https://wals.info/chapter/79"
421
+ }
422
+ ],
423
+ "related": [
424
+ "paradigm",
425
+ "inflection"
426
+ ]
427
+ },
428
+ {
429
+ "term": "root-and-pattern morphology",
430
+ "also": [
431
+ "root pattern",
432
+ "rootPattern",
433
+ "templatic morphology",
434
+ "nonconcatenative morphology",
435
+ "root-pattern morphology"
436
+ ],
437
+ "plain": "A word-building style, typical of Arabic and Hebrew, where a consonant root like k-t-b 'write' is threaded through vowel templates: kitāb 'book', kātib 'writer', maktab 'office'. The root and the pattern each carry meaning, but neither is a contiguous chunk.",
438
+ "mt_relevance": "Standard subword tokenization cannot see the shared root across these forms, weakening generalization in Semitic-language MT.",
439
+ "example": "Surfaced on cards as linguisticChallenges.rootPattern (e.g. Arabic, Hebrew, Maltese).",
440
+ "citations": [
441
+ {
442
+ "source": "SIL Glossary of Linguistic Terms",
443
+ "url": "https://glossary.sil.org/term/root-1"
444
+ }
445
+ ],
446
+ "related": [
447
+ "root",
448
+ "ablaut",
449
+ "tokenization"
450
+ ]
451
+ },
452
+ {
453
+ "term": "vowel harmony",
454
+ "also": [
455
+ "vowel-harmony"
456
+ ],
457
+ "plain": "A rule that all vowels in a word must agree in some property, such as front/back or rounded/unrounded. In Turkish, suffix vowels change shape to match the stem: ev-ler 'houses' but at-lar 'horses'.",
458
+ "mt_relevance": "Each suffix has several surface forms, multiplying the token variants MT must learn for one grammatical ending.",
459
+ "citations": [
460
+ {
461
+ "source": "SIL Glossary of Linguistic Terms",
462
+ "url": "https://glossary.sil.org/term/vowel-harmony"
463
+ }
464
+ ],
465
+ "related": [
466
+ "advanced tongue root",
467
+ "suffix",
468
+ "agglutinative"
469
+ ]
470
+ },
471
+ {
472
+ "term": "consonant mutation",
473
+ "also": [
474
+ "initial mutation",
475
+ "consonant mutations"
476
+ ],
477
+ "plain": "A grammatical change to a word's first consonant triggered by the word before it. Welsh cath 'cat' becomes gath after the article: y gath 'the cat'. The dictionary form and the spoken form can look quite different.",
478
+ "mt_relevance": "Mutation makes the same word appear under several spellings, fragmenting its statistics in MT training data.",
479
+ "citations": [
480
+ {
481
+ "source": "SIL Glossary of Linguistic Terms",
482
+ "url": "https://glossary.sil.org/term/consonant-mutation"
483
+ }
484
+ ],
485
+ "related": [
486
+ "lenition",
487
+ "sandhi"
488
+ ]
489
+ },
490
+ {
491
+ "term": "lenition",
492
+ "also": [
493
+ "lenited"
494
+ ],
495
+ "plain": "The softening of a consonant, often between vowels or under grammatical triggers — a 'k' weakening toward 'g' or 'h'. In Celtic languages lenition is part of the grammar, not just pronunciation.",
496
+ "mt_relevance": "Grammatical lenition changes word spellings in context, so surface text diverges from dictionary forms.",
497
+ "citations": [
498
+ {
499
+ "source": "SIL Glossary of Linguistic Terms",
500
+ "url": "https://glossary.sil.org/term/lenition"
501
+ }
502
+ ],
503
+ "related": [
504
+ "consonant mutation",
505
+ "sandhi"
506
+ ]
507
+ },
508
+ {
509
+ "term": "gemination",
510
+ "also": [
511
+ "geminate",
512
+ "geminated",
513
+ "double consonant"
514
+ ],
515
+ "plain": "Holding a consonant longer to make a different word — Italian pala 'shovel' vs palla 'ball'. Some scripts write the doubling, others leave it to the reader.",
516
+ "mt_relevance": "When the script omits gemination, distinct words collapse into one spelling and MT loses the contrast.",
517
+ "citations": [
518
+ {
519
+ "source": "SIL Glossary of Linguistic Terms",
520
+ "url": "https://glossary.sil.org/term/gemination"
521
+ }
522
+ ],
523
+ "related": [
524
+ "vowel length",
525
+ "orthography"
526
+ ]
527
+ },
528
+ {
529
+ "term": "sandhi",
530
+ "also": [
531
+ "tone sandhi",
532
+ "external sandhi"
533
+ ],
534
+ "plain": "Sound changes that happen where words or morphemes meet, like 'don't you' becoming 'dontcha'. In tone languages, tones themselves can change in context (tone sandhi).",
535
+ "mt_relevance": "Sandhi makes written or transcribed forms context-dependent, complicating consistent tokenization.",
536
+ "citations": [
537
+ {
538
+ "source": "SIL Glossary of Linguistic Terms",
539
+ "url": "https://glossary.sil.org/term/sandhi"
540
+ }
541
+ ],
542
+ "related": [
543
+ "tone",
544
+ "lenition",
545
+ "consonant mutation"
546
+ ]
547
+ },
548
+ {
549
+ "term": "umlaut",
550
+ "also": [
551
+ "i-mutation"
552
+ ],
553
+ "plain": "A vowel change caused historically by a following vowel, surviving as grammar: German Apfel 'apple' → Äpfel 'apples'. Related to ablaut but with a different historical origin.",
554
+ "mt_relevance": "Umlaut puts plural or tense marking inside the stem, invisible to affix-based segmentation.",
555
+ "citations": [
556
+ {
557
+ "source": "SIL Glossary of Linguistic Terms",
558
+ "url": "https://glossary.sil.org/term/umlaut"
559
+ }
560
+ ],
561
+ "related": [
562
+ "ablaut",
563
+ "inflection"
564
+ ]
565
+ },
566
+ {
567
+ "term": "tokenization",
568
+ "also": [
569
+ "tokenizer",
570
+ "subword",
571
+ "subword segmentation",
572
+ "tokenize",
573
+ "tokenization and alignment"
574
+ ],
575
+ "plain": "Splitting text into the units (tokens) a translation model actually processes. Modern systems split rare words into subword pieces; how well those pieces line up with real morphemes varies hugely by language.",
576
+ "mt_relevance": "Bad tokenization is a root cause of MT failure for morphologically rich and low-resource languages.",
577
+ "citations": [
578
+ {
579
+ "source": "Wikipedia: Byte pair encoding (standard reference)",
580
+ "url": "https://en.wikipedia.org/wiki/Byte_pair_encoding"
581
+ }
582
+ ],
583
+ "related": [
584
+ "morpheme",
585
+ "compounding",
586
+ "agglutinative"
587
+ ]
588
+ },
589
+ {
590
+ "term": "grammatical case",
591
+ "also": [
592
+ "case",
593
+ "case system",
594
+ "case marking",
595
+ "cases",
596
+ "case morphology",
597
+ "borderline case-marking"
598
+ ],
599
+ "plain": "Marking nouns and pronouns to show their role in the sentence — who acts, who is acted on, where, with what. English keeps only traces (I/me/my); Finnish has fifteen cases; many languages have none.",
600
+ "mt_relevance": "Case-marking languages allow free word order, so MT must read roles from endings rather than position — and generate the right endings in return.",
601
+ "citations": [
602
+ {
603
+ "source": "WALS Online, chapter 49 (Iggesen)",
604
+ "url": "https://wals.info/chapter/49"
605
+ }
606
+ ],
607
+ "related": [
608
+ "nominative",
609
+ "ergativity",
610
+ "oblique",
611
+ "morphosyntactic alignment"
612
+ ]
613
+ },
614
+ {
615
+ "term": "nominative",
616
+ "also": [
617
+ "accusative",
618
+ "nominative-accusative",
619
+ "nominative–accusative"
620
+ ],
621
+ "plain": "In the most familiar alignment system, the subject of any verb takes nominative case and the direct object takes accusative. Most European languages work this way, so it is what MT training data overwhelmingly reflects.",
622
+ "mt_relevance": "Systems trained mostly on nominative–accusative languages misassign roles when translating ergative languages.",
623
+ "citations": [
624
+ {
625
+ "source": "WALS Online, chapter 98 (Comrie)",
626
+ "url": "https://wals.info/chapter/98"
627
+ }
628
+ ],
629
+ "related": [
630
+ "ergativity",
631
+ "grammatical case",
632
+ "morphosyntactic alignment"
633
+ ]
634
+ },
635
+ {
636
+ "term": "ergativity",
637
+ "also": [
638
+ "ergative",
639
+ "ergative-absolutive",
640
+ "ergative–absolutive",
641
+ "ergative alignment",
642
+ "ergative case",
643
+ "split ergativity"
644
+ ],
645
+ "plain": "An alignment system where the subject of an intransitive verb ('she sleeps') is marked like the object of a transitive one ('saw her'), while transitive subjects get special ergative marking. Basque, many Mayan, Australian and Caucasian languages work this way.",
646
+ "mt_relevance": "Ergative marking reverses the role cues MT expects, causing who-did-what-to-whom errors.",
647
+ "citations": [
648
+ {
649
+ "source": "WALS Online, chapter 98 (Comrie)",
650
+ "url": "https://wals.info/chapter/98"
651
+ }
652
+ ],
653
+ "related": [
654
+ "absolutive",
655
+ "nominative",
656
+ "morphosyntactic alignment"
657
+ ]
658
+ },
659
+ {
660
+ "term": "absolutive",
661
+ "also": [
662
+ "absolutive case"
663
+ ],
664
+ "plain": "The unmarked case in an ergative system, covering intransitive subjects and transitive objects. It is usually the citation form of the noun.",
665
+ "mt_relevance": "Absolutive nouns look identical in two different roles, so MT must use the verb to disambiguate.",
666
+ "citations": [
667
+ {
668
+ "source": "SIL Glossary of Linguistic Terms",
669
+ "url": "https://glossary.sil.org/term/absolutive-case"
670
+ }
671
+ ],
672
+ "related": [
673
+ "ergativity",
674
+ "grammatical case"
675
+ ]
676
+ },
677
+ {
678
+ "term": "morphosyntactic alignment",
679
+ "also": [
680
+ "alignment",
681
+ "alignment of verbal person marking"
682
+ ],
683
+ "plain": "The system a language uses to group the three core roles — intransitive subject, transitive subject, and object — for marking purposes. Nominative–accusative and ergative–absolutive are the two big patterns; some languages split between them or use animacy-driven systems.",
684
+ "mt_relevance": "Alignment mismatch between source and target is a structural translation problem, not a vocabulary one.",
685
+ "citations": [
686
+ {
687
+ "source": "WALS Online, chapter 100 (Siewierska)",
688
+ "url": "https://wals.info/chapter/100"
689
+ }
690
+ ],
691
+ "related": [
692
+ "ergativity",
693
+ "nominative",
694
+ "agreement"
695
+ ]
696
+ },
697
+ {
698
+ "term": "genitive",
699
+ "also": [
700
+ "genitive case",
701
+ "possessive case"
702
+ ],
703
+ "plain": "The case that marks possession or close association, like English 's or 'of'. Languages differ in whether the genitive phrase comes before or after the noun it modifies.",
704
+ "mt_relevance": "Genitive order differences ('the king's horse' vs 'horse of-king') require systematic reordering inside noun phrases.",
705
+ "citations": [
706
+ {
707
+ "source": "WALS Online, chapter 86 (Dryer)",
708
+ "url": "https://wals.info/chapter/86"
709
+ }
710
+ ],
711
+ "related": [
712
+ "grammatical case",
713
+ "possession",
714
+ "word order"
715
+ ]
716
+ },
717
+ {
718
+ "term": "dative",
719
+ "also": [
720
+ "dative case"
721
+ ],
722
+ "plain": "The case marking the recipient or beneficiary — the 'to whom' of giving and telling. German wem, Latin cui; English expresses it with word order or 'to'.",
723
+ "mt_relevance": "MT must decide between dative case marking and prepositional phrasing depending on the target language.",
724
+ "citations": [
725
+ {
726
+ "source": "SIL Glossary of Linguistic Terms",
727
+ "url": "https://glossary.sil.org/term/dative-case"
728
+ }
729
+ ],
730
+ "related": [
731
+ "grammatical case",
732
+ "benefactive"
733
+ ]
734
+ },
735
+ {
736
+ "term": "locative",
737
+ "also": [
738
+ "locative case"
739
+ ],
740
+ "plain": "A case meaning 'at/in/on' a place, expressed by a noun ending rather than a preposition. Finnish and Hungarian split location into several precise locative cases (inside, on top, near, motion toward, motion from).",
741
+ "mt_relevance": "One English preposition can map to several locative cases, forcing MT to choose by context.",
742
+ "citations": [
743
+ {
744
+ "source": "SIL Glossary of Linguistic Terms",
745
+ "url": "https://glossary.sil.org/term/locative-case"
746
+ }
747
+ ],
748
+ "related": [
749
+ "grammatical case",
750
+ "adposition"
751
+ ]
752
+ },
753
+ {
754
+ "term": "instrumental",
755
+ "also": [
756
+ "instrumental case",
757
+ "instrumentals"
758
+ ],
759
+ "plain": "A case meaning 'using/by means of' — Russian marks 'with a hammer' with an ending instead of a preposition. Polysynthetic languages may build the instrument right into the verb.",
760
+ "mt_relevance": "Instrumental meaning shifts between case endings, prepositions, and verb-internal marking across languages.",
761
+ "citations": [
762
+ {
763
+ "source": "SIL Glossary of Linguistic Terms",
764
+ "url": "https://glossary.sil.org/term/instrumental-case"
765
+ }
766
+ ],
767
+ "related": [
768
+ "grammatical case",
769
+ "polysynthesis"
770
+ ]
771
+ },
772
+ {
773
+ "term": "comitative",
774
+ "also": [
775
+ "comitative case"
776
+ ],
777
+ "plain": "A case or marker meaning 'together with someone'. Some languages use the same marker for 'with' (accompaniment) and 'and' (coordination), which can be ambiguous to outsiders.",
778
+ "mt_relevance": "Comitative/coordination overlap means 'X with Y' and 'X and Y' can be the same construction in the source.",
779
+ "citations": [
780
+ {
781
+ "source": "WALS Online, chapter 63 (Stassen)",
782
+ "url": "https://wals.info/chapter/63"
783
+ }
784
+ ],
785
+ "related": [
786
+ "grammatical case",
787
+ "instrumental"
788
+ ]
789
+ },
790
+ {
791
+ "term": "oblique",
792
+ "also": [
793
+ "obliques",
794
+ "oblique argument",
795
+ "oblique phrase"
796
+ ],
797
+ "plain": "Any phrase in the clause that is neither subject nor direct object — typically locations, instruments, recipients and other 'extras', often marked with a case or adposition. In WALS, 'X' in orders like VOX stands for the oblique phrase.",
798
+ "mt_relevance": "Where obliques sit in the sentence varies by language and must be reordered correctly around verb and object.",
799
+ "citations": [
800
+ {
801
+ "source": "WALS Online, chapter 84 (Dryer & Gensler)",
802
+ "url": "https://wals.info/chapter/84"
803
+ }
804
+ ],
805
+ "related": [
806
+ "grammatical case",
807
+ "word order",
808
+ "adposition"
809
+ ]
810
+ },
811
+ {
812
+ "term": "differential object marking",
813
+ "also": [
814
+ "DOM",
815
+ "differential marking"
816
+ ],
817
+ "plain": "Marking some direct objects but not others, usually depending on animacy or definiteness. Spanish adds 'a' before human objects (veo a María) but not things (veo la casa).",
818
+ "mt_relevance": "MT must decide per-object whether the marker belongs, using animacy cues the source may not show.",
819
+ "citations": [
820
+ {
821
+ "source": "SIL Glossary of Linguistic Terms",
822
+ "url": "https://glossary.sil.org/term/differential-object-marking"
823
+ }
824
+ ],
825
+ "related": [
826
+ "animacy",
827
+ "definiteness",
828
+ "grammatical case"
829
+ ]
830
+ },
831
+ {
832
+ "term": "grammatical gender",
833
+ "also": [
834
+ "gender",
835
+ "gender system",
836
+ "gender agreement",
837
+ "noun gender",
838
+ "gendered"
839
+ ],
840
+ "plain": "Sorting all nouns into classes (often called masculine/feminine/neuter) that force matching forms on articles, adjectives, and sometimes verbs. The assignment is grammatical, not biological — a German table is masculine, a Spanish one feminine.",
841
+ "mt_relevance": "Translating into a gendered language forces choices about people and things the source leaves unspecified, a major source of MT bias.",
842
+ "citations": [
843
+ {
844
+ "source": "WALS Online, chapter 30 (Corbett)",
845
+ "url": "https://wals.info/chapter/30"
846
+ }
847
+ ],
848
+ "related": [
849
+ "noun class",
850
+ "agreement",
851
+ "animacy"
852
+ ]
853
+ },
854
+ {
855
+ "term": "noun class",
856
+ "also": [
857
+ "noun classes",
858
+ "noun class agreement",
859
+ "noun-class"
860
+ ],
861
+ "plain": "A gender-like system with many classes — Bantu languages typically have 10–20, sorting nouns by shape, animacy, size and more. Each class triggers its own agreement prefixes across the sentence.",
862
+ "mt_relevance": "Every noun choice ripples agreement markers through the whole clause, so one wrong class produces many visible errors.",
863
+ "citations": [
864
+ {
865
+ "source": "WALS Online, chapter 30 (Corbett)",
866
+ "url": "https://wals.info/chapter/30"
867
+ }
868
+ ],
869
+ "related": [
870
+ "grammatical gender",
871
+ "agreement",
872
+ "classifier"
873
+ ]
874
+ },
875
+ {
876
+ "term": "animacy",
877
+ "also": [
878
+ "animate",
879
+ "inanimate",
880
+ "animacy hierarchy"
881
+ ],
882
+ "plain": "A grammatical distinction between living (or living-like) and non-living referents. In Algonquian languages every noun is grammatically animate or inanimate, and verbs change shape entirely depending on which they combine with.",
883
+ "mt_relevance": "Animacy drives verb choice and agreement in many Indigenous languages, with no overt cue in an English source.",
884
+ "example": "Plains Cree (crk): verb conjugation changes completely based on whether subject/object nouns are animate or inanimate (card field linguisticChallenges.animacy).",
885
+ "citations": [
886
+ {
887
+ "source": "SIL Glossary of Linguistic Terms",
888
+ "url": "https://glossary.sil.org/term/animacy"
889
+ }
890
+ ],
891
+ "related": [
892
+ "grammatical gender",
893
+ "obviation",
894
+ "differential object marking"
895
+ ]
896
+ },
897
+ {
898
+ "term": "obviation",
899
+ "also": [
900
+ "obviative",
901
+ "fourth person",
902
+ "proximate"
903
+ ],
904
+ "plain": "A system, central to Algonquian languages like Plains Cree, that ranks third persons in a stretch of discourse: one is 'proximate' (in focus) and any others are 'obviative' (marked as backgrounded, sometimes called fourth person). It tracks who is who without pronouns like 'he₁ vs he₂'.",
905
+ "mt_relevance": "English has no obviation, so MT into Cree must invent proximate/obviative assignments and keep them consistent across sentences.",
906
+ "example": "Plains Cree (crk) marks obviative referents on nouns and verbs; see the crk card and the crk method's documentation.",
907
+ "citations": [
908
+ {
909
+ "source": "SIL Glossary of Linguistic Terms",
910
+ "url": "https://glossary.sil.org/term/obviative-person"
911
+ }
912
+ ],
913
+ "related": [
914
+ "animacy",
915
+ "agreement",
916
+ "conjunct order"
917
+ ]
918
+ },
919
+ {
920
+ "term": "clusivity",
921
+ "also": [
922
+ "inclusive",
923
+ "exclusive",
924
+ "inclusive/exclusive",
925
+ "inclusive-exclusive",
926
+ "inclusiveExclusive"
927
+ ],
928
+ "plain": "A distinction between two kinds of 'we': inclusive (me + you, maybe others) and exclusive (me + others, but not you). Hundreds of languages make this distinction obligatorily.",
929
+ "mt_relevance": "Translating English 'we' requires choosing inclusive or exclusive with no source-side cue — and the wrong choice can be socially serious.",
930
+ "citations": [
931
+ {
932
+ "source": "WALS Online, chapter 39 (Cysouw)",
933
+ "url": "https://wals.info/chapter/39"
934
+ }
935
+ ],
936
+ "related": [
937
+ "grammatical number",
938
+ "T-V distinction"
939
+ ]
940
+ },
941
+ {
942
+ "term": "grammatical number",
943
+ "also": [
944
+ "plural marking",
945
+ "plurality",
946
+ "plural",
947
+ "number marking",
948
+ "obligatory plural"
949
+ ],
950
+ "plain": "How a language marks how many — singular, plural, and sometimes dual (exactly two) or paucal (a few). Some languages mark number on every noun; others leave it to context entirely.",
951
+ "mt_relevance": "When the source does not mark number, MT must guess it; when the target requires dual forms, MT must supply them.",
952
+ "citations": [
953
+ {
954
+ "source": "WALS Online, chapter 33 (Dryer)",
955
+ "url": "https://wals.info/chapter/33"
956
+ }
957
+ ],
958
+ "related": [
959
+ "dual",
960
+ "clusivity",
961
+ "agreement"
962
+ ]
963
+ },
964
+ {
965
+ "term": "dual",
966
+ "also": [
967
+ "dual number",
968
+ "trial"
969
+ ],
970
+ "plain": "A grammatical number meaning exactly two, distinct from both singular and plural. Slovene, Arabic and many Oceanic languages have it; a few languages add trial (exactly three) in pronouns.",
971
+ "mt_relevance": "MT into dual-marking languages must detect two-ness that the source expresses only by counting words.",
972
+ "citations": [
973
+ {
974
+ "source": "SIL Glossary of Linguistic Terms",
975
+ "url": "https://glossary.sil.org/term/dual-number"
976
+ }
977
+ ],
978
+ "related": [
979
+ "grammatical number",
980
+ "clusivity"
981
+ ]
982
+ },
983
+ {
984
+ "term": "classifier",
985
+ "also": [
986
+ "classifiers",
987
+ "numeral classifier",
988
+ "numeral classifiers",
989
+ "measure word",
990
+ "classifier system",
991
+ "counter system",
992
+ "counters",
993
+ "noun classifiers"
994
+ ],
995
+ "plain": "A word required when counting or referring to nouns, sorting them by type — like English 'three head of cattle' or 'two sheets of paper', but obligatory for everything. Chinese, Japanese, Thai and many other languages cannot count without the right classifier.",
996
+ "mt_relevance": "MT must select the correct classifier for each noun; the wrong one is instantly visible to native speakers.",
997
+ "citations": [
998
+ {
999
+ "source": "WALS Online, chapter 55 (Gil)",
1000
+ "url": "https://wals.info/chapter/55"
1001
+ }
1002
+ ],
1003
+ "related": [
1004
+ "noun class",
1005
+ "grammatical number"
1006
+ ]
1007
+ },
1008
+ {
1009
+ "term": "definiteness",
1010
+ "also": [
1011
+ "definite",
1012
+ "indefinite",
1013
+ "specific articles"
1014
+ ],
1015
+ "plain": "Whether a noun refers to something the listener can already identify ('the dog') or not ('a dog'). Many languages never mark this distinction; others mark it with articles, affixes, or word order.",
1016
+ "mt_relevance": "Translating from an article-less language forces MT to decide 'the' vs 'a' for every noun phrase from context alone.",
1017
+ "citations": [
1018
+ {
1019
+ "source": "WALS Online, chapter 37 (Dryer)",
1020
+ "url": "https://wals.info/chapter/37"
1021
+ }
1022
+ ],
1023
+ "related": [
1024
+ "article",
1025
+ "differential object marking"
1026
+ ]
1027
+ },
1028
+ {
1029
+ "term": "article",
1030
+ "also": [
1031
+ "articles",
1032
+ "definite article",
1033
+ "indefinite article",
1034
+ "article system"
1035
+ ],
1036
+ "plain": "A small word that marks a noun as definite ('the') or indefinite ('a'). About a third of the world's languages have no articles at all, and some use the numeral 'one' or demonstratives instead.",
1037
+ "mt_relevance": "Article insertion and deletion is one of the most frequent edit categories in MT between article and article-less languages.",
1038
+ "citations": [
1039
+ {
1040
+ "source": "WALS Online, chapter 38 (Dryer)",
1041
+ "url": "https://wals.info/chapter/38"
1042
+ }
1043
+ ],
1044
+ "related": [
1045
+ "definiteness",
1046
+ "demonstrative"
1047
+ ]
1048
+ },
1049
+ {
1050
+ "term": "demonstrative",
1051
+ "also": [
1052
+ "demonstratives",
1053
+ "deixis"
1054
+ ],
1055
+ "plain": "Pointing words like 'this' and 'that'. Languages divide pointing space differently — some have a three-way near-me / near-you / far split, others add visibility or elevation.",
1056
+ "mt_relevance": "Demonstrative systems rarely map one-to-one, so MT must approximate distance and visibility distinctions.",
1057
+ "citations": [
1058
+ {
1059
+ "source": "WALS Online, chapter 41 (Diessel)",
1060
+ "url": "https://wals.info/chapter/41"
1061
+ }
1062
+ ],
1063
+ "related": [
1064
+ "article",
1065
+ "definiteness"
1066
+ ]
1067
+ },
1068
+ {
1069
+ "term": "possession",
1070
+ "also": [
1071
+ "possessive",
1072
+ "possessives",
1073
+ "alienable",
1074
+ "inalienable",
1075
+ "possessive affixes"
1076
+ ],
1077
+ "plain": "How a language expresses 'my X / your X'. Many languages distinguish inalienable possession (body parts, kin — things you cannot give away) from alienable, marking them with different constructions.",
1078
+ "mt_relevance": "Inalienable possession often requires obligatory possessor marking that English sources omit.",
1079
+ "citations": [
1080
+ {
1081
+ "source": "WALS Online, chapter 58 (Nichols & Bickel)",
1082
+ "url": "https://wals.info/chapter/58"
1083
+ }
1084
+ ],
1085
+ "related": [
1086
+ "genitive",
1087
+ "head marking"
1088
+ ]
1089
+ },
1090
+ {
1091
+ "term": "head marking",
1092
+ "also": [
1093
+ "head-marking",
1094
+ "headMarking"
1095
+ ],
1096
+ "plain": "Putting the grammatical marking on the head of a phrase — the verb shows who its subject and object are, the possessed noun shows who owns it. The dependents (the nouns themselves) can stay bare.",
1097
+ "mt_relevance": "Head-marking concentrates clause grammar on the verb, the mirror image of the dependent-marking languages most MT data comes from.",
1098
+ "citations": [
1099
+ {
1100
+ "source": "WALS Online, chapter 23 (Nichols & Bickel)",
1101
+ "url": "https://wals.info/chapter/23"
1102
+ }
1103
+ ],
1104
+ "related": [
1105
+ "dependent marking",
1106
+ "agreement",
1107
+ "polysynthesis"
1108
+ ]
1109
+ },
1110
+ {
1111
+ "term": "dependent marking",
1112
+ "also": [
1113
+ "dependent-marking",
1114
+ "dependentMarking"
1115
+ ],
1116
+ "plain": "Putting grammatical marking on the dependents of a phrase — case endings on nouns, possessive endings on the possessor — while the head stays unmarked. Most European languages lean this way.",
1117
+ "mt_relevance": "Dependent marking puts the grammatical signal in noun endings, which tokenizers must segment correctly.",
1118
+ "citations": [
1119
+ {
1120
+ "source": "WALS Online, chapter 23 (Nichols & Bickel)",
1121
+ "url": "https://wals.info/chapter/23"
1122
+ }
1123
+ ],
1124
+ "related": [
1125
+ "head marking",
1126
+ "grammatical case"
1127
+ ]
1128
+ },
1129
+ {
1130
+ "term": "evidentiality",
1131
+ "also": [
1132
+ "evidential",
1133
+ "evidentials",
1134
+ "evidential marking"
1135
+ ],
1136
+ "plain": "Grammar that forces speakers to say how they know something — saw it, heard about it, inferred it. In Quechua or Turkish, leaving out the evidential is like leaving out tense in English.",
1137
+ "mt_relevance": "MT into evidential languages must state an information source the original text never specifies.",
1138
+ "citations": [
1139
+ {
1140
+ "source": "WALS Online, chapter 77 (de Haan)",
1141
+ "url": "https://wals.info/chapter/77"
1142
+ }
1143
+ ],
1144
+ "related": [
1145
+ "mood",
1146
+ "tense"
1147
+ ]
1148
+ },
1149
+ {
1150
+ "term": "tense",
1151
+ "also": [
1152
+ "tenses",
1153
+ "past tense",
1154
+ "future tense",
1155
+ "tense-aspect"
1156
+ ],
1157
+ "plain": "Grammatical marking that locates an event in time — past, present, future. Some languages have no grammatical tense at all (Mandarin), while others distinguish several degrees of past remoteness.",
1158
+ "mt_relevance": "Tenseless source text forces MT to infer time reference; remoteness systems demand finer distinctions than the source provides.",
1159
+ "citations": [
1160
+ {
1161
+ "source": "WALS Online, chapter 66 (Dahl & Velupillai)",
1162
+ "url": "https://wals.info/chapter/66"
1163
+ }
1164
+ ],
1165
+ "related": [
1166
+ "aspect",
1167
+ "mood",
1168
+ "evidentiality"
1169
+ ]
1170
+ },
1171
+ {
1172
+ "term": "aspect",
1173
+ "also": [
1174
+ "aspectual",
1175
+ "aspect marking"
1176
+ ],
1177
+ "plain": "Grammatical marking for how an event unfolds in time — completed, ongoing, habitual, repeated — independent of when it happened. Russian verbs come in perfective/imperfective pairs; English uses 'was doing' vs 'did'.",
1178
+ "mt_relevance": "Aspect choices must be made on every verb when translating into aspect-marking languages, often without source cues.",
1179
+ "citations": [
1180
+ {
1181
+ "source": "WALS Online, chapter 65 (Dahl & Velupillai)",
1182
+ "url": "https://wals.info/chapter/65"
1183
+ }
1184
+ ],
1185
+ "related": [
1186
+ "perfective",
1187
+ "imperfective",
1188
+ "tense"
1189
+ ]
1190
+ },
1191
+ {
1192
+ "term": "perfective",
1193
+ "also": [
1194
+ "perfective aspect"
1195
+ ],
1196
+ "plain": "Aspect presenting an event as a complete whole — 'she wrote the letter' viewed as one finished fact. Often paired with imperfective in a grammatical opposition.",
1197
+ "mt_relevance": "Choosing perfective vs imperfective wrongly is among the most common MT errors into Slavic languages.",
1198
+ "citations": [
1199
+ {
1200
+ "source": "SIL Glossary of Linguistic Terms",
1201
+ "url": "https://glossary.sil.org/term/perfective-aspect"
1202
+ }
1203
+ ],
1204
+ "related": [
1205
+ "imperfective",
1206
+ "aspect"
1207
+ ]
1208
+ },
1209
+ {
1210
+ "term": "imperfective",
1211
+ "also": [
1212
+ "imperfective aspect"
1213
+ ],
1214
+ "plain": "Aspect presenting an event from the inside — ongoing, habitual, or repeated, like 'she was writing' or 'she used to write'. The counterpart of perfective.",
1215
+ "mt_relevance": "Imperfective readings (ongoing vs habitual) must be disambiguated by context for correct translation.",
1216
+ "citations": [
1217
+ {
1218
+ "source": "SIL Glossary of Linguistic Terms",
1219
+ "url": "https://glossary.sil.org/term/imperfective-aspect"
1220
+ }
1221
+ ],
1222
+ "related": [
1223
+ "perfective",
1224
+ "aspect"
1225
+ ]
1226
+ },
1227
+ {
1228
+ "term": "mood",
1229
+ "also": [
1230
+ "modality",
1231
+ "grammatical mood"
1232
+ ],
1233
+ "plain": "Grammatical marking for the speaker's stance toward an event — fact, wish, command, possibility. Indicative, subjunctive and imperative are the familiar European moods; other languages mark finer shades.",
1234
+ "mt_relevance": "Mood selection (especially subjunctive) follows target-language rules that cannot be copied from the source.",
1235
+ "citations": [
1236
+ {
1237
+ "source": "SIL Glossary of Linguistic Terms",
1238
+ "url": "https://glossary.sil.org/term/mood-and-modality"
1239
+ }
1240
+ ],
1241
+ "related": [
1242
+ "subjunctive",
1243
+ "evidentiality",
1244
+ "tense"
1245
+ ]
1246
+ },
1247
+ {
1248
+ "term": "subjunctive",
1249
+ "also": [
1250
+ "subjunctive mood"
1251
+ ],
1252
+ "plain": "A verb mood for non-asserted content — wishes, doubts, hypotheticals, and clauses after certain verbs. Romance languages require it in many subordinate clauses where English uses plain forms.",
1253
+ "mt_relevance": "Subjunctive triggers are target-language-specific, so MT must apply grammar rules rather than translate forms.",
1254
+ "citations": [
1255
+ {
1256
+ "source": "SIL Glossary of Linguistic Terms",
1257
+ "url": "https://glossary.sil.org/term/subjunctive-mood"
1258
+ }
1259
+ ],
1260
+ "related": [
1261
+ "mood"
1262
+ ]
1263
+ },
1264
+ {
1265
+ "term": "negation",
1266
+ "also": [
1267
+ "negative morpheme",
1268
+ "negator",
1269
+ "negative marker",
1270
+ "standard negation"
1271
+ ],
1272
+ "plain": "How a language says 'not'. Strategies include particles (English not), affixes on the verb, special negative verbs, and two-part constructions like French ne…pas. Position varies: before the verb, after it, or both.",
1273
+ "mt_relevance": "Negation errors invert meaning entirely, making correct negative placement one of the highest-stakes MT requirements.",
1274
+ "citations": [
1275
+ {
1276
+ "source": "WALS Online, chapter 112 (Dryer)",
1277
+ "url": "https://wals.info/chapter/112"
1278
+ }
1279
+ ],
1280
+ "related": [
1281
+ "particle",
1282
+ "word order"
1283
+ ]
1284
+ },
1285
+ {
1286
+ "term": "passive",
1287
+ "also": [
1288
+ "passive voice",
1289
+ "passive constructions"
1290
+ ],
1291
+ "plain": "A construction that promotes the object to subject and demotes or drops the doer: 'the window was broken (by the boy)'. Many languages lack a passive entirely or use other strategies to background the agent.",
1292
+ "mt_relevance": "Passive-less target languages force MT to restructure passives into actives, inventing or recovering the agent.",
1293
+ "citations": [
1294
+ {
1295
+ "source": "WALS Online, chapter 107 (Siewierska)",
1296
+ "url": "https://wals.info/chapter/107"
1297
+ }
1298
+ ],
1299
+ "related": [
1300
+ "grammatical voice",
1301
+ "valency"
1302
+ ]
1303
+ },
1304
+ {
1305
+ "term": "grammatical voice",
1306
+ "also": [
1307
+ "voice",
1308
+ "voice system"
1309
+ ],
1310
+ "plain": "The grammatical system controlling which participant is the subject — active, passive, middle, and in some language families much richer systems. Voice reshapes the whole clause around a chosen perspective.",
1311
+ "mt_relevance": "Voice mismatches require restructuring whole clauses, not substituting words.",
1312
+ "citations": [
1313
+ {
1314
+ "source": "SIL Glossary of Linguistic Terms",
1315
+ "url": "https://glossary.sil.org/term/voice-2"
1316
+ }
1317
+ ],
1318
+ "related": [
1319
+ "passive",
1320
+ "focus system",
1321
+ "valency"
1322
+ ]
1323
+ },
1324
+ {
1325
+ "term": "valency",
1326
+ "also": [
1327
+ "valence",
1328
+ "valency patterns",
1329
+ "valencyPatterns"
1330
+ ],
1331
+ "plain": "How many participants a verb requires and how it encodes them — 'sleep' takes one, 'give' takes three. Languages disagree about which participants particular verbs take and how they are marked.",
1332
+ "mt_relevance": "Valency mismatches make literal translations assign the wrong roles or drop required participants.",
1333
+ "citations": [
1334
+ {
1335
+ "source": "ValPaL (Valency Patterns Leipzig)",
1336
+ "url": "https://valpal.info"
1337
+ }
1338
+ ],
1339
+ "related": [
1340
+ "grammatical voice",
1341
+ "agreement",
1342
+ "benefactive"
1343
+ ]
1344
+ },
1345
+ {
1346
+ "term": "transitive",
1347
+ "also": [
1348
+ "transitivity",
1349
+ "transitive verb",
1350
+ "transitive verbs"
1351
+ ],
1352
+ "plain": "Describes a verb that takes a direct object — 'she reads the book'. Transitivity is grammatically central: many languages mark the subject, the object, and the verb itself differently depending on whether the clause is transitive.",
1353
+ "mt_relevance": "Transitivity governs case marking and agreement, so misjudging whether a verb is transitive scrambles who-did-what-to-whom in the output.",
1354
+ "citations": [
1355
+ {
1356
+ "source": "SIL Glossary of Linguistic Terms",
1357
+ "url": "https://glossary.sil.org/term/transitive-verb"
1358
+ }
1359
+ ],
1360
+ "related": [
1361
+ "intransitive",
1362
+ "valency",
1363
+ "ergativity"
1364
+ ]
1365
+ },
1366
+ {
1367
+ "term": "intransitive",
1368
+ "also": [
1369
+ "intransitivity",
1370
+ "intransitive verb",
1371
+ "intransitive verbs"
1372
+ ],
1373
+ "plain": "Describes a verb that takes no direct object — 'she sleeps', 'the sun rose'. In ergative languages the single argument of an intransitive verb is marked like the object of a transitive one, not like its subject.",
1374
+ "mt_relevance": "Intransitive clauses trigger different case and agreement patterns, especially in ergative languages, so MT must track transitivity to choose the right forms.",
1375
+ "citations": [
1376
+ {
1377
+ "source": "SIL Glossary of Linguistic Terms",
1378
+ "url": "https://glossary.sil.org/term/intransitive-verb"
1379
+ }
1380
+ ],
1381
+ "related": [
1382
+ "transitive",
1383
+ "valency",
1384
+ "ergativity"
1385
+ ]
1386
+ },
1387
+ {
1388
+ "term": "antipassive",
1389
+ "also": [
1390
+ "antipassives",
1391
+ "antipassive voice",
1392
+ "antipassive construction"
1393
+ ],
1394
+ "plain": "A construction, common in ergative languages, that removes or demotes the object of a transitive verb, leaving an intransitive clause — the mirror image of the passive. Where the passive backgrounds the doer, the antipassive backgrounds the thing acted upon.",
1395
+ "mt_relevance": "Antipassives restructure the clause and change case marking, so translating them into a passive-oriented language means rebuilding the whole sentence rather than swapping words.",
1396
+ "citations": [
1397
+ {
1398
+ "source": "SIL Glossary of Linguistic Terms",
1399
+ "url": "https://glossary.sil.org/term/antipassive-voice"
1400
+ }
1401
+ ],
1402
+ "related": [
1403
+ "passive",
1404
+ "ergativity",
1405
+ "valency"
1406
+ ]
1407
+ },
1408
+ {
1409
+ "term": "causative",
1410
+ "also": [
1411
+ "causatives",
1412
+ "causation",
1413
+ "causative marker",
1414
+ "causative construction"
1415
+ ],
1416
+ "plain": "A construction that adds a causer to an event — turning 'the pot broke' into 'she broke the pot' or 'she made it break'. Many languages build causatives with a dedicated verb affix rather than a separate verb like English 'make'.",
1417
+ "mt_relevance": "Causative meaning can hide inside a single verb form, so MT must recognise it and unpack it into a periphrastic construction — or fold one back in — for the target language.",
1418
+ "citations": [
1419
+ {
1420
+ "source": "WALS Online, chapter 111 (Song)",
1421
+ "url": "https://wals.info/chapter/111"
1422
+ }
1423
+ ],
1424
+ "related": [
1425
+ "valency",
1426
+ "affix",
1427
+ "benefactive"
1428
+ ]
1429
+ },
1430
+ {
1431
+ "term": "pronoun",
1432
+ "also": [
1433
+ "pronouns",
1434
+ "personal pronoun",
1435
+ "personal pronouns",
1436
+ "pronominal"
1437
+ ],
1438
+ "plain": "A word that stands in for a noun phrase — I, you, she, they, this, who. Languages differ sharply in how many pronouns they have and which distinctions they force: gender, formality, or the inclusive/exclusive 'we', among others.",
1439
+ "mt_relevance": "Pronoun systems rarely line up between languages, so MT must supply distinctions the source leaves unspecified (gender, formality) or collapse ones the target cannot express.",
1440
+ "citations": [
1441
+ {
1442
+ "source": "SIL Glossary of Linguistic Terms",
1443
+ "url": "https://glossary.sil.org/term/pronoun"
1444
+ }
1445
+ ],
1446
+ "related": [
1447
+ "clusivity",
1448
+ "grammatical gender",
1449
+ "T-V distinction"
1450
+ ]
1451
+ },
1452
+ {
1453
+ "term": "numeral",
1454
+ "also": [
1455
+ "numerals",
1456
+ "numeral system",
1457
+ "numeral systems",
1458
+ "number word",
1459
+ "number words"
1460
+ ],
1461
+ "plain": "A word expressing a number — one, seventeen, thousand. Languages build higher numerals on different bases (ten, twenty, or mixtures) and may require a classifier or a special counting form with each number.",
1462
+ "mt_relevance": "Numbers are high-stakes content, and translating number words across different counting bases or classifier requirements is an easy place for MT to introduce errors.",
1463
+ "citations": [
1464
+ {
1465
+ "source": "WALS Online, chapter 131 (Comrie)",
1466
+ "url": "https://wals.info/chapter/131"
1467
+ }
1468
+ ],
1469
+ "related": [
1470
+ "vigesimal",
1471
+ "classifier"
1472
+ ]
1473
+ },
1474
+ {
1475
+ "term": "ideophone",
1476
+ "also": [
1477
+ "ideophones",
1478
+ "ideophonic",
1479
+ "mimetic",
1480
+ "mimetics"
1481
+ ],
1482
+ "plain": "A vivid word that evokes a sensation — a sound, movement, texture, or light — through its own shape, like English 'zigzag' or Japanese kirakira 'glitteringly'. Many African and Asian languages have large, grammatically distinct classes of ideophones.",
1483
+ "mt_relevance": "Ideophones are expressive and highly language-specific, so MT tends to drop or flatten them, losing meaning that native speakers hear as precise and vivid.",
1484
+ "citations": [
1485
+ {
1486
+ "source": "SIL Glossary of Linguistic Terms",
1487
+ "url": "https://glossary.sil.org/term/ideophone"
1488
+ }
1489
+ ],
1490
+ "related": [
1491
+ "reduplication",
1492
+ "particle"
1493
+ ]
1494
+ },
1495
+ {
1496
+ "term": "serial verb construction",
1497
+ "also": [
1498
+ "serial verbs",
1499
+ "serialVerbs",
1500
+ "verb serialization",
1501
+ "serial verb"
1502
+ ],
1503
+ "plain": "Stringing several verbs together in one clause with no 'and' or 'to' between them — 'take knife cut bread' for 'cut the bread with a knife'. Common in West African, Southeast Asian and creole languages.",
1504
+ "mt_relevance": "Serial verbs must be decomposed into prepositions or subordinate clauses when translating into European languages, and rebuilt going the other way.",
1505
+ "citations": [
1506
+ {
1507
+ "source": "SIL Glossary of Linguistic Terms",
1508
+ "url": "https://glossary.sil.org/term/serial-verb"
1509
+ }
1510
+ ],
1511
+ "related": [
1512
+ "grammatical voice",
1513
+ "particle"
1514
+ ]
1515
+ },
1516
+ {
1517
+ "term": "participle",
1518
+ "also": [
1519
+ "participles",
1520
+ "participial"
1521
+ ],
1522
+ "plain": "A verb form that acts like an adjective or builds compound tenses — 'the running water', 'has eaten'. Languages differ in how many participles they have and what they are used for.",
1523
+ "mt_relevance": "Participial clauses often replace relative clauses in other languages, requiring structural conversion.",
1524
+ "citations": [
1525
+ {
1526
+ "source": "SIL Glossary of Linguistic Terms",
1527
+ "url": "https://glossary.sil.org/term/participle"
1528
+ }
1529
+ ],
1530
+ "related": [
1531
+ "gerund",
1532
+ "relative clause"
1533
+ ]
1534
+ },
1535
+ {
1536
+ "term": "gerund",
1537
+ "also": [
1538
+ "gerunds"
1539
+ ],
1540
+ "plain": "A verb form used as a noun, like 'swimming' in 'swimming is fun'. Other languages use infinitives, verbal nouns, or special converb forms where English uses gerunds.",
1541
+ "mt_relevance": "English gerunds map to several different constructions depending on the target language.",
1542
+ "citations": [
1543
+ {
1544
+ "source": "SIL Glossary of Linguistic Terms",
1545
+ "url": "https://glossary.sil.org/term/gerund"
1546
+ }
1547
+ ],
1548
+ "related": [
1549
+ "participle"
1550
+ ]
1551
+ },
1552
+ {
1553
+ "term": "copula",
1554
+ "also": [
1555
+ "zero copula",
1556
+ "copular"
1557
+ ],
1558
+ "plain": "The linking verb 'be' in sentences like 'she is a doctor'. Many languages omit it (zero copula) in the present tense — Russian and Arabic say literally 'she doctor'.",
1559
+ "mt_relevance": "MT must insert copulas translating out of zero-copula languages and delete them going in.",
1560
+ "citations": [
1561
+ {
1562
+ "source": "WALS Online, chapter 120 (Stassen)",
1563
+ "url": "https://wals.info/chapter/120"
1564
+ }
1565
+ ],
1566
+ "related": [
1567
+ "negation",
1568
+ "word order"
1569
+ ]
1570
+ },
1571
+ {
1572
+ "term": "agreement",
1573
+ "also": [
1574
+ "concord",
1575
+ "object agreement",
1576
+ "verbal agreement",
1577
+ "agreement patterns"
1578
+ ],
1579
+ "plain": "When one word's form must match another's grammatical features — verbs matching their subjects, adjectives matching their nouns in gender and number. Some languages mark agreement with both subject and object on the verb.",
1580
+ "mt_relevance": "Generated text must satisfy all agreement chains; a single wrong feature produces multiple visible errors.",
1581
+ "citations": [
1582
+ {
1583
+ "source": "SIL Glossary of Linguistic Terms",
1584
+ "url": "https://glossary.sil.org/term/agreement"
1585
+ }
1586
+ ],
1587
+ "related": [
1588
+ "grammatical gender",
1589
+ "polypersonal agreement",
1590
+ "inflection"
1591
+ ]
1592
+ },
1593
+ {
1594
+ "term": "polypersonal agreement",
1595
+ "also": [
1596
+ "polypersonalism",
1597
+ "polypersonal"
1598
+ ],
1599
+ "plain": "Verb agreement with more than one participant at once — the verb carries markers for both subject and object (and sometimes more). Basque, Georgian, and Algonquian languages do this systematically.",
1600
+ "mt_relevance": "The verb form encodes who acts on whom, so MT must resolve both roles before it can produce a single correct verb.",
1601
+ "citations": [
1602
+ {
1603
+ "source": "WALS Online, chapter 102 (Siewierska)",
1604
+ "url": "https://wals.info/chapter/102"
1605
+ }
1606
+ ],
1607
+ "related": [
1608
+ "agreement",
1609
+ "polysynthesis",
1610
+ "obviation"
1611
+ ]
1612
+ },
1613
+ {
1614
+ "term": "conjunct order",
1615
+ "also": [
1616
+ "conjunct",
1617
+ "independent order",
1618
+ "conjunct verb"
1619
+ ],
1620
+ "plain": "In Algonquian languages like Plains Cree, verbs come in distinct inflectional 'orders': the independent order for main statements and the conjunct order mainly for subordinate clauses, questions, and certain discourse contexts. The two use entirely different ending sets.",
1621
+ "mt_relevance": "Choosing independent vs conjunct forms is a clause-type decision English gives no direct cue for.",
1622
+ "example": "Plains Cree (crk): described in the card's formality notes and the crk method's documentation.",
1623
+ "citations": [
1624
+ {
1625
+ "source": "Wikipedia: Plains Cree (standard reference; see also Wolvengrey 2011)",
1626
+ "url": "https://en.wikipedia.org/wiki/Plains_Cree"
1627
+ }
1628
+ ],
1629
+ "related": [
1630
+ "obviation",
1631
+ "mood",
1632
+ "agreement"
1633
+ ]
1634
+ },
1635
+ {
1636
+ "term": "directional",
1637
+ "also": [
1638
+ "directionals",
1639
+ "directional affixes",
1640
+ "directional prefixes"
1641
+ ],
1642
+ "plain": "Verb marking that builds direction of motion into the verb itself — toward the speaker, away, upriver, uphill. Common in Mayan, Tibeto-Burman, and many Papuan languages.",
1643
+ "mt_relevance": "Directional meaning packed into verbs must be unpacked into adverbs or prepositions in the target.",
1644
+ "citations": [
1645
+ {
1646
+ "source": "SIL Glossary of Linguistic Terms",
1647
+ "url": "https://glossary.sil.org/term/directional"
1648
+ }
1649
+ ],
1650
+ "related": [
1651
+ "affix",
1652
+ "valency"
1653
+ ]
1654
+ },
1655
+ {
1656
+ "term": "benefactive",
1657
+ "also": [
1658
+ "benefactives"
1659
+ ],
1660
+ "plain": "Marking that an action is done for someone's benefit — 'I baked her a cake'. Some languages mark this with a verb affix (an applicative) rather than word order or a preposition.",
1661
+ "mt_relevance": "Beneficiaries can hide inside verb morphology, so MT must detect and re-express them as separate phrases.",
1662
+ "citations": [
1663
+ {
1664
+ "source": "SIL Glossary of Linguistic Terms",
1665
+ "url": "https://glossary.sil.org/term/benefactive-case"
1666
+ }
1667
+ ],
1668
+ "related": [
1669
+ "dative",
1670
+ "valency"
1671
+ ]
1672
+ },
1673
+ {
1674
+ "term": "interrogative",
1675
+ "also": [
1676
+ "interrogatives",
1677
+ "question particle",
1678
+ "polar question",
1679
+ "question marker"
1680
+ ],
1681
+ "plain": "The grammar of asking questions. Yes/no questions may be marked by a particle, a verb form, word-order change, or intonation alone; content questions differ in whether 'who/what' moves to the front.",
1682
+ "mt_relevance": "If the source marks questions only by intonation, written input gives MT no signal that a question is being asked.",
1683
+ "citations": [
1684
+ {
1685
+ "source": "WALS Online, chapter 116 (Dryer)",
1686
+ "url": "https://wals.info/chapter/116"
1687
+ }
1688
+ ],
1689
+ "related": [
1690
+ "particle",
1691
+ "word order"
1692
+ ]
1693
+ },
1694
+ {
1695
+ "term": "word order",
1696
+ "also": [
1697
+ "basic word order",
1698
+ "dominant word order",
1699
+ "constituent order",
1700
+ "SOV",
1701
+ "SVO",
1702
+ "VSO",
1703
+ "VOS",
1704
+ "OVS",
1705
+ "wordOrder",
1706
+ "free word order",
1707
+ "flexible word order"
1708
+ ],
1709
+ "plain": "The typical arrangement of subject (S), object (O), and verb (V) in a plain statement. SOV (Japanese) and SVO (English) cover most languages; VSO (Welsh, Tagalog-type) is the third common type, and some languages have no fixed order at all.",
1710
+ "mt_relevance": "Word-order mismatch is the single largest driver of reordering errors between language pairs.",
1711
+ "citations": [
1712
+ {
1713
+ "source": "WALS Online, chapter 81 (Dryer)",
1714
+ "url": "https://wals.info/chapter/81"
1715
+ }
1716
+ ],
1717
+ "related": [
1718
+ "oblique",
1719
+ "adposition",
1720
+ "isolating"
1721
+ ]
1722
+ },
1723
+ {
1724
+ "term": "adposition",
1725
+ "also": [
1726
+ "adpositions",
1727
+ "preposition",
1728
+ "prepositions",
1729
+ "postposition",
1730
+ "postpositions"
1731
+ ],
1732
+ "plain": "The cover term for prepositions (before their noun: 'in the house') and postpositions (after it: Japanese uchi de 'house in'). A language's choice correlates strongly with its verb–object order.",
1733
+ "mt_relevance": "Preposition-to-postposition conversion flips the bracketing of every spatial and temporal phrase.",
1734
+ "citations": [
1735
+ {
1736
+ "source": "WALS Online, chapter 85 (Dryer)",
1737
+ "url": "https://wals.info/chapter/85"
1738
+ }
1739
+ ],
1740
+ "related": [
1741
+ "word order",
1742
+ "locative",
1743
+ "grammatical case"
1744
+ ]
1745
+ },
1746
+ {
1747
+ "term": "relative clause",
1748
+ "also": [
1749
+ "relative clauses",
1750
+ "relativization"
1751
+ ],
1752
+ "plain": "A clause that modifies a noun: 'the book that I read'. Languages place it before or after the noun, and use strategies from relative pronouns to gaps to special verb forms.",
1753
+ "mt_relevance": "Prenominal relative clauses (Japanese, Turkish) require inverting long stretches of text relative to English order.",
1754
+ "citations": [
1755
+ {
1756
+ "source": "WALS Online, chapter 90 (Dryer)",
1757
+ "url": "https://wals.info/chapter/90"
1758
+ }
1759
+ ],
1760
+ "related": [
1761
+ "word order",
1762
+ "participle"
1763
+ ]
1764
+ },
1765
+ {
1766
+ "term": "focus system",
1767
+ "also": [
1768
+ "voice focus",
1769
+ "voiceFocus",
1770
+ "focus marker",
1771
+ "symmetrical voice",
1772
+ "Austronesian voice",
1773
+ "focus construction"
1774
+ ],
1775
+ "plain": "A clause system, best known from Philippine languages like Tagalog, where verb morphology selects which participant — actor, patient, location, instrument — is the grammatical pivot of the sentence. Often called symmetrical voice.",
1776
+ "mt_relevance": "Focus choice changes verb form, marker placement and word order at once, so MT cannot map clauses word-by-word.",
1777
+ "example": "Surfaced on cards as linguisticChallenges.voiceFocus for many Austronesian languages.",
1778
+ "citations": [
1779
+ {
1780
+ "source": "SIL Glossary of Linguistic Terms",
1781
+ "url": "https://glossary.sil.org/term/focus"
1782
+ }
1783
+ ],
1784
+ "related": [
1785
+ "grammatical voice",
1786
+ "word order"
1787
+ ]
1788
+ },
1789
+ {
1790
+ "term": "particle",
1791
+ "also": [
1792
+ "particles",
1793
+ "sentence-final particle",
1794
+ "discourse particle",
1795
+ "topic marker"
1796
+ ],
1797
+ "plain": "A small, uninflected function word that adds grammatical or attitudinal meaning — question markers, topic markers, politeness softeners. East Asian languages make heavy use of sentence-final particles.",
1798
+ "mt_relevance": "Particles carry meaning (questionhood, attitude, topic) that MT must re-express by entirely different means.",
1799
+ "citations": [
1800
+ {
1801
+ "source": "SIL Glossary of Linguistic Terms",
1802
+ "url": "https://glossary.sil.org/term/particle"
1803
+ }
1804
+ ],
1805
+ "related": [
1806
+ "clitic",
1807
+ "interrogative",
1808
+ "register"
1809
+ ]
1810
+ },
1811
+ {
1812
+ "term": "tone",
1813
+ "also": [
1814
+ "tonal",
1815
+ "tone system",
1816
+ "tonal language",
1817
+ "tones",
1818
+ "lexical tone",
1819
+ "toneSystem",
1820
+ "contour tone"
1821
+ ],
1822
+ "plain": "Using voice pitch to distinguish words: Mandarin mā 'mother' vs mǎ 'horse'. Simple systems contrast two levels; complex systems (many West African and Southeast Asian languages) use several levels and contours.",
1823
+ "mt_relevance": "Tone is usually invisible in romanized or unmarked text, collapsing distinct words into one spelling for the MT system.",
1824
+ "citations": [
1825
+ {
1826
+ "source": "WALS Online, chapter 13 (Maddieson)",
1827
+ "url": "https://wals.info/chapter/13"
1828
+ }
1829
+ ],
1830
+ "related": [
1831
+ "pitch accent",
1832
+ "diacritic",
1833
+ "sandhi"
1834
+ ]
1835
+ },
1836
+ {
1837
+ "term": "pitch accent",
1838
+ "also": [
1839
+ "pitch-accent"
1840
+ ],
1841
+ "plain": "A system where pitch distinguishes words, but only one syllable per word carries the distinctive pitch — Japanese háshi 'chopsticks' vs hashí 'bridge'. Lighter than full tone, heavier than pure stress.",
1842
+ "mt_relevance": "Like tone, pitch accent is rarely written, so homographs multiply in text.",
1843
+ "citations": [
1844
+ {
1845
+ "source": "SIL Glossary of Linguistic Terms",
1846
+ "url": "https://glossary.sil.org/term/pitch-accent"
1847
+ }
1848
+ ],
1849
+ "related": [
1850
+ "tone",
1851
+ "mora"
1852
+ ]
1853
+ },
1854
+ {
1855
+ "term": "mora",
1856
+ "also": [
1857
+ "moraic",
1858
+ "morae"
1859
+ ],
1860
+ "plain": "A timing unit smaller than the syllable: a short syllable counts one mora, a long vowel or a final consonant adds another. Japanese rhythm, poetry, and even abbreviations count morae, not syllables.",
1861
+ "mt_relevance": "Mora-based phonology shapes how loanwords and names are adapted, affecting transliteration quality.",
1862
+ "citations": [
1863
+ {
1864
+ "source": "SIL Glossary of Linguistic Terms",
1865
+ "url": "https://glossary.sil.org/term/mora"
1866
+ }
1867
+ ],
1868
+ "related": [
1869
+ "vowel length",
1870
+ "pitch accent"
1871
+ ]
1872
+ },
1873
+ {
1874
+ "term": "ejective",
1875
+ "also": [
1876
+ "ejectives",
1877
+ "glottalized consonant",
1878
+ "glottalized consonants"
1879
+ ],
1880
+ "plain": "A consonant produced with a burst of air from the closed glottis instead of the lungs, giving a sharp popping quality. Common in languages of the Caucasus, the Americas, and East Africa; written with an apostrophe (k', t').",
1881
+ "mt_relevance": "Ejective marks are often dropped in casual typing, merging distinct words in the input text.",
1882
+ "citations": [
1883
+ {
1884
+ "source": "WALS Online, chapter 7 (Maddieson)",
1885
+ "url": "https://wals.info/chapter/7"
1886
+ }
1887
+ ],
1888
+ "related": [
1889
+ "glottal stop",
1890
+ "phoneme",
1891
+ "orthography"
1892
+ ]
1893
+ },
1894
+ {
1895
+ "term": "glottal stop",
1896
+ "also": [
1897
+ "glottal",
1898
+ "ʔ",
1899
+ "okina",
1900
+ "ʻokina"
1901
+ ],
1902
+ "plain": "The catch in the throat in the middle of 'uh-oh'. In many languages it is a full consonant that distinguishes words — Hawaiian writes it as the ʻokina (ʻ), and dropping it changes meanings.",
1903
+ "mt_relevance": "Glottal stops are frequently omitted or typed with the wrong apostrophe character, fragmenting words across spellings.",
1904
+ "example": "Hawaiian (haw): the ʻokina is a phonemic consonant; card orthography notes flag apostrophe-variant issues.",
1905
+ "citations": [
1906
+ {
1907
+ "source": "SIL Glossary of Linguistic Terms",
1908
+ "url": "https://glossary.sil.org/term/glottal-stop"
1909
+ }
1910
+ ],
1911
+ "related": [
1912
+ "ejective",
1913
+ "diacritic",
1914
+ "orthography"
1915
+ ]
1916
+ },
1917
+ {
1918
+ "term": "uvular",
1919
+ "also": [
1920
+ "uvulars",
1921
+ "uvular consonants",
1922
+ "uvular consonant"
1923
+ ],
1924
+ "plain": "Consonants made at the very back of the mouth against the uvula, like the Arabic q or French r. Rarer than velar k/g sounds and a signature of certain language areas.",
1925
+ "mt_relevance": "Uvulars are often romanized inconsistently (q/k/kh), splitting one word into several text forms.",
1926
+ "citations": [
1927
+ {
1928
+ "source": "WALS Online, chapter 6 (Maddieson)",
1929
+ "url": "https://wals.info/chapter/6"
1930
+ }
1931
+ ],
1932
+ "related": [
1933
+ "pharyngeal",
1934
+ "phoneme"
1935
+ ]
1936
+ },
1937
+ {
1938
+ "term": "click",
1939
+ "also": [
1940
+ "clicks",
1941
+ "click consonants",
1942
+ "click consonant"
1943
+ ],
1944
+ "plain": "Consonants made by sucking air in, like the English 'tsk-tsk' — but used as ordinary speech sounds. Khoisan and some southern Bantu languages (Zulu, Xhosa) have full sets of click consonants.",
1945
+ "mt_relevance": "Clicks are written with unusual characters (ǃ, ǂ, c, q, x) that tokenizers and fonts often mishandle.",
1946
+ "citations": [
1947
+ {
1948
+ "source": "WALS Online, chapter 19 (Maddieson)",
1949
+ "url": "https://wals.info/chapter/19"
1950
+ }
1951
+ ],
1952
+ "related": [
1953
+ "phoneme",
1954
+ "orthography"
1955
+ ]
1956
+ },
1957
+ {
1958
+ "term": "implosive",
1959
+ "also": [
1960
+ "implosives"
1961
+ ],
1962
+ "plain": "Consonants made with air briefly sucked inward at the throat, like the ɓ and ɗ of Hausa or Vietnamese. They sound like emphatic b/d to untrained ears but are distinct phonemes.",
1963
+ "mt_relevance": "Implosive letters (ɓ, ɗ) are often typed as plain b/d, merging distinct words in text corpora.",
1964
+ "citations": [
1965
+ {
1966
+ "source": "WALS Online, chapter 7 (Maddieson)",
1967
+ "url": "https://wals.info/chapter/7"
1968
+ }
1969
+ ],
1970
+ "related": [
1971
+ "ejective",
1972
+ "phoneme"
1973
+ ]
1974
+ },
1975
+ {
1976
+ "term": "nasal",
1977
+ "also": [
1978
+ "nasals",
1979
+ "nasal consonant",
1980
+ "nasal consonants",
1981
+ "nasal stop"
1982
+ ],
1983
+ "plain": "A consonant made with air flowing out through the nose, like m, n, or the ng in 'sing'. Nearly every language has nasal consonants, and some also spread nasality onto neighbouring vowels.",
1984
+ "mt_relevance": "As a purely phonetic category, nasal consonants matter for speech and transliteration rather than text translation, which works on written words.",
1985
+ "citations": [
1986
+ {
1987
+ "source": "SIL Glossary of Linguistic Terms",
1988
+ "url": "https://glossary.sil.org/term/nasal"
1989
+ }
1990
+ ],
1991
+ "related": [
1992
+ "nasal vowel",
1993
+ "phoneme"
1994
+ ]
1995
+ },
1996
+ {
1997
+ "term": "fricative",
1998
+ "also": [
1999
+ "fricatives",
2000
+ "sibilant",
2001
+ "sibilants"
2002
+ ],
2003
+ "plain": "A consonant made by forcing air through a narrow gap so it hisses, like f, s, sh, or th. Fricatives are among the most common consonant types in the world's languages.",
2004
+ "mt_relevance": "Fricatives are a phonetic category that affects pronunciation and the transliteration of names, not the grammar of text-based MT.",
2005
+ "citations": [
2006
+ {
2007
+ "source": "SIL Glossary of Linguistic Terms",
2008
+ "url": "https://glossary.sil.org/term/fricative"
2009
+ }
2010
+ ],
2011
+ "related": [
2012
+ "phoneme",
2013
+ "romanization"
2014
+ ]
2015
+ },
2016
+ {
2017
+ "term": "retroflex",
2018
+ "also": [
2019
+ "retroflexes",
2020
+ "retroflex consonant",
2021
+ "retroflex consonants",
2022
+ "retroflexion"
2023
+ ],
2024
+ "plain": "A consonant made by curling the tongue tip up and back toward the roof of the mouth, giving a distinctive hollow sound. Retroflex consonants are characteristic of South Asian languages such as Hindi and Tamil.",
2025
+ "mt_relevance": "Retroflexion is mostly a matter of speech and transliteration; in text it appears only as distinct letters or diacritics that MT must preserve.",
2026
+ "citations": [
2027
+ {
2028
+ "source": "SIL Glossary of Linguistic Terms",
2029
+ "url": "https://glossary.sil.org/term/retroflex"
2030
+ }
2031
+ ],
2032
+ "related": [
2033
+ "phoneme",
2034
+ "diacritic"
2035
+ ]
2036
+ },
2037
+ {
2038
+ "term": "glottalized",
2039
+ "also": [
2040
+ "glottalization",
2041
+ "glottalic",
2042
+ "creaky voice",
2043
+ "creaky-voiced"
2044
+ ],
2045
+ "plain": "Produced with a constriction or closure of the glottis, adding a catch or creaky quality to a consonant or vowel. Ejectives and implosives are glottalized consonants, and some languages also have glottalized (creaky-voiced) sonorants and vowels.",
2046
+ "mt_relevance": "Glottalization is a phonetic property that matters for speech and transliteration; in text it survives only as special letters or apostrophes that MT must keep intact.",
2047
+ "citations": [
2048
+ {
2049
+ "source": "WALS Online, chapter 7 (Maddieson)",
2050
+ "url": "https://wals.info/chapter/7"
2051
+ }
2052
+ ],
2053
+ "related": [
2054
+ "ejective",
2055
+ "glottal stop",
2056
+ "phoneme"
2057
+ ]
2058
+ },
2059
+ {
2060
+ "term": "pharyngeal",
2061
+ "also": [
2062
+ "pharyngeals",
2063
+ "pharyngeal consonants"
2064
+ ],
2065
+ "plain": "Consonants made by squeezing the throat (pharynx), like Arabic ʿayn (ع). They are rare worldwide and hard for non-native speakers to hear or produce.",
2066
+ "mt_relevance": "Pharyngeals are romanized many ways (ʿ, ', 3, or nothing), creating spelling chaos in informal text.",
2067
+ "citations": [
2068
+ {
2069
+ "source": "SIL Glossary of Linguistic Terms",
2070
+ "url": "https://glossary.sil.org/term/pharyngeal"
2071
+ }
2072
+ ],
2073
+ "related": [
2074
+ "uvular",
2075
+ "glottal stop",
2076
+ "romanization"
2077
+ ]
2078
+ },
2079
+ {
2080
+ "term": "nasal vowel",
2081
+ "also": [
2082
+ "nasal vowels",
2083
+ "nasalization",
2084
+ "nasalized",
2085
+ "vowel nasalization"
2086
+ ],
2087
+ "plain": "A vowel pronounced with air flowing through the nose, as in French bon or Portuguese são. Where it is contrastive, oral and nasal vowels distinguish different words.",
2088
+ "mt_relevance": "Nasalization marks (ã, ę, ą) are commonly dropped in informal typing, merging word pairs.",
2089
+ "citations": [
2090
+ {
2091
+ "source": "WALS Online, chapter 10 (Hajek)",
2092
+ "url": "https://wals.info/chapter/10"
2093
+ }
2094
+ ],
2095
+ "related": [
2096
+ "diacritic",
2097
+ "phoneme"
2098
+ ]
2099
+ },
2100
+ {
2101
+ "term": "vowel length",
2102
+ "also": [
2103
+ "long vowel",
2104
+ "long vowels",
2105
+ "vowel-length"
2106
+ ],
2107
+ "plain": "Holding a vowel longer to make a different word — Finnish tuli 'fire' vs tuuli 'wind'. Scripts mark it with double letters, macrons (ā), or not at all.",
2108
+ "mt_relevance": "When length marking is optional (e.g. Hawaiian kahakō, Arabic vowels), the same word appears in multiple spellings.",
2109
+ "example": "Hawaiian (haw): the kahakō (macron) marks long vowels and is often omitted in casual text.",
2110
+ "citations": [
2111
+ {
2112
+ "source": "SIL Glossary of Linguistic Terms",
2113
+ "url": "https://glossary.sil.org/term/vowel-length"
2114
+ }
2115
+ ],
2116
+ "related": [
2117
+ "mora",
2118
+ "diacritic",
2119
+ "gemination"
2120
+ ]
2121
+ },
2122
+ {
2123
+ "term": "phoneme",
2124
+ "also": [
2125
+ "phonemes",
2126
+ "phonemic",
2127
+ "phoneme inventory",
2128
+ "consonant inventory",
2129
+ "vowel inventory"
2130
+ ],
2131
+ "plain": "A speech sound that distinguishes words in a particular language — swap one phoneme for another and you get a different word (pat vs bat). A language's phoneme inventory ranges from about a dozen sounds to well over a hundred.",
2132
+ "mt_relevance": "Inventory size and content determine how foreign names and loanwords get reshaped in the language.",
2133
+ "citations": [
2134
+ {
2135
+ "source": "WALS Online, chapter 1 (Maddieson)",
2136
+ "url": "https://wals.info/chapter/1"
2137
+ }
2138
+ ],
2139
+ "related": [
2140
+ "tone",
2141
+ "orthography"
2142
+ ]
2143
+ },
2144
+ {
2145
+ "term": "downstep",
2146
+ "also": [
2147
+ "downdrift"
2148
+ ],
2149
+ "plain": "In many African tone languages, a step-down in pitch that affects all following high tones in the phrase. It is a tonal landmark that can itself distinguish meanings.",
2150
+ "mt_relevance": "Downstep is essentially never written, so tonal information is systematically absent from text data.",
2151
+ "citations": [
2152
+ {
2153
+ "source": "SIL Glossary of Linguistic Terms",
2154
+ "url": "https://glossary.sil.org/term/downstep"
2155
+ }
2156
+ ],
2157
+ "related": [
2158
+ "tone"
2159
+ ]
2160
+ },
2161
+ {
2162
+ "term": "advanced tongue root",
2163
+ "also": [
2164
+ "ATR",
2165
+ "ATR harmony",
2166
+ "[+ATR]"
2167
+ ],
2168
+ "plain": "A vowel quality made by pushing the tongue root forward, expanding the throat. Many African languages split their vowels into +ATR and -ATR sets and require words to stay within one set (ATR harmony).",
2169
+ "mt_relevance": "ATR distinctions are inconsistently marked in orthographies, splitting words across spellings.",
2170
+ "citations": [
2171
+ {
2172
+ "source": "SIL Glossary of Linguistic Terms",
2173
+ "url": "https://glossary.sil.org/term/advanced-tongue-root"
2174
+ }
2175
+ ],
2176
+ "related": [
2177
+ "vowel harmony",
2178
+ "phoneme"
2179
+ ]
2180
+ },
2181
+ {
2182
+ "term": "script",
2183
+ "also": [
2184
+ "writing system",
2185
+ "scripts"
2186
+ ],
2187
+ "plain": "The set of symbols a language is written in — Latin, Cyrillic, Arabic, Han characters, and many more. One language can use several scripts (Serbian), and one script can serve hundreds of languages.",
2188
+ "mt_relevance": "Script identity drives every downstream text process: encoding, tokenization, and which MT models even accept the input.",
2189
+ "citations": [
2190
+ {
2191
+ "source": "Wikipedia: Writing system (standard reference)",
2192
+ "url": "https://en.wikipedia.org/wiki/Writing_system"
2193
+ }
2194
+ ],
2195
+ "related": [
2196
+ "orthography",
2197
+ "abugida",
2198
+ "syllabary",
2199
+ "writing direction"
2200
+ ]
2201
+ },
2202
+ {
2203
+ "term": "orthography",
2204
+ "also": [
2205
+ "orthographic",
2206
+ "spelling system",
2207
+ "orthographies",
2208
+ "orthographic status"
2209
+ ],
2210
+ "plain": "The agreed rules for writing a language in its script — which letters, diacritics, and spellings are correct. Some languages have multiple competing orthographies or none standardized at all.",
2211
+ "mt_relevance": "Competing or unstandardized orthographies split scarce training data into incompatible spelling variants.",
2212
+ "citations": [
2213
+ {
2214
+ "source": "SIL Glossary of Linguistic Terms",
2215
+ "url": "https://glossary.sil.org/term/orthography"
2216
+ }
2217
+ ],
2218
+ "related": [
2219
+ "script",
2220
+ "diacritic",
2221
+ "romanization"
2222
+ ]
2223
+ },
2224
+ {
2225
+ "term": "abugida",
2226
+ "also": [
2227
+ "alphasyllabary",
2228
+ "abugidas"
2229
+ ],
2230
+ "plain": "A script where each symbol is a consonant with a built-in default vowel, and other vowels are marked by modifying the base sign — as in Devanagari, Ethiopic, or Thai. Between an alphabet and a syllabary.",
2231
+ "mt_relevance": "Abugida vowel marks are combining characters that naïve text processing can strip or reorder, corrupting words.",
2232
+ "citations": [
2233
+ {
2234
+ "source": "Wikipedia: Abugida (standard reference)",
2235
+ "url": "https://en.wikipedia.org/wiki/Abugida"
2236
+ }
2237
+ ],
2238
+ "related": [
2239
+ "syllabary",
2240
+ "script",
2241
+ "diacritic"
2242
+ ]
2243
+ },
2244
+ {
2245
+ "term": "syllabary",
2246
+ "also": [
2247
+ "syllabic script",
2248
+ "syllabaries"
2249
+ ],
2250
+ "plain": "A script with one symbol per syllable rather than per sound — Japanese kana or Cherokee. Works best for languages with simple syllable structures.",
2251
+ "mt_relevance": "Syllabaries change the granularity of text: tokenizers see syllables, not consonants and vowels.",
2252
+ "citations": [
2253
+ {
2254
+ "source": "Wikipedia: Syllabary (standard reference)",
2255
+ "url": "https://en.wikipedia.org/wiki/Syllabary"
2256
+ }
2257
+ ],
2258
+ "related": [
2259
+ "abugida",
2260
+ "syllabics",
2261
+ "script"
2262
+ ]
2263
+ },
2264
+ {
2265
+ "term": "syllabics",
2266
+ "also": [
2267
+ "Canadian Aboriginal Syllabics",
2268
+ "UCAS",
2269
+ "Cans"
2270
+ ],
2271
+ "plain": "The script family used for Cree, Inuktitut, Ojibwe and other Indigenous Canadian languages, where each character encodes a consonant and its rotation encodes the vowel. Invented in the 1840s and still in active community use.",
2272
+ "mt_relevance": "Many Cree/Inuktitut texts exist in both syllabics and roman orthography, so MT pipelines need reliable script conversion.",
2273
+ "example": "Plains Cree (crk): card script is Cans (Canadian Aboriginal Syllabics) with a roman-orthography converter.",
2274
+ "citations": [
2275
+ {
2276
+ "source": "Wikipedia: Canadian Aboriginal syllabics (standard reference)",
2277
+ "url": "https://en.wikipedia.org/wiki/Canadian_Aboriginal_syllabics"
2278
+ }
2279
+ ],
2280
+ "related": [
2281
+ "syllabary",
2282
+ "romanization",
2283
+ "script"
2284
+ ]
2285
+ },
2286
+ {
2287
+ "term": "romanization",
2288
+ "also": [
2289
+ "transliteration",
2290
+ "romanized",
2291
+ "latinization",
2292
+ "scriptConverter"
2293
+ ],
2294
+ "plain": "Writing a language in Latin letters instead of its native script, by rule (transliteration) or by sound. One language often has several competing romanization standards.",
2295
+ "mt_relevance": "Romanized and native-script text behave as different languages to an MT model unless explicitly converted.",
2296
+ "citations": [
2297
+ {
2298
+ "source": "SIL Glossary of Linguistic Terms",
2299
+ "url": "https://glossary.sil.org/term/transliteration"
2300
+ }
2301
+ ],
2302
+ "related": [
2303
+ "script",
2304
+ "orthography",
2305
+ "tone"
2306
+ ]
2307
+ },
2308
+ {
2309
+ "term": "diacritic",
2310
+ "also": [
2311
+ "diacritics",
2312
+ "accent marks",
2313
+ "tone marks",
2314
+ "combining marks"
2315
+ ],
2316
+ "plain": "Marks added to letters — accents, tildes, dots, macrons — to indicate tone, length, nasality, or different sounds. Some orthographies depend on them completely; users often omit them in casual typing.",
2317
+ "mt_relevance": "Diacritic-stripped text is a pervasive data-quality problem: it merges distinct words and mismatches clean training data.",
2318
+ "citations": [
2319
+ {
2320
+ "source": "SIL Glossary of Linguistic Terms",
2321
+ "url": "https://glossary.sil.org/term/diacritic"
2322
+ }
2323
+ ],
2324
+ "related": [
2325
+ "orthography",
2326
+ "tone",
2327
+ "vowel length"
2328
+ ]
2329
+ },
2330
+ {
2331
+ "term": "writing direction",
2332
+ "also": [
2333
+ "right-to-left",
2334
+ "RTL",
2335
+ "left-to-right",
2336
+ "LTR",
2337
+ "writingDirection",
2338
+ "bidirectional text"
2339
+ ],
2340
+ "plain": "Which way the script runs: left-to-right (Latin), right-to-left (Arabic, Hebrew), or historically top-to-bottom (Mongolian, classical Chinese). Mixed-direction text needs special handling.",
2341
+ "mt_relevance": "RTL and bidirectional text break naive string handling — numbers, punctuation and embedded Latin words reorder unpredictably.",
2342
+ "citations": [
2343
+ {
2344
+ "source": "Wikipedia: Bidirectional text (standard reference)",
2345
+ "url": "https://en.wikipedia.org/wiki/Bidirectional_text"
2346
+ }
2347
+ ],
2348
+ "related": [
2349
+ "script",
2350
+ "orthography"
2351
+ ]
2352
+ },
2353
+ {
2354
+ "term": "register",
2355
+ "also": [
2356
+ "registers",
2357
+ "speech register",
2358
+ "register-levels",
2359
+ "speech levels"
2360
+ ],
2361
+ "plain": "A variety of a language tied to social context — formal, casual, ceremonial, technical. Some languages grammaticalize registers: Javanese has distinct vocabulary sets for different politeness levels.",
2362
+ "mt_relevance": "MT must hold a consistent register; mixing formal and casual forms in one output reads as broken or rude.",
2363
+ "citations": [
2364
+ {
2365
+ "source": "SIL Glossary of Linguistic Terms",
2366
+ "url": "https://glossary.sil.org/term/register"
2367
+ }
2368
+ ],
2369
+ "related": [
2370
+ "T-V distinction",
2371
+ "honorific",
2372
+ "diglossia"
2373
+ ]
2374
+ },
2375
+ {
2376
+ "term": "T-V distinction",
2377
+ "also": [
2378
+ "tu/vous distinction",
2379
+ "T-V",
2380
+ "formal/informal pronouns",
2381
+ "tu-vous",
2382
+ "formal you"
2383
+ ],
2384
+ "plain": "Having two (or more) words for 'you' depending on social distance, like French tu/vous or German du/Sie. The choice signals respect, intimacy, or hierarchy and is hard to undo once made.",
2385
+ "mt_relevance": "English 'you' gives no cue, so MT must choose formality from context — a frequent and socially visible error.",
2386
+ "example": "French (fra): card formality data distinguishes tu/vous usage contexts.",
2387
+ "citations": [
2388
+ {
2389
+ "source": "WALS Online, chapter 45 (Helmbrecht)",
2390
+ "url": "https://wals.info/chapter/45"
2391
+ }
2392
+ ],
2393
+ "related": [
2394
+ "register",
2395
+ "honorific",
2396
+ "politeness"
2397
+ ]
2398
+ },
2399
+ {
2400
+ "term": "honorific",
2401
+ "also": [
2402
+ "honorifics",
2403
+ "respectful speech",
2404
+ "respect forms",
2405
+ "honorific system"
2406
+ ],
2407
+ "plain": "Grammatical or lexical forms that encode respect toward the listener or the person discussed — Japanese keigo, Korean speech levels, special kin-respect vocabularies. Often obligatory, not optional politeness.",
2408
+ "mt_relevance": "Honorific selection requires social knowledge (who outranks whom) that the source text rarely states.",
2409
+ "citations": [
2410
+ {
2411
+ "source": "SIL Glossary of Linguistic Terms",
2412
+ "url": "https://glossary.sil.org/term/honorifics"
2413
+ }
2414
+ ],
2415
+ "related": [
2416
+ "T-V distinction",
2417
+ "register",
2418
+ "politeness"
2419
+ ]
2420
+ },
2421
+ {
2422
+ "term": "politeness",
2423
+ "also": [
2424
+ "politeness distinctions",
2425
+ "formality",
2426
+ "formality system",
2427
+ "politeness levels"
2428
+ ],
2429
+ "plain": "The linguistic encoding of social relationships — through pronoun choice, verb endings, particles, or vocabulary. Languages range from no grammatical politeness to elaborate multi-level systems.",
2430
+ "mt_relevance": "A translation can be lexically perfect and still fail by choosing the wrong politeness level for the situation.",
2431
+ "citations": [
2432
+ {
2433
+ "source": "WALS Online, chapter 45 (Helmbrecht)",
2434
+ "url": "https://wals.info/chapter/45"
2435
+ }
2436
+ ],
2437
+ "related": [
2438
+ "T-V distinction",
2439
+ "honorific",
2440
+ "register"
2441
+ ]
2442
+ },
2443
+ {
2444
+ "term": "diglossia",
2445
+ "also": [
2446
+ "diglossic"
2447
+ ],
2448
+ "plain": "A stable situation where a community uses two varieties of a language for different purposes — a 'high' variety for writing and formal speech (Modern Standard Arabic) and a 'low' one for daily life (the spoken dialects). Speakers switch by context, not by choice.",
2449
+ "mt_relevance": "Training data skews toward the written 'high' variety, so MT fails on the spoken variety people actually use.",
2450
+ "citations": [
2451
+ {
2452
+ "source": "SIL Glossary of Linguistic Terms",
2453
+ "url": "https://glossary.sil.org/term/diglossia"
2454
+ }
2455
+ ],
2456
+ "related": [
2457
+ "register",
2458
+ "code-switching",
2459
+ "macrolanguage"
2460
+ ]
2461
+ },
2462
+ {
2463
+ "term": "code-switching",
2464
+ "also": [
2465
+ "code switching",
2466
+ "codeswitching",
2467
+ "codeSwitching",
2468
+ "code-mixing"
2469
+ ],
2470
+ "plain": "Alternating between two or more languages within one conversation or sentence, as bilingual speakers naturally do. It follows grammatical patterns rather than being random mixing.",
2471
+ "mt_relevance": "Mixed-language input confuses language detection and produces garbled output unless the system handles both codes.",
2472
+ "citations": [
2473
+ {
2474
+ "source": "SIL Glossary of Linguistic Terms",
2475
+ "url": "https://glossary.sil.org/term/code-switching"
2476
+ }
2477
+ ],
2478
+ "related": [
2479
+ "diglossia",
2480
+ "loanword"
2481
+ ]
2482
+ },
2483
+ {
2484
+ "term": "creole",
2485
+ "also": [
2486
+ "creoles",
2487
+ "creole language",
2488
+ "creolization"
2489
+ ],
2490
+ "plain": "A full natural language that grew out of intense language contact, typically drawing most vocabulary from one language (the lexifier) while developing its own grammar. Haitian Creole, Tok Pisin, and Papiamento are examples.",
2491
+ "mt_relevance": "Creoles look deceptively like their lexifier in writing, so MT systems trained on the lexifier mistranslate them systematically.",
2492
+ "citations": [
2493
+ {
2494
+ "source": "APiCS Online (Michaelis et al., eds.)",
2495
+ "url": "https://apics-online.info"
2496
+ }
2497
+ ],
2498
+ "related": [
2499
+ "pidgin",
2500
+ "lexifier",
2501
+ "loanword"
2502
+ ]
2503
+ },
2504
+ {
2505
+ "term": "pidgin",
2506
+ "also": [
2507
+ "pidgins"
2508
+ ],
2509
+ "plain": "A simplified contact language with no native speakers, created for trade or work between groups with no common tongue. When children grow up speaking one natively, it becomes a creole.",
2510
+ "mt_relevance": "Pidgins have high variability and thin text data, making consistent MT especially hard.",
2511
+ "citations": [
2512
+ {
2513
+ "source": "APiCS Online (Michaelis et al., eds.)",
2514
+ "url": "https://apics-online.info"
2515
+ }
2516
+ ],
2517
+ "related": [
2518
+ "creole",
2519
+ "lexifier"
2520
+ ]
2521
+ },
2522
+ {
2523
+ "term": "lexifier",
2524
+ "also": [
2525
+ "lexifier language",
2526
+ "lexified"
2527
+ ],
2528
+ "plain": "The language that supplied most of a creole's or pidgin's vocabulary — English for Jamaican Patois, French for Haitian Creole, Portuguese for Papiamento. The grammar, however, is the creole's own.",
2529
+ "mt_relevance": "Shared vocabulary with the lexifier masks deep grammatical differences that MT must not gloss over.",
2530
+ "citations": [
2531
+ {
2532
+ "source": "APiCS Online (Michaelis et al., eds.)",
2533
+ "url": "https://apics-online.info"
2534
+ }
2535
+ ],
2536
+ "related": [
2537
+ "creole",
2538
+ "pidgin",
2539
+ "loanword"
2540
+ ]
2541
+ },
2542
+ {
2543
+ "term": "loanword",
2544
+ "also": [
2545
+ "loanwords",
2546
+ "borrowing",
2547
+ "borrowings",
2548
+ "calque",
2549
+ "English loanwords"
2550
+ ],
2551
+ "plain": "A word taken from another language and adapted to local sound patterns, like 'sushi' in English or 'le weekend' in French. A calque borrows the structure instead, translating piece by piece ('skyscraper' → French 'gratte-ciel').",
2552
+ "mt_relevance": "Deciding whether to keep, adapt, or translate a loanword is a recurring choice in localization, especially for technical terms.",
2553
+ "citations": [
2554
+ {
2555
+ "source": "WOLD (World Loanword Database)",
2556
+ "url": "https://wold.clld.org"
2557
+ }
2558
+ ],
2559
+ "related": [
2560
+ "code-switching",
2561
+ "lexifier"
2562
+ ]
2563
+ },
2564
+ {
2565
+ "term": "macrolanguage",
2566
+ "also": [
2567
+ "macrolanguages"
2568
+ ],
2569
+ "plain": "An ISO 639-3 bookkeeping category: a single code (like 'ara' Arabic or 'zho' Chinese) that covers several distinct member languages treated as one for historical or political reasons. Each member also has its own code.",
2570
+ "mt_relevance": "Data labeled with a macrolanguage code mixes mutually unintelligible varieties, contaminating training and evaluation.",
2571
+ "citations": [
2572
+ {
2573
+ "source": "SIL ISO 639-3: macrolanguage scope",
2574
+ "url": "https://iso639-3.sil.org/about/scope#Macrolanguages"
2575
+ }
2576
+ ],
2577
+ "related": [
2578
+ "ISO 639-3",
2579
+ "diglossia",
2580
+ "language isolate"
2581
+ ]
2582
+ },
2583
+ {
2584
+ "term": "endonym",
2585
+ "also": [
2586
+ "autonym",
2587
+ "native name",
2588
+ "exonym",
2589
+ "endonyms"
2590
+ ],
2591
+ "plain": "A community's own name for its language (endonym/autonym), as opposed to the name outsiders use (exonym). 'Deutsch' is the endonym for what English calls 'German'; many Indigenous communities prefer their endonyms.",
2592
+ "mt_relevance": "Name mismatches across databases cause language-identification and data-merging errors in MT pipelines.",
2593
+ "citations": [
2594
+ {
2595
+ "source": "Glottolog",
2596
+ "url": "https://glottolog.org"
2597
+ }
2598
+ ],
2599
+ "related": [
2600
+ "ISO 639-3",
2601
+ "glottocode"
2602
+ ]
2603
+ },
2604
+ {
2605
+ "term": "language isolate",
2606
+ "also": [
2607
+ "isolate",
2608
+ "isolates",
2609
+ "isIsolate"
2610
+ ],
2611
+ "plain": "A language with no demonstrated relatives — a family of one. Basque, Ainu, and Burushaski are famous examples; isolates are surprisingly common worldwide.",
2612
+ "mt_relevance": "Isolates cannot borrow training signal from related languages, removing a key low-resource MT strategy (transfer learning).",
2613
+ "citations": [
2614
+ {
2615
+ "source": "Glottolog: language families",
2616
+ "url": "https://glottolog.org/glottolog/family"
2617
+ }
2618
+ ],
2619
+ "related": [
2620
+ "glottocode",
2621
+ "macrolanguage"
2622
+ ]
2623
+ },
2624
+ {
2625
+ "term": "language vitality",
2626
+ "also": [
2627
+ "vitality",
2628
+ "endangerment status",
2629
+ "endangered",
2630
+ "endangered language",
2631
+ "dormant",
2632
+ "sleeping language",
2633
+ "vigorous",
2634
+ "threatened"
2635
+ ],
2636
+ "plain": "How robustly a language is being passed to children and used across life domains. Scales like EGIDS and Glottolog's endangerment status run from 'vigorous' through 'threatened' and 'moribund' to 'dormant' (no fluent speakers, but potential for revival).",
2637
+ "mt_relevance": "Vitality predicts data availability and, for community-driven MT, what role technology should play (e.g. revitalization support, not replacement).",
2638
+ "citations": [
2639
+ {
2640
+ "source": "Glottolog: Agglomerated Endangerment Status",
2641
+ "url": "https://glottolog.org/langdoc/status"
2642
+ }
2643
+ ],
2644
+ "related": [
2645
+ "moribund",
2646
+ "language isolate"
2647
+ ]
2648
+ },
2649
+ {
2650
+ "term": "moribund",
2651
+ "also": [],
2652
+ "plain": "A vitality status meaning the language is no longer being learned by children; the remaining fluent speakers are all older adults. Without intervention, such a language becomes dormant within a generation.",
2653
+ "mt_relevance": "Moribund languages have shrinking speaker pools to validate MT output, raising the stakes of every data decision.",
2654
+ "citations": [
2655
+ {
2656
+ "source": "Glottolog: Agglomerated Endangerment Status",
2657
+ "url": "https://glottolog.org/langdoc/status"
2658
+ }
2659
+ ],
2660
+ "related": [
2661
+ "language vitality"
2662
+ ]
2663
+ },
2664
+ {
2665
+ "term": "glottocode",
2666
+ "also": [
2667
+ "Glottolog code",
2668
+ "glottocodes"
2669
+ ],
2670
+ "plain": "A unique identifier from the Glottolog database (like stan1293 for English) covering languages, dialects, and families. More fine-grained than ISO codes and revised continuously by linguists.",
2671
+ "mt_relevance": "Glottocodes let pipelines join typological databases precisely, even for varieties without ISO codes.",
2672
+ "citations": [
2673
+ {
2674
+ "source": "Glottolog",
2675
+ "url": "https://glottolog.org"
2676
+ }
2677
+ ],
2678
+ "related": [
2679
+ "ISO 639-3",
2680
+ "endonym"
2681
+ ]
2682
+ },
2683
+ {
2684
+ "term": "ISO 639-3",
2685
+ "also": [
2686
+ "ISO 639",
2687
+ "iso639_3",
2688
+ "language code",
2689
+ "three-letter code"
2690
+ ],
2691
+ "plain": "The international standard of three-letter codes for the world's languages (eng, crk, haw), maintained by SIL. It aims to cover every known language, living or extinct, with one code each.",
2692
+ "mt_relevance": "ISO 639-3 codes are the join keys of multilingual NLP — wrong or ambiguous codes silently corrupt datasets.",
2693
+ "citations": [
2694
+ {
2695
+ "source": "SIL ISO 639-3 Registration Authority",
2696
+ "url": "https://iso639-3.sil.org"
2697
+ }
2698
+ ],
2699
+ "related": [
2700
+ "glottocode",
2701
+ "macrolanguage"
2702
+ ]
2703
+ },
2704
+ {
2705
+ "term": "vigesimal",
2706
+ "also": [
2707
+ "base-20",
2708
+ "vigesimal system"
2709
+ ],
2710
+ "plain": "A counting system based on twenty rather than ten. Maya, Yoruba, Nahuatl, and Danish (partly) count this way — 'eighty' is literally 'four twenties' in French (quatre-vingts).",
2711
+ "mt_relevance": "Number-word translation across different bases is error-prone, and numbers are high-stakes content.",
2712
+ "example": "Surfaced on cards via numeralSystem.baseType from Numeralbank data.",
2713
+ "citations": [
2714
+ {
2715
+ "source": "Numeralbank (channumerals)",
2716
+ "url": "https://github.com/numeralbank/channumerals"
2717
+ }
2718
+ ],
2719
+ "related": [
2720
+ "classifier"
2721
+ ]
2722
+ },
2723
+ {
2724
+ "term": "kinship terms",
2725
+ "also": [
2726
+ "kin terms",
2727
+ "kinship terminology",
2728
+ "kinship system"
2729
+ ],
2730
+ "plain": "The vocabulary for family relations, which different languages slice very differently — separate words for older vs younger siblings, or for maternal vs paternal uncles. Some systems encode the speaker's own position too.",
2731
+ "mt_relevance": "English 'uncle' or 'cousin' may have no single equivalent, forcing MT to choose among precise kin terms without the needed family facts.",
2732
+ "citations": [
2733
+ {
2734
+ "source": "SIL Glossary of Linguistic Terms",
2735
+ "url": "https://glossary.sil.org/term/kinship-terminology"
2736
+ }
2737
+ ],
2738
+ "related": [
2739
+ "honorific",
2740
+ "clusivity"
2741
+ ]
2742
+ },
2743
+ {
2744
+ "term": "chrF",
2745
+ "also": [
2746
+ "chrF++",
2747
+ "chrf"
2748
+ ],
2749
+ "plain": "An automatic MT evaluation metric that scores character-level overlap between a system's output and reference translations. Because it works on characters rather than words, it is fairer to morphologically rich languages than word-based metrics.",
2750
+ "mt_relevance": "chrF is the preferred surface metric for agglutinative and polysynthetic languages where word-level matching breaks down.",
2751
+ "citations": [
2752
+ {
2753
+ "source": "Popović 2015, chrF: character n-gram F-score (ACL Anthology)",
2754
+ "url": "https://aclanthology.org/W15-3049/"
2755
+ }
2756
+ ],
2757
+ "related": [
2758
+ "COMET",
2759
+ "tokenization"
2760
+ ]
2761
+ },
2762
+ {
2763
+ "term": "COMET",
2764
+ "also": [
2765
+ "AfriCOMET",
2766
+ "comet score"
2767
+ ],
2768
+ "plain": "A neural MT evaluation metric that uses a trained multilingual model to predict human quality judgments, rather than counting surface overlap. Variants like AfriCOMET extend coverage to African languages.",
2769
+ "mt_relevance": "COMET correlates better with human judgment than surface metrics, but only for languages its underlying model has seen.",
2770
+ "citations": [
2771
+ {
2772
+ "source": "Rei et al. 2020, COMET (ACL Anthology)",
2773
+ "url": "https://aclanthology.org/2020.emnlp-main.213/"
2774
+ }
2775
+ ],
2776
+ "related": [
2777
+ "chrF"
2778
+ ]
2779
+ },
2780
+ {
2781
+ "term": "FST",
2782
+ "also": [
2783
+ "finite-state transducer",
2784
+ "finite state transducer",
2785
+ "FST-based"
2786
+ ],
2787
+ "plain": "A finite-state transducer: a rule-based computational model that maps between word forms and their grammatical analyses. For morphologically complex languages, hand-built FSTs can analyze and generate word forms that statistical systems never saw.",
2788
+ "mt_relevance": "FSTs provide reliable morphological analysis and validation for languages too data-poor to learn morphology from text alone.",
2789
+ "example": "Plains Cree (crk): the eval pack uses an FST-based semantic validator (card field evalMetrics.lyss-sem).",
2790
+ "citations": [
2791
+ {
2792
+ "source": "GiellaLT infrastructure (UiT)",
2793
+ "url": "https://giellalt.github.io"
2794
+ }
2795
+ ],
2796
+ "related": [
2797
+ "morphology",
2798
+ "paradigm"
2799
+ ]
2800
+ },
2801
+ {
2802
+ "term": "treebank",
2803
+ "also": [
2804
+ "treebanks",
2805
+ "UD treebank",
2806
+ "Universal Dependencies"
2807
+ ],
2808
+ "plain": "A corpus of sentences hand-annotated with grammatical structure (parse trees). The Universal Dependencies project maintains treebanks in a shared format for 150+ languages.",
2809
+ "mt_relevance": "Treebank existence signals serious NLP infrastructure for a language and enables syntax-aware evaluation.",
2810
+ "citations": [
2811
+ {
2812
+ "source": "Universal Dependencies",
2813
+ "url": "https://universaldependencies.org"
2814
+ }
2815
+ ],
2816
+ "related": [
2817
+ "parallel corpus",
2818
+ "tokenization"
2819
+ ]
2820
+ },
2821
+ {
2822
+ "term": "parallel corpus",
2823
+ "also": [
2824
+ "parallel text",
2825
+ "bitext",
2826
+ "parallel corpora",
2827
+ "parallel data"
2828
+ ],
2829
+ "plain": "A collection of texts paired with their translations, aligned sentence by sentence. Parallel corpora are the primary fuel for training and evaluating MT systems.",
2830
+ "mt_relevance": "The size and domain of available parallel data is the strongest single predictor of MT quality for a language pair.",
2831
+ "citations": [
2832
+ {
2833
+ "source": "OPUS, the open parallel corpus collection",
2834
+ "url": "https://opus.nlpl.eu"
2835
+ }
2836
+ ],
2837
+ "related": [
2838
+ "treebank",
2839
+ "chrF"
2840
+ ]
2841
+ },
2842
+ {
2843
+ "term": "BLEU",
2844
+ "also": [
2845
+ "sacreBLEU",
2846
+ "BLEU score",
2847
+ "corpus BLEU",
2848
+ "bleu"
2849
+ ],
2850
+ "plain": "BLEU (Bilingual Evaluation Understudy) scores a translation by how many word n-grams it shares with one or more reference translations. It is the oldest and most-cited MT metric, but because it matches whole words it is harsh and unreliable for morphologically rich and low-resource languages. Higher is better, on a 0–100 scale.",
2851
+ "mt_relevance": "BLEU is reported for comparison with published tables, but for the languages Champollion focuses on it correlates poorly with human judgement — read it next to chrF++ and COMET, never alone.",
2852
+ "citations": [
2853
+ {
2854
+ "source": "Papineni et al. 2002, BLEU (ACL Anthology)",
2855
+ "url": "https://aclanthology.org/P02-1040/"
2856
+ },
2857
+ {
2858
+ "source": "Post 2018, A Call for Clarity in Reporting BLEU Scores (sacreBLEU)",
2859
+ "url": "https://aclanthology.org/W18-6319/"
2860
+ }
2861
+ ],
2862
+ "related": [
2863
+ "chrF",
2864
+ "spBLEU",
2865
+ "TER"
2866
+ ]
2867
+ },
2868
+ {
2869
+ "term": "spBLEU",
2870
+ "also": [
2871
+ "sentencepiece BLEU",
2872
+ "FLORES BLEU",
2873
+ "spbleu"
2874
+ ],
2875
+ "plain": "spBLEU is BLEU computed after splitting text with the FLORES-200 SentencePiece tokenizer — a single fixed subword model shared across all 200 languages. That makes scores comparable across scripts and writing systems where ordinary word-level BLEU is not.",
2876
+ "mt_relevance": "spBLEU is the lingua-franca number in the NLLB / FLORES low-resource literature — handy for cross-language comparison, but still a surface metric that cannot see meaning.",
2877
+ "citations": [
2878
+ {
2879
+ "source": "NLLB Team 2022, No Language Left Behind / FLORES-200",
2880
+ "url": "https://arxiv.org/abs/2207.04672"
2881
+ }
2882
+ ],
2883
+ "related": [
2884
+ "BLEU",
2885
+ "chrF"
2886
+ ]
2887
+ },
2888
+ {
2889
+ "term": "TER",
2890
+ "also": [
2891
+ "translation edit rate"
2892
+ ],
2893
+ "plain": "Translation Edit Rate counts the edits — insertions, deletions, substitutions, and block shifts — needed to turn a system's output into the reference, as a percentage of the reference length. Unlike most metrics, LOWER is better.",
2894
+ "mt_relevance": "TER is an intuitive 'how much would a human have to fix this?' signal, but like other surface metrics it penalises valid paraphrase and rich morphology.",
2895
+ "citations": [
2896
+ {
2897
+ "source": "Snover et al. 2006, A Study of Translation Edit Rate (AMTA)",
2898
+ "url": "https://aclanthology.org/2006.amta-papers.25/"
2899
+ }
2900
+ ],
2901
+ "related": [
2902
+ "BLEU",
2903
+ "chrF"
2904
+ ]
2905
+ },
2906
+ {
2907
+ "term": "data contamination",
2908
+ "also": [
2909
+ "contamination",
2910
+ "contamination posture",
2911
+ "contamination grade",
2912
+ "contaminated",
2913
+ "relative-only",
2914
+ "relative only"
2915
+ ],
2916
+ "plain": "A test set is 'contaminated' when its sentences have likely already been seen during a model's training — for example because the corpus is public and widely scraped. A contaminated benchmark cannot prove a model is good; it may just be recalling answers it memorised. 'High' contamination means likely-seen, so the result is treated as a relative comparison between systems rather than an absolute quality claim ('relative-only').",
2917
+ "mt_relevance": "Most public benchmarks are broadly trained on, so Champollion scores them on a relative-only lane and reserves absolute rankings for sealed, never-published test sets.",
2918
+ "citations": [
2919
+ {
2920
+ "source": "Sainz et al. 2023, NLP Evaluation in Trouble: data contamination (EMNLP Findings)",
2921
+ "url": "https://aclanthology.org/2023.findings-emnlp.722/"
2922
+ }
2923
+ ],
2924
+ "related": [
2925
+ "held-out test set",
2926
+ "parallel corpus"
2927
+ ]
2928
+ },
2929
+ {
2930
+ "term": "MT complexity",
2931
+ "also": [
2932
+ "MT complexity index",
2933
+ "complexity index",
2934
+ "complexity score",
2935
+ "MT difficulty",
2936
+ "translation difficulty"
2937
+ ],
2938
+ "plain": "A Champollion-computed 0–100 estimate of how hard a language is to machine-translate, combining four signals: commercial MT API coverage, availability of parallel corpora, typological distance from well-resourced languages, and depth of linguistic documentation. Higher means harder. It is a guide to difficulty, not a measurement of any system's output.",
2939
+ "mt_relevance": "The index is a derived guide that helps prioritise effort — it is NOT a benchmark result. Measured translation quality lives on the leaderboard, keyed by (method, dataset, metric).",
2940
+ "citations": [
2941
+ {
2942
+ "source": "Joshi et al. 2020, The State and Fate of Linguistic Diversity (ACL)",
2943
+ "url": "https://aclanthology.org/2020.acl-main.560/"
2944
+ }
2945
+ ],
2946
+ "related": [
2947
+ "parallel corpus",
2948
+ "language vitality"
2949
+ ]
2950
+ },
2951
+ {
2952
+ "term": "held-out test set",
2953
+ "also": [
2954
+ "held-out",
2955
+ "held out",
2956
+ "blind test set",
2957
+ "secret test set",
2958
+ "sealed test set",
2959
+ "hidden test set"
2960
+ ],
2961
+ "plain": "A set of translation examples kept hidden from a system during development and used only to score it. If the system never saw these sentences, the score reflects genuine ability rather than memorisation. Champollion's strongest tier is a 'sealed' test set a community keeps private — a method is scored against it without anyone, including us, seeing the data.",
2962
+ "mt_relevance": "Held-out (especially sealed) test sets are the only way to make an absolute quality claim; public, contaminated sets support relative comparison only.",
2963
+ "citations": [
2964
+ {
2965
+ "source": "WMT shared tasks — blind test sets (Conference on Machine Translation)",
2966
+ "url": "https://www2.statmt.org/wmt24/"
2967
+ }
2968
+ ],
2969
+ "related": [
2970
+ "data contamination",
2971
+ "parallel corpus"
2972
+ ]
2973
+ }
2974
+ ]
2975
+ }