champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,532 @@
|
|
|
1
|
+
# Language Card Field Reference
|
|
2
|
+
|
|
3
|
+
> **Version:** 1.0
|
|
4
|
+
> **Date:** 2026-06-09
|
|
5
|
+
> **Schema:** `cli/shared/schemas/language-card.schema.json` (v2020-12)
|
|
6
|
+
> **Audience:** Contributors, enrichment script authors, downstream consumers
|
|
7
|
+
|
|
8
|
+
This document describes every field in the Champollion language card schema. For each field you will find the JSON key, type, data source(s), a plain-English description, and a real value from an existing card.
|
|
9
|
+
|
|
10
|
+
For the enrichment philosophy, provenance rules, and merge semantics see [DATA-ENRICHMENT.md](../DATA-ENRICHMENT.md).
|
|
11
|
+
For license obligations see [ATTRIBUTION.md](./ATTRIBUTION.md).
|
|
12
|
+
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
## 1. Core Identity
|
|
16
|
+
|
|
17
|
+
These fields uniquely identify a language and provide its basic naming and coding information.
|
|
18
|
+
|
|
19
|
+
| Field | Type | Source(s) | Description | Example |
|
|
20
|
+
|-------|------|-----------|-------------|---------|
|
|
21
|
+
| `code` | `string` | ISO 639-3 | Primary language identifier. ISO 639-3 three-letter code for individual languages (`fra`, `crk`), BCP 47 with region for regional variants (`por-PT`), or prefixed codes for genera/families (`genus-cree`, `family-algic`). Conlangs use `x-` prefix. **Required.** | `"crk"` (Plains Cree) |
|
|
22
|
+
| `name` | `string` | ISO 639-3 | English display name of the language. **Required.** | `"Plains Cree"` |
|
|
23
|
+
| `nativeName` | `string \| null` | Wikidata P1705 | Name in the language's own script (endonym). For dual orthographies, both forms separated by ` / `. Must render in native script — never romanized. Null for conlangs without established endonyms. | `"nêhiyawêwin / ᓀᐦᐃᔭᐍᐏᐣ"` (crk) |
|
|
24
|
+
| `bcp47` | `string \| null` | LinguaMeta | BCP 47 language tag. Null for languages without a registered subtag (430+ cards have null). | `"yo-Latn-NG"` (yor), `"es"` (spa) |
|
|
25
|
+
| `iso639_1` | `string \| null` | ISO 639-3 | ISO 639-1 two-letter code. Null if none exists. Pattern: `^[a-z]{2}$`. | `"es"` (spa), `null` (crk) |
|
|
26
|
+
| `iso639_3` | `string \| null` | ISO 639-3 | ISO 639-3 three-letter code. Null for conlangs. Pattern: `^[a-z]{3}$`. | `"yor"` (Yoruba) |
|
|
27
|
+
| `glottocode` | `string \| null` | Glottolog 5.3 | Glottolog identifier for cross-referencing the Glottolog language database. Pattern: `^[a-z]{4}[0-9]{4}$`. | `"plai1258"` (crk), `"stan1288"` (spa) |
|
|
28
|
+
| `isoScope` | `string \| null` | ISO 639-3 | ISO 639-3 scope: `"I"` (individual), `"M"` (macrolanguage), `"S"` (special). Null if not in ISO 639-3. | `"I"` (spa, yor), `null` (crk) |
|
|
29
|
+
| `isoType` | `string \| null` | ISO 639-3 | ISO 639-3 type: `"L"` (living), `"E"` (extinct), `"A"` (ancient), `"H"` (historical), `"C"` (constructed), `"S"` (special). Null if not in ISO 639-3. | `"L"` (cmn, yor) |
|
|
30
|
+
| `script` | `string \| null` | Wikidata P282, LinguaMeta | Primary ISO 15924 script code. Pattern: `^[A-Z][a-z]{3}$`. Null for unwritten languages or when unknown (1,400+ null). | `"Cans"` (crk), `"Latn"` (spa, yor), `"Hans"` (cmn) |
|
|
31
|
+
| `scripts` | `array \| null` | LinguaMeta | All scripts used by this language. Each entry has `code` (ISO 15924), `name` (human-readable), and `primary` (boolean). Many languages use multiple scripts. | `[{"code": "Cans", "name": "Unified Canadian Aboriginal Syllabics", "primary": true}, {"code": "Latn", "name": "Latin (SRO)", "primary": false}]` (crk) |
|
|
32
|
+
| `dir` | `string \| null` | Derived from script | Text directionality: `"ltr"`, `"rtl"`, or `null` for unwritten languages (620+ null). | `"ltr"` (all example cards) |
|
|
33
|
+
| `aliases` | `array` | Manual / LinguaMeta | Alternative locale codes that resolve to this card (e.g., `"no"` → `"nb"`, `"iw"` → `"he"`). | `["zh", "zh-CN", "zh-Hans", "zho"]` (cmn), `["es"]` (spa) |
|
|
34
|
+
| `alternateNames` | `array` | Glottolog, WALS, LinguaMeta, ElCat | Alternative English names for search and display. | `["Cree (Plains)", "ᓀᐦᐃᔭᐍᐏᐣ", "Cree"]` (crk) |
|
|
35
|
+
|
|
36
|
+
---
|
|
37
|
+
|
|
38
|
+
## 2. Classification
|
|
39
|
+
|
|
40
|
+
Genealogical classification placing the language in the world's language family tree.
|
|
41
|
+
|
|
42
|
+
| Field | Type | Source(s) | Description | Example |
|
|
43
|
+
|-------|------|-----------|-------------|---------|
|
|
44
|
+
| `classification` | `object \| null` | Glottolog 5.3 | Full genealogical classification. Contains `family`, `familyGlottocode`, `genus`, `genusGlottocode`, and `ancestry`. Null for conlangs. | See sub-fields below |
|
|
45
|
+
| `classification.family` | `string` | Glottolog 5.3 | Top-level language family. | `"Algic"` (crk), `"Indo-European"` (spa), `"Niger-Congo"` (yor), `"Sino-Tibetan"` (cmn) |
|
|
46
|
+
| `classification.familyGlottocode` | `string` | Glottolog 5.3 | Glottocode of the top-level family node. | `"algi1248"` (crk), `"indo1319"` (spa) |
|
|
47
|
+
| `classification.genus` | `string` | Glottolog 5.3 / WALS | WALS-style genus — lowest-level grouping sharing runtime properties. Used for genus card inheritance. | `"Plains Creeic"` (crk), `"Castilic"` (spa), `"Yoruboid"` (yor), `"Mandarinic"` (cmn) |
|
|
48
|
+
| `classification.genusGlottocode` | `string` | Glottolog 5.3 | Glottocode of the genus node. | `"plai1264"` (crk), `"cast1243"` (spa) |
|
|
49
|
+
| `classification.ancestry` | `array` | Glottolog 5.3 | Full ancestry chain from top-level family to genus. | `["Algic", "Algonquian-Blackfoot", "Algonquian", "Cree-Montagnais-Naskapi", "Cree", "Plains Creeic"]` (crk) |
|
|
50
|
+
| `macroarea` | `string \| null` | Derived from coordinates / Glottolog | Glottolog macroarea. One of: `"Africa"`, `"Australia"`, `"Eurasia"`, `"North America"`, `"Papunesia"`, `"South America"`, or `null`. | `"Africa"` (yor), `"South America"` (spa), `"Eurasia"` (cmn), `null` (crk) |
|
|
51
|
+
| `isIsolate` | `boolean` | Glottolog 5.3 | Whether this language has no known genetic relatives. | `false` (all example cards) |
|
|
52
|
+
| `macrolanguage` | `string \| null` | LinguaMeta | ISO 639-3 macrolanguage code if part of a macrolanguage umbrella. Null otherwise. Pattern: `^[a-z]{3}$`. | `"zho"` (cmn), `"cre"` (crk), `null` (spa, yor) |
|
|
53
|
+
| `extends` | `string \| null` | Manual | Locale code of another card/family this card inherits from. | `"macrolanguage-zho"` (cmn), `"genus-cree"` (crk), `"genus-romance"` (spa), `null` (yor) |
|
|
54
|
+
|
|
55
|
+
---
|
|
56
|
+
|
|
57
|
+
## 3. Vitality & Speakers
|
|
58
|
+
|
|
59
|
+
Endangerment status and speaker population data.
|
|
60
|
+
|
|
61
|
+
| Field | Type | Source(s) | Description | Example |
|
|
62
|
+
|-------|------|-----------|-------------|---------|
|
|
63
|
+
| `vitality` | `object \| null` | LinguaMeta, ElCat, UNESCO | Language vitality and endangerment status. Used for LRL prioritization. | See sub-fields below |
|
|
64
|
+
| `vitality.unescoStatus` | `string \| null` | LinguaMeta / UNESCO | UNESCO classification: `"safe"`, `"vulnerable"`, `"definitely-endangered"`, `"severely-endangered"`, `"critically-endangered"`, `"extinct"`. | `"safe"` (spa, cmn, yor), `"severely-endangered"` (crk) |
|
|
65
|
+
| `vitality.egids` | `string \| null` | Manual / LinguaMeta | Ethnologue EGIDS level (0–10). | `"6b"` (crk), `"2"` (yor), `null` (spa) |
|
|
66
|
+
| `vitality.speakerCount` | `string \| number \| null` | Derived from speakerEstimates | Approximate total speaker count. Uses ranges: `"~50M"`, `"20K-25K"`. | `"~490M L1"` (spa), `20000` (crk), `"~47M (L1 + L2)"` (yor) |
|
|
67
|
+
| `vitality.trend` | `string \| null` | Derived / ElCat / UNESCO | Speaker population trend: `"growing"`, `"stable"`, `"declining"`, `"rapidly-declining"`, `"moribund"`. | `"growing"` (spa, yor), `"stable"` (cmn), `"declining"` (crk) |
|
|
68
|
+
| `vitality.notes` | `string \| null` | Manual | Context notes about endangerment. | `"Intergenerational transmission breaking down in most communities..."` (crk) |
|
|
69
|
+
| `speakerEstimates` | `array` | Wikidata, LinguaMeta, Ethnologue | Speaker count estimates from multiple sources. Each entry has `source`, `count` (integer), optional `date` (ISO 8601), and optional `type` (`"L1"`, `"L2"`, `"total"`). | `[{"source": "wikidata", "count": 37800000, "date": "2026-06-07"}, {"count": 28000000, "source": "linguameta", "date": "2024"}]` (yor) |
|
|
70
|
+
| `dialectCount` | `integer \| null` | Glottolog 5.3 | Number of recognized dialects or varieties. | `3` (crk), `22` (yor), `31` (cmn), `38` (spa) |
|
|
71
|
+
|
|
72
|
+
---
|
|
73
|
+
|
|
74
|
+
## 4. Typological Profile
|
|
75
|
+
|
|
76
|
+
Structural/typological features describing the grammar and phonology of the language.
|
|
77
|
+
|
|
78
|
+
### 4.1 `typologicalProfile`
|
|
79
|
+
|
|
80
|
+
| Field | Type | Source(s) | Description | Example |
|
|
81
|
+
|-------|------|-----------|-------------|---------|
|
|
82
|
+
| `typologicalProfile` | `object \| null` | Grambank, WALS, AUTOTYP, WACL | Typological features. Auto-populated by `enrich-grambank-typology.mjs`. | See sub-fields below |
|
|
83
|
+
| `.featuresDocumented` | `integer` | Grambank | Number of Grambank features documented. | `195` (cmn), `174` (crk), `155` (spa), `140` (yor) |
|
|
84
|
+
| `.featuresCoverage` | `number` | Grambank | Fraction of Grambank features documented (0.0–1.0). | `1` (cmn), `0.89` (crk) |
|
|
85
|
+
| `.wordOrderDominant` | `string \| null` | Grambank / WALS | Dominant word order. | `"SVO"` (spa, yor), `"SOV"` (cmn) |
|
|
86
|
+
| `.hasDefiniteArticle` | `boolean \| null` | Grambank | Whether the language has a definite article. | `true` (spa), `false` (cmn, crk, yor) |
|
|
87
|
+
| `.hasIndefiniteArticle` | `boolean \| null` | Grambank | Whether the language has an indefinite article. | `true` (spa), `false` (cmn, crk, yor) |
|
|
88
|
+
| `.hasGenderSystem` | `boolean \| null` | Grambank / WALS | Whether the language has a grammatical gender system. | `true` (spa, cmn, crk), `false` (yor) |
|
|
89
|
+
| `.hasCaseMorphology` | `boolean \| null` | Grambank | Whether the language has morphological case marking. | `true` (cmn, crk), `false` (spa, yor) |
|
|
90
|
+
| `.hasEvidentiality` | `boolean \| null` | Grambank | Whether the language has grammatical evidentiality. | `false` (all example cards) |
|
|
91
|
+
| `.hasToneSystem` | `boolean \| null` | Grambank | Whether the language uses lexical or grammatical tone. | `true` (crk, yor), `false` (cmn, spa) |
|
|
92
|
+
| `.source` | `string \| null` | — | Data source and version. | `"grambank-1.0.3"` (cmn, crk), `"wals-2024"` (spa, yor) |
|
|
93
|
+
| `.headMarking` | `boolean \| null` | AUTOTYP | Whether the language uses head-marking for S/A arguments. | `true` (crk, spa), `false` (cmn, yor) |
|
|
94
|
+
| `.dependentMarking` | `boolean \| null` | AUTOTYP | Whether the language uses dependent-marking. | `false` (all example cards) |
|
|
95
|
+
| `.hasNumeralClassifiers` | `boolean \| null` | WALS 55A | Whether the language uses numeral classifiers. | `true` (cmn), `false` (crk, yor) |
|
|
96
|
+
| `.numeralClassifierType` | `string \| null` | WALS 55A | Classifier type: `"Obligatory"`, `"Optional"`, `"Absent"`. | `"Obligatory"` (cmn), `"Absent"` (crk, yor) |
|
|
97
|
+
| `.caseCount` | `integer \| null` | WALS 49A | Number of grammatical cases. | `0` (cmn, spa, yor) |
|
|
98
|
+
| `.genderCount` | `string \| integer \| null` | WALS 30A | Number of grammatical genders. | `"2"` (spa), `0` (cmn, yor) |
|
|
99
|
+
| `.inflectionalStrategy` | `string \| null` | WALS 26A | Prefixing vs. suffixing strategy. | `"Strongly suffixing"` (cmn, spa), `"Equal prefixing and suffixing"` (crk), `"Little affixation"` (yor) |
|
|
100
|
+
| `.morphologicalSynthesis` | `string \| null` | Champollion-derived (derive-morphological-synthesis.mjs) | Synthesis-degree enum: `"analytic"`, `"synthetic"`, `"polysynthetic"`. Derived STRICTLY from cited on-card signals — the WALS 22A categories-per-word value (`encyclopedic.typology.verbSynthesis` / `linguisticChallenges.morphologicalComplexity`; 0–1 → analytic, 2–7 → synthetic, 8+ → polysynthetic), WALS 26A `"Little affixation"` (analytic), and cited polysynthesis prose (polysynthetic). Absent when signals are missing or conflict. Provenance MUST be a `derived:` stamp (lint R6). | `"polysynthetic"` (crk) |
|
|
101
|
+
| `.ordinalNumerals` | `string \| null` | WALS 53A | How ordinal numerals are formed. | `"One-th, two-th, three-th"` (cmn, yor), `"First, second, three-th"` (spa) |
|
|
102
|
+
| `.obligatoryNumberMarking` | `boolean \| null` | Grambank GB024 | Whether number marking on nouns is obligatory. | `true` (cmn), `false` (crk) |
|
|
103
|
+
| `.hasNounClassifiers` | `boolean \| null` | Grambank GB522 | Whether the language has noun classifiers. | `true` (cmn, crk) |
|
|
104
|
+
| `.classifierLanguage` | `boolean \| null` | WACL | Whether classified as a classifier language. | `true` (all example cards) |
|
|
105
|
+
| `.valencyPatterns` | `boolean \| null` | ValPaL | Whether valency pattern data is available. | `true` (cmn, yor) |
|
|
106
|
+
|
|
107
|
+
### 4.2 `phonologicalInventory`
|
|
108
|
+
|
|
109
|
+
| Field | Type | Source(s) | Description | Example |
|
|
110
|
+
|-------|------|-----------|-------------|---------|
|
|
111
|
+
| `phonologicalInventory` | `object \| null` | PHOIBLE 2.0 | Phoneme inventory. Auto-populated by `enrich-phoible-phonemes.mjs`. Null if undocumented. | `null` (crk) |
|
|
112
|
+
| `.consonants` | `integer` | PHOIBLE | Number of consonant phonemes. | `25` (cmn), `19` (spa), `18` (yor) |
|
|
113
|
+
| `.vowels` | `integer` | PHOIBLE | Number of vowel phonemes. | `17` (cmn), `7` (spa), `11` (yor) |
|
|
114
|
+
| `.tones` | `integer` | PHOIBLE | Number of tonal contrasts (0 for non-tonal). | `2` (cmn), `0` (spa, yor) |
|
|
115
|
+
| `.totalPhonemes` | `integer` | PHOIBLE | Total phoneme count (consonants + vowels + tones). | `42` (cmn), `26` (spa), `29` (yor) |
|
|
116
|
+
| `.isTonal` | `boolean` | PHOIBLE | Whether the language uses lexical/grammatical tone. | `true` (cmn), `false` (spa, yor) |
|
|
117
|
+
| `.inventorySize` | `string \| null` | PHOIBLE | Qualitative classification: `"small"`, `"moderately-small"`, `"average"`, `"moderately-large"`, `"large"`. | `"large"` (cmn), `"average"` (spa, yor) |
|
|
118
|
+
| `.source` | `string \| null` | — | Data source. | `"phoible-2.0"` (all) |
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
## 5. Orthography & Writing
|
|
123
|
+
|
|
124
|
+
Fields describing the language's writing system, keyboard availability, and script conversion.
|
|
125
|
+
|
|
126
|
+
| Field | Type | Source(s) | Description | Example |
|
|
127
|
+
|-------|------|-----------|-------------|---------|
|
|
128
|
+
| `orthographicStatus` | `string \| null` | Derived (CLDR + script + keyboard) | Status of the writing system: `"has-orthography"`, `"no-orthography"`, `"disputed"`, `"emerging"`, `"historical-only"`, or `null`. | `"has-orthography"` (yor), `"developing"` (crk, spa), `"unwritten"` (cmn) |
|
|
129
|
+
| `script` | `string \| null` | Wikidata P282, LinguaMeta | See §1 Core Identity above. | `"Cans"` (crk), `"Latn"` (spa) |
|
|
130
|
+
| `scriptUnicodeName` | `string \| null` | Derived via enrich-script-unicode-names.mjs | Unicode script block name mapped from the ISO 15924 `script` code. Used by `code_switching` metric plugin to detect script mixing. | `"Canadian_Aboriginal"` (crk), `"Latin"` (spa, yor), `"CJK"` (cmn) |
|
|
131
|
+
| `keyboardSupport` | `object \| null` | Keyman API | Keyboard layout availability. Null if no data. | `{"keymanKeyboards": 1, "keyboardNames": ["Pan Africa Mnemonic (SIL)"], "source": "keyman-api"}` (yor) |
|
|
132
|
+
| `keyboardSupport.keymanKeyboards` | `integer` | Keyman API | Number of Keyman keyboard layouts available. | `1` (cmn, yor) |
|
|
133
|
+
| `keyboardSupport.keyboardNames` | `array` | Keyman API | Names of available keyboards. | `["Pinyin Mandarin"]` (cmn) |
|
|
134
|
+
| `scriptConverter` | `string \| null` | Manual | Key in the `SCRIPT_CONVERTERS` registry (from `scripts.js`) if the language has a deterministic script conversion step. Null if not applicable. | `"crk"` (crk), `null` (spa, cmn, yor) |
|
|
135
|
+
| `orthographies` | `array \| null` | Derived (derive-orthographies.mjs) from `scripts[]` + `orthographicStatus` + `shared/curated-orthography-conventions.json` | Structured writing-convention entries, one per script: `script` (ISO 15924, required), optional `scheme` (named convention, e.g. `"SRO"`), optional `longVowelMarking` (`"circumflex"`, `"macron"`, …), optional `canonicalForMt` (the pipeline's working form — derived for sole-script written cards, curated otherwise; the `primary` display flag is NOT this signal), `source` (required). Optional keys are omitted when unknown — never guessed. | `[{"script": "Cans", "canonicalForMt": false, ...}, {"script": "Latn", "scheme": "SRO", "longVowelMarking": "circumflex", "canonicalForMt": true, ...}]` (crk) |
|
|
136
|
+
|
|
137
|
+
---
|
|
138
|
+
|
|
139
|
+
## 6. Corpus & Resource Availability
|
|
140
|
+
|
|
141
|
+
Information about available corpora, NLP resources, digital presence, and archive holdings.
|
|
142
|
+
|
|
143
|
+
### 6.1 `corpusAvailability`
|
|
144
|
+
|
|
145
|
+
| Field | Type | Source(s) | Description | Example |
|
|
146
|
+
|-------|------|-----------|-------------|---------|
|
|
147
|
+
| `corpusAvailability` | `object \| null` | OPUS, Lexibank, UD, ASJP, etc. | Available parallel and monolingual corpora. | See sub-fields below |
|
|
148
|
+
| `.opus` | `object \| null` | OPUS NLP API | OPUS parallel corpus data. | `{"corpora": 164, "corpusNames": ["NLLB", "CCMatrix", "OpenSubtitles", "MultiParaCrawl", "ParaCrawl"], "languagePairs": 665, "totalAlignmentPairs": 7936408771, "source": "opus-nlpl-api"}` (spa) |
|
|
149
|
+
| `.opus.corpora` | `integer` | OPUS | Number of OPUS corpora containing this language. | `164` (spa), `12` (yor), `1` (cmn, crk) |
|
|
150
|
+
| `.opus.corpusNames` | `array` | OPUS | Names of the most significant corpora. | `["Tatoeba"]` (cmn, crk) |
|
|
151
|
+
| `.lexibank` | `object \| null` | Lexibank | Lexical dataset availability. Has `datasets` (count) and `totalForms`. | `{"datasets": 6, "totalForms": 3582}` (spa), `null` (crk) |
|
|
152
|
+
| `.ud` | `object \| null` | Universal Dependencies | Treebank availability with `treebanks` (count), `treebankNames`, and `source`. | `{"treebanks": 7, "treebankNames": ["GSD", "GSDSimp", "CFL", "HK", "PUD"], "source": "universal-dependencies"}` (cmn) |
|
|
153
|
+
| `.asjpWordlists` | `integer` | ASJP | Number of ASJP basic vocabulary word lists. | `184` (cmn), `2` (spa), `21` (yor), `1` (crk) |
|
|
154
|
+
| `.asjpForms` | `integer` | ASJP | Number of ASJP lexical forms. | `9516` (cmn), `226` (spa), `1440` (yor), `39` (crk) |
|
|
155
|
+
| `.wiktionaryStructuredDump` | `boolean` | Kaikki/Wiktextract | Whether a structured Wiktionary dump is available. | `true` (cmn, spa, yor) |
|
|
156
|
+
| `.openMultilingualWordnet` | `boolean` | OMW | Whether linked wordnet data is available. | `true` (cmn, spa) |
|
|
157
|
+
| `.unimorphParadigms` | `boolean` | UniMorph | Whether normalized morphological paradigms are available. | `true` (spa) |
|
|
158
|
+
| `.huggingFaceDatasets` | `integer` | HuggingFace | Number of HuggingFace datasets tagged for this language. | `32` (spa), `7` (yor), `4` (cmn), `2` (crk) |
|
|
159
|
+
| `.diachronicAtlas` | `boolean` | DIACL | Whether diachronic comparative lexical data is available. | `true` (cmn, spa) |
|
|
160
|
+
| `.intercontinentalDictionarySeries` | `boolean` | IDS | Whether IDS concept-aligned dictionary data is available. | `true` (spa) |
|
|
161
|
+
| `.semanticShiftDatabase` | `boolean` | DatSemShift | Whether semantic shift pattern data is available. | `true` (cmn, spa, yor) |
|
|
162
|
+
| `.lexibankDatasets` | `array` | Lexibank (batch) | Names of Lexibank family-level datasets covering this language. | `["dyenindoeuropean", "ielexfinal", "joophonosemantic", ...]` (spa) |
|
|
163
|
+
|
|
164
|
+
### 6.2 `digitalPresence`
|
|
165
|
+
|
|
166
|
+
| Field | Type | Source(s) | Description | Example |
|
|
167
|
+
|-------|------|-----------|-------------|---------|
|
|
168
|
+
| `digitalPresence` | `object \| null` | Wikipedia, Tatoeba, Common Voice, Wikimedia Incubator | Digital presence indicators. | See sub-fields below |
|
|
169
|
+
| `.wikipedia` | `object \| null` | Wikimedia SiteMatrix + SiteInfo | Wikipedia edition data: `code`, `url`, `articles`, `activeUsers`, `totalEdits`. | `{"code": "es", "url": "https://es.wikipedia.org", "articles": 2118142, "activeUsers": 43996, ...}` (spa) |
|
|
170
|
+
| `.tatoeba` | `object \| null` | Tatoeba | Sentence count. | `{"sentences": 443270, "source": "tatoeba"}` (spa), `{"sentences": 52, "source": "tatoeba"}` (crk) |
|
|
171
|
+
| `.commonVoice` | `object \| null` | Common Voice 20.0 | Speech data: `validatedHours`, `totalHours`, `speakers`, `sentences`, `locale`. | `{"validatedHours": 6.5, "totalHours": 9.2, "speakers": 135, "sentences": 5419, "locale": "yo", ...}` (yor) |
|
|
172
|
+
| `.incubatorWikiPages` | `integer` | Wikimedia Incubator | Page count in Wikimedia Incubator test wiki (for languages without full Wikipedia). | `9` (crk), `1` (cmn) |
|
|
173
|
+
|
|
174
|
+
### 6.3 `resources`
|
|
175
|
+
|
|
176
|
+
| Field | Type | Source(s) | Description | Example |
|
|
177
|
+
|-------|------|-----------|-------------|---------|
|
|
178
|
+
| `resources` | `object \| null` | Manual, Masakhane, ABVD, NorthEuraLex | NLP resources: corpora, models, FSTs, tools. | See sub-fields below |
|
|
179
|
+
| `.fsts` | `array` | Manual / GiellaLT | Morphological analyzers and FSTs. Each entry has `name`, `type`, `url`, optional `notes`, and optional `install` metadata for automated download. **See §6.3.1 — getting this field wrong fails silently.** | `[{"name": "GiellaLT Plains Cree FST (lang-crk)", "type": "morphological-analyzer", "install": {"repo": "giellalt/lang-crk", "format": "giellalt-nightly-apt", "aptPool": ".../g/giella-crk/", "debFile": "giella-crk_0.2.0+g4278~e1f96fea-1~sid1_all.deb", "debSha256": "de10b471…", "langCommit": "e1f96fea…"}}]` (crk) |
|
|
180
|
+
| `.corpora` | `array` | Manual / OPUS | Parallel and monolingual corpora. Each entry has `name`, `type` (`"parallel"`, `"monolingual"`, `"speech"`, `"nmt"`), optional `url`, `size`, `pairLanguages`, `license`, `domain`, `exposure`. | `[{"name": "MENYO-20k", "type": "parallel", "size": "~20K pairs", "domain": "mixed", "exposure": "open-web"}]` (yor) |
|
|
181
|
+
| `.models` | `array` | Manual | Pretrained NLP/MT models. Each entry has `name`, `url`, `type`. | `[{"name": "NLLB-200 (spa_Latn)", "type": "nmt", ...}]` (spa) |
|
|
182
|
+
| `.tools` | `array` | Manual | Other NLP tools (tokenizers, diacritic restorers, etc.). | `[{"name": "jieba (Chinese text segmentation)", "type": "tokenizer", ...}]` (cmn) |
|
|
183
|
+
| `.lexical` | `array` | ABVD, NorthEuraLex (via Lexibank) | Lexical databases with `type: "lexical-database"`, `name`, `description`, `source`. | `[{"name": "ABVD", "description": "102 concepts documented", ...}, {"name": "NorthEuraLex", "description": "1010/1016 concepts documented", ...}]` (cmn) |
|
|
184
|
+
| `.nlp` | `array` | Masakhane, IndicNLP, AmericasNLP | NLP community benchmark indicators with `type: "nlp-benchmark"`. | `[{"name": "Masakhane", "description": "MT, NER benchmarks for African languages", ...}]` (yor) |
|
|
185
|
+
| `.dictionaries` | `array` | derive-dictionaries.mjs (promotes `encyclopedic.resources.dictionaries` + dictionary-shaped `resources.lexical` entries; flags from `shared/curated-dictionary-flags.json`) | Schematized dictionary/lexical-database pointers: `name` (required), `url`, optional `license`, optional `machineReadable`, optional `redistributable` (`false` = pointer-only, content must never be copied/redistributed), `source`. Existence pointers only — never dictionary content, never scores. | `[{"name": "itwêwina (Plains Cree Dictionary)", "url": "https://itwewina.altlab.app/", "machineReadable": true, "redistributable": false, ...}]` (crk) |
|
|
186
|
+
| `.grammars` | `array` | enrich-grammars-from-glottolog.mjs (Glottolog 5.3 MED citation records, pinned dump) | Bibliographic reference grammars — the MED-best few (≤3), filtered to grammar-typed records (`hhtype` grammar/grammar_sketch): optional `author`, optional `year`, `title` (required), `url` (stable Glottolog reference URL), `type`. `documentationDepth.med` proves a grammar exists; these entries name it. Citation metadata only. | `[{"author": "Edwards, Mary", "year": 1961, "title": "Cree: an intensive language course", "url": "https://glottolog.org/resource/reference/id/701784", "type": "grammar"}]` (crk) |
|
|
187
|
+
|
|
188
|
+
### 6.3.1 `resources.fsts[].install` — FST install channels
|
|
189
|
+
|
|
190
|
+
> ⚠️ **GitHub Releases is NOT GiellaLT/ALTLab's distribution channel.** They ship
|
|
191
|
+
> continuously; the release tags on `lang-*` repos are **vestigial**.
|
|
192
|
+
> **"No release since YEAR" does not mean "no update since YEAR."**
|
|
193
|
+
>
|
|
194
|
+
> `crk` sat on lang-crk's *newest* release tag — `fst-v2021.7.8` (2021) — for five
|
|
195
|
+
> years. The cost was invisible until someone measured it: the 2021 build uses `y`
|
|
196
|
+
> on the analysis side, current lang-crk uses **`ý`**, so the generator returned
|
|
197
|
+
> `None` **silently** for every ý-lemma. **4,989 of 28,268 lemmas (17.6%) were
|
|
198
|
+
> ungeneratable.** Surfaces were byte-identical either way (`ayisiyiniw` both
|
|
199
|
+
> times), so nothing ever looked broken. A stale FST pin does not throw.
|
|
200
|
+
|
|
201
|
+
| `format` | Channel | Pin | Use when |
|
|
202
|
+
|---|---|---|---|
|
|
203
|
+
| `giellalt-nightly-apt` | GiellaLT/Apertium nightly apt pool (`apertium.projectjj.com`) | `debFile` (carries the upstream commit) + `debSha256` | **Default for GiellaLT languages.** This is where upstream actually ships. |
|
|
204
|
+
| `legacy-zip` | GitHub Releases zip | `releaseTag` | Only when upstream genuinely publishes usable releases — and only after checking the tag is not stale. |
|
|
205
|
+
| `divvun-macos-pkg` | Divvun speller `.pkg` from a GitHub release | `releaseTag` + `bundlePattern` | Divvun-packaged spellers. |
|
|
206
|
+
| `divvun` | — | — | Divvun manager required; the installer refuses and tells the user. |
|
|
207
|
+
| `manual` | — | — | No automated path exists. |
|
|
208
|
+
|
|
209
|
+
**`giellalt-nightly-apt` fields:** `aptPool` (pool directory URL), `debFile` (exact
|
|
210
|
+
filename — **this is the pin**), `debSha256` (verified *before* extraction; install
|
|
211
|
+
aborts on mismatch), `langCommit` (upstream source commit), `include` (exact `.hfstol`
|
|
212
|
+
basenames to install), `stripSuffix` (e.g. `-giellaltbuild`).
|
|
213
|
+
|
|
214
|
+
**Three rules learned the hard way:**
|
|
215
|
+
|
|
216
|
+
1. **Lexicon and binary move together, in one commit.** If a project extracts a
|
|
217
|
+
lexicon from `lang-X` source, the binary must be pinned to the *same* commit.
|
|
218
|
+
`crk`'s `data/lexc_source/*.lexc` are byte-identical to `e1f96fea`, so the card
|
|
219
|
+
pins `e1f96fea`. That is the whole reason the ý-skew existed.
|
|
220
|
+
2. **`include` explicitly; never glob everything.** A GiellaLT `.deb` ships ~31
|
|
221
|
+
transducers of several **kinds** (`strict` / `relaxed` / `-gt-desc` descriptive /
|
|
222
|
+
`-gt-norm` normative / `.Cans` syllabics). Installing all of them wastes ~250MB
|
|
223
|
+
and invites comparing across kinds, which is **never valid** — it once produced a
|
|
224
|
+
fictitious "16.7% regression rate" that was really 2.1%.
|
|
225
|
+
3. **Newer is not automatically better.** Re-pinning `crk` from 2021 → `e1f96fea`
|
|
226
|
+
*lost* `namôya`, `mwêstas`, `nisis`, `okîsikâw` (upstream deleted them). Any
|
|
227
|
+
re-pin needs a regression run, not just a version bump.
|
|
228
|
+
|
|
229
|
+
**Nightly pools are pruned.** A pinned `debFile` can eventually 404. When it does,
|
|
230
|
+
re-pin deliberately and re-verify — **do not fall back to a release tag.**
|
|
231
|
+
|
|
232
|
+
### 6.4 `archivePresence`
|
|
233
|
+
|
|
234
|
+
| Field | Type | Source(s) | Description | Example |
|
|
235
|
+
|-------|------|-----------|-------------|---------|
|
|
236
|
+
| `archivePresence` | `object \| null` | PARADISEC, OLAC, ELAR, AILLA, Rosetta Project, Kaipuleohone | Presence in language documentation archives. Null if not present in any archive. | See sub-fields below |
|
|
237
|
+
| `.paradisec` | `object \| null` | PARADISEC (OAI-PMH) | PARADISEC holdings: `itemCount`, `collectionCount`, `mediaTypes`. | `{"itemCount": 485, "collectionCount": null, "mediaTypes": ["audio", "image", "text", "video"]}` (cmn) |
|
|
238
|
+
| `.olacResourceCount` | `integer` | OLAC Aggregator | Total resource count across OLAC archives. | `2624` (spa), `168` (cmn), `33` (yor), `13` (crk) |
|
|
239
|
+
| `.olacArchiveCount` | `integer` | OLAC Aggregator | Number of distinct OLAC archives holding resources. | `6` (spa), `4` (cmn, yor), `2` (crk) |
|
|
240
|
+
| `.olacArchives` | `array` | OLAC Aggregator | Archive domain names. | `["ethnologue.com", "gial.edu", "refdb.wals.info", "www.mpi.nl"]` (cmn, yor) |
|
|
241
|
+
| `.rosettaProjectItems` | `integer` | Rosetta Project (Internet Archive) | Number of items in the Rosetta Project. | `151` (spa), `2` (cmn), `1` (crk, yor) |
|
|
242
|
+
| `.kaipuleohoneItems` | `integer` | Kaipuleohone (UH) | Number of items in the Kaipuleohone archive. | `13` (spa) |
|
|
243
|
+
| `.ailla` | `boolean` | AILLA | Whether the language has entries in the Archive of Indigenous Languages of Latin America. | `true` (cmn, spa, yor) |
|
|
244
|
+
|
|
245
|
+
---
|
|
246
|
+
|
|
247
|
+
## 7. Evaluation
|
|
248
|
+
|
|
249
|
+
Fields for MT evaluation — benchmark datasets, metric model support, pipeline readiness.
|
|
250
|
+
|
|
251
|
+
### 7.1 `evalDatasets`
|
|
252
|
+
|
|
253
|
+
| Field | Type | Source(s) | Description | Example |
|
|
254
|
+
|-------|------|-----------|-------------|---------|
|
|
255
|
+
| `evalDatasets` | `array` | Manual | IDs of evaluation datasets from the companion eval harness (the MT Eval Arena). Metadata only — not consumed at runtime. | `["edtekla-dev-v1"]` (crk), `["flores-plus-devtest"]` (spa, yor), `[]` (cmn) |
|
|
256
|
+
|
|
257
|
+
### 7.2 `pipelineReadiness`
|
|
258
|
+
|
|
259
|
+
| Field | Type | Source(s) | Description | Example |
|
|
260
|
+
|-------|------|-----------|-------------|---------|
|
|
261
|
+
| `pipelineReadiness` | `object \| null` | Derived | Readiness assessment for the Champollion FST-gated translation pipeline. | See sub-fields below |
|
|
262
|
+
| `.tier` | `string` | Derived | Readiness tier: `"tier-1-ready"`, `"tier-2-feasible"`, `"tier-3-buildable"`, `"watch-list"`, `"not-applicable"`. (In practice, computed values include `"strong"`, `"good"`, `"low"`.) | `"strong"` (spa, yor), `"good"` (cmn), `"low"` (crk) |
|
|
263
|
+
| `.score` | `integer` | Derived | Numeric readiness score. | `85` (spa, yor), `60` (cmn), `25` (crk) |
|
|
264
|
+
| `.hasFST` | `boolean` | Derived | Whether a usable FST/morphological analyzer exists. | (present in schema; some cards use `.components` instead) |
|
|
265
|
+
| `.hasParallelCorpus` | `boolean` | Derived | Whether a parallel corpus >10K pairs exists. | (present in schema) |
|
|
266
|
+
| `.hasEvalBenchmark` | `boolean` | Derived | Whether a held-out evaluation benchmark exists (not FLORES — contaminated). | (present in schema) |
|
|
267
|
+
| `.components` | `object` | Derived | Component availability flags (`commonVoice`, `wikipedia`, `opus`, `lexibank`, `ud`, `keyboard`, `resources`, `googleTranslate`, `deepl`, `nllb`). | `{"commonVoice": false, "wikipedia": true, "opus": true, ...}` (cmn) |
|
|
268
|
+
| `.blockers` | `array \| null` | Derived | Key blockers preventing higher tier. | (present in schema) |
|
|
269
|
+
|
|
270
|
+
### 7.3 `methodSupport`
|
|
271
|
+
|
|
272
|
+
| Field | Type | Source(s) | Description | Example |
|
|
273
|
+
|-------|------|-----------|-------------|---------|
|
|
274
|
+
| `methodSupport` | `object` | API verification | Which translation APIs/methods support this language. Single Source of Truth for method availability. Each key is a method, value is an object with `supported` (boolean) plus optional metadata. | See sub-fields below |
|
|
275
|
+
| `.googleTranslate` | `object` | API verification | Google Cloud Translation / Google Translate. | `{"supported": true, "verifiedDate": "2026-06-07"}` (spa, yor), `{"supported": false}` (crk) |
|
|
276
|
+
| `.deepl` | `object` | API verification | DeepL Translation API. Includes optional `formality` (boolean) for formality parameter support. | `{"supported": true, "formality": true}` (spa), `{"supported": false}` (crk, yor) |
|
|
277
|
+
| `.microsoftTranslator` | `object` | API verification | Microsoft Azure Cognitive Services Translator. | `{"supported": true}` (spa, yor), `{"supported": false}` (crk) |
|
|
278
|
+
| `.libreTranslate` | `object` | API verification | LibreTranslate open-source translation. | `{"supported": true}` (spa), `{"supported": false}` (crk, yor) |
|
|
279
|
+
| `.nllb` | `object` | API verification | Meta's NLLB-200 model. Includes optional `code` (NLLB language code), `variety`, `qualityNotes`. | `{"supported": true, "code": "spa_Latn"}` (spa), `{"supported": true, "code": "yor_Latn"}` (yor), `{"supported": false}` (crk) |
|
|
280
|
+
| `.llm` | `object` | API verification | General LLM-based translation. | `{"supported": true}` (all example cards) |
|
|
281
|
+
|
|
282
|
+
### 7.4 `metricModelSupport`
|
|
283
|
+
|
|
284
|
+
| Field | Type | Source(s) | Description | Example |
|
|
285
|
+
|-------|------|-----------|-------------|---------|
|
|
286
|
+
| `metricModelSupport` | `object \| null` | enrich-metric-model-support.mjs | Which MT evaluation models produce reliable scores. Drives automatic model selection in `metrics_comet.py`. | See sub-fields below |
|
|
287
|
+
| `.xlmr` | `object \| null` | XLM-R training data analysis | XLM-R quality tier: `"high"`, `"medium"`, `"low"`. High = well-represented, standard COMET is reliable. | `{"tier": "high", "note": "Top-100 XLM-R training language by CommonCrawl volume"}` (cmn, spa) |
|
|
288
|
+
| `.africomet` | `object \| null` | AfriCOMET | Whether AfriCOMET covers this language. When true, AfriCOMET is preferred over standard COMET. | `{"supported": true, "model": "masakhane/africomet-mtl"}` (yor), `null` (crk) |
|
|
289
|
+
|
|
290
|
+
### 7.5 `metricPlugins`
|
|
291
|
+
|
|
292
|
+
| Field | Type | Source(s) | Description | Example |
|
|
293
|
+
|-------|------|-----------|-------------|---------|
|
|
294
|
+
| `metricPlugins` | `object \| null` | Derived | Declares which per-language metric plugin resource packs are available. Each key is a plugin pack name, value is `true` if a resource file exists at `plugins/resources/{packName}/{code}.json`. | `null` (all example cards) |
|
|
295
|
+
|
|
296
|
+
### 7.6 `omt1600`
|
|
297
|
+
|
|
298
|
+
| Field | Type | Source(s) | Description | Example |
|
|
299
|
+
|-------|------|-----------|-------------|---------|
|
|
300
|
+
| `omt1600` | `object \| null` | Manual | Meta OMT-1600 benchmark coverage. | See sub-fields below |
|
|
301
|
+
| `.covered` | `boolean` | Manual | Whether the language is covered by OMT-1600. | `true` (crk, spa, yor) |
|
|
302
|
+
| `.tier` | `string \| null` | Manual | Resource tier in OMT-1600 (`"R1"` high-resource through `"R5"` very-low-resource). | `"R1"` (crk), `"R3"` (yor), `"R5"` (spa) |
|
|
303
|
+
| `.evalMetrics` | `array` | Manual | Evaluation metrics used. | `["chrF++", "BLASER-3"]` (all) |
|
|
304
|
+
| `.notes` | `string \| null` | Manual | Context notes. | `"Plains Cree: no web-crawled bitext..."` (crk) |
|
|
305
|
+
|
|
306
|
+
---
|
|
307
|
+
|
|
308
|
+
## 8. Sociolinguistic
|
|
309
|
+
|
|
310
|
+
Formality, gender, register presets, code-switching, and contact influences.
|
|
311
|
+
|
|
312
|
+
### 8.1 `formality`
|
|
313
|
+
|
|
314
|
+
| Field | Type | Source(s) | Description | Example |
|
|
315
|
+
|-------|------|-----------|-------------|---------|
|
|
316
|
+
| `formality` | `object \| null` | Manual, WALS 45A (Helmbrecht) | Structured description of the formality system. Null for languages with no formal/informal distinction. | See sub-fields below |
|
|
317
|
+
| `.system` | `string` | Manual / WALS | Category of formality system. Examples: `"T-V"`, `"speech-levels"`, `"keigo"`, `"register-levels"`, `"none"`. | `"T-V"` (spa), `"register-levels"` (cmn, crk, yor) |
|
|
318
|
+
| `.description` | `string` | Manual | Human-readable explanation of how formality works, written for a developer. | `"Spanish has a complex T-V system that varies by region..."` (spa) |
|
|
319
|
+
| `.default` | `string` | Manual | Register preset key to use by default. Must match a key in `registers`. | `"neutral-latam"` (spa), `"professional"` (cmn, yor), `"standard"` (crk) |
|
|
320
|
+
|
|
321
|
+
### 8.2 `gender`
|
|
322
|
+
|
|
323
|
+
| Field | Type | Source(s) | Description | Example |
|
|
324
|
+
|-------|------|-----------|-------------|---------|
|
|
325
|
+
| `gender` | `object \| null` | Manual, WALS 44A | Grammatical gender information. Null if not applicable. | See sub-fields below |
|
|
326
|
+
| `.grammatical` | `boolean` | Manual / WALS | Whether the language has grammatical gender. **Required.** | `true` (spa), `false` (cmn, yor) |
|
|
327
|
+
| `.inclusiveGuidance` | `string \| null` | Manual | Guidance for gender-inclusive language, injected into register prompts. | `"Spanish has grammatical gender. For inclusive language, avoid gendered defaults..."` (spa) |
|
|
328
|
+
|
|
329
|
+
### 8.3 `registers`
|
|
330
|
+
|
|
331
|
+
| Field | Type | Source(s) | Description | Example |
|
|
332
|
+
|-------|------|-----------|-------------|---------|
|
|
333
|
+
| `registers` | `object` | Manual | Named register presets. Keys are preset identifiers. At least one required when present. Each preset has `label`, `description`, `prompt`, and optional `deeplFormality`. | See below |
|
|
334
|
+
| `registers.{key}.label` | `string` | Manual | Short display label shown in CLI. | `"Neutro (Latin American)"` (spa: `neutral-latam`) |
|
|
335
|
+
| `registers.{key}.description` | `string` | Manual | One-sentence explanation of when to use this preset. | `"Standard for international Spanish-language products..."` (spa: `neutral-latam`) |
|
|
336
|
+
| `registers.{key}.prompt` | `string` | Manual | The register instruction injected into the LLM system prompt. This steers translation tone. | `"Neutral Latin American Spanish. Professional register using usted-form..."` (spa: `neutral-latam`) |
|
|
337
|
+
| `registers.{key}.deeplFormality` | `string` | Manual | DeepL API formality value: `"prefer_more"`, `"prefer_less"`, `"default"`. Only relevant when `deepl.formality` is `true`. | `"prefer_more"` (spa: `formal-usted`), `"prefer_less"` (spa: `informal-tu`) |
|
|
338
|
+
|
|
339
|
+
### 8.4 `codeSwitching`
|
|
340
|
+
|
|
341
|
+
| Field | Type | Source(s) | Description | Example |
|
|
342
|
+
|-------|------|-----------|-------------|---------|
|
|
343
|
+
| `codeSwitching` | `object \| null` | Manual | Active code-switching patterns that affect MT I/O. Different from `contactInfluences` (historical). | `null` (cmn, crk, spa) |
|
|
344
|
+
| `.contactLanguage` | `string` | Manual | Contact language name. **Required.** | `"English"` (yor) |
|
|
345
|
+
| `.contactIso639_3` | `string \| null` | Manual | ISO 639-3 code of the contact language. | `"eng"` (yor) |
|
|
346
|
+
| `.mixedVarietyName` | `string \| null` | Manual | Named mixed variety (e.g., `"Jopará"`, `"Taglish"`, `"Hinglish"`). | `null` (yor) |
|
|
347
|
+
| `.prevalence` | `string` | Manual | How common code-switching is: `"rare"`, `"common"`, `"dominant"`. **Required.** | `"common"` (yor) |
|
|
348
|
+
| `.morphologicalIntegration` | `boolean` | Manual | Whether borrowed words take target-language morphology. **Required.** | `false` (yor) |
|
|
349
|
+
| `.pipelineStrategy` | `string \| null` | Manual | Recommended strategy: `"hybrid-fst"`, `"language-id-preprocessing"`, `"ignore"`. | `"ignore"` (yor) |
|
|
350
|
+
|
|
351
|
+
### 8.5 `contactInfluences`
|
|
352
|
+
|
|
353
|
+
| Field | Type | Source(s) | Description | Example |
|
|
354
|
+
|-------|------|-----------|-------------|---------|
|
|
355
|
+
| `contactInfluences` | `array \| object \| null` | Manual, WOLD, SegBo, AfBo | Universal contact history — borrowing layers, superstrates, substrates. Affects ALL languages. | See sub-fields below |
|
|
356
|
+
| `[].source` | `string` | Manual / WOLD | Contact language name. **Required.** | `"French"` (crk), `"Arabic"` (yor) |
|
|
357
|
+
| `[].sourceIso639_3` | `string \| null` | Manual | ISO 639-3 code of the contact language. | `"fra"` (crk), `"ara"` (yor) |
|
|
358
|
+
| `[].type` | `string` | Manual / WOLD | Contact type: `"superstrate"`, `"substrate"`, `"adstrate"`, `"learned_borrowing"`, `"lexical_borrowing"`, `"relexification"`. **Required.** | `"superstrate"` (crk: English), `"lexical_borrowing"` (crk: French; yor: Arabic, Portuguese) |
|
|
359
|
+
| `[].domains` | `array` | Manual | Domains affected (e.g., `"legal"`, `"culinary"`, `"religious"`). | `["education", "government", "technology", "commerce"]` (crk: English) |
|
|
360
|
+
| `[].depth` | `string` | Manual / WOLD | Depth: `"light"`, `"moderate"`, `"heavy"`, `"structural"`, `"defining"`. **Required.** | `"deep"` (crk: English), `"moderate"` (crk: French; yor: Arabic), `"light"` (yor: Portuguese) |
|
|
361
|
+
| `[].period` | `string \| null` | Manual | Historical period. | `"1670–1870"` (crk: French), `"colonial + ongoing"` (yor: English) |
|
|
362
|
+
| `[].notes` | `string \| null` | Manual | Free-form notes. | `"Fur trade era borrowings, many fully nativized..."` (crk: French) |
|
|
363
|
+
|
|
364
|
+
### 8.6 `arealContext`
|
|
365
|
+
|
|
366
|
+
| Field | Type | Source(s) | Description | Example |
|
|
367
|
+
|-------|------|-----------|-------------|---------|
|
|
368
|
+
| `arealContext` | `object \| null` | D-PLACE (Ethnographic Atlas), areal linguistics | Sprachbund membership, contact zones, ethnographic context. | See sub-fields below |
|
|
369
|
+
| `.dominantSubsistence` | `string` | D-PLACE EA042 | Dominant subsistence economy. | `"Hunting"` (crk), `"Intensive agriculture"` (cmn, spa) |
|
|
370
|
+
| `.settlementPattern` | `string` | D-PLACE EA030 | Settlement pattern. | `"Nomadic"` (crk), `"Villages/towns"` (cmn, spa) |
|
|
371
|
+
| `.politicalComplexity` | `string` | D-PLACE EA033 | Jurisdictional hierarchy level. | `"Acephalous"` (crk), `"Four levels"` (cmn), `"Three levels"` (spa) |
|
|
372
|
+
| `.communitySize` | `string` | D-PLACE EA031 | Mean community size. | `"Missing data"` (crk), `"50000+"` (cmn, spa) |
|
|
373
|
+
| `.dplaceRegion` | `string` | D-PLACE | Ethnographic region. | `"North-Central U.S.A."` (crk), `"China"` (cmn), `"Southwestern Europe"` (spa) |
|
|
374
|
+
| `.arealZone` | `string` | Manual / areal linguistics | Sprachbund or convergence area. | `"Mainland Southeast Asian Sprachbund"` (cmn) |
|
|
375
|
+
| `.arealFeatures` | `string` | Manual | Description of areal convergence features. | `"Tonal convergence, classifier systems, topic-prominence..."` (cmn) |
|
|
376
|
+
| `.contacts` | `array` | Manual | Areal contact descriptions. | (cmn has entries for Classical Chinese, Sanskrit/Pali) |
|
|
377
|
+
|
|
378
|
+
---
|
|
379
|
+
|
|
380
|
+
## 9. Linguistic Challenges
|
|
381
|
+
|
|
382
|
+
MT-relevant linguistic challenges for the language.
|
|
383
|
+
|
|
384
|
+
| Field | Type | Source(s) | Description | Example |
|
|
385
|
+
|-------|------|-----------|-------------|---------|
|
|
386
|
+
| `linguisticChallenges` | `object \| null` | Manual, WALS, Grambank, PHOIBLE | Keys are challenge IDs, values are description strings. Covers polysynthesis, animacy, tonal diacritics, tokenization, gender agreement, and more. | See examples below |
|
|
387
|
+
|
|
388
|
+
**Common challenge keys found across cards:**
|
|
389
|
+
|
|
390
|
+
| Key | Source(s) | Description | Example Value |
|
|
391
|
+
|-----|-----------|-------------|---------------|
|
|
392
|
+
| `polysynthesis` | Manual | Polysynthetic morphology challenges. | `"Cree is highly polysynthetic. A single verb can incorporate subject/object pronouns..."` (crk) |
|
|
393
|
+
| `animacy` | Manual | Animacy-based verb conjugation. | `"Verb conjugation changes completely based on whether the subject/object nouns are animate or inanimate."` (crk) |
|
|
394
|
+
| `tonalDiacritics` | Manual | Tonal diacritics are essential for disambiguation. | `"Yoruba has three lexical tones... 'ọkọ' without diacritics could mean husband, hoe, vehicle, or spear."` (yor) |
|
|
395
|
+
| `tokenization` | Manual | Word segmentation challenges. | `"Chinese has no spaces between words, making tokenization non-trivial."` (cmn) |
|
|
396
|
+
| `wordOrder` | WALS / Manual | Word order alignment with English. | `"SVO. Subject-Verb-Object order aligns well with English..."` (spa, yor, cmn) |
|
|
397
|
+
| `morphologicalComplexity` | WALS | Inflectional synthesis level. | `"High inflectional synthesis (WALS level 4: \"6-7 categories per word\")."` (crk, yor) |
|
|
398
|
+
| `genderAgreement` | WALS / Grambank | Grammatical gender agreement challenges. | `"Grammatical gender system (WALS: \"Two\"). Agreement patterns affect adjectives..."` (spa, crk) |
|
|
399
|
+
| `serialVerbs` | Grambank | Serial verb construction challenges. | `"Uses serial verb constructions (Grambank)."` (cmn, crk) |
|
|
400
|
+
| `reduplication` | WALS | Reduplication as a grammatical process. | `"Uses productive reduplication (WALS: \"Productive full and partial reduplication\")."` (cmn, crk, yor) |
|
|
401
|
+
| `scriptChallenges` | Derived from script properties | Script-specific processing challenges. | `"Canadian Aboriginal Syllabics: each symbol represents a consonant-vowel syllable..."` (crk) |
|
|
402
|
+
| `regionalVocabularyDivergence` | Manual | Vocabulary differences across regions. | `"Spanish has the largest regional vocabulary split of any European language..."` (spa) |
|
|
403
|
+
| `subjunctiveMood` | Manual | Subjunctive mood mapping challenges. | `"Spanish has a productive subjunctive mood..."` (spa) |
|
|
404
|
+
|
|
405
|
+
---
|
|
406
|
+
|
|
407
|
+
## 10. Cultural Context
|
|
408
|
+
|
|
409
|
+
| Field | Type | Source(s) | Description | Example |
|
|
410
|
+
|-------|------|-----------|-------------|---------|
|
|
411
|
+
| `culturalAphorism` | `object \| null` | Manual (verified) | An iconic proverb or saying from the language community. Must be a real, documented proverb — never fabricated. Null if no verifiable proverb is available. | See sub-fields below |
|
|
412
|
+
| `.text` | `string` | Manual | The aphorism in the original language/script. **Required.** | `"En boca cerrada no entran moscas"` (spa) |
|
|
413
|
+
| `.transliteration` | `string \| null` | Manual | Romanized form if non-Latin script. Null for Latin-script languages. | `"Xué wú zhǐ jìng"` (cmn), `null` (spa, crk, yor) |
|
|
414
|
+
| `.translation` | `string` | Manual | English translation. **Required.** | `"Flies don't enter a closed mouth"` (spa) |
|
|
415
|
+
| `.literal` | `string \| null` | Manual | Literal word-for-word translation if the idiomatic meaning differs significantly. | `"In mouth closed not enter flies"` (spa), `"Learning without end boundary"` (cmn) |
|
|
416
|
+
| `.source` | `string \| null` | Manual | Attribution or cultural context. | `"Traditional Spanish proverb (refrán). Documented in Correas, G. (1627)."` (spa) |
|
|
417
|
+
| `encyclopedic` | `object \| null` | Multiple (Wikidata, WALS, PHOIBLE, manual) | Encyclopedic metadata: family, demographics, dialect information, typological summary, resource links. Free-form structure. | Contains sub-objects: `typology`, `demographics`, `dialects`, `history`, `resources`, `wikidataDescription`, `phonology` |
|
|
418
|
+
|
|
419
|
+
---
|
|
420
|
+
|
|
421
|
+
## 11. Enrichment Fields
|
|
422
|
+
|
|
423
|
+
Fields added during Waves III–IV data enrichment.
|
|
424
|
+
|
|
425
|
+
### 11.1 `colexificationProfile`
|
|
426
|
+
|
|
427
|
+
| Field | Type | Source(s) | Description | Example |
|
|
428
|
+
|-------|------|-----------|-------------|---------|
|
|
429
|
+
| `colexificationProfile` | `object \| null` | CLICS³ | Cross-linguistic colexification data — which semantic concepts share the same word form. Null if not covered. 1,783 cards enriched. | See sub-fields below |
|
|
430
|
+
| `.conceptsDocumented` | `integer \| null` | CLICS³ | Number of Concepticon concept sets documented. | `2872` (cmn), `2516` (spa) |
|
|
431
|
+
| `.colexificationCount` | `integer \| null` | CLICS³ | Number of colexification pairs (meanings sharing a word form). | `443` (cmn), `530` (spa) |
|
|
432
|
+
| `.notableColexifications` | `array \| null` | CLICS³ | Notable colexification patterns as `{concepts: [A, B]}` pairs. | `[{"concepts": ["SMALL (NOT TALL)", "LOW"]}, {"concepts": ["HUSBAND", "WIFE"]}]` (cmn) |
|
|
433
|
+
| `.source` | `string \| null` | — | Data source identifier. | `"clics3-2020"` |
|
|
434
|
+
|
|
435
|
+
### 11.2 `numeralSystem`
|
|
436
|
+
|
|
437
|
+
| Field | Type | Source(s) | Description | Example |
|
|
438
|
+
|-------|------|-----------|-------------|---------|
|
|
439
|
+
| `numeralSystem` | `object \| null` | Numeralbank (Chan's database) | Numeral system properties. Null if undocumented. 4,088 cards enriched. | See sub-fields below |
|
|
440
|
+
| `.base` | `integer \| null` | Numeralbank | Primary counting base (10 = decimal, 20 = vigesimal, 5 = quinary, etc.). | `10` (all example cards) |
|
|
441
|
+
| `.baseType` | `string \| null` | Numeralbank | Named base type: `"decimal"`, `"vigesimal"`, `"quinary"`, `"octal"`, `"duodecimal"`, `"senary"`, `"mixed"`, `"body-part"`, `"restricted"`. | `"decimal"` (all example cards) |
|
|
442
|
+
| `.highestDocumented` | `integer \| null` | Numeralbank | Highest numeral documented in the source data. | `2000` (all example cards) |
|
|
443
|
+
| `.bodyPartCounting` | `boolean \| null` | Numeralbank | Whether the language uses a body-part counting system. | `null` (all example cards) |
|
|
444
|
+
| `.source` | `string \| null` | — | Data source identifier. | `"numeralbank-channumerals"` |
|
|
445
|
+
|
|
446
|
+
### 11.3 `databaseCoverage`
|
|
447
|
+
|
|
448
|
+
| Field | Type | Source(s) | Description | Example |
|
|
449
|
+
|-------|------|-----------|-------------|---------|
|
|
450
|
+
| `databaseCoverage` | `object \| null` | Derived | Coverage in major linguistic databases. Boolean flags for each database plus `totalDatabases` count. | `{"grambank": true, "wals": true, "phoible": false, "cldr": false, "linguameta": true, "opus": true, "totalDatabases": 4}` (crk) |
|
|
451
|
+
|
|
452
|
+
### 11.4 `documentationDepth`
|
|
453
|
+
|
|
454
|
+
| Field | Type | Source(s) | Description | Example |
|
|
455
|
+
|-------|------|-----------|-------------|---------|
|
|
456
|
+
| `documentationDepth` | `object \| null` | Glottolog (via OLAC) | Most Extensive Description (MED) level. Indicates depth of grammatical documentation. | See sub-fields below |
|
|
457
|
+
| `.med` | `string \| null` | Glottolog CLDF | MED category: `"long_grammar"`, `"grammar"`, `"grammar_sketch"`, `"phonology"`, `"wordlist"`. | `"long_grammar"` (all example cards) |
|
|
458
|
+
| `.medLevel` | `integer \| null` | Glottolog CLDF | Numeric MED level (0 = highest documentation). | `0` (all example cards) |
|
|
459
|
+
| `.source` | `string \| null` | — | Data source. | `"glottolog-5.3"` |
|
|
460
|
+
|
|
461
|
+
---
|
|
462
|
+
|
|
463
|
+
## 12. Internal Fields
|
|
464
|
+
|
|
465
|
+
Fields used for provenance tracking, generation metadata, and internal pipeline state. Not typically consumed by downstream users.
|
|
466
|
+
|
|
467
|
+
### 12.1 Provenance
|
|
468
|
+
|
|
469
|
+
| Field | Type | Source(s) | Description | Example |
|
|
470
|
+
|-------|------|-----------|-------------|---------|
|
|
471
|
+
| `dataSources` | `array \| null` | All enrichment scripts | Provenance tracking — which data sources contributed to this card. Required for license compliance. | `["iso639-3-2024", "glottolog-5.3", "wikidata", "wals-2024", "linguameta-2024", ...]` (cmn — 49 sources) |
|
|
472
|
+
| `_fieldSources` | `object` | All enrichment scripts | Per-field provenance. Maps each card field to the data source(s) that populated it. Machine-verifiable — enforced by the test suite. | `{"code": "iso639-3-2024", "classification": "glottolog-5.3", "nativeName": "wikidata-P1705", "numeralSystem": "numeralbank-channumerals", ...}` |
|
|
473
|
+
|
|
474
|
+
### 12.2 Generation Metadata
|
|
475
|
+
|
|
476
|
+
| Field | Type | Source(s) | Description | Example |
|
|
477
|
+
|-------|------|-----------|-------------|---------|
|
|
478
|
+
| `_generated` | `object \| null` | Generation scripts | Generation metadata for auto-generated cards: `by` (script name), `at` (ISO 8601 timestamp), `sources` (data sources used), `completeness`, `lastEnriched`, `derivedFields`. Null for hand-curated cards. | `{"by": "generate-all-cards.mjs", "at": "2026-06-07T07:16:44.778Z", "sources": ["iso639-3", "glottolog-5.3", "wikidata"], "completeness": "substantial"}` (cmn) |
|
|
479
|
+
| `_migration` | `object \| null` | Migration scripts | Migration metadata. Records `mergedFrom` / `previousCode` and `mergedAt` / `migratedAt` when cards were consolidated or renamed. | `{"mergedFrom": "zh", "mergedAt": "2026-06-08T03:26:23.376Z"}` (cmn), `{"previousCode": "es", "migratedAt": "2026-06-07T07:16:44.778Z"}` (spa) |
|
|
480
|
+
|
|
481
|
+
### 12.3 Status & Geography
|
|
482
|
+
|
|
483
|
+
| Field | Type | Source(s) | Description | Example |
|
|
484
|
+
|-------|------|-----------|-------------|---------|
|
|
485
|
+
| `supportTier` | `string \| null` | Derived | Champollion support tier: `"supported"`, `"experimental"`, `"cataloged"`, `"community"`, or `null`. | `"supported"` (spa, yor), `"developing"` (cmn, crk) |
|
|
486
|
+
| `countries` | `array` | Glottolog 5.3 | ISO 3166-1 alpha-2 country codes where the language is spoken. | `["CA", "US"]` (crk), `["BJ", "NG"]` (yor), `["CN", "KP", "LA", "MM", "MN", "RU", "TW", "VN"]` (cmn) |
|
|
487
|
+
| `coordinates` | `object \| null` | Glottolog 5.3 | Geographic coordinates of the language's primary area: `lat` (latitude), `lng` (longitude), `source`. | `{"lat": 7.15345, "lng": 3.67225, "source": "glottolog-5.3"}` (yor), `null` (crk) |
|
|
488
|
+
| `regions` | `array \| null` | Manual | Detailed geographic regions where the language is spoken. Each entry has `country`, `countryCode`, `officialStatus`, optional `region`, `speakerEstimate`, `coordinates` [lon, lat], `admin1Codes`. | `[{"country": "Canada", "countryCode": "CA", "officialStatus": "recognized", "region": "Saskatchewan, Alberta, Manitoba", "speakerEstimate": "~20,000", "coordinates": [-106.6, 52.1], "admin1Codes": ["CA-SK", "CA-AB", "CA-MB"]}]` (crk) |
|
|
489
|
+
| `varieties` | `array \| null` | Manual | Major dialect/variety differences relevant to MT. Each entry has `name`, optional `iso639_3`, `region`, `fstCoverage`, `corpusCoverage`, `nllbCoverage`, `mutualIntelligibility`, `notes`. | `[]` (all example cards currently empty) |
|
|
490
|
+
|
|
491
|
+
### 12.4 Other Internal Fields
|
|
492
|
+
|
|
493
|
+
| Field | Type | Source(s) | Description | Example |
|
|
494
|
+
|-------|------|-----------|-------------|---------|
|
|
495
|
+
| `humanReviewed` | `object \| null` | Manual | Whether the card has been reviewed by a human. Contains `reviewed` (boolean), `reviewer` (GitHub handle), `date` (ISO 8601). Null if not yet reviewed. | `null` (all example cards) |
|
|
496
|
+
| `firstDocumented` | `string \| null` | Manual | ISO 8601 date when this card was first created. | `null` (all example cards) |
|
|
497
|
+
| `lastDocumented` | `string \| null` | Manual | ISO 8601 date when this card was last significantly updated. | `null` (all example cards) |
|
|
498
|
+
| `notes` | `string \| null` | Manual | Free-form notes for developers and contributors. | `"Low-resource language under active development..."` (crk) |
|
|
499
|
+
| `rules` | `object \| null` | Manual | Executable rules for typography (`quoteStart`, `quoteEnd`, `usesSpaces`, `punctuationSpacing`), plurals (`categories`, `guidance`), capitalization (`hasCase`, `uiConventions`), and variables (`syntax`, `guidance`). | `{"typography": {"quoteStart": "\"", "quoteEnd": "\"", "usesSpaces": true}, "capitalization": {"hasCase": true}, "plurals": {"categories": ["one", "many", "other"]}}` (spa) |
|
|
500
|
+
|
|
501
|
+
---
|
|
502
|
+
|
|
503
|
+
## Source Label Conventions
|
|
504
|
+
|
|
505
|
+
Source labels in `_fieldSources` follow the format `<database-name>-<version>[-<component>]`:
|
|
506
|
+
|
|
507
|
+
| Label | Meaning |
|
|
508
|
+
|-------|---------|
|
|
509
|
+
| `iso639-3-2024` | ISO 639-3 Registration Authority (2024) |
|
|
510
|
+
| `glottolog-5.3` | Glottolog version 5.3 |
|
|
511
|
+
| `glottolog-5.x-languoid` | Glottolog 5.x, languoid component |
|
|
512
|
+
| `wikidata-P1705` | Wikidata native label property |
|
|
513
|
+
| `wikidata-P282` | Wikidata writing system property |
|
|
514
|
+
| `linguameta-2024` | LinguaMeta (Google Research 2024) |
|
|
515
|
+
| `grambank-1.0.3` | Grambank v1.0.3 (Skirgård et al. 2023) |
|
|
516
|
+
| `wals-2024` | WALS Online (Dryer & Haspelmath) |
|
|
517
|
+
| `phoible-2.0` | PHOIBLE 2.0 (Moran & McCloy) |
|
|
518
|
+
| `numeralbank-channumerals` | Numeralbank / Chan's Numeral Systems |
|
|
519
|
+
| `clics3-2020` | CLICS³ (Rzymski et al. 2020) |
|
|
520
|
+
| `dplace-ea-2016` | D-PLACE Ethnographic Atlas |
|
|
521
|
+
| `autotyp-2023` | AUTOTYP (Bickel et al. 2023) |
|
|
522
|
+
| `opus-nlpl-api` | OPUS parallel corpus API |
|
|
523
|
+
| `keyman-api` | Keyman keyboard database |
|
|
524
|
+
| `common-voice-20.0` | Mozilla Common Voice v20.0 |
|
|
525
|
+
| `paradisec-olac-2026` | PARADISEC via OAI-PMH |
|
|
526
|
+
| `olac-aggregator-2026` | OLAC aggregator harvest |
|
|
527
|
+
| `asjp-v20` | ASJP v20 (Wichmann et al.) |
|
|
528
|
+
| `derived-from-script` | Computed from `script` field |
|
|
529
|
+
| `derived-from-coordinates` | Computed from `coordinates` field |
|
|
530
|
+
| `manual-curation` | Hand-written by human contributor |
|
|
531
|
+
| `manual-curation-verified` | Hand-written and independently verified |
|
|
532
|
+
| `api-verification-2026` | Verified against live API (2026) |
|