champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,1308 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://champollion.dev/schemas/language-card.schema.json",
|
|
4
|
+
"title": "champollion Language Card",
|
|
5
|
+
"description": "Complete reference for a single language's translation configuration, including metadata, formality system, register presets, method support, and eval dataset links. BOUNDARY INVARIANT: a language card asserts language PROPERTIES (classification, phonology, typology, vitality, speaker counts, script) and resource EXISTENCE/CAPABILITY (a corpus/model/FST/metric/method is AVAILABLE — including methodSupport, evalDatasets, metricPlugins listing metric NAMES, and pipelineReadiness.source='derived') ONLY — each value cited and labeled derived-vs-asserted. Any MEASURED score of method output (chrF/BLEU/COMET/TER values, FST acceptance rate, '% morphologically valid') is a RUN RESULT keyed by (method, dataset, metric) and belongs on the leaderboard/edge — it is FORBIDDEN on a language card. Champollion is an INDEX, not a truth-arbiter: every value faithfully reports what its cited source actually says, and where sources DISAGREE the card shows ALL of them attributed (e.g. speakerEstimates[]) rather than manufacturing a single winner. These invariants are enforced by lint-language-cards.mjs rules R1 (tone-consistent-with-phoible), R2 (speaker-count-supported-by-source), R3 (no-run-results), R4 (no-metric-mislabel), R5 (family-name-matches-glottolog: classification.family must be Glottolog's own name for the cited familyGlottocode; 'Language isolate' exempt), R6 (derived-fields-carry-derived-provenance: Champollion-computed fields like dialectCount and scriptUnicodeName must carry derived: provenance, never a bare upstream name).",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"required": [
|
|
8
|
+
"code",
|
|
9
|
+
"name"
|
|
10
|
+
],
|
|
11
|
+
"properties": {
|
|
12
|
+
"extends": {
|
|
13
|
+
"type": ["string", "null"],
|
|
14
|
+
"description": "Locale code of another card/family this card inherits from."
|
|
15
|
+
},
|
|
16
|
+
"code": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"description": "Language identifier. ISO 639-3 three-letter code (e.g., 'fra', 'crk') for individual languages, BCP 47 with region for regional variants (e.g., 'por-PT', 'spa-MX'), or prefixed codes for genera/families (e.g., 'genus-cree', 'family-algic'). For conlangs, use 'x-' prefix.",
|
|
19
|
+
"pattern": "^[a-z]{2,3}(-[A-Z][a-z]{3})?(-[A-Z]{2})?$|^x-[a-z-]+$|^tlh$|^(genus|family|macrolanguage)-[a-z-]+$|^[a-z0-9]{4}[0-9]{4}$"
|
|
20
|
+
},
|
|
21
|
+
"name": {
|
|
22
|
+
"type": ["string", "object"],
|
|
23
|
+
"description": "English display name of the language. ATLAS SHAPE: this field may instead carry an ATTRIBUTION ENVELOPE {agreement, consensus, values:[{value, source}]} wherever the ingested sources disagree — the standing rule is to show every source attributed rather than elect a winner. The envelope is why the type list admits 'object'; mini-schema has no anyOf, and the SEMANTICS are enforced by the R1-R6 card-integrity rules, not here."
|
|
24
|
+
},
|
|
25
|
+
"nativeName": {
|
|
26
|
+
"type": ["string", "null"],
|
|
27
|
+
"description": "Name of the language in its own script (endonym). For languages with dual orthographies (e.g., SRO and Syllabics for Cree, Latin and Cyrillic for Serbian), include both separated by ' / '. Must render correctly in the language's native script — e.g., 'Русский' not 'Russkiy', 'العربية' not 'al-Arabiyyah', 'nêhiyawêwin / ᓀᐦᐃᔭᐍᐏᐣ' not just 'nehiyawewin'. Null for conlangs without established endonyms."
|
|
28
|
+
},
|
|
29
|
+
"iso639_1": {
|
|
30
|
+
"type": ["string", "null"],
|
|
31
|
+
"description": "ISO 639-1 two-letter code. Null if no ISO 639-1 code exists (e.g., conlangs, some indigenous languages).",
|
|
32
|
+
"pattern": "^[a-z]{2}$"
|
|
33
|
+
},
|
|
34
|
+
"iso639_3": {
|
|
35
|
+
"type": ["string", "null"],
|
|
36
|
+
"description": "ISO 639-3 three-letter code. Null for conlangs.",
|
|
37
|
+
"pattern": "^[a-z]{3}$"
|
|
38
|
+
},
|
|
39
|
+
"bcp47": {
|
|
40
|
+
"type": ["string", "null"],
|
|
41
|
+
"description": "BCP 47 language tag (e.g., 'fr', 'en-US', 'iu-Cans-CA'). Null for languages without a registered BCP 47 subtag (many sign languages, small oral languages). 430+ languages in the database have null bcp47.",
|
|
42
|
+
"minLength": 2
|
|
43
|
+
},
|
|
44
|
+
"alternateNames": {
|
|
45
|
+
"type": "array",
|
|
46
|
+
"description": "Alternative English names for this language. Used for search and display.",
|
|
47
|
+
"items": { "type": "string" }
|
|
48
|
+
},
|
|
49
|
+
"isoScope": {
|
|
50
|
+
"type": ["string", "null"],
|
|
51
|
+
"description": "ISO 639-3 scope: 'I' (individual), 'M' (macrolanguage), 'S' (special). Null if not in ISO 639-3. This is the canonical 'scope' field per docs/LANGUAGE_TAXONOMY.md — only ISO 639-3 individual languages (scope 'I') are benchmark targets; 'M' cards are navigation hubs, never benchmarked.",
|
|
52
|
+
"enum": [null, "I", "M", "S", "Individual", "Macrolanguage", "Special"]
|
|
53
|
+
},
|
|
54
|
+
"isoType": {
|
|
55
|
+
"type": ["string", "null"],
|
|
56
|
+
"description": "ISO 639-3 type: 'L' (living), 'E' (extinct), 'A' (ancient), 'H' (historical), 'C' (constructed), 'S' (special). Null if not in ISO 639-3. This is the canonical 'languageType' field per docs/LANGUAGE_TAXONOMY.md — ISO's own type assignment is the SSOT for living/extinct/historical/ancient/constructed status. Source: cli/data/iso639-3/iso-639-3.tab (Language_Type column).",
|
|
57
|
+
"enum": [null, "L", "E", "A", "H", "C", "S"]
|
|
58
|
+
},
|
|
59
|
+
"modality": {
|
|
60
|
+
"type": ["string", "null"],
|
|
61
|
+
"description": "Primary language modality: 'spoken' or 'signed'. Sign languages (e.g., 'ase' American Sign Language, 'psd' Plains Indian Sign Language) are natural languages with full grammars, native acquisition, and language communities — modality 'signed' marks them as such, never as a lesser category. Writing is NOT a modality — it is an orthography attribute (see scripts[] and orthographicStatus); a language with no written form is still fully 'spoken' or 'signed'. Null when unknown or not yet derived. Whistled/drummed speech surrogates and tactile signing are not separate modalities in this model (acknowledged simplification — see docs/LANGUAGE_TAXONOMY.md). Derived by derive-taxonomy-fields.mjs from the Glottolog 'Sign Language' pseudo-family plus the ISO 639-3 reference-name heuristic.",
|
|
62
|
+
"enum": [null, "spoken", "signed"]
|
|
63
|
+
},
|
|
64
|
+
"isIsolate": {
|
|
65
|
+
"type": "boolean",
|
|
66
|
+
"description": "Whether this language is a language isolate (no known genetic relatives). Default: false."
|
|
67
|
+
},
|
|
68
|
+
"script": {
|
|
69
|
+
"type": ["string", "null"],
|
|
70
|
+
"description": "Primary ISO 15924 script code (e.g., 'Latn', 'Arab', 'Cans', 'Jpan'). Null for unwritten languages, languages with no standardized orthography, or when unknown. 1,400+ languages in the database have null script.",
|
|
71
|
+
"pattern": "^[A-Z][a-z]{3}$"
|
|
72
|
+
},
|
|
73
|
+
"scriptUnicodeName": {
|
|
74
|
+
"type": ["string", "null"],
|
|
75
|
+
"description": "Unicode script block name corresponding to the primary ISO 15924 script code. Derived from 'script' via a standard mapping (e.g., 'Latn' → 'Latin', 'Cans' → 'Canadian_Aboriginal', 'Arab' → 'Arabic', 'Jpan' → 'CJK'). Used by the code_switching metric plugin to detect script mixing. Auto-populated by enrich-script-unicode-names.mjs."
|
|
76
|
+
},
|
|
77
|
+
"dir": {
|
|
78
|
+
"type": ["string", "null"],
|
|
79
|
+
"description": "Text directionality. Null for unwritten languages or when the writing direction cannot be determined (e.g., no standardized orthography). 620+ languages in the database have null dir.",
|
|
80
|
+
"enum": ["ltr", "rtl", null]
|
|
81
|
+
},
|
|
82
|
+
"formality": {
|
|
83
|
+
"type": ["object", "null"],
|
|
84
|
+
"description": "Structured description of the language's formality system. Null for languages with no formal/informal distinction relevant to translation.",
|
|
85
|
+
"properties": {
|
|
86
|
+
"system": {
|
|
87
|
+
"type": "string",
|
|
88
|
+
"description": "Category of formality system. Examples: 'T-V' (French, German), 'speech-levels' (Korean), 'keigo' (Japanese), 'particles-and-pronouns' (Thai), 'pronoun-system' (Vietnamese), 'register-levels' (Arabic), 'none'."
|
|
89
|
+
},
|
|
90
|
+
"description": {
|
|
91
|
+
"type": "string",
|
|
92
|
+
"description": "Human-readable explanation of how formality works in this language, written for a developer who doesn't speak it."
|
|
93
|
+
},
|
|
94
|
+
"default": {
|
|
95
|
+
"type": "string",
|
|
96
|
+
"description": "The register preset key to use by default. Must match a key in the registers object."
|
|
97
|
+
}
|
|
98
|
+
},
|
|
99
|
+
"required": ["system", "description", "default"]
|
|
100
|
+
},
|
|
101
|
+
"gender": {
|
|
102
|
+
"type": ["object", "null"],
|
|
103
|
+
"description": "Grammatical gender information. Null for languages with no gender considerations. THREE writer generations coexist (reconciled 2026-07-07, no key is required): (a) the curated translation-guidance form { grammatical, inclusiveGuidance }; (b) the Grambank-derived form { hasGrammaticalGender, genderCount, sexBased, nounClassSystem, phonologicallyPredictable, source } from enrich-gender-from-grambank.mjs; (c) the WALS 44A pronoun-system form { system, description, source }.",
|
|
104
|
+
"properties": {
|
|
105
|
+
"grammatical": {
|
|
106
|
+
"type": "boolean",
|
|
107
|
+
"description": "Whether the language has grammatical gender (e.g., French, German, Arabic) vs. no grammatical gender (e.g., Finnish, Korean, Turkish)."
|
|
108
|
+
},
|
|
109
|
+
"inclusiveGuidance": {
|
|
110
|
+
"type": ["string", "null"],
|
|
111
|
+
"description": "Guidance for gender-inclusive language. Injected into all register prompts. Null if not applicable."
|
|
112
|
+
},
|
|
113
|
+
"hasGrammaticalGender": {
|
|
114
|
+
"type": ["boolean", "null"],
|
|
115
|
+
"description": "Grambank-derived: whether any grammatical gender/noun-class system is documented."
|
|
116
|
+
},
|
|
117
|
+
"genderCount": {
|
|
118
|
+
"type": ["integer", "null"],
|
|
119
|
+
"description": "Grambank-derived: number of gender/noun-class distinctions, when documented."
|
|
120
|
+
},
|
|
121
|
+
"sexBased": { "type": ["boolean", "null"], "description": "Grambank-derived: sex-based gender system." },
|
|
122
|
+
"nounClassSystem": { "type": ["boolean", "null"], "description": "Grambank-derived: broader noun-class system." },
|
|
123
|
+
"phonologicallyPredictable": { "type": ["boolean", "null"], "description": "Grambank-derived: assignment predictable from phonology." },
|
|
124
|
+
"system": { "type": ["string", "null"], "description": "WALS 44A pronoun gender system label." },
|
|
125
|
+
"description": { "type": ["string", "null"], "description": "Cited prose description (WALS form)." },
|
|
126
|
+
"source": { "type": ["string", "null"], "description": "Source id that asserted this block (e.g. 'grambank-2023', 'wals-44A')." }
|
|
127
|
+
}
|
|
128
|
+
},
|
|
129
|
+
"registers": {
|
|
130
|
+
"type": ["object", "null"],
|
|
131
|
+
"description": "Named register presets specific to this language's formality system. Keys are preset identifiers (e.g., 'formal-vous', 'polite-haeyo'). Null on the (majority of) cataloged cards whose register research has not been done yet — a card with registers MUST have at least one preset, but null is the honest 'not populated' state (type widened 2026-07-07).",
|
|
132
|
+
"minProperties": 1,
|
|
133
|
+
"additionalProperties": {
|
|
134
|
+
"type": "object",
|
|
135
|
+
"required": ["label", "description", "prompt"],
|
|
136
|
+
"properties": {
|
|
137
|
+
"label": {
|
|
138
|
+
"type": "string",
|
|
139
|
+
"description": "Short display label shown in the CLI wizard and status output."
|
|
140
|
+
},
|
|
141
|
+
"description": {
|
|
142
|
+
"type": "string",
|
|
143
|
+
"description": "One-sentence explanation of when to use this preset."
|
|
144
|
+
},
|
|
145
|
+
"prompt": {
|
|
146
|
+
"type": "string",
|
|
147
|
+
"description": "The actual register instruction injected into the LLM system prompt. This is the text that steers translation tone."
|
|
148
|
+
},
|
|
149
|
+
"deeplFormality": {
|
|
150
|
+
"type": "string",
|
|
151
|
+
"enum": ["prefer_more", "prefer_less", "default"],
|
|
152
|
+
"description": "DeepL API formality value for this preset. Only relevant when methodSupport.deepl.formality is true. 'prefer_more' for formal, 'prefer_less' for casual, 'default' for neutral."
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
},
|
|
157
|
+
"aliases": {
|
|
158
|
+
"type": "array",
|
|
159
|
+
"description": "Alternative locale codes that should resolve to this language card (e.g., 'no' for 'nb', 'iw' for 'he').",
|
|
160
|
+
"items": { "type": "string" }
|
|
161
|
+
},
|
|
162
|
+
"methodSupport": {
|
|
163
|
+
"type": "object",
|
|
164
|
+
"description": "Which translation APIs/methods support this language. Each entry is an object with 'supported' (boolean) plus optional metadata (dateAdded, qualityNotes, variety, NLLB code, etc.). This is the Single Source of Truth for method availability AND detail.",
|
|
165
|
+
"properties": {
|
|
166
|
+
"googleTranslate": {
|
|
167
|
+
"type": "object",
|
|
168
|
+
"description": "Google Cloud Translation API / Google Translate consumer product.",
|
|
169
|
+
"properties": {
|
|
170
|
+
"supported": { "type": "boolean" },
|
|
171
|
+
"dateAdded": { "type": ["string", "null"], "description": "When support was added (e.g., '2006', '2024-07')." },
|
|
172
|
+
"qualityNotes": { "type": ["string", "null"], "description": "Known quality assessment." }
|
|
173
|
+
},
|
|
174
|
+
"required": ["supported"]
|
|
175
|
+
},
|
|
176
|
+
"deepl": {
|
|
177
|
+
"type": "object",
|
|
178
|
+
"description": "DeepL Translation API.",
|
|
179
|
+
"properties": {
|
|
180
|
+
"supported": { "type": "boolean" },
|
|
181
|
+
"formality": {
|
|
182
|
+
"type": "boolean",
|
|
183
|
+
"description": "Whether DeepL supports the formality parameter for this language. When true, the DeepL method sends prefer_more/prefer_less based on the active register preset."
|
|
184
|
+
}
|
|
185
|
+
},
|
|
186
|
+
"required": ["supported"]
|
|
187
|
+
},
|
|
188
|
+
"microsoftTranslator": {
|
|
189
|
+
"type": "object",
|
|
190
|
+
"description": "Microsoft Azure Cognitive Services Translator.",
|
|
191
|
+
"properties": {
|
|
192
|
+
"supported": { "type": "boolean" }
|
|
193
|
+
},
|
|
194
|
+
"required": ["supported"]
|
|
195
|
+
},
|
|
196
|
+
"libreTranslate": {
|
|
197
|
+
"type": "object",
|
|
198
|
+
"description": "LibreTranslate open-source translation.",
|
|
199
|
+
"properties": {
|
|
200
|
+
"supported": { "type": "boolean" }
|
|
201
|
+
},
|
|
202
|
+
"required": ["supported"]
|
|
203
|
+
},
|
|
204
|
+
"nllb": {
|
|
205
|
+
"type": "object",
|
|
206
|
+
"description": "Meta's NLLB-200 (No Language Left Behind) model.",
|
|
207
|
+
"properties": {
|
|
208
|
+
"supported": { "type": "boolean" },
|
|
209
|
+
"code": { "type": ["string", "null"], "description": "NLLB-200 language code (e.g., 'fra_Latn', 'yor_Latn', 'grn_Latn')." },
|
|
210
|
+
"variety": { "type": ["string", "null"], "description": "Which variety is covered if ambiguous (e.g., 'Ayacucho Quechua' for que_Latn)." },
|
|
211
|
+
"qualityNotes": { "type": ["string", "null"], "description": "Known quality assessment." }
|
|
212
|
+
},
|
|
213
|
+
"required": ["supported"]
|
|
214
|
+
},
|
|
215
|
+
"llm": {
|
|
216
|
+
"type": "object",
|
|
217
|
+
"description": "General LLM-based translation (via OpenRouter, direct API, etc.).",
|
|
218
|
+
"properties": {
|
|
219
|
+
"supported": { "type": "boolean" }
|
|
220
|
+
},
|
|
221
|
+
"required": ["supported"]
|
|
222
|
+
}
|
|
223
|
+
},
|
|
224
|
+
"additionalProperties": {
|
|
225
|
+
"type": "object",
|
|
226
|
+
"description": "Additional method support entries beyond the standard set.",
|
|
227
|
+
"properties": {
|
|
228
|
+
"supported": { "type": "boolean" }
|
|
229
|
+
},
|
|
230
|
+
"required": ["supported"]
|
|
231
|
+
}
|
|
232
|
+
},
|
|
233
|
+
"metricModelSupport": {
|
|
234
|
+
"type": ["object", "null"],
|
|
235
|
+
"description": "Which MT evaluation models produce reliable scores for this language. Drives automatic model selection in metrics_comet.py. Auto-populated by enrich-metric-model-support.mjs.",
|
|
236
|
+
"properties": {
|
|
237
|
+
"xlmr": {
|
|
238
|
+
"description": "XLM-R quality tier. NOTE: the harness code (language_cards.is_xlmr_high_resource) reads an OBJECT with a 'tier' field; a bare string/null is accepted for back-compat. 'high' = standard COMET is reliable.",
|
|
239
|
+
"oneOf": [
|
|
240
|
+
{ "type": "null" },
|
|
241
|
+
{ "type": "string", "enum": ["high", "medium", "low"] },
|
|
242
|
+
{
|
|
243
|
+
"type": "object",
|
|
244
|
+
"properties": {
|
|
245
|
+
"tier": { "type": ["string", "null"], "enum": [null, "high", "medium", "low"] },
|
|
246
|
+
"source": { "type": ["string", "null"] }
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
]
|
|
250
|
+
},
|
|
251
|
+
"africomet": {
|
|
252
|
+
"description": "AfriCOMET coverage. NOTE: the harness code (language_cards.has_africomet / get_metric_model_for) reads an OBJECT with 'supported' (bool) and 'model'; a bare boolean/null is accepted for back-compat.",
|
|
253
|
+
"oneOf": [
|
|
254
|
+
{ "type": "null" },
|
|
255
|
+
{ "type": "boolean" },
|
|
256
|
+
{
|
|
257
|
+
"type": "object",
|
|
258
|
+
"properties": {
|
|
259
|
+
"supported": { "type": "boolean" },
|
|
260
|
+
"model": { "type": ["string", "null"] },
|
|
261
|
+
"note": { "type": ["string", "null"] },
|
|
262
|
+
"source": { "type": ["string", "null"] }
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
]
|
|
266
|
+
},
|
|
267
|
+
"qe": {
|
|
268
|
+
"description": "Reference-free QE model coverage (read by language_cards.get_qe_model_for). Used by the no-reference scoring profile — e.g. masakhane/africomet-qe-stl for African languages.",
|
|
269
|
+
"type": ["object", "null"],
|
|
270
|
+
"properties": {
|
|
271
|
+
"supported": { "type": "boolean" },
|
|
272
|
+
"model": { "type": ["string", "null"] },
|
|
273
|
+
"note": { "type": ["string", "null"] },
|
|
274
|
+
"source": { "type": ["string", "null"] }
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
},
|
|
279
|
+
"scoringProfile": {
|
|
280
|
+
"type": ["object", "null"],
|
|
281
|
+
"description": "Card-driven composite scoring profile (SCORING_SPEC §4.3). Resolved by language_cards.resolve_scoring_profile() and consumed by scoring.compute_composite_score(). When absent, the harness falls back to the legacy default (fst-coverage when an FST scored the run, else surface-only).",
|
|
282
|
+
"properties": {
|
|
283
|
+
"basis": {
|
|
284
|
+
"type": ["string", "null"],
|
|
285
|
+
"description": "Named profile from scoring.PROFILE_REGISTRY selecting the composite weight table.",
|
|
286
|
+
"enum": [null, "fst-coverage", "surface-only", "comet-validated"]
|
|
287
|
+
},
|
|
288
|
+
"lane": {
|
|
289
|
+
"type": ["string", "null"],
|
|
290
|
+
"description": "Composite lane: 'public' (Apache/MIT-clean, commercial-safe) or 'nc-research' (may include CC-BY-NC QE metrics).",
|
|
291
|
+
"enum": [null, "public", "nc-research"]
|
|
292
|
+
},
|
|
293
|
+
"metrics": {
|
|
294
|
+
"type": ["object", "null"],
|
|
295
|
+
"description": "Per-metric overrides keyed by metric id: { weight, applies, normalize, reference, model, interpretation }. Overrides the named profile's defaults for this language.",
|
|
296
|
+
"additionalProperties": { "type": "object" }
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
},
|
|
300
|
+
"evalMetrics": {
|
|
301
|
+
"type": ["object", "null"],
|
|
302
|
+
"description": "Language-specific eval-standard metric plugins to load (referee metrics, e.g. CRK's LYSS linter/semantic validator). Each key is a metric id; the value declares how to load it. Consumed by plugin_discovery._load_language_card_metrics().",
|
|
303
|
+
"additionalProperties": {
|
|
304
|
+
"type": "object",
|
|
305
|
+
"properties": {
|
|
306
|
+
"module": { "type": "string", "description": "Importable module path for the metric class." },
|
|
307
|
+
"class": { "type": "string", "description": "Metric class implementing the MetricPlugin protocol." },
|
|
308
|
+
"description": { "type": ["string", "null"] },
|
|
309
|
+
"dependencies": { "type": ["array", "null"], "items": { "type": "string" } },
|
|
310
|
+
"spacy_models": { "type": ["array", "null"], "items": { "type": "string" } }
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
},
|
|
314
|
+
"evalStandard": {
|
|
315
|
+
"type": ["object", "null"],
|
|
316
|
+
"description": "How to fetch the EXTERNAL package that provides this language's evalMetrics (e.g. champollion-lyss for Cree). The harness core ships no language-specific scorer code; it installs this on demand (auto-fetch) and fails loud if it cannot. Consumed by plugin_discovery._ensure_eval_standard_installed().",
|
|
317
|
+
"properties": {
|
|
318
|
+
"package": { "type": "string", "description": "Distribution name, e.g. 'champollion-lyss'." },
|
|
319
|
+
"import": { "type": "string", "description": "Top-level import package, e.g. 'champollion_lyss'." },
|
|
320
|
+
"pip": { "type": "string", "description": "pip install target — a PEP 508 spec or VCS URL, e.g. 'champollion-lyss @ git+https://github.com/.../champollion-LYSS.git@main'." },
|
|
321
|
+
"description": { "type": ["string", "null"] }
|
|
322
|
+
}
|
|
323
|
+
},
|
|
324
|
+
"evalPack": {
|
|
325
|
+
"type": ["object", "null"],
|
|
326
|
+
"description": "On-demand dependency manifest for this language's eval, consumed by setup_wizard.install_lang(). Declares pip deps, post-install steps, and whether an FST is required.",
|
|
327
|
+
"properties": {
|
|
328
|
+
"pythonDeps": {
|
|
329
|
+
"type": ["object", "array", "null"],
|
|
330
|
+
"description": "Pip dependencies. Current form (2026): an object map of package name → pip requirement spec (e.g. { 'pyhfst': 'pyhfst>=1.4' }), which is what setup_wizard.install_lang() consumes. The legacy array-of-spec-strings form is still accepted."
|
|
331
|
+
},
|
|
332
|
+
"postInstall": {
|
|
333
|
+
"type": ["array", "null"],
|
|
334
|
+
"items": {
|
|
335
|
+
"type": ["string", "object"],
|
|
336
|
+
"description": "A post-install step: either a bare shell command string, or an object { command, label } (e.g. crk's spaCy model download)."
|
|
337
|
+
}
|
|
338
|
+
},
|
|
339
|
+
"requiresFst": { "type": ["boolean", "null"] },
|
|
340
|
+
"description": { "type": ["string", "null"] }
|
|
341
|
+
}
|
|
342
|
+
},
|
|
343
|
+
"metricPlugins": {
|
|
344
|
+
"type": ["object", "null"],
|
|
345
|
+
"description": "Declares which per-language metric plugin resource packs are available for this language. Each key is a plugin pack name, value is true if a resource file exists at plugins/resources/{packName}/{code}.json. Example: { 'formalityMarkers': true } means formality analysis has markers for this language.",
|
|
346
|
+
"additionalProperties": {
|
|
347
|
+
"type": "boolean"
|
|
348
|
+
}
|
|
349
|
+
},
|
|
350
|
+
"databaseCoverage": {
|
|
351
|
+
"type": ["object", "null"],
|
|
352
|
+
"description": "Coverage in major linguistic databases and datasets (e.g., OPUS, NLLB, Tatoeba)."
|
|
353
|
+
},
|
|
354
|
+
"corpusAvailability": {
|
|
355
|
+
"type": ["object", "null"],
|
|
356
|
+
"description": "Available parallel and monolingual corpora. OPUS corpus counts, alignment pair counts, corpus names.",
|
|
357
|
+
"properties": {
|
|
358
|
+
"opus": {
|
|
359
|
+
"type": ["object", "boolean", "null"],
|
|
360
|
+
"description": "OPUS coverage. Object with corpus details when the OPUS API returned data; the boolean form (typically false) is the checked-but-not-available summary written by the corpus-availability enrichment for languages OPUS does not carry.",
|
|
361
|
+
"properties": {
|
|
362
|
+
"corpora": { "type": "integer", "description": "Number of OPUS corpora containing this language." },
|
|
363
|
+
"corpusNames": { "type": "array", "items": { "type": "string" }, "description": "Names of the most significant corpora." },
|
|
364
|
+
"source": { "type": ["string", "null"] }
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
},
|
|
369
|
+
"keyboardSupport": {
|
|
370
|
+
"type": ["object", "null"],
|
|
371
|
+
"description": "Keyboard layout availability from the Keyman API and other sources.",
|
|
372
|
+
"properties": {
|
|
373
|
+
"keymanKeyboards": { "type": "integer", "description": "Number of Keyman keyboard layouts available." },
|
|
374
|
+
"keyboardNames": { "type": "array", "items": { "type": "string" } },
|
|
375
|
+
"source": { "type": ["string", "null"] }
|
|
376
|
+
}
|
|
377
|
+
},
|
|
378
|
+
"omt1600": {
|
|
379
|
+
"type": ["object", "null"],
|
|
380
|
+
"description": "Meta OMT-1600 benchmark coverage. Whether this language is in Meta's 1600-language translation model benchmark.",
|
|
381
|
+
"properties": {
|
|
382
|
+
"covered": { "type": "boolean", "description": "Whether the language is covered by OMT-1600." },
|
|
383
|
+
"tier": {
|
|
384
|
+
"type": ["string", "null"],
|
|
385
|
+
"enum": ["high", "mid", "low", "very_low", "zero", null],
|
|
386
|
+
"description": "Resource tier in the OMT-1600 paper's own vocabulary (arXiv:2603.16309 §3.3, Figure 3.2 buckets 0_high…4_zero): 'high' (>50M parallel documents), 'mid' (>1M), 'low' (40K–1M), 'very_low' — the paper's 'extremely low' (1K–40K), 'zero' (<1K). NEVER 'R1'…'R5': those are Met-BOUQuET annotation ROUNDS in the paper's tables, not resource tiers. The paper publishes no per-language tier table, so a per-language tier is uncitable — null it rather than inferring or remapping one."
|
|
387
|
+
},
|
|
388
|
+
"evalMetrics": { "type": "array", "items": { "type": "string" }, "description": "Evaluation metrics used for this language." },
|
|
389
|
+
"notes": { "type": ["string", "null"] }
|
|
390
|
+
}
|
|
391
|
+
},
|
|
392
|
+
"scriptConverter": {
|
|
393
|
+
"type": ["string", "null"],
|
|
394
|
+
"description": "Key in SCRIPT_CONVERTERS registry (from scripts.js) if this language has a deterministic script conversion step. Null if not applicable."
|
|
395
|
+
},
|
|
396
|
+
"evalDatasets": {
|
|
397
|
+
"type": "array",
|
|
398
|
+
"description": "IDs of evaluation datasets from the companion eval harness (the MT Eval Arena (arena/)). Not consumed by champollion at runtime — metadata only.",
|
|
399
|
+
"items": { "type": "string" }
|
|
400
|
+
},
|
|
401
|
+
"notes": {
|
|
402
|
+
"type": ["string", "null"],
|
|
403
|
+
"description": "Free-form notes for developers and contributors."
|
|
404
|
+
},
|
|
405
|
+
"rules": {
|
|
406
|
+
"type": ["object", "null"],
|
|
407
|
+
"description": "Executable rules and constraints for typography, plurals, casing, and variables used in validation.",
|
|
408
|
+
"properties": {
|
|
409
|
+
"typography": {
|
|
410
|
+
"type": "object",
|
|
411
|
+
"properties": {
|
|
412
|
+
"quoteStart": { "type": "string" },
|
|
413
|
+
"quoteEnd": { "type": "string" },
|
|
414
|
+
"usesSpaces": { "type": "boolean" },
|
|
415
|
+
"punctuationSpacing": {
|
|
416
|
+
"type": "object",
|
|
417
|
+
"properties": {
|
|
418
|
+
"doublePunctuation": { "type": "string", "enum": ["none", "space", "thin-nbsp"] }
|
|
419
|
+
}
|
|
420
|
+
}
|
|
421
|
+
}
|
|
422
|
+
},
|
|
423
|
+
"plurals": {
|
|
424
|
+
"type": "object",
|
|
425
|
+
"properties": {
|
|
426
|
+
"categories": {
|
|
427
|
+
"type": "array",
|
|
428
|
+
"items": { "type": "string" }
|
|
429
|
+
},
|
|
430
|
+
"guidance": { "type": "string" }
|
|
431
|
+
}
|
|
432
|
+
},
|
|
433
|
+
"capitalization": {
|
|
434
|
+
"type": "object",
|
|
435
|
+
"properties": {
|
|
436
|
+
"hasCase": { "type": "boolean" },
|
|
437
|
+
"uiConventions": {
|
|
438
|
+
"type": "object",
|
|
439
|
+
"properties": {
|
|
440
|
+
"buttons": { "type": "string", "enum": ["title-case", "sentence-case", "all-caps", "none"] },
|
|
441
|
+
"headings": { "type": "string", "enum": ["title-case", "sentence-case", "all-caps", "none"] }
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
},
|
|
446
|
+
"variables": {
|
|
447
|
+
"type": "object",
|
|
448
|
+
"properties": {
|
|
449
|
+
"syntax": { "type": "string" },
|
|
450
|
+
"guidance": { "type": "string" }
|
|
451
|
+
}
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
},
|
|
455
|
+
"humanReviewed": {
|
|
456
|
+
"type": ["object", "null"],
|
|
457
|
+
"description": "Whether this card has been reviewed by a human for accuracy. Absent or null means not yet reviewed.",
|
|
458
|
+
"properties": {
|
|
459
|
+
"reviewed": {
|
|
460
|
+
"type": "boolean",
|
|
461
|
+
"description": "True if a human has verified the card's accuracy."
|
|
462
|
+
},
|
|
463
|
+
"reviewer": {
|
|
464
|
+
"type": "string",
|
|
465
|
+
"description": "GitHub handle or name of the reviewer."
|
|
466
|
+
},
|
|
467
|
+
"date": {
|
|
468
|
+
"type": "string",
|
|
469
|
+
"format": "date",
|
|
470
|
+
"description": "Date of last human review (ISO 8601, e.g. '2026-05-24')."
|
|
471
|
+
}
|
|
472
|
+
},
|
|
473
|
+
"required": ["reviewed"]
|
|
474
|
+
},
|
|
475
|
+
"supportTier": {
|
|
476
|
+
"type": ["string", "null"],
|
|
477
|
+
"description": "Champollion support tier for this language. Vocabulary is the deriveSupportTier() ladder in cli/scripts/derive-card-fields.mjs: 'supported' (formality+registers+challenges populated) > 'developing' (resources or 3+ supported methods) > 'emerging' (vitality/speaker data) > 'cataloged' (identity+classification only). (Enum reconciled to the writer 2026-07-07 — the earlier 'experimental'/'community' values were never produced.)",
|
|
478
|
+
"enum": [null, "supported", "developing", "emerging", "cataloged"]
|
|
479
|
+
},
|
|
480
|
+
"firstDocumented": {
|
|
481
|
+
"type": ["string", "integer", "null"],
|
|
482
|
+
"description": "When the LANGUAGE was first documented: a bare year (integer, from enrich-documentation-dates.mjs) or an ISO 8601 date string."
|
|
483
|
+
},
|
|
484
|
+
"lastDocumented": {
|
|
485
|
+
"type": ["string", "integer", "null"],
|
|
486
|
+
"description": "When the LANGUAGE was last documented: a bare year (integer, from enrich-documentation-dates.mjs) or an ISO 8601 date string."
|
|
487
|
+
},
|
|
488
|
+
"_generated": {
|
|
489
|
+
"type": ["object", "null"],
|
|
490
|
+
"description": "Generation metadata for auto-generated cards. Records which script generated this card, when, and from what sources.",
|
|
491
|
+
"properties": {
|
|
492
|
+
"by": { "type": "string", "description": "Script that generated this card." },
|
|
493
|
+
"at": { "type": "string", "description": "ISO 8601 timestamp of generation." },
|
|
494
|
+
"sources": { "type": "array", "items": { "type": "string" }, "description": "Data sources used." }
|
|
495
|
+
}
|
|
496
|
+
},
|
|
497
|
+
"_fieldSources": {
|
|
498
|
+
"type": ["object", "null"],
|
|
499
|
+
"description": "Per-field provenance map: field name → the source id(s) that produced that field's value. A value is a single source string ('phoible-2.0'), an array of source strings (composite fields like linguisticChallenges), or a nested object mirroring the field's shape (e.g. encyclopedic.typology). Derived values use a 'derived:'/'template-generated:'/'champollion-derived' prefix; any value Champollion computes must carry champollion-derived provenance, never an upstream's name (see the license-boundaries doctrine and the fact-provenance audit). This is the provenance layer the card-integrity gate enforces ('every fact cited'); it is intentionally heterogeneous, so values are not strictly typed here. Source ids must resolve in shared/licenses.json (the license-source-resolves rule). NOTE: provenance is field-group-coarse today — a single wrong bit inside a multi-source field can inherit a citation it did not earn; the card-integrity rules (R1–R2) cross-check the actual value against its claimed source rather than trusting this map alone."
|
|
500
|
+
},
|
|
501
|
+
"glottocode": {
|
|
502
|
+
"type": ["string", "null"],
|
|
503
|
+
"description": "Glottolog identifier for cross-referencing with the Glottolog language database (https://glottolog.org).",
|
|
504
|
+
"pattern": "^[a-z0-9]{4}[0-9]{4}$"
|
|
505
|
+
},
|
|
506
|
+
"macrolanguage": {
|
|
507
|
+
"type": ["string", "null"],
|
|
508
|
+
"description": "ISO 639-3 macrolanguage code if this language belongs to a macrolanguage umbrella (e.g., 'cre' for Cree varieties, 'ara' for Arabic varieties, 'zho' for Chinese varieties). Null if the language is not part of a macrolanguage. This is the canonical 'memberOf' field per docs/LANGUAGE_TAXONOMY.md. Source: cli/data/iso639-3/iso-639-3-macrolanguages.tab (M_Id column, I_Status 'A' rows only).",
|
|
509
|
+
"pattern": "^[a-z]{3}$"
|
|
510
|
+
},
|
|
511
|
+
"members": {
|
|
512
|
+
"type": ["array", "null"],
|
|
513
|
+
"description": "For macrolanguage hub cards only (isoScope 'M' or code 'macrolanguage-*'): the ISO 639-3 codes of the individual member languages this hub links to (e.g., for 'cre': ['crj', 'crk', 'crl', 'crm', 'csw', 'cwd']). Inverse of the members' own 'macrolanguage' field. Source: cli/data/iso639-3/iso-639-3-macrolanguages.tab (I_Id column, I_Status 'A' rows only). Hub cards are a typed navigation layer — never benchmark targets; benchmarks bind to individual member languages because corpora, FSTs, and methods are variety-specific. Null/absent on individual-language cards.",
|
|
514
|
+
"items": { "type": "string", "pattern": "^[a-z]{3}$" }
|
|
515
|
+
},
|
|
516
|
+
"taxonomyNotes": {
|
|
517
|
+
"type": ["string", "null"],
|
|
518
|
+
"description": "Free-text affordance for known taxonomy disputes affecting this language: cases where ISO 639-3 and Glottolog disagree on splitting/lumping, contested language-vs-dialect status, or naming disputes. Policy (docs/LANGUAGE_TAXONOMY.md): display both authorities' positions, never adjudicate between them; communities name themselves (OCAP). Null when no dispute is documented — null means 'no note recorded', not 'no dispute exists'."
|
|
519
|
+
},
|
|
520
|
+
"classification": {
|
|
521
|
+
"type": ["object", "null"],
|
|
522
|
+
"description": "Genealogical classification from Glottolog + WALS. Auto-populated by build-language-tree.mjs --enrich. Null for conlangs.",
|
|
523
|
+
"properties": {
|
|
524
|
+
"family": {
|
|
525
|
+
"type": ["string", "object", "null"],
|
|
526
|
+
"description": "Top-level language family (e.g., 'Indo-European', 'Algic', 'Sino-Tibetan'). Source: Glottolog. Null — with glottologBucket + note carrying the citation — when Glottolog files the entry under a housekeeping bucket (e.g. 'Bookkeeping', book1242) rather than a genealogical family: the bucket is never displayed as a family (lint rule no-bookkeeping-family)."
|
|
527
|
+
},
|
|
528
|
+
"glottologBucket": {
|
|
529
|
+
"type": "string",
|
|
530
|
+
"description": "Glottocode of the Glottolog housekeeping pseudo-family the entry is filed under (e.g. 'book1242' = Bookkeeping: spurious, unattested, or retired entries). Present iff family is null because Glottolog asserts no genealogical classification for the entry.",
|
|
531
|
+
"pattern": "^[a-z0-9]{4}[0-9]{4}$"
|
|
532
|
+
},
|
|
533
|
+
"note": {
|
|
534
|
+
"type": "string",
|
|
535
|
+
"description": "Cited plain-language note on classification status — e.g. that Glottolog files the entry under its Bookkeeping bucket. Faithful-to-source per the Language-Card Boundary Invariant: reports what the upstream actually says, never adjudicates."
|
|
536
|
+
},
|
|
537
|
+
"familyGlottocode": {
|
|
538
|
+
"type": "string",
|
|
539
|
+
"description": "Glottocode of the top-level family node. Glottocodes are four alphanumerics + four digits — the alpha-only prefix assumption was wrong (real codes like 'b10b1234', '3adt1234' exist).",
|
|
540
|
+
"pattern": "^[a-z0-9]{4}[0-9]{4}$"
|
|
541
|
+
},
|
|
542
|
+
"genus": {
|
|
543
|
+
"type": "string",
|
|
544
|
+
"description": "WALS-style genus — the lowest-level grouping of related languages that share runtime properties (e.g., 'Plains Creeic', 'Arabic', 'Mandarinic'). Used for genus card inheritance. OPTIONAL: build-language-tree.mjs only writes a genus when the languoid has a family-level ancestor distinct from its top family — isolates and single-branch families legitimately have none."
|
|
545
|
+
},
|
|
546
|
+
"genusGlottocode": {
|
|
547
|
+
"type": "string",
|
|
548
|
+
"description": "Glottocode of the genus node. Optional, present iff genus is.",
|
|
549
|
+
"pattern": "^[a-z0-9]{4}[0-9]{4}$"
|
|
550
|
+
},
|
|
551
|
+
"ancestry": {
|
|
552
|
+
"type": "array",
|
|
553
|
+
"description": "Full ancestry chain from top-level family to genus (e.g., ['Algic', 'Algonquian-Blackfoot', 'Algonquian', 'Cree-Montagnais-Naskapi', 'Cree', 'Plains Creeic']). Isolates carry a single '<Name> (isolate)' entry; Glottolog housekeeping-bucket entries (family: null) carry an empty array — the bucket is not ancestry. Source: Glottolog.",
|
|
554
|
+
"items": { "type": "string" }
|
|
555
|
+
}
|
|
556
|
+
},
|
|
557
|
+
"_comment_required": "family is NOT required: a language Glottolog files under a housekeeping bucket (Unclassifiable, Pidgin, Artificial Language, ...) asserts glottologBucket and no family, because Glottolog makes no descent claim for it. 403 cards are in that position. mini-schema has no conditional required, and demanding family here would force the projector to invent one.",
|
|
558
|
+
"required": []
|
|
559
|
+
},
|
|
560
|
+
"contactInfluences": {
|
|
561
|
+
"type": ["array", "object", "null"],
|
|
562
|
+
"description": "Universal contact history — borrowing layers, superstrates, substrates. Affects ALL languages, not just creoles. English has French superstrate, Latin learned borrowings, Norse adstrate. Japanese has Chinese learned borrowings. This data informs MT method design and cognate detection. TWO GENERATIONS coexist (reconciled 2026-07-07): (a) the array-of-influences form below; (b) the areal-context object form written by the areal enrichment ({ arealZone, arealFeatures, contacts: [{ language, type, period, description }] }). The 'type'/'depth' vocabularies are curated free text, not closed enums — enrichment produces values like 'colonial/modern', 'trade lingua franca', 'deep', 'surface' alongside the canonical superstrate/substrate/adstrate + light…defining ladders.",
|
|
563
|
+
"items": {
|
|
564
|
+
"type": "object",
|
|
565
|
+
"properties": {
|
|
566
|
+
"source": {
|
|
567
|
+
"type": "string",
|
|
568
|
+
"description": "Name of the contact language (e.g., 'French', 'Latin', 'Old Norse'). Areal-derived entries use 'language' instead."
|
|
569
|
+
},
|
|
570
|
+
"language": {
|
|
571
|
+
"type": "string",
|
|
572
|
+
"description": "Contact language name (areal-enrichment vocabulary for 'source')."
|
|
573
|
+
},
|
|
574
|
+
"sourceIso639_3": {
|
|
575
|
+
"type": ["string", "null"],
|
|
576
|
+
"description": "ISO 639-3 code of the contact language, if applicable.",
|
|
577
|
+
"pattern": "^[a-z]{3}$"
|
|
578
|
+
},
|
|
579
|
+
"type": {
|
|
580
|
+
"type": "string",
|
|
581
|
+
"description": "Type of contact influence. Canonical values: superstrate, substrate, adstrate, learned_borrowing, lexical_borrowing, relexification — plus curated descriptive values (e.g. 'colonial/modern', 'religious/literary', 'trade lingua franca')."
|
|
582
|
+
},
|
|
583
|
+
"domains": {
|
|
584
|
+
"type": "array",
|
|
585
|
+
"description": "Domains affected by this contact (e.g., ['legal', 'culinary', 'scientific']).",
|
|
586
|
+
"items": { "type": "string" }
|
|
587
|
+
},
|
|
588
|
+
"depth": {
|
|
589
|
+
"type": "string",
|
|
590
|
+
"description": "Depth of contact influence. Canonical ladder: light, moderate, heavy, structural, defining — plus curated values ('deep', 'surface')."
|
|
591
|
+
},
|
|
592
|
+
"period": {
|
|
593
|
+
"type": ["string", "null"],
|
|
594
|
+
"description": "Historical period of contact (e.g., 'post-1066', '7th–9th century')."
|
|
595
|
+
},
|
|
596
|
+
"description": {
|
|
597
|
+
"type": ["string", "null"],
|
|
598
|
+
"description": "Prose description of this contact influence (areal-enrichment form)."
|
|
599
|
+
},
|
|
600
|
+
"notes": {
|
|
601
|
+
"type": ["string", "null"],
|
|
602
|
+
"description": "Free-form notes about this contact influence."
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
},
|
|
606
|
+
"properties": {
|
|
607
|
+
"arealZone": { "type": ["string", "null"], "description": "Named convergence area (e.g. 'West African Convergence Area')." },
|
|
608
|
+
"arealFeatures": { "type": ["string", "null"], "description": "Cited prose summary of shared areal features." },
|
|
609
|
+
"contacts": {
|
|
610
|
+
"type": "array",
|
|
611
|
+
"description": "Contact influences in the areal-enrichment shape ({ language, type, period, description }).",
|
|
612
|
+
"items": { "type": "object" }
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
},
|
|
616
|
+
"scripts": {
|
|
617
|
+
"type": ["array", "null"],
|
|
618
|
+
"description": "ISO 15924 script tracking. Many languages use multiple scripts (e.g., Cree uses Cans + Latn, Serbian uses Cyrl + Latn). ATLAS SHAPE: entries may be bare ISO 15924 code STRINGS; the richer object form carries code/name/primary/source.",
|
|
619
|
+
"items": {
|
|
620
|
+
"type": ["object", "string"],
|
|
621
|
+
"required": ["code"],
|
|
622
|
+
"properties": {
|
|
623
|
+
"code": {
|
|
624
|
+
"type": "string",
|
|
625
|
+
"description": "ISO 15924 four-letter script code (e.g., 'Latn', 'Cans', 'Arab').",
|
|
626
|
+
"pattern": "^[A-Z][a-z]{3}$"
|
|
627
|
+
},
|
|
628
|
+
"name": {
|
|
629
|
+
"type": "string",
|
|
630
|
+
"description": "Human-readable script name. Optional — bulk-derived entries (derive-scripts-from-script.mjs / linguameta) carry only code+primary+source; names resolve via scriptUnicodeName / the ISO 15924 table."
|
|
631
|
+
},
|
|
632
|
+
"primary": {
|
|
633
|
+
"type": "boolean",
|
|
634
|
+
"description": "Whether this is the primary working script for computational purposes. Optional on secondary-script entries from linguameta enrichment."
|
|
635
|
+
},
|
|
636
|
+
"source": {
|
|
637
|
+
"type": ["string", "null"],
|
|
638
|
+
"description": "Source id that asserted this script (e.g. 'linguameta', 'cldr')."
|
|
639
|
+
}
|
|
640
|
+
}
|
|
641
|
+
}
|
|
642
|
+
},
|
|
643
|
+
"orthographicStatus": {
|
|
644
|
+
"type": ["string", "null"],
|
|
645
|
+
"description": "Status of the language's writing system. Vocabulary is the derive-orthographic-status.mjs decision ladder: standardized (official CLDR status) | de-facto-standard | has-orthography (known CLDR writing system / keyboard) | developing (script assigned, not standardized) | unwritten (no known script). (Enum reconciled to the writer 2026-07-07 — the earlier no-orthography/disputed/emerging/historical-only values were never produced.)",
|
|
646
|
+
"enum": [null, "standardized", "de-facto-standard", "has-orthography", "developing", "unwritten"]
|
|
647
|
+
},
|
|
648
|
+
"orthographies": {
|
|
649
|
+
"type": ["array", "null"],
|
|
650
|
+
"description": "Structured writing-convention entries, one per script the card's scripts[] asserts. Restructures scripts[] + orthographicStatus into the convention-level facts MT pipelines need (scheme name, long-vowel marking, canonical working form) — the conventions were previously prose-only. Populated by derive-orthographies.mjs; convention details (scheme/longVowelMarking) come only from the script entry's own name or the curated register (shared/curated-orthography-conventions.json). Optional keys are OMITTED when unknown — never guessed (index, not arbiter).",
|
|
651
|
+
"items": {
|
|
652
|
+
"type": "object",
|
|
653
|
+
"required": ["script", "source"],
|
|
654
|
+
"properties": {
|
|
655
|
+
"script": {
|
|
656
|
+
"type": "string",
|
|
657
|
+
"description": "ISO 15924 four-letter script code (e.g., 'Latn', 'Cans').",
|
|
658
|
+
"pattern": "^[A-Z][a-z]{3}$"
|
|
659
|
+
},
|
|
660
|
+
"scheme": {
|
|
661
|
+
"type": "string",
|
|
662
|
+
"description": "Named orthographic scheme/convention within the script (e.g., 'SRO' — Standard Roman Orthography for Cree). Omitted when no named scheme is documented."
|
|
663
|
+
},
|
|
664
|
+
"longVowelMarking": {
|
|
665
|
+
"type": "string",
|
|
666
|
+
"description": "How the scheme marks long vowels (e.g., 'circumflex', 'macron', 'double-vowel', 'none'). Omitted when not documented."
|
|
667
|
+
},
|
|
668
|
+
"canonicalForMt": {
|
|
669
|
+
"type": "boolean",
|
|
670
|
+
"description": "Whether this orthography is the canonical working form for MT/pipeline I/O — a Champollion capability designation (derived: sole-script cards, or curated, e.g. crk SRO-circumflex matching the GiellaLT FST). Omitted when undetermined."
|
|
671
|
+
},
|
|
672
|
+
"source": {
|
|
673
|
+
"type": "string",
|
|
674
|
+
"description": "Source id asserting this entry (the scripts[] entry's own source, or the curated-register citation)."
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
}
|
|
678
|
+
},
|
|
679
|
+
"dataSources": {
|
|
680
|
+
"type": ["array", "null"],
|
|
681
|
+
"description": "Provenance tracking — which data sources contributed to this card (e.g., ['glottolog-5.3', 'cldr-48', 'wals-2024']). Required for license compliance.",
|
|
682
|
+
"items": { "type": "string" }
|
|
683
|
+
},
|
|
684
|
+
"linguisticChallenges": {
|
|
685
|
+
"type": ["object", "null"],
|
|
686
|
+
"description": "MT-relevant linguistic challenges for this language (e.g., polysynthesis, animacy, tonal diacritics). Keys are challenge IDs, values are cited prose descriptions. PROVENANCE: each claim must cite a real source that supports it. For TONE the sole authority is PHOIBLE phonologicalInventory — a tone challenge must NOT appear when the card's own PHOIBLE block has isTonal=false/tones=0 (Grambank has NO tone feature; the v1 'phonemic tone present (Grambank)' string was a misread of GB079='verb prefixes'). Enforced by lint rule R1.",
|
|
687
|
+
"additionalProperties": { "type": "string" }
|
|
688
|
+
},
|
|
689
|
+
"typologicalProfile": {
|
|
690
|
+
"type": ["object", "null"],
|
|
691
|
+
"description": "Typological features from Grambank. Includes word order, morphological properties, and feature coverage. Auto-populated by enrich-grambank-typology.mjs.",
|
|
692
|
+
"properties": {
|
|
693
|
+
"featuresDocumented": {
|
|
694
|
+
"type": "integer",
|
|
695
|
+
"description": "Number of Grambank features documented for this language."
|
|
696
|
+
},
|
|
697
|
+
"featuresCoverage": {
|
|
698
|
+
"type": "number",
|
|
699
|
+
"description": "Fraction of Grambank features documented (0.0–1.0)."
|
|
700
|
+
},
|
|
701
|
+
"wordOrderDominant": {
|
|
702
|
+
"type": ["string", "null"],
|
|
703
|
+
"description": "Dominant word order (e.g., 'SVO', 'SOV', 'OV-flexible')."
|
|
704
|
+
},
|
|
705
|
+
"hasDefiniteArticle": { "type": ["boolean", "null"] },
|
|
706
|
+
"hasIndefiniteArticle": { "type": ["boolean", "null"] },
|
|
707
|
+
"hasGenderSystem": { "type": ["boolean", "null"] },
|
|
708
|
+
"hasCaseMorphology": { "type": ["boolean", "null"] },
|
|
709
|
+
"hasEvidentiality": { "type": ["boolean", "null"] },
|
|
710
|
+
"hasToneSystem": { "type": ["boolean", "null"] },
|
|
711
|
+
"morphologicalSynthesis": {
|
|
712
|
+
"type": ["string", "null"],
|
|
713
|
+
"enum": [null, "analytic", "synthetic", "polysynthetic"],
|
|
714
|
+
"description": "Champollion-DERIVED synthesis-degree enum — the 'does this language reward FST-driven morphological handling' signal. Computed by derive-morphological-synthesis.mjs strictly from cited signals already on the card: the WALS 22A categories-per-word value (encyclopedic.typology.verbSynthesis / linguisticChallenges.morphologicalComplexity; 0–1 → analytic, 2–7 → synthetic, 8+ → polysynthetic), WALS 26A 'Little affixation' (analytic), and explicitly cited polysynthesis prose (polysynthetic). Conflicting signals or no signal → field ABSENT (unknown is never defaulted). Provenance MUST be a derived: stamp in _fieldSources['typologicalProfile.morphologicalSynthesis'] naming the on-card signal — never a bare upstream name (CLAUDE.md derived-values doctrine; lint R6)."
|
|
715
|
+
},
|
|
716
|
+
"source": {
|
|
717
|
+
"type": ["string", "null"],
|
|
718
|
+
"description": "Data source and version (e.g., 'grambank-1.0.3')."
|
|
719
|
+
}
|
|
720
|
+
}
|
|
721
|
+
},
|
|
722
|
+
"phonologicalInventory": {
|
|
723
|
+
"type": ["object", "null"],
|
|
724
|
+
"description": "Phoneme inventory from PHOIBLE. Includes consonant/vowel counts, tone status, and inventory size classification. Auto-populated by enrich-phoible-phonemes.mjs.",
|
|
725
|
+
"properties": {
|
|
726
|
+
"consonants": {
|
|
727
|
+
"type": "integer",
|
|
728
|
+
"description": "Number of consonant phonemes."
|
|
729
|
+
},
|
|
730
|
+
"vowels": {
|
|
731
|
+
"type": "integer",
|
|
732
|
+
"description": "Number of vowel phonemes."
|
|
733
|
+
},
|
|
734
|
+
"tones": {
|
|
735
|
+
"type": "integer",
|
|
736
|
+
"description": "Number of tonal contrasts (0 for non-tonal)."
|
|
737
|
+
},
|
|
738
|
+
"totalPhonemes": {
|
|
739
|
+
"type": "integer",
|
|
740
|
+
"description": "Total phoneme count (consonants + vowels + tones)."
|
|
741
|
+
},
|
|
742
|
+
"isTonal": {
|
|
743
|
+
"type": "boolean",
|
|
744
|
+
"description": "Whether the language uses lexical or grammatical tone."
|
|
745
|
+
},
|
|
746
|
+
"inventorySize": {
|
|
747
|
+
"type": ["string", "null"],
|
|
748
|
+
"description": "Qualitative inventory size classification.",
|
|
749
|
+
"enum": [null, "small", "moderately-small", "average", "moderately-large", "large"]
|
|
750
|
+
},
|
|
751
|
+
"source": {
|
|
752
|
+
"type": ["string", "null"],
|
|
753
|
+
"description": "Data source and version (e.g., 'phoible-2.0')."
|
|
754
|
+
}
|
|
755
|
+
}
|
|
756
|
+
},
|
|
757
|
+
"encyclopedic": {
|
|
758
|
+
"type": ["object", "null"],
|
|
759
|
+
"description": "Encyclopedic metadata: family, demographics, dialect information, external resource links."
|
|
760
|
+
},
|
|
761
|
+
"resources": {
|
|
762
|
+
"type": ["object", "array", "null"],
|
|
763
|
+
"description": "NLP resources available for this language: corpora, models, FSTs, tools. TODO(2026-07-07, post-ACL migration): ~3,100 cards still carry the legacy flat-array shape (a bare list of resource entries) — the array type is accepted until derive-resources-from-coverage.mjs migrates them to the keyed object form below; the linter tracks this as the card-schema-resources-shape warning.",
|
|
764
|
+
"items": { "type": "object" },
|
|
765
|
+
"properties": {
|
|
766
|
+
"fsts": {
|
|
767
|
+
"type": "array",
|
|
768
|
+
"description": "Morphological analyzers and FSTs available for this language.",
|
|
769
|
+
"items": {
|
|
770
|
+
"type": "object",
|
|
771
|
+
"required": ["name", "type"],
|
|
772
|
+
"properties": {
|
|
773
|
+
"name": { "type": "string", "description": "Human-readable name of the FST/tool." },
|
|
774
|
+
"url": { "type": ["string", "null"], "description": "URL to the tool or its releases." },
|
|
775
|
+
"type": { "type": "string", "description": "Tool type (e.g., 'morphological-analyzer', 'tokenizer', 'spellchecker')." },
|
|
776
|
+
"technology": {
|
|
777
|
+
"type": ["string", "null"],
|
|
778
|
+
"description": "Underlying technology (e.g., 'hfst', 'xfst', 'foma', 'python', 'neural', 'spacy', 'lttoolbox')."
|
|
779
|
+
},
|
|
780
|
+
"coverage": {
|
|
781
|
+
"type": ["string", "null"],
|
|
782
|
+
"description": "Known naïve coverage percentage if available (e.g., '86-91%', '~96%')."
|
|
783
|
+
},
|
|
784
|
+
"license": {
|
|
785
|
+
"type": ["string", "null"],
|
|
786
|
+
"description": "License (e.g., 'GPL-3.0', 'MIT', 'Apache-2.0')."
|
|
787
|
+
},
|
|
788
|
+
"maintainer": {
|
|
789
|
+
"type": ["string", "null"],
|
|
790
|
+
"description": "Primary maintainer or organization."
|
|
791
|
+
},
|
|
792
|
+
"status": {
|
|
793
|
+
"type": ["string", "null"],
|
|
794
|
+
"description": "Maintenance status.",
|
|
795
|
+
"enum": [null, "active", "maintained", "legacy", "archived"]
|
|
796
|
+
},
|
|
797
|
+
"install": {
|
|
798
|
+
"type": ["object", "null"],
|
|
799
|
+
"description": "Installation metadata for automated FST download. Replaces the former hardcoded GIELLALT_FST_REGISTRY. Null if the FST must be installed manually. ⚠️ CHANNEL WARNING: GitHub Releases is NOT GiellaLT/ALTLab's distribution channel — they ship continuously and their lang-* release tags are VESTIGIAL. 'No release since YEAR' does not mean 'no update since YEAR'. crk sat on lang-crk's newest release tag (fst-v2021.7.8, 2021) for five years: because that build uses 'y' on the analysis side while the current lexicon uses 'ý', generation returned None SILENTLY for every ý-lemma — 4,989 of 28,268 lemmas (17.6%) were ungeneratable, with identical surfaces, so nothing looked wrong. Prefer 'giellalt-nightly-apt' for GiellaLT languages; a releaseTag is a point in time you must justify, never 'latest'.",
|
|
800
|
+
"properties": {
|
|
801
|
+
"repo": {
|
|
802
|
+
"type": "string",
|
|
803
|
+
"description": "Upstream repository in 'owner/repo' format (e.g., 'giellalt/lang-crk'). For giellalt-nightly-apt this identifies the SOURCE repo — the artifact is NOT fetched from it."
|
|
804
|
+
},
|
|
805
|
+
"releaseTag": {
|
|
806
|
+
"type": "string",
|
|
807
|
+
"description": "GitHub release tag to download from (e.g., 'fst-v2021.7.8'). Used by legacy-zip / divvun-macos-pkg only. ⚠️ A POINT IN TIME, not 'latest' — verify it is not stale before relying on it."
|
|
808
|
+
},
|
|
809
|
+
"assetPattern": {
|
|
810
|
+
"type": "string",
|
|
811
|
+
"description": "Substring pattern to match the release asset filename."
|
|
812
|
+
},
|
|
813
|
+
"aptPool": {
|
|
814
|
+
"type": "string",
|
|
815
|
+
"description": "giellalt-nightly-apt: directory URL of the GiellaLT/Apertium nightly apt pool (e.g. 'https://apertium.projectjj.com/apt/nightly/pool/main/g/giella-crk/')."
|
|
816
|
+
},
|
|
817
|
+
"debFile": {
|
|
818
|
+
"type": "string",
|
|
819
|
+
"description": "giellalt-nightly-apt: exact .deb filename — THIS IS THE VERSION PIN. The filename carries the upstream commit (e.g. 'giella-crk_0.2.0+g4278~e1f96fea-1~sid1_all.deb'). Nightly pools are pruned over time; if it 404s, re-pin deliberately and re-verify — never fall back to a release tag."
|
|
820
|
+
},
|
|
821
|
+
"debSha256": {
|
|
822
|
+
"type": "string",
|
|
823
|
+
"description": "giellalt-nightly-apt: SHA256 of the .deb, verified BEFORE extraction. The install aborts on mismatch.",
|
|
824
|
+
"pattern": "^[0-9a-f]{64}$"
|
|
825
|
+
},
|
|
826
|
+
"langCommit": {
|
|
827
|
+
"type": "string",
|
|
828
|
+
"description": "giellalt-nightly-apt: full upstream source commit the build was made from. Record it beside any lexicon extracted from the same commit — lexicon and binary MUST move together."
|
|
829
|
+
},
|
|
830
|
+
"include": {
|
|
831
|
+
"type": "array",
|
|
832
|
+
"description": "giellalt-nightly-apt: exact .hfstol basenames to install from the package. A GiellaLT .deb ships ~31 transducers of several different KINDS; installing all of them would waste ~250MB and invite comparing across kinds (strict / relaxed / -gt-desc descriptive / -gt-norm normative), which is never valid.",
|
|
833
|
+
"items": { "type": "string" }
|
|
834
|
+
},
|
|
835
|
+
"stripSuffix": {
|
|
836
|
+
"type": "string",
|
|
837
|
+
"description": "giellalt-nightly-apt: suffix to strip from extracted filenames (e.g. '-giellaltbuild') so cached names match the conventional ones."
|
|
838
|
+
},
|
|
839
|
+
"format": {
|
|
840
|
+
"type": "string",
|
|
841
|
+
"description": "Download and extraction format. 'giellalt-nightly-apt' is the CURRENT channel for GiellaLT languages.",
|
|
842
|
+
"enum": ["giellalt-nightly-apt", "legacy-zip", "divvun", "divvun-macos-pkg", "manual"]
|
|
843
|
+
},
|
|
844
|
+
"bundlePattern": {
|
|
845
|
+
"type": ["string", "null"],
|
|
846
|
+
"description": "For divvun-macos-pkg format: glob pattern to locate the FST bundle inside the .pkg payload."
|
|
847
|
+
},
|
|
848
|
+
"maturity": {
|
|
849
|
+
"type": ["string", "null"],
|
|
850
|
+
"description": "FST quality/completeness level.",
|
|
851
|
+
"enum": [null, "production", "beta", "stub"]
|
|
852
|
+
}
|
|
853
|
+
},
|
|
854
|
+
"required": ["repo", "format"]
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
}
|
|
858
|
+
},
|
|
859
|
+
"corpora": {
|
|
860
|
+
"type": "array",
|
|
861
|
+
"description": "Parallel and monolingual corpora available for this language.",
|
|
862
|
+
"items": {
|
|
863
|
+
"type": "object",
|
|
864
|
+
"required": ["name", "type"],
|
|
865
|
+
"properties": {
|
|
866
|
+
"name": { "type": "string", "description": "Corpus name." },
|
|
867
|
+
"type": { "type": "string", "description": "Corpus type. 'ner' covers named-entity annotation sets (e.g. MasakhaNER).", "enum": ["parallel", "monolingual", "speech", "nmt", "ner"] },
|
|
868
|
+
"url": { "type": ["string", "null"], "description": "URL to the corpus." },
|
|
869
|
+
"size": { "type": ["string", "null"], "description": "Approximate size (e.g., '31K pairs', '97 hours', '2.3M sentences')." },
|
|
870
|
+
"pairLanguages": {
|
|
871
|
+
"type": ["array", "null"],
|
|
872
|
+
"description": "Language pairs available (e.g., ['en-crk', 'fr-crk']). Only for parallel/nmt type.",
|
|
873
|
+
"items": { "type": "string" }
|
|
874
|
+
},
|
|
875
|
+
"license": { "type": ["string", "null"], "description": "License (e.g., 'CC-BY-4.0', 'CC-BY-NC-4.0')." },
|
|
876
|
+
"domain": {
|
|
877
|
+
"type": ["string", "null"],
|
|
878
|
+
"description": "Primary domain (e.g., 'news', 'bible', 'government', 'web', 'mixed', 'conversation')."
|
|
879
|
+
},
|
|
880
|
+
"exposure": {
|
|
881
|
+
"type": ["string", "null"],
|
|
882
|
+
"description": "Data exposure level for contamination assessment. 'open-web': publicly available and easily scraped (likely in web-crawled training sets). 'proprietary-distributed': obtainable but not freely scrapable (behind auth/license). 'private-secret': held-out, never published (only available to project members).",
|
|
883
|
+
"enum": [null, "open-web", "proprietary-distributed", "private-secret"]
|
|
884
|
+
},
|
|
885
|
+
"notes": { "type": ["string", "null"], "description": "Free-form notes." }
|
|
886
|
+
}
|
|
887
|
+
}
|
|
888
|
+
},
|
|
889
|
+
"models": {
|
|
890
|
+
"type": "array",
|
|
891
|
+
"description": "Pretrained NLP/MT models available for this language.",
|
|
892
|
+
"items": {
|
|
893
|
+
"type": "object",
|
|
894
|
+
"properties": {
|
|
895
|
+
"name": { "type": "string" },
|
|
896
|
+
"url": { "type": ["string", "null"] },
|
|
897
|
+
"type": { "type": ["string", "null"] }
|
|
898
|
+
}
|
|
899
|
+
}
|
|
900
|
+
},
|
|
901
|
+
"tools": {
|
|
902
|
+
"type": "array",
|
|
903
|
+
"description": "Other NLP tools (tokenizers, diacritic restorers, language ID, etc.).",
|
|
904
|
+
"items": {
|
|
905
|
+
"type": "object",
|
|
906
|
+
"properties": {
|
|
907
|
+
"name": { "type": "string" },
|
|
908
|
+
"url": { "type": ["string", "null"] },
|
|
909
|
+
"type": { "type": ["string", "null"] }
|
|
910
|
+
}
|
|
911
|
+
}
|
|
912
|
+
},
|
|
913
|
+
"dictionaries": {
|
|
914
|
+
"type": "array",
|
|
915
|
+
"description": "Dictionaries and machine-readable lexical databases available for this language — resource EXISTENCE pointers only (name + URL + license flags), never dictionary content and never scores. Populated by derive-dictionaries.mjs, which promotes the free-form encyclopedic.resources.dictionaries entries and dictionary-shaped resources.lexical entries into this schematized form.",
|
|
916
|
+
"items": {
|
|
917
|
+
"type": "object",
|
|
918
|
+
"required": ["name"],
|
|
919
|
+
"properties": {
|
|
920
|
+
"name": { "type": "string", "description": "Dictionary/lexical-database name (e.g., 'itwêwina (Plains Cree Dictionary)')." },
|
|
921
|
+
"url": { "type": ["string", "null"], "description": "URL to the resource. Null only when no stable URL is known." },
|
|
922
|
+
"license": { "type": ["string", "null"], "description": "License if actually known (e.g., 'CC-BY-4.0'). Omitted when unknown — never guessed." },
|
|
923
|
+
"machineReadable": { "type": "boolean", "description": "Whether the content is available in structured, machine-readable form (database/API/CLDF), not just human-browsable pages. Omitted when unknown." },
|
|
924
|
+
"redistributable": { "type": "boolean", "description": "Whether the underlying content may be redistributed. false = pointer-only resource: the content must never be copied into corpora, cards, exports, or public repos (e.g. itwêwina — the Wolvengrey CW / Maskwacîs MD / AECD content is not openly licensed, permission pending). Omitted when unknown." },
|
|
925
|
+
"source": { "type": ["string", "null"], "description": "Source id or card field this entry was promoted from (e.g., 'encyclopedic.resources.dictionaries', 'abvd-lexibank')." }
|
|
926
|
+
}
|
|
927
|
+
}
|
|
928
|
+
},
|
|
929
|
+
"grammars": {
|
|
930
|
+
"type": "array",
|
|
931
|
+
"description": "Bibliographic reference grammars — citation metadata only (author/year/title/URL), the MED-best few (≤3) per language, filtered to grammar-type records. Populated by enrich-grammars-from-glottolog.mjs from Glottolog's MED (Most Extensive Description) citation records: documentationDepth.med proves a grammar exists, these entries name it. Resource existence, never content and never scores.",
|
|
932
|
+
"items": {
|
|
933
|
+
"type": "object",
|
|
934
|
+
"required": ["title"],
|
|
935
|
+
"properties": {
|
|
936
|
+
"author": { "type": ["string", "null"], "description": "Author(s) as given by the source record (Glottolog reference record, or a publisher archive record such as SIL REAP)." },
|
|
937
|
+
"year": { "type": ["integer", "string", "null"], "description": "Publication year as given by the reference record." },
|
|
938
|
+
"title": { "type": "string", "description": "Title of the reference grammar." },
|
|
939
|
+
"url": { "type": ["string", "null"], "description": "Stable reference URL — a Glottolog reference page (https://glottolog.org/resource/reference/id/<refid>) or a publisher archive page (e.g., SIL REAP https://www.sil.org/resources/archives/<id>)." },
|
|
940
|
+
"type": { "type": ["string", "null"], "description": "hhtype-style label of the record (Glottolog hhtype vocabulary, e.g., 'grammar', 'grammar_sketch', 'phonology'); null when the work does not fit that vocabulary." }
|
|
941
|
+
}
|
|
942
|
+
}
|
|
943
|
+
}
|
|
944
|
+
}
|
|
945
|
+
},
|
|
946
|
+
"experts": {
|
|
947
|
+
"type": ["array", "null"],
|
|
948
|
+
"description": "Researchers, institutions, and projects behind the resources this card cites. Used for collaboration outreach. 100% fact-based: every entry is derived from data already on the card (a cited resource, FST maintainer, corpus publisher) and carries a `source` field naming the card field/resource it came from. Never invented — public professional info only. Auto-populated by derive-experts.mjs.",
|
|
949
|
+
"items": {
|
|
950
|
+
"type": "object",
|
|
951
|
+
"required": ["name", "type", "role", "source"],
|
|
952
|
+
"properties": {
|
|
953
|
+
"name": {
|
|
954
|
+
"type": "string",
|
|
955
|
+
"description": "Name of the person, institution, or project (e.g., 'GiellaLT', 'University of Alberta ALTLab', 'Arok Wolvengrey')."
|
|
956
|
+
},
|
|
957
|
+
"type": {
|
|
958
|
+
"type": "string",
|
|
959
|
+
"description": "Kind of contact: an individual researcher, a formal institution (university, academy, lab), a community organization, or an open project/team.",
|
|
960
|
+
"enum": ["researcher", "institution", "community_org", "project"]
|
|
961
|
+
},
|
|
962
|
+
"affiliation": {
|
|
963
|
+
"type": ["string", "null"],
|
|
964
|
+
"description": "Host institution if applicable (e.g., 'UiT The Arctic University of Norway' for GiellaLT). Null if unknown or not applicable."
|
|
965
|
+
},
|
|
966
|
+
"role": {
|
|
967
|
+
"type": "string",
|
|
968
|
+
"description": "Relationship to this language's resources (e.g., 'FST maintainer', 'dictionary author', 'corpus publisher', 'language archive')."
|
|
969
|
+
},
|
|
970
|
+
"url": {
|
|
971
|
+
"type": ["string", "null"],
|
|
972
|
+
"description": "Public homepage or project URL. Null if none known."
|
|
973
|
+
},
|
|
974
|
+
"orcid": {
|
|
975
|
+
"type": ["string", "null"],
|
|
976
|
+
"description": "ORCID identifier for individual researchers (e.g., '0000-0002-1825-0097'). Null if unknown or not a person."
|
|
977
|
+
},
|
|
978
|
+
"source": {
|
|
979
|
+
"type": "string",
|
|
980
|
+
"description": "Which card field or resource this entry was derived from (e.g., 'resources.fsts[GiellaLT Plains Cree FST (lang-crk)].install.repo', 'encyclopedic.resources.foundations', 'digitalPresence.tatoeba'). Required for fact-traceability."
|
|
981
|
+
}
|
|
982
|
+
}
|
|
983
|
+
}
|
|
984
|
+
},
|
|
985
|
+
"vitality": {
|
|
986
|
+
"type": ["object", "null"],
|
|
987
|
+
"description": "Language vitality and endangerment status. Used for LRL prioritization and partnership planning.",
|
|
988
|
+
"properties": {
|
|
989
|
+
"unescoStatus": {
|
|
990
|
+
"type": ["string", "null"],
|
|
991
|
+
"description": "UNESCO vitality classification.",
|
|
992
|
+
"enum": [null, "safe", "vulnerable", "definitely-endangered", "severely-endangered", "critically-endangered", "extinct"]
|
|
993
|
+
},
|
|
994
|
+
"egids": {
|
|
995
|
+
"type": ["string", "null"],
|
|
996
|
+
"description": "Ethnologue EGIDS level (0-10). See https://www.ethnologue.com/about/language-status"
|
|
997
|
+
},
|
|
998
|
+
"speakerCount": {
|
|
999
|
+
"type": ["string", "integer", "null"],
|
|
1000
|
+
"description": "The single displayed speaker count. Most cards carry the integer reconciled by derive-speaker-vitality-bridge.mjs; a string range ('~50M', '20K-25K') is also allowed. INDEX-NOT-ARBITER: this is one CITED value — it must equal an entry in speakerEstimates[] (the full attributed spread), or be a deliberate curated/blended figure carrying champollion-derived provenance in speakerCountSource. Never a bare upstream name whose own estimate disagrees with the number. Enforced by lint rule R2."
|
|
1001
|
+
},
|
|
1002
|
+
"speakerCountSource": {
|
|
1003
|
+
"type": ["string", "null"],
|
|
1004
|
+
"description": "Provenance of vitality.speakerCount specifically (distinct from vitality.source, which sources the vitality STATUS block). Either the source name of the speakerEstimates[] entry the count equals (e.g. 'wikidata', 'linguameta'), or 'champollion-derived' for a curated/blended figure. Stamped by derive-speaker-vitality-bridge.mjs."
|
|
1005
|
+
},
|
|
1006
|
+
"source": {
|
|
1007
|
+
"type": ["string", "null"],
|
|
1008
|
+
"description": "Source of the vitality STATUS fields (unescoStatus/egids/aes/elcat). NOT the source of speakerCount — see speakerCountSource."
|
|
1009
|
+
},
|
|
1010
|
+
"trend": {
|
|
1011
|
+
"type": ["string", "null"],
|
|
1012
|
+
"description": "Direction of speaker population trend. 'critically-declining' and 'shifting' (language shift in progress) come from the ELCat-derived trend bridge (enum reconciled 2026-07-07).",
|
|
1013
|
+
"enum": [null, "growing", "stable", "declining", "rapidly-declining", "critically-declining", "shifting", "moribund"]
|
|
1014
|
+
},
|
|
1015
|
+
"notes": {
|
|
1016
|
+
"type": ["string", "null"],
|
|
1017
|
+
"description": "Context notes (e.g., 'Resilient despite small population — co-official in Paraguay')."
|
|
1018
|
+
}
|
|
1019
|
+
}
|
|
1020
|
+
},
|
|
1021
|
+
"speakerEstimates": {
|
|
1022
|
+
"type": ["array", "object"],
|
|
1023
|
+
"description": "Speaker count estimates from multiple sources. Each entry has a source, count, and optional date. ATLAS SHAPE: may instead be an ATTRIBUTION ENVELOPE {agreement, consensus, values:[{value, source}]} — 2,173 languages have sources that genuinely disagree on speaker counts, and the envelope reports all of them rather than picking one. Semantics are enforced by the R1-R6 card-integrity rules (R2 requires a displayed vitality.speakerCount to match a cited estimate), not by this structural schema.",
|
|
1024
|
+
"items": {
|
|
1025
|
+
"type": "object",
|
|
1026
|
+
"properties": {
|
|
1027
|
+
"source": { "type": "string", "description": "Data source (e.g., 'wikidata', 'linguameta', 'ethnologue')." },
|
|
1028
|
+
"count": { "type": "integer", "description": "Estimated number of speakers." },
|
|
1029
|
+
"date": { "type": ["string", "null"], "description": "Date of the estimate (ISO 8601)." },
|
|
1030
|
+
"type": { "type": ["string", "null"], "description": "Type of speaker count: 'L1', 'L2', 'total'." }
|
|
1031
|
+
}
|
|
1032
|
+
}
|
|
1033
|
+
},
|
|
1034
|
+
"documentationDepth": {
|
|
1035
|
+
"type": ["object", "null"],
|
|
1036
|
+
"description": "Most Extensive Description (MED) level from Glottolog. Indicates depth of grammatical documentation.",
|
|
1037
|
+
"properties": {
|
|
1038
|
+
"med": { "type": ["string", "null"], "description": "MED category (e.g., 'long_grammar', 'grammar', 'grammar_sketch', 'phonology', 'wordlist')." },
|
|
1039
|
+
"medLevel": { "type": ["integer", "null"], "description": "Numeric MED level (0=highest documentation)." },
|
|
1040
|
+
"source": { "type": ["string", "null"], "description": "Data source and version." }
|
|
1041
|
+
}
|
|
1042
|
+
},
|
|
1043
|
+
"digitalPresence": {
|
|
1044
|
+
"type": ["object", "null"],
|
|
1045
|
+
"description": "Digital presence indicators: Wikipedia edition, CommonVoice hours, Tatoeba sentence count, etc.",
|
|
1046
|
+
"properties": {
|
|
1047
|
+
"wikipedia": {
|
|
1048
|
+
"type": ["object", "null"],
|
|
1049
|
+
"properties": {
|
|
1050
|
+
"code": { "type": "string" },
|
|
1051
|
+
"url": { "type": "string" },
|
|
1052
|
+
"source": { "type": ["string", "null"] }
|
|
1053
|
+
}
|
|
1054
|
+
}
|
|
1055
|
+
}
|
|
1056
|
+
},
|
|
1057
|
+
"dialectCount": {
|
|
1058
|
+
"type": ["integer", "null"],
|
|
1059
|
+
"description": "Number of recognized dialects or varieties. From Glottolog child language count or manual research."
|
|
1060
|
+
},
|
|
1061
|
+
"pipelineReadiness": {
|
|
1062
|
+
"type": ["object", "null"],
|
|
1063
|
+
"description": "Readiness assessment for the Champollion FST-gated translation pipeline, as computed by derive-pipeline-readiness.mjs: a 0–100 score over weighted resource-existence components, bucketed into tiers. source is always 'derived' (this is a Champollion computation, never an upstream assertion). (Schema reconciled to the writer 2026-07-07 — the earlier tier-1-ready/hasFST shape was a design draft the generator never produced.)",
|
|
1064
|
+
"properties": {
|
|
1065
|
+
"score": {
|
|
1066
|
+
"type": "integer",
|
|
1067
|
+
"description": "Weighted resource-existence score (0–100).",
|
|
1068
|
+
"minimum": 0,
|
|
1069
|
+
"maximum": 100
|
|
1070
|
+
},
|
|
1071
|
+
"tier": {
|
|
1072
|
+
"type": "string",
|
|
1073
|
+
"description": "Score bucket: minimal < low < moderate < good < strong.",
|
|
1074
|
+
"enum": ["minimal", "low", "moderate", "good", "strong"]
|
|
1075
|
+
},
|
|
1076
|
+
"components": {
|
|
1077
|
+
"type": "object",
|
|
1078
|
+
"description": "Per-resource existence booleans that fed the score (commonVoice, wikipedia, opus, lexibank, ud, keyboard, resources, googleTranslate, deepl, nllb).",
|
|
1079
|
+
"additionalProperties": { "type": "boolean" }
|
|
1080
|
+
},
|
|
1081
|
+
"source": {
|
|
1082
|
+
"type": "string",
|
|
1083
|
+
"description": "Always 'derived' — provenance marker.",
|
|
1084
|
+
"enum": ["derived"]
|
|
1085
|
+
}
|
|
1086
|
+
},
|
|
1087
|
+
"required": ["score", "tier"]
|
|
1088
|
+
},
|
|
1089
|
+
"regions": {
|
|
1090
|
+
"type": ["array", "null"],
|
|
1091
|
+
"description": "Geographic regions where this language is actively spoken. Each entry represents a country or territory with its official status and optional speaker estimate.",
|
|
1092
|
+
"items": {
|
|
1093
|
+
"type": "object",
|
|
1094
|
+
"required": ["country", "countryCode"],
|
|
1095
|
+
"properties": {
|
|
1096
|
+
"country": {
|
|
1097
|
+
"type": "string",
|
|
1098
|
+
"description": "Country or territory display name."
|
|
1099
|
+
},
|
|
1100
|
+
"countryCode": {
|
|
1101
|
+
"type": "string",
|
|
1102
|
+
"description": "ISO 3166-1 alpha-2 country code (e.g., 'US', 'PH', 'NZ').",
|
|
1103
|
+
"pattern": "^[A-Z]{2}$"
|
|
1104
|
+
},
|
|
1105
|
+
"officialStatus": {
|
|
1106
|
+
"type": ["string", "null"],
|
|
1107
|
+
"description": "Official status of the language in this country.",
|
|
1108
|
+
"enum": [null, "official", "co-official", "recognized", "regional", "widely-spoken", "minority", "diaspora"]
|
|
1109
|
+
},
|
|
1110
|
+
"region": {
|
|
1111
|
+
"type": ["string", "null"],
|
|
1112
|
+
"description": "Specific region within the country (e.g., 'Bicol Region', 'North Island')."
|
|
1113
|
+
},
|
|
1114
|
+
"speakerEstimate": {
|
|
1115
|
+
"type": ["string", "null"],
|
|
1116
|
+
"description": "Approximate speaker population in this region (e.g., '~2.5M', '800K')."
|
|
1117
|
+
},
|
|
1118
|
+
"coordinates": {
|
|
1119
|
+
"type": ["array", "null"],
|
|
1120
|
+
"description": "Geographic centroid [longitude, latitude] for map marker placement. Use the center of the actual speaking area, not the country capital. E.g., [-106.6, 52.1] for Saskatchewan (Plains Cree), not [-75.7, 45.4] (Ottawa).",
|
|
1121
|
+
"items": { "type": "number" },
|
|
1122
|
+
"minItems": 2,
|
|
1123
|
+
"maxItems": 2
|
|
1124
|
+
},
|
|
1125
|
+
"admin1Codes": {
|
|
1126
|
+
"type": ["array", "null"],
|
|
1127
|
+
"description": "ISO 3166-2 codes for the specific provinces/states where the language is spoken. E.g., ['CA-SK', 'CA-AB', 'CA-MB'] for Plains Cree. Used for map boundary highlighting.",
|
|
1128
|
+
"items": { "type": "string" }
|
|
1129
|
+
}
|
|
1130
|
+
}
|
|
1131
|
+
}
|
|
1132
|
+
},
|
|
1133
|
+
"macroarea": {
|
|
1134
|
+
"type": ["string", "null"],
|
|
1135
|
+
"description": "Glottolog macroarea: the broad geographic region where the language is primarily spoken. A language can genuinely span two — Acehnese is Eurasia and Papunesia, Egyptian Arabic is Africa and Eurasia — and Glottolog says so, so the card joins them with ';' rather than picking one (8 of 8,685 cards). The joined form is kept, not reshaped into an array, because CLDF's own LanguageTable Macroarea column and the Supabase index row are both single-valued; splitting here would have to be undone at every consumer. The pattern below validates each member against the same vocabulary the enum used to.",
|
|
1136
|
+
"pattern": "^(Africa|Australia|Eurasia|North America|Papunesia|South America)(;(Africa|Australia|Eurasia|North America|Papunesia|South America))*$"
|
|
1137
|
+
},
|
|
1138
|
+
"coordinates": {
|
|
1139
|
+
"type": ["object", "null"],
|
|
1140
|
+
"description": "Geographic coordinates of the language's primary area. From Glottolog or manual assignment.",
|
|
1141
|
+
"properties": {
|
|
1142
|
+
"lat": { "type": "number", "description": "Latitude in decimal degrees." },
|
|
1143
|
+
"lng": { "type": "number", "description": "Longitude in decimal degrees." },
|
|
1144
|
+
"source": { "type": ["string", "null"], "description": "Data source (e.g., 'glottolog-5.3')." }
|
|
1145
|
+
}
|
|
1146
|
+
},
|
|
1147
|
+
"countries": {
|
|
1148
|
+
"type": "array",
|
|
1149
|
+
"description": "ISO 3166-1 alpha-2 country codes where this language is spoken.",
|
|
1150
|
+
"items": { "type": "string", "pattern": "^[A-Z]{2}$" }
|
|
1151
|
+
},
|
|
1152
|
+
"arealContext": {
|
|
1153
|
+
"type": ["object", "null"],
|
|
1154
|
+
"description": "Areal-linguistic context: Sprachbund membership, contact zones, and convergence area details."
|
|
1155
|
+
},
|
|
1156
|
+
"numeralSystem": {
|
|
1157
|
+
"type": ["object", "null"],
|
|
1158
|
+
"description": "Numeral system properties: counting base, system type, and complexity. From Numeralbank/Chan's database. Null if undocumented.",
|
|
1159
|
+
"properties": {
|
|
1160
|
+
"base": {
|
|
1161
|
+
"type": ["string", "integer", "null"],
|
|
1162
|
+
"description": "VERBATIM from Chan's Base column, and a STRING because Chan records 94 distinct bases including 'binary', 'body tally', 'quinary-vigesimal' and 'complicated'. Coercing these to an integer destroyed exactly the numeral systems that change an MT pipeline. Primary counting base (e.g., 10 for decimal, 20 for vigesimal, 5 for quinary). Null if not determinable."
|
|
1163
|
+
},
|
|
1164
|
+
"baseType": {
|
|
1165
|
+
"type": ["string", "null"],
|
|
1166
|
+
"description": "Named base type: 'decimal', 'vigesimal', 'quinary', 'octal', 'duodecimal', 'senary', 'mixed', 'body-part', 'restricted'. Null if undocumented.",
|
|
1167
|
+
"enum": ["decimal", "vigesimal", "quinary", "octal", "duodecimal", "senary", "mixed", "body-part", "restricted", null]
|
|
1168
|
+
},
|
|
1169
|
+
"highestDocumented": {
|
|
1170
|
+
"type": ["integer", "null"],
|
|
1171
|
+
"description": "Highest numeral documented in the source data. Null if unknown."
|
|
1172
|
+
},
|
|
1173
|
+
"bodyPartCounting": {
|
|
1174
|
+
"type": ["boolean", "null"],
|
|
1175
|
+
"description": "Whether the language uses a body-part counting system. Null if unknown."
|
|
1176
|
+
},
|
|
1177
|
+
"source": {
|
|
1178
|
+
"type": ["string", "null"],
|
|
1179
|
+
"description": "Data source identifier (e.g., 'numeralbank-2023', 'channumerals')."
|
|
1180
|
+
}
|
|
1181
|
+
}
|
|
1182
|
+
},
|
|
1183
|
+
"colexificationProfile": {
|
|
1184
|
+
"type": ["object", "null"],
|
|
1185
|
+
"description": "Cross-linguistic colexification data from CLICS³. Documents which semantic concepts share the same word form in this language. Null if not covered in CLICS³.",
|
|
1186
|
+
"properties": {
|
|
1187
|
+
"conceptsDocumented": {
|
|
1188
|
+
"type": ["integer", "null"],
|
|
1189
|
+
"description": "Number of Concepticon concept sets documented for this language."
|
|
1190
|
+
},
|
|
1191
|
+
"colexificationCount": {
|
|
1192
|
+
"type": ["integer", "null"],
|
|
1193
|
+
"description": "Number of colexification pairs (meanings sharing the same word form)."
|
|
1194
|
+
},
|
|
1195
|
+
"notableColexifications": {
|
|
1196
|
+
"type": ["array", "null"],
|
|
1197
|
+
"description": "Notable colexification patterns (e.g., HAND-ARM, EAR-LEAF).",
|
|
1198
|
+
"items": {
|
|
1199
|
+
"type": "object",
|
|
1200
|
+
"properties": {
|
|
1201
|
+
"concepts": {
|
|
1202
|
+
"type": "array",
|
|
1203
|
+
"items": { "type": "string" },
|
|
1204
|
+
"description": "Pair of Concepticon concept labels that share a word form."
|
|
1205
|
+
}
|
|
1206
|
+
}
|
|
1207
|
+
}
|
|
1208
|
+
},
|
|
1209
|
+
"source": {
|
|
1210
|
+
"type": ["string", "null"],
|
|
1211
|
+
"description": "Data source identifier (e.g., 'clics3-2020')."
|
|
1212
|
+
}
|
|
1213
|
+
}
|
|
1214
|
+
},
|
|
1215
|
+
"archivePresence": {
|
|
1216
|
+
"type": ["object", "null"],
|
|
1217
|
+
"description": "Presence in language documentation archives (PARADISEC, ELAR, AILLA, etc.). Tracks how much archived material exists for this language. Null if not present in any archive.",
|
|
1218
|
+
"properties": {
|
|
1219
|
+
"paradisec": {
|
|
1220
|
+
"type": ["object", "null"],
|
|
1221
|
+
"description": "PARADISEC (Pacific And Regional Archive) holdings.",
|
|
1222
|
+
"properties": {
|
|
1223
|
+
"itemCount": { "type": ["integer", "null"], "description": "Number of archived items (recordings, transcripts, etc.). Null when the OAI harvest matched the language but returned no usable count." },
|
|
1224
|
+
"collectionCount": { "type": ["integer", "null"], "description": "Number of distinct collections. Null when not reported by the harvest." },
|
|
1225
|
+
"mediaTypes": {
|
|
1226
|
+
"type": ["array", "null"],
|
|
1227
|
+
"items": { "type": "string" },
|
|
1228
|
+
"description": "Types of media available (e.g., 'audio', 'video', 'text'). Null when not reported."
|
|
1229
|
+
}
|
|
1230
|
+
}
|
|
1231
|
+
},
|
|
1232
|
+
"source": {
|
|
1233
|
+
"type": ["string", "null"],
|
|
1234
|
+
"description": "Data source identifier (e.g., 'paradisec-oai-2025')."
|
|
1235
|
+
}
|
|
1236
|
+
}
|
|
1237
|
+
},
|
|
1238
|
+
"culturalAphorism": {
|
|
1239
|
+
"type": ["object", "null"],
|
|
1240
|
+
"description": "An iconic proverb or saying that encapsulates the language community's worldview. Must be a real, documented proverb — never fabricated. Null if no verifiable proverb is available.",
|
|
1241
|
+
"properties": {
|
|
1242
|
+
"text": {
|
|
1243
|
+
"type": "string",
|
|
1244
|
+
"description": "The aphorism in the original language/script (e.g., Arabic script for Arabic, Devanagari for Hindi)."
|
|
1245
|
+
},
|
|
1246
|
+
"transliteration": {
|
|
1247
|
+
"type": ["string", "null"],
|
|
1248
|
+
"description": "Romanized form if the original script is non-Latin. Null for Latin-script languages."
|
|
1249
|
+
},
|
|
1250
|
+
"translation": {
|
|
1251
|
+
"type": "string",
|
|
1252
|
+
"description": "English translation of the aphorism."
|
|
1253
|
+
},
|
|
1254
|
+
"literal": {
|
|
1255
|
+
"type": ["string", "null"],
|
|
1256
|
+
"description": "Literal word-for-word translation if the idiomatic meaning differs significantly. Null if translation is already literal."
|
|
1257
|
+
},
|
|
1258
|
+
"source": {
|
|
1259
|
+
"type": ["string", "null"],
|
|
1260
|
+
"description": "Attribution or cultural context (e.g., 'Traditional proverb', 'ʻŌlelo Noʻeau', 'Whakatauki')."
|
|
1261
|
+
}
|
|
1262
|
+
},
|
|
1263
|
+
"required": ["text", "translation"]
|
|
1264
|
+
},
|
|
1265
|
+
"varieties": {
|
|
1266
|
+
"type": ["array", "null"],
|
|
1267
|
+
"description": "Major dialect/variety differences relevant to MT. Critical for macrolanguages where FSTs and corpora may cover different varieties.",
|
|
1268
|
+
"items": {
|
|
1269
|
+
"type": "object",
|
|
1270
|
+
"required": ["name"],
|
|
1271
|
+
"properties": {
|
|
1272
|
+
"name": { "type": "string", "description": "Variety name (e.g., 'Cusco Quechua', 'Ayacucho Quechua')." },
|
|
1273
|
+
"iso639_3": { "type": ["string", "null"], "description": "ISO 639-3 code if the variety has its own code." },
|
|
1274
|
+
"region": { "type": ["string", "null"], "description": "Primary region." },
|
|
1275
|
+
"fstCoverage": { "type": ["boolean", "null"], "description": "Whether this variety is covered by the listed FST(s)." },
|
|
1276
|
+
"corpusCoverage": { "type": ["boolean", "null"], "description": "Whether parallel corpora exist for this variety." },
|
|
1277
|
+
"nllbCoverage": { "type": ["boolean", "null"], "description": "Whether NLLB-200 covers this variety." },
|
|
1278
|
+
"mutualIntelligibility": { "type": ["string", "null"], "description": "Intelligibility with the card's primary variety." },
|
|
1279
|
+
"notes": { "type": ["string", "null"] }
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
},
|
|
1283
|
+
"codeSwitching": {
|
|
1284
|
+
"type": ["object", "null"],
|
|
1285
|
+
"description": "Active code-switching patterns that affect MT input/output processing. Different from contactInfluences (historical) — this tracks live, current, active code-switching.",
|
|
1286
|
+
"properties": {
|
|
1287
|
+
"contactLanguage": { "type": "string", "description": "Name of the contact language." },
|
|
1288
|
+
"contactIso639_3": { "type": ["string", "null"], "description": "ISO 639-3 code of the contact language." },
|
|
1289
|
+
"mixedVarietyName": { "type": ["string", "null"], "description": "Named mixed variety (e.g., 'Jopará', 'Taglish', 'Hinglish')." },
|
|
1290
|
+
"prevalence": {
|
|
1291
|
+
"type": "string",
|
|
1292
|
+
"description": "How common code-switching is in real-world input.",
|
|
1293
|
+
"enum": ["rare", "common", "dominant"]
|
|
1294
|
+
},
|
|
1295
|
+
"morphologicalIntegration": {
|
|
1296
|
+
"type": "boolean",
|
|
1297
|
+
"description": "Whether borrowed words take target-language morphology (e.g., Guaraní prefixes on Spanish roots)."
|
|
1298
|
+
},
|
|
1299
|
+
"pipelineStrategy": {
|
|
1300
|
+
"type": ["string", "null"],
|
|
1301
|
+
"description": "Recommended pipeline strategy (e.g., 'hybrid-fst', 'language-id-preprocessing', 'ignore')."
|
|
1302
|
+
},
|
|
1303
|
+
"notes": { "type": ["string", "null"] }
|
|
1304
|
+
},
|
|
1305
|
+
"required": ["contactLanguage", "prevalence", "morphologicalIntegration"]
|
|
1306
|
+
}
|
|
1307
|
+
}
|
|
1308
|
+
}
|