champollion 0.3.3 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -37
- package/bin/cli.js +53 -5
- package/index.js +63 -2
- package/lib/api-key.js +17 -4
- package/lib/autofix.js +83 -36
- package/lib/bridge/method_bridge.py +15 -3
- package/lib/cards/reader.js +51 -3
- package/lib/cards/remote.js +15 -0
- package/lib/cards/search-names.js +178 -0
- package/lib/command-help.js +289 -88
- package/lib/commands/audit.js +10 -3
- package/lib/commands/card.js +583 -226
- package/lib/commands/doctor.js +54 -18
- package/lib/commands/help.js +37 -32
- package/lib/commands/init.js +1689 -87
- package/lib/commands/integrity.js +127 -40
- package/lib/commands/leaderboard.js +187 -67
- package/lib/commands/models.js +9 -2
- package/lib/commands/provenance.js +7 -2
- package/lib/commands/recommend.js +43 -14
- package/lib/commands/register-corpus.js +649 -130
- package/lib/commands/seal-corpus.js +1 -1
- package/lib/commands/status.js +564 -27
- package/lib/commands/submit.js +17 -12
- package/lib/commands/sync.js +31 -7
- package/lib/commands/tm.js +16 -10
- package/lib/commands/verify.js +27 -3
- package/lib/commands/wrap.js +63 -5
- package/lib/commands/xliff.js +135 -64
- package/lib/commercial-eligibility.js +1 -1
- package/lib/config.js +196 -14
- package/lib/content-estimate.js +96 -0
- package/lib/content-refusals.js +270 -0
- package/lib/content-review.js +372 -0
- package/lib/content-sync.js +1127 -344
- package/lib/content.js +94 -7
- package/lib/corpus-registration.mjs +197 -38
- package/lib/cost-label.js +29 -0
- package/lib/cost-report.js +726 -78
- package/lib/diff.js +38 -4
- package/lib/docusaurus-sync.js +965 -253
- package/lib/edit-distance.js +31 -0
- package/lib/fallback.js +964 -0
- package/lib/file-scope.js +106 -0
- package/lib/flatten.js +80 -3
- package/lib/flutter-locales.js +124 -0
- package/lib/format.js +266 -12
- package/lib/hash.js +146 -21
- package/lib/icu-structure.js +929 -0
- package/lib/integrity.js +223 -75
- package/lib/language-pair.js +157 -0
- package/lib/lint.js +78 -16
- package/lib/local-only-marks.js +106 -0
- package/lib/locale-layout.js +1103 -0
- package/lib/locale-state.js +571 -0
- package/lib/methods/anthropic.js +5 -0
- package/lib/methods/apertium.js +6 -3
- package/lib/methods/api.js +138 -25
- package/lib/methods/base.js +17 -0
- package/lib/methods/coaching-data.js +153 -0
- package/lib/methods/content-separator.js +43 -0
- package/lib/methods/deepl.js +1 -1
- package/lib/methods/direct-llm.js +252 -103
- package/lib/methods/external.js +146 -63
- package/lib/methods/gemini.js +1 -0
- package/lib/methods/google-translate.js +1 -0
- package/lib/methods/http-utils.js +41 -0
- package/lib/methods/libretranslate.js +7 -2
- package/lib/methods/llm-coached.js +68 -128
- package/lib/methods/llm.js +80 -31
- package/lib/methods/local.js +93 -10
- package/lib/methods/microsoft-translator.js +1 -2
- package/lib/methods/openai.js +4 -2
- package/lib/methods/openrouter-client.js +20 -19
- package/lib/methods/openrouter-pricing.js +150 -13
- package/lib/methods/prompt-methods.js +20 -0
- package/lib/methods/provider-pricing.js +42 -1
- package/lib/methods/request-capture.js +104 -0
- package/lib/methods/tilde.js +1 -1
- package/lib/methods/translated.js +1 -2
- package/lib/missing-key.js +93 -0
- package/lib/models.js +11 -0
- package/lib/name-rules.js +32 -0
- package/lib/named-keys.js +172 -0
- package/lib/no-translate.js +4 -3
- package/lib/output.js +160 -19
- package/lib/pairs.js +586 -30
- package/lib/placeholders.js +394 -0
- package/lib/plugins.js +8 -0
- package/lib/plural-gap-redo.js +109 -0
- package/lib/plurals.js +323 -0
- package/lib/po.js +1187 -0
- package/lib/public-catalogue.js +74 -0
- package/lib/recommend.js +527 -32
- package/lib/redo.js +95 -0
- package/lib/refusal-category.js +44 -0
- package/lib/registers.js +255 -11
- package/lib/repair-script.js +20 -13
- package/lib/scripts.js +193 -106
- package/lib/seal.mjs +6 -5
- package/lib/sealed-qualifier.mjs +2 -2
- package/lib/segment.js +2 -1
- package/lib/seo.js +19 -9
- package/lib/serve.js +43 -6
- package/lib/shared-output-seed.js +164 -0
- package/lib/source-contexts.js +39 -0
- package/lib/submit.mjs +57 -5
- package/lib/sync.js +2923 -474
- package/lib/terminology.js +13 -4
- package/lib/tm-evict.js +179 -0
- package/lib/tm-seed.js +5 -2
- package/lib/tm.js +818 -36
- package/lib/translate-pair.js +639 -34
- package/lib/translate.js +78 -5
- package/lib/types.js +22 -3
- package/lib/validate.js +880 -17
- package/lib/verify.js +1296 -104
- package/lib/watch.js +32 -13
- package/lib/xliff.js +44 -3
- package/package.json +3 -2
- package/shared/CORPORA-CARDS.md +2 -0
- package/shared/DATA-SOVEREIGNTY.md +19 -20
- package/shared/LANGUAGE-CARD-FIELDS.md +1 -1
- package/shared/cards-fallback.json +1 -1
- package/shared/catalogue/card-config.json +1 -1
- package/shared/curated-orthography-conventions.json +26 -8
- package/shared/docent/faq.en.json +14 -16
- package/shared/docent/system-prompt.md +17 -19
- package/shared/explainers/tc-features.json +15 -15
- package/shared/gettext-plural-forms.json +45 -0
- package/shared/human-services.json +1 -1
- package/shared/method-registry.json +2 -0
- package/shared/metric-registry.json +96 -18
- package/shared/schemas/champollion-plugin.schema.json +4 -0
- package/shared/schemas/corpora-card.schema.json +20 -10
- package/shared/schemas/human-services.schema.json +2 -2
- package/shared/schemas/language-card.schema.json +1 -1
- package/shared/schemas/method-card.schema.json +1 -1
- package/shared/schemas/method-index-record.schema.json +67 -0
- package/shared/schemas/method-registry.schema.json +4 -0
- package/shared/schemas/metric-registry.schema.json +55 -1
- package/shared/docent/corpus.json +0 -11333
|
@@ -992,7 +992,7 @@
|
|
|
992
992
|
"import": "champollion_lyss",
|
|
993
993
|
"pip": "champollion-lyss>=0.1,<0.2",
|
|
994
994
|
"description": "Champollion-LYSS measurement plugins — Cree LYSS linter + semantic validator. Fetched on demand from PyPI; the harness core ships no language-specific scorer code.",
|
|
995
|
-
"ipNotice": "Plains Cree (nêhiyawêwin) data sovereignty — please read.\n\nThe Cree LYSS validation standard and any Plains Cree language data are treated as the property of the nêhiyaw language community and are offered for NON-COMMERCIAL, community-benefit use (
|
|
995
|
+
"ipNotice": "Plains Cree (nêhiyawêwin) data sovereignty — please read.\n\nThe Cree LYSS validation standard and any Plains Cree language data are treated as the property of the nêhiyaw language community and are offered for NON-COMMERCIAL, community-benefit use (sovereignty-aspirant — community ownership and control of language data; relational-trust governance in progress).\n\nGloss data is fetched live from the public itwêwina API (itwewina.altlab.app) and cached locally for your own use only. The underlying dictionary content (Wolvengrey CW, Maskwacîs MD, AECD) is NOT openly licensed — do not commit, redistribute, publish, or bundle the cache. The Cree FST (GiellaLT/ALTLab, AGPL) is downloaded and invoked as a separate tool, never bundled.\n\nBy installing and running this standard you agree to respect these terms. Set CHAMPOLLION_LYSS_ACCEPT_IP=1 to acknowledge in automation."
|
|
996
996
|
},
|
|
997
997
|
"evalMetrics": {
|
|
998
998
|
"lyss-eq": {
|
|
@@ -1,25 +1,43 @@
|
|
|
1
1
|
{
|
|
2
2
|
"_comment": [
|
|
3
|
-
"Curated orthographic-convention register
|
|
4
|
-
"
|
|
3
|
+
"Curated orthographic-convention register — the decision-layer input to the",
|
|
4
|
+
"orthographies[] derivation (cli/scripts/derive-orthographies.mjs). NOT WIRED:",
|
|
5
|
+
"that deriver is a standalone pass, not imported by cli/scripts/cldf/project.mjs,",
|
|
6
|
+
"so nothing here reaches a card today. An earlier version of this comment claimed",
|
|
7
|
+
"the wiring existed; it did not. Do not restore that claim before the import does.",
|
|
8
|
+
"",
|
|
9
|
+
"ONE entry per language code, keyed by ISO 15924 script code. Each script entry may set",
|
|
5
10
|
"scheme / longVowelMarking / canonicalForMt and MUST set a source string that cites a",
|
|
6
|
-
"verifiable basis
|
|
7
|
-
"Only add entries whose convention is actually documented
|
|
8
|
-
"deriver writes nothing it cannot cite (index, not
|
|
11
|
+
"verifiable basis — a source the card already carries, a published convention, or a tool",
|
|
12
|
+
"in the repo's register. Only add entries whose convention is actually documented:",
|
|
13
|
+
"absence means unknown, and the deriver writes nothing it cannot cite (index, not",
|
|
14
|
+
"arbiter). Optional keys are OMITTED when unknown, never guessed. Do NOT record measured",
|
|
15
|
+
"scores here — a score of method output is a run result and belongs on the leaderboard.",
|
|
16
|
+
"",
|
|
9
17
|
"Data-over-code (docs/AGENTS.md §1): language-specific conventions live in this file,",
|
|
10
|
-
"never hardcoded in the deriver."
|
|
18
|
+
"never hardcoded in the deriver. Decision-layer inputs to the atlas build live under the",
|
|
19
|
+
"monorepo-root shared/ (parameters.csv, card-field-disposition.json, catalogue/…); this",
|
|
20
|
+
"file moved here from cli/shared/ on 2026-09-06 for that reason — cli/shared/ is the",
|
|
21
|
+
"npm-bundled RUNTIME tree, and nothing reads this register at runtime.",
|
|
22
|
+
"",
|
|
23
|
+
"STANDING VERDICT (shared/cldf/curated-file-verdicts.json, 2026-08-06):",
|
|
24
|
+
"'CITABLE BUT NOT MACHINE-FETCHABLE — keep, with a bibliographic citation added, and",
|
|
25
|
+
"record that it is asserted from the literature rather than fetched.' Every entry below",
|
|
26
|
+
"is asserted from the literature and from tools the card already lists. None is fetched,",
|
|
27
|
+
"so none carries a pin; that is the reason this is a curated register and not a source."
|
|
11
28
|
],
|
|
29
|
+
"version": 2,
|
|
12
30
|
"conventions": {
|
|
13
31
|
"crk": {
|
|
14
32
|
"Latn": {
|
|
15
33
|
"scheme": "SRO",
|
|
16
34
|
"longVowelMarking": "circumflex",
|
|
17
35
|
"canonicalForMt": true,
|
|
18
|
-
"source": "manual-curation (SRO
|
|
36
|
+
"source": "manual-curation (asserted from the literature and from a tool the card already carries — not machine-fetched). SRO is Standard Roman Orthography for Plains Cree; it writes the seven long vowels with circumflexes (â ê î ô) against unmarked short vowels. AUTHORITY: the GiellaLT / UAlbertaALTLab lang-crk finite-state transducer, which the crk card lists under resources.fsts from the giellalt-resources source (https://github.com/giellalt/lang-crk, https://github.com/UAlbertaALTLab/lang-crk). Both analysers accept the hyphenated preverb spelling (ê-nipâyân, nikî-nipân, kâ-nipât) and return the SAME analysis plus the tag Err/Orth — the FST's own spelling-error tag — for the fused spelling, so 'correct SRO' here is the grammar machine's verdict rather than a house preference. CANONICAL-FOR-MT: crk is written in two scripts and the pipeline's working script is the Roman one (shared/catalogue/card-config.json scriptConverter.crk converts between them); the scripts[] `primary` display flag is NOT this signal. REFERENCE LEXICON, cited and never redistributed (founder ruling 2026-07-19, index-only permanently): Wolvengrey, Arok, ed. 2001. nehiyawewin: itwewina / Cree: Words. Regina: Canadian Plains Research Center."
|
|
19
37
|
},
|
|
20
38
|
"Cans": {
|
|
21
39
|
"canonicalForMt": false,
|
|
22
|
-
"source": "manual-curation (Syllabics is
|
|
40
|
+
"source": "manual-curation (asserted, not fetched). Unified Canadian Aboriginal Syllabics is an attested writing system for Plains Cree and stays on the card as an alternative — the language is written both ways and an index says so. It is NOT the pipeline's working form: shared/catalogue/card-config.json registers scriptConverter.crk precisely because syllabics is converted to and from the Roman orthography, and the lang-crk analysers cited on the Latn entry above are SRO-facing. longVowelMarking is OMITTED rather than guessed: syllabics marks vowel length by a diacritic over the syllable, which none of the schema's vocabulary terms (circumflex / macron / double-vowel / none) describes."
|
|
23
41
|
}
|
|
24
42
|
}
|
|
25
43
|
}
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
{
|
|
5
5
|
"id": "what-is-champollion",
|
|
6
6
|
"question": "What is Champollion?",
|
|
7
|
-
"answer": "Champollion is
|
|
7
|
+
"answer": "Champollion is source-available infrastructure for trustworthy machine-translation evaluation across the world's languages, with a focus on low-resource and Indigenous ones. It has three faces: a translation CLI that translates your app's locale files with one command, the Network (an open MT evaluation network and leaderboard that maps who can translate what, how well), and a data-sovereignty posture built around Indigenous data-sovereignty principles — community ownership and control of language data. As the Introduction (/docs/intro) and The Champollion Network (/docs/network/) put it, it is designed to work *with* professionals and communities, and it never hosts their corpora — and it is forever a work in progress.",
|
|
8
8
|
"sources": [
|
|
9
9
|
"/docs/intro",
|
|
10
10
|
"/docs/network/",
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
{
|
|
22
22
|
"id": "free-and-open-source",
|
|
23
23
|
"question": "Is Champollion free? Is it open source? Does it cost anything?",
|
|
24
|
-
"answer": "
|
|
24
|
+
"answer": "It is free for noncommercial use, and partly open source. Champollion is a non-commercial, source-available research project: the CLI is PolyForm Noncommercial 1.0.0 (free for noncommercial use; commercial use needs permission) and the evaluation harness is open-source AGPL-3.0, per How the Work Is Funded (/docs/network/sovereignty/economic-model). There is no paid API, no metering, and no revenue share — the only costs you'd incur are the API tokens you spend on your chosen model provider when you translate or benchmark. Champollion itself takes nothing.",
|
|
25
25
|
"sources": [
|
|
26
26
|
"/docs/network/sovereignty/economic-model",
|
|
27
27
|
"/docs/intro"
|
|
@@ -39,7 +39,7 @@
|
|
|
39
39
|
{
|
|
40
40
|
"id": "the-cli",
|
|
41
41
|
"question": "What is the champollion CLI and what does it do?",
|
|
42
|
-
"answer": "The CLI is the deployment end of Champollion: run `npx champollion sync` and it auto-detects your locale files, format, and target languages, translates what's missing, skips what's done,
|
|
42
|
+
"answer": "The CLI is the deployment end of Champollion: run `npx champollion sync` and it auto-detects your locale files, format, and target languages, translates what's missing, skips what's done, checks every result for broken output (empty, echoed, looping or wrong-script — not wrong meaning) through a quality gate, and writes clean output. Per How It Works (/docs/how-it-works), it tracks every source string with SHA-256 hashes so re-runs only re-translate what changed, caches results in Translation Memory to save cost, and lets you pick a different translation method per language pair. It's a full i18n pipeline — sync, watch, lint, verify, audit, XLIFF export, and more (/docs/intro).",
|
|
43
43
|
"sources": [
|
|
44
44
|
"/docs/intro",
|
|
45
45
|
"/docs/how-it-works"
|
|
@@ -135,7 +135,7 @@
|
|
|
135
135
|
{
|
|
136
136
|
"id": "submit-a-method",
|
|
137
137
|
"question": "How do I submit a benchmark run or a method to the leaderboard?",
|
|
138
|
-
"answer": "Install the harness with `pip install mt-eval`, run it against a registered corpus (for example `mt-eval run --corpus eval-amh-fra-globalvoices-test-v1 --model gemini-pro`), review the run card it produces, then publish with `mt-eval publish` — the full walkthrough is Submit a Method (/docs/network/getting-started/submit-a-method). Your run first appears as self-benchmarked; the server then re-scores your outputs against the sha-pinned corpus and, when it reproduces, promotes it to Champollion Verified. A hosted upload API and web UI are planned but not yet live, so `mt-eval publish` or a pull request to the harness repo are the working paths today.",
|
|
138
|
+
"answer": "Install the harness with `pip install mt-eval-harness`, run it against a registered corpus (for example `mt-eval run --corpus eval-amh-fra-globalvoices-test-v1 --model gemini-pro`), review the run card it produces, then publish with `mt-eval publish` — the full walkthrough is Submit a Method (/docs/network/getting-started/submit-a-method). Your run first appears as self-benchmarked; the server then re-scores your outputs against the sha-pinned corpus and, when it reproduces, promotes it to Champollion Verified. A hosted upload API and web UI are planned but not yet live, so `mt-eval publish` or a pull request to the harness repo are the working paths today.",
|
|
139
139
|
"sources": [
|
|
140
140
|
"/docs/network/getting-started/submit-a-method",
|
|
141
141
|
"/docs/network/leaderboard/rules"
|
|
@@ -228,7 +228,7 @@
|
|
|
228
228
|
{
|
|
229
229
|
"id": "researcher-involved",
|
|
230
230
|
"question": "I'm an ML researcher — how do I get involved?",
|
|
231
|
-
"answer": "Build a method and benchmark it. The Network Agent Guide (/docs/network/getting-started/agent-guide) walks through installing the harness (`pip install mt-eval`), running baselines against real corpora, and implementing the simple `TranslationMethod` protocol — coached LLM, FST-gated pipeline, fine-tuned model, or anything that produces translations. You get standardized, reproducible, fingerprinted scoring against a shared corpus catalogue, and every run adds a point to a shared map of what works. Get Involved (/get-involved) points to the developer paths, including running the public benchmark queue.",
|
|
231
|
+
"answer": "Build a method and benchmark it. The Network Agent Guide (/docs/network/getting-started/agent-guide) walks through installing the harness (`pip install mt-eval-harness`), running baselines against real corpora, and implementing the simple `TranslationMethod` protocol — coached LLM, FST-gated pipeline, fine-tuned model, or anything that produces translations. You get standardized, reproducible, fingerprinted scoring against a shared corpus catalogue, and every run adds a point to a shared map of what works. Get Involved (/get-involved) points to the developer paths, including running the public benchmark queue.",
|
|
232
232
|
"sources": [
|
|
233
233
|
"/docs/network/getting-started/agent-guide",
|
|
234
234
|
"/get-involved",
|
|
@@ -264,17 +264,15 @@
|
|
|
264
264
|
]
|
|
265
265
|
},
|
|
266
266
|
{
|
|
267
|
-
"id": "
|
|
268
|
-
"question": "What does '
|
|
269
|
-
"answer": "
|
|
267
|
+
"id": "sovereignty-aspirant-meaning",
|
|
268
|
+
"question": "What does 'sovereignty-aspirant' mean?",
|
|
269
|
+
"answer": "Champollion's posture is sovereignty-aspirant — built around Indigenous data-sovereignty principles: community ownership and control of language data. As Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) puts it, the design is built so communities *can* exercise ownership and control of their data and anything derived from it — but whether it *achieves* that is for communities to decide, which is exactly why the objections channel exists. We say sovereignty-aspirant deliberately, rather than claiming compliance with or certification under any named framework.",
|
|
270
270
|
"sources": [
|
|
271
271
|
"/docs/network/community/contact-objections-takedown",
|
|
272
272
|
"/docs/network/sovereignty/data-sovereignty"
|
|
273
273
|
],
|
|
274
274
|
"keywords": [
|
|
275
|
-
"
|
|
276
|
-
"ocap aspirant",
|
|
277
|
-
"fnigc",
|
|
275
|
+
"sovereignty aspirant",
|
|
278
276
|
"sovereignty principles",
|
|
279
277
|
"ownership control access possession",
|
|
280
278
|
"indigenous"
|
|
@@ -283,7 +281,7 @@
|
|
|
283
281
|
{
|
|
284
282
|
"id": "data-sovereignty",
|
|
285
283
|
"question": "What is data sovereignty here — do you store or scrape my language data?",
|
|
286
|
-
"answer": "No. Data Stewardship (/docs/network/sovereignty/data-sovereignty) states the position: language data is biodata, so the people who provide a corpus hold the keys to it and to anything measured against it. Champollion never holds the data — corpora are registered as hash-pinned metadata cards and fetched from the steward's own hosting at evaluation time; take your archive offline and evaluation simply stops. Every license and community restriction is respected by gate, not by promise (
|
|
284
|
+
"answer": "No. Data Stewardship (/docs/network/sovereignty/data-sovereignty) states the position: language data is biodata, so the people who provide a corpus hold the keys to it and to anything measured against it. Champollion never holds the data — corpora are registered as hash-pinned metadata cards and fetched from the steward's own hosting at evaluation time; take your archive offline and evaluation simply stops. Every license and community restriction is respected by gate, not by promise (local pre-push checks and database triggers), and the design is informed by Indigenous data-sovereignty principles — community ownership and control of language data — including the CARE Principles, Te Mana Raraunga, and Te Hiku Media's Kaitiakitanga License.",
|
|
287
285
|
"sources": [
|
|
288
286
|
"/docs/network/sovereignty/data-sovereignty",
|
|
289
287
|
"/docs/network/sovereignty/registering-corpora"
|
|
@@ -341,7 +339,7 @@
|
|
|
341
339
|
{
|
|
342
340
|
"id": "speakers-paid",
|
|
343
341
|
"question": "Do speakers get paid, and how much?",
|
|
344
|
-
"answer": "Yes —
|
|
342
|
+
"answer": "Yes, once the work is funded — no speaker work is funded or underway today. Paying speakers is treated as non-negotiable, and payment is unconditional (you're paid whether or not your ratings are used). How Speakers Get Paid (/docs/network/perspectives/how-speakers-get-paid) publishes the rates: bilingual speaker work at $50–65 CAD per hour, with corpus curation budgeted at $2,500–6,000 and a full metric-validation round at $1,475–1,920. Payment does not buy your data — you are paid for the work and remain the steward of what you build. These are the rates we intend to pay for Plains Cree work; rates for future languages are set with the partner community and published before the work starts.",
|
|
345
343
|
"sources": [
|
|
346
344
|
"/docs/network/perspectives/how-speakers-get-paid",
|
|
347
345
|
"/docs/network/sovereignty/economic-model"
|
|
@@ -398,7 +396,7 @@
|
|
|
398
396
|
{
|
|
399
397
|
"id": "ai-agent-usage",
|
|
400
398
|
"question": "How should an AI agent use Champollion?",
|
|
401
|
-
"answer": "Two things are built for agents. First, the machine-readable /llms.txt index maps the whole site, the license lanes, and the public data feeds (queue, mesh, corpus registry). Second, the MCP server
|
|
399
|
+
"answer": "Two things are built for agents. First, the machine-readable /llms.txt index maps the whole site, the license lanes, and the public data feeds (queue, mesh, corpus registry). Second, the MCP server `champollion-mcp-server` (on npm) is the agent-facing door — translate with Translation-Memory caching and a deterministic quality gate, browse the benchmark queue, estimate cost, run benchmarks, read the leaderboard, and query per-language metric-trust evidence, all as tools. The Agent Guides (/docs/guides/agent-guide for the CLI, /docs/network/getting-started/agent-guide for benchmarking) give step-by-step recipes. The best pattern is to set up your own agent with a workspace and your own API key — tokens aren't free.",
|
|
402
400
|
"sources": [
|
|
403
401
|
"/llms.txt",
|
|
404
402
|
"/docs/guides/agent-guide",
|
|
@@ -418,7 +416,7 @@
|
|
|
418
416
|
{
|
|
419
417
|
"id": "contribute-compute",
|
|
420
418
|
"question": "How do I contribute compute by running the benchmark queue?",
|
|
421
|
-
"answer": "The leaderboard has empty squares — (language pair, model, condition) combinations nobody has measured — kept in a public queue at /queue.json. Install the harness (`pipx install mt-eval`), set one provider API key, and run the highest-value open items with `mt-eval queue --top 5` (add `--dry-run` to see the plan and spend nothing first); the full guide is Contributing Compute (/docs/network/getting-started/contributing-compute) and the Contribute page (/contribute). No account is needed — results publish as 'anonymous' unless you sign in to put your name on the board — and every run is a real, citable contribution to low-resource MT evaluation.",
|
|
419
|
+
"answer": "The leaderboard has empty squares — (language pair, model, condition) combinations nobody has measured — kept in a public queue at /queue.json. Install the harness (`pipx install mt-eval-harness`), set one provider API key, and run the highest-value open items with `mt-eval queue --top 5` (add `--dry-run` to see the plan and spend nothing first); the full guide is Contributing Compute (/docs/network/getting-started/contributing-compute) and the Contribute page (/contribute). No account is needed — results publish as 'anonymous' unless you sign in to put your name on the board — and every run is a real, citable contribution to low-resource MT evaluation.",
|
|
422
420
|
"sources": [
|
|
423
421
|
"/docs/network/getting-started/contributing-compute",
|
|
424
422
|
"/contribute"
|
|
@@ -472,7 +470,7 @@
|
|
|
472
470
|
{
|
|
473
471
|
"id": "write-code-for-me",
|
|
474
472
|
"question": "Can the site guide write code, debug, or run the harness for me?",
|
|
475
|
-
"answer": "No — the site guide is an index and a docent, not an engineering service; Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) is explicit that this isn't a place to request development help or custom scripts. What it can do is point you at the tooling so your own setup does the work: the /llms.txt cookbook, the MCP server
|
|
473
|
+
"answer": "No — the site guide is an index and a docent, not an engineering service; Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) is explicit that this isn't a place to request development help or custom scripts. What it can do is point you at the tooling so your own setup does the work: the /llms.txt cookbook, the MCP server `champollion-mcp-server`, and the Agent Guides (/docs/guides/agent-guide and /docs/network/getting-started/agent-guide). The recommended path is to set up your own agent with a workspace and your own API key — tokens aren't free, and your agent can then run the harness or the CLI for you.",
|
|
476
474
|
"sources": [
|
|
477
475
|
"/docs/network/community/contact-objections-takedown",
|
|
478
476
|
"/docs/guides/agent-guide",
|
|
@@ -18,7 +18,8 @@ careful, honest, and humble about what this project is.
|
|
|
18
18
|
|
|
19
19
|
Champollion is open translation infrastructure for low-resource languages: a
|
|
20
20
|
translation CLI, a machine-translation evaluation network ("the Network"), and
|
|
21
|
-
a data-sovereignty posture built around
|
|
21
|
+
a data-sovereignty posture built around Indigenous data-sovereignty
|
|
22
|
+
principles — community ownership and control of language data.
|
|
22
23
|
It is a permanent work in progress, built to be flexible and to support MT
|
|
23
24
|
developers — and especially low-resource-language speaker communities.
|
|
24
25
|
|
|
@@ -40,22 +41,19 @@ which is provided to you fresh with each question under RETRIEVED CONTEXT below.
|
|
|
40
41
|
the retrieved context. If two sources disagree, present both — never pick a
|
|
41
42
|
winner or manufacture a consensus.
|
|
42
43
|
|
|
43
|
-
## Sovereignty: how to talk about
|
|
44
|
-
|
|
45
|
-
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
-
|
|
56
|
-
or that it "follows OCAP®." We do the work so that others can judge it —
|
|
57
|
-
self-certification is worth nothing. ("OCAP®-forward" was a retired earlier
|
|
58
|
-
term; the only sanctioned form is "OCAP®-aspirant".)
|
|
44
|
+
## Sovereignty: how to talk about data sovereignty and communities
|
|
45
|
+
|
|
46
|
+
- The project's posture is built around Indigenous data-sovereignty
|
|
47
|
+
principles — community ownership and control of language data and anything
|
|
48
|
+
derived from it.
|
|
49
|
+
- The project is **sovereignty-aspirant**: the design is built so communities
|
|
50
|
+
*can* exercise ownership and control of their data and anything derived from
|
|
51
|
+
it. Whether it *achieves* sovereignty is **for communities to decide, not for
|
|
52
|
+
us to claim.** Say this nuance plainly and without defensiveness when asked.
|
|
53
|
+
- **Never** say the project is compliant with, certified under, or aligned with
|
|
54
|
+
any named data-governance framework. We do the work so that others can judge
|
|
55
|
+
it — self-certification is worth nothing. (The only sanctioned house term is
|
|
56
|
+
"sovereignty-aspirant".)
|
|
59
57
|
- **Never name a specific nation, community, or organization as a partner, key
|
|
60
58
|
custodian, or endorser.** If asked who the custodians are, say they are
|
|
61
59
|
"community key custodians (in confirmation)" and that no one is named publicly
|
|
@@ -75,7 +73,7 @@ You are a guide, not a developer, a translator, or a general assistant.
|
|
|
75
73
|
pair-programming session — and that tokens aren't free. Point them to the
|
|
76
74
|
agent-native path instead: set up their own agent with its own workspace and
|
|
77
75
|
give it `champollion.dev/llms.txt` and the MCP server
|
|
78
|
-
(
|
|
76
|
+
(`champollion-mcp-server`), which expose the queue, language metadata, and
|
|
79
77
|
translation/benchmark tools as real tools an agent can drive. Offer to tour
|
|
80
78
|
them to the exact docs.
|
|
81
79
|
- **No translation-on-demand.** You don't translate documents or sentences for
|
|
@@ -115,7 +113,7 @@ quickly."
|
|
|
115
113
|
|
|
116
114
|
If the visitor is itself an agent (or asks how an agent should use Champollion),
|
|
117
115
|
skip the tour and point it straight at `champollion.dev/llms.txt` and the MCP
|
|
118
|
-
server
|
|
116
|
+
server `champollion-mcp-server`, and briefly name what those expose (benchmark
|
|
119
117
|
queue, language metadata, translate/benchmark tools, the coaching "open problem"
|
|
120
118
|
path). That's the token-efficient path for both of us.
|
|
121
119
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"_meta": {
|
|
3
|
-
"generated": "2026-
|
|
3
|
+
"generated": "2026-09-27T08:15:46.589Z",
|
|
4
4
|
"generator": "cli/scripts/build-explainers.mjs",
|
|
5
5
|
"description": "Typological feature explainers: per-feature plain-English names, value code → label maps, MT-relevance notes and citations. Join from a fact via propertyIndex[\"<source>|<property>\"], or from a card typologicalProfile key via propertyIndex[\"card|<key>\"].",
|
|
6
6
|
"counts": {
|
|
@@ -7832,8 +7832,8 @@
|
|
|
7832
7832
|
"id": "derived-grambank-hasObliqueCase",
|
|
7833
7833
|
"source": "grambank-1.0.3",
|
|
7834
7834
|
"property": "hasObliqueCase",
|
|
7835
|
-
"name": "Oblique case (
|
|
7836
|
-
"question": "Whether
|
|
7835
|
+
"name": "Oblique case (GB072)",
|
|
7836
|
+
"question": "Whether full nouns take case marking for oblique (non-core) roles such as location or instrument.",
|
|
7837
7837
|
"values": {
|
|
7838
7838
|
"true": {
|
|
7839
7839
|
"label": "Yes",
|
|
@@ -7844,10 +7844,10 @@
|
|
|
7844
7844
|
"plain": "The language does not have this feature."
|
|
7845
7845
|
}
|
|
7846
7846
|
},
|
|
7847
|
-
"mt_relevance": "
|
|
7847
|
+
"mt_relevance": "Oblique case endings carry roles English expresses with prepositions (in, with, to), so each preposition must be mapped to the right ending.",
|
|
7848
7848
|
"citation": {
|
|
7849
7849
|
"authors": "Skirgård, Hedvig, Hannah J. Haynie, Damián E. Blasi, Hedvig Skirgård et al. (Grambank Consortium)",
|
|
7850
|
-
"title": "Oblique case (
|
|
7850
|
+
"title": "Oblique case (GB072) — derived from Grambank v1.0.3",
|
|
7851
7851
|
"source_url": "https://grambank.clld.org"
|
|
7852
7852
|
}
|
|
7853
7853
|
},
|
|
@@ -7878,8 +7878,8 @@
|
|
|
7878
7878
|
"id": "derived-grambank-marksPresentTense",
|
|
7879
7879
|
"source": "grambank-1.0.3",
|
|
7880
7880
|
"property": "marksPresentTense",
|
|
7881
|
-
"name": "Marks present tense (
|
|
7882
|
-
"question": "Whether
|
|
7881
|
+
"name": "Marks present tense (GB082)",
|
|
7882
|
+
"question": "Whether verbs carry overt morphological marking of present tense.",
|
|
7883
7883
|
"values": {
|
|
7884
7884
|
"true": {
|
|
7885
7885
|
"label": "Yes",
|
|
@@ -7893,7 +7893,7 @@
|
|
|
7893
7893
|
"mt_relevance": "Present-tense marking interacts with aspect: rendering it as a plain English present often asserts a habituality the source did not.",
|
|
7894
7894
|
"citation": {
|
|
7895
7895
|
"authors": "Skirgård, Hedvig, Hannah J. Haynie, Damián E. Blasi, Hedvig Skirgård et al. (Grambank Consortium)",
|
|
7896
|
-
"title": "Marks present tense (
|
|
7896
|
+
"title": "Marks present tense (GB082) — derived from Grambank v1.0.3",
|
|
7897
7897
|
"source_url": "https://grambank.clld.org"
|
|
7898
7898
|
}
|
|
7899
7899
|
},
|
|
@@ -8127,7 +8127,7 @@
|
|
|
8127
8127
|
"plain": "absent"
|
|
8128
8128
|
}
|
|
8129
8129
|
},
|
|
8130
|
-
"mt_relevance": "
|
|
8130
|
+
"mt_relevance": "A polite/familiar \"you\" must be chosen on every second-person reference; English marks neither, so MT has to infer the relationship between speaker and addressee.",
|
|
8131
8131
|
"citation": {
|
|
8132
8132
|
"authors": "Skirgård, Hedvig, Hannah J. Haynie, Damián E. Blasi, Hedvig Skirgård et al. (Grambank Consortium)",
|
|
8133
8133
|
"title": "Is there a politeness distinction in 2nd person forms? (Grambank GB415)",
|
|
@@ -19258,11 +19258,11 @@
|
|
|
19258
19258
|
"name": "Noun classifiers",
|
|
19259
19259
|
"question": "Whether nouns combine with classifier words that sort them by type.",
|
|
19260
19260
|
"values": {},
|
|
19261
|
-
"mt_relevance": "
|
|
19261
|
+
"mt_relevance": "Numeral classifiers must be chosen per noun; the wrong classifier is immediately visible to native speakers.",
|
|
19262
19262
|
"citation": {
|
|
19263
19263
|
"authors": "Champollion enrichment pipeline (derived)",
|
|
19264
|
-
"title": "Noun classifiers — derived from Grambank
|
|
19265
|
-
"source_url": "https://grambank.clld.org/parameters/
|
|
19264
|
+
"title": "Noun classifiers — derived from Grambank GB057 (numeral classifiers)",
|
|
19265
|
+
"source_url": "https://grambank.clld.org/parameters/GB057"
|
|
19266
19266
|
}
|
|
19267
19267
|
},
|
|
19268
19268
|
"derived-classifierLanguage": {
|
|
@@ -19272,11 +19272,11 @@
|
|
|
19272
19272
|
"name": "Classifier language",
|
|
19273
19273
|
"question": "Whether the language routinely uses classifiers when counting or referring to nouns.",
|
|
19274
19274
|
"values": {},
|
|
19275
|
-
"mt_relevance": "
|
|
19275
|
+
"mt_relevance": "Numeral classifiers must be chosen per noun; the wrong classifier is immediately visible to native speakers.",
|
|
19276
19276
|
"citation": {
|
|
19277
19277
|
"authors": "Champollion enrichment pipeline (derived)",
|
|
19278
|
-
"title": "Classifier language — derived from Grambank
|
|
19279
|
-
"source_url": "https://grambank.clld.org/parameters/
|
|
19278
|
+
"title": "Classifier language — derived from Grambank GB057 / WALS 55A",
|
|
19279
|
+
"source_url": "https://grambank.clld.org/parameters/GB057"
|
|
19280
19280
|
}
|
|
19281
19281
|
},
|
|
19282
19282
|
"derived-hasNumeralClassifiers": {
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_about": "The Plural-Forms header GNU gettext's msginit writes for a new catalog, per language: its built-in table (gettext-tools plural-table.c). champollion gives a catalog it creates the same header, so the catalog's msgstr[] slots are the ones gettext and Django select at runtime. Looked up as msginit does: the full code first (pt_BR), then the language (pt). A language not listed gets no header from msginit; champollion derives one from CLDR (lib/po.js synthesizePluralForms). An existing catalog's own header is never replaced.",
|
|
3
|
+
"_source": "Probed 2026-10-03 with `msginit --no-translator --no-wrap --locale=<code>` (GNU gettext-tools 1.0, GETTEXTCLDRDIR unset) for every ISO 639-1 code and pt_BR, pt_PT, en_GB, de_AT, es_AR, fr_CA, zh_CN, zh_TW, sr_Latn; region variants not listed resolved to their language's entry.",
|
|
4
|
+
"forms": {
|
|
5
|
+
"be": "nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && n%10<=4 && (n%100<10 || n%100>=20) ? 1 : 2);",
|
|
6
|
+
"bg": "nplurals=2; plural=(n != 1);",
|
|
7
|
+
"ca": "nplurals=2; plural=(n != 1);",
|
|
8
|
+
"cs": "nplurals=3; plural=(n==1) ? 0 : (n>=2 && n<=4) ? 1 : 2;",
|
|
9
|
+
"da": "nplurals=2; plural=(n != 1);",
|
|
10
|
+
"de": "nplurals=2; plural=(n != 1);",
|
|
11
|
+
"el": "nplurals=2; plural=(n != 1);",
|
|
12
|
+
"en": "nplurals=2; plural=(n != 1);",
|
|
13
|
+
"eo": "nplurals=2; plural=(n != 1);",
|
|
14
|
+
"es": "nplurals=2; plural=(n != 1);",
|
|
15
|
+
"et": "nplurals=2; plural=(n != 1);",
|
|
16
|
+
"fi": "nplurals=2; plural=(n != 1);",
|
|
17
|
+
"fo": "nplurals=2; plural=(n != 1);",
|
|
18
|
+
"fr": "nplurals=2; plural=(n > 1);",
|
|
19
|
+
"ga": "nplurals=3; plural=n==1 ? 0 : n==2 ? 1 : 2;",
|
|
20
|
+
"he": "nplurals=2; plural=(n != 1);",
|
|
21
|
+
"hr": "nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && n%10<=4 && (n%100<10 || n%100>=20) ? 1 : 2);",
|
|
22
|
+
"hu": "nplurals=2; plural=(n != 1);",
|
|
23
|
+
"it": "nplurals=2; plural=(n != 1);",
|
|
24
|
+
"ja": "nplurals=1; plural=0;",
|
|
25
|
+
"ko": "nplurals=1; plural=0;",
|
|
26
|
+
"lt": "nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && (n%100<10 || n%100>=20) ? 1 : 2);",
|
|
27
|
+
"lv": "nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n != 0 ? 1 : 2);",
|
|
28
|
+
"nb": "nplurals=2; plural=(n != 1);",
|
|
29
|
+
"nl": "nplurals=2; plural=(n != 1);",
|
|
30
|
+
"nn": "nplurals=2; plural=(n != 1);",
|
|
31
|
+
"no": "nplurals=2; plural=(n != 1);",
|
|
32
|
+
"pl": "nplurals=3; plural=(n==1 ? 0 : n%10>=2 && n%10<=4 && (n%100<10 || n%100>=20) ? 1 : 2);",
|
|
33
|
+
"pt": "nplurals=2; plural=(n != 1);",
|
|
34
|
+
"pt_BR": "nplurals=2; plural=(n > 1);",
|
|
35
|
+
"ro": "nplurals=3; plural=n==1 ? 0 : (n==0 || (n%100 > 0 && n%100 < 20)) ? 1 : 2;",
|
|
36
|
+
"ru": "nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && n%10<=4 && (n%100<10 || n%100>=20) ? 1 : 2);",
|
|
37
|
+
"sk": "nplurals=3; plural=(n==1) ? 0 : (n>=2 && n<=4) ? 1 : 2;",
|
|
38
|
+
"sl": "nplurals=4; plural=(n%100==1 ? 0 : n%100==2 ? 1 : n%100==3 || n%100==4 ? 2 : 3);",
|
|
39
|
+
"sr": "nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && n%10<=4 && (n%100<10 || n%100>=20) ? 1 : 2);",
|
|
40
|
+
"sv": "nplurals=2; plural=(n != 1);",
|
|
41
|
+
"tr": "nplurals=2; plural=(n != 1);",
|
|
42
|
+
"uk": "nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && n%10<=4 && (n%100<10 || n%100>=20) ? 1 : 2);",
|
|
43
|
+
"vi": "nplurals=1; plural=0;"
|
|
44
|
+
}
|
|
45
|
+
}
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
"sovereignty_flags": {
|
|
14
14
|
"off_community_processing": false,
|
|
15
15
|
"attribution_required": true,
|
|
16
|
-
"
|
|
16
|
+
"sovereignty_review": "pending"
|
|
17
17
|
},
|
|
18
18
|
"nc_terms": "Community-set terms; non-commercial use pending custodian confirmation.",
|
|
19
19
|
"consent_attested": false,
|
|
@@ -65,6 +65,7 @@
|
|
|
65
65
|
"OPENAI_API_KEY"
|
|
66
66
|
],
|
|
67
67
|
"default_base_url": "http://localhost:11434/v1",
|
|
68
|
+
"keyless": true,
|
|
68
69
|
"homepage": "https://github.com/ollama/ollama",
|
|
69
70
|
"license": "Varies (self-hosted / OpenAI-compatible gateway)",
|
|
70
71
|
"commercialReady": false,
|
|
@@ -147,6 +148,7 @@
|
|
|
147
148
|
"APERTIUM_API_KEY"
|
|
148
149
|
],
|
|
149
150
|
"default_base_url": "https://apertium.org/apy",
|
|
151
|
+
"keyless": true,
|
|
150
152
|
"homepage": "https://www.apertium.org",
|
|
151
153
|
"license": "GPL-3.0+ (Apertium)",
|
|
152
154
|
"commercialReady": false,
|