champollion 0.3.3 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,7 @@
4
4
  {
5
5
  "id": "what-is-champollion",
6
6
  "question": "What is Champollion?",
7
- "answer": "Champollion is open-source infrastructure for trustworthy machine translation across the world's languages, with a focus on low-resource and Indigenous ones. It has three faces: a translation CLI that translates your app's locale files with one command, the Network (an open MT evaluation network and leaderboard that maps who can translate what, how well), and a data-sovereignty posture built around the First Nations principles of OCAP®. As the Introduction (/docs/intro) and The Champollion Network (/docs/network/) put it, it is built *with* professionals and communities, never scraped from them — and it is forever a work in progress.",
7
+ "answer": "Champollion is source-available infrastructure for trustworthy machine-translation evaluation across the world's languages, with a focus on low-resource and Indigenous ones. It has three faces: a translation CLI that translates your app's locale files with one command, the Network (an open MT evaluation network and leaderboard that maps who can translate what, how well), and a data-sovereignty posture built around Indigenous data-sovereignty principles — community ownership and control of language data. As the Introduction (/docs/intro) and The Champollion Network (/docs/network/) put it, it is designed to work *with* professionals and communities, and it never hosts their corpora — and it is forever a work in progress.",
8
8
  "sources": [
9
9
  "/docs/intro",
10
10
  "/docs/network/",
@@ -21,7 +21,7 @@
21
21
  {
22
22
  "id": "free-and-open-source",
23
23
  "question": "Is Champollion free? Is it open source? Does it cost anything?",
24
- "answer": "Yes on both counts. Champollion is a non-commercial, open-source research project: the CLI is Apache-2.0 and the evaluation harness is AGPL-3.0, per How the Work Is Funded (/docs/network/sovereignty/economic-model). There is no paid API, no metering, and no revenue share — the only costs you'd incur are the API tokens you spend on your chosen model provider when you translate or benchmark. Champollion itself takes nothing.",
24
+ "answer": "It is free for noncommercial use, and partly open source. Champollion is a non-commercial, source-available research project: the CLI is PolyForm Noncommercial 1.0.0 (free for noncommercial use; commercial use needs permission) and the evaluation harness is open-source AGPL-3.0, per How the Work Is Funded (/docs/network/sovereignty/economic-model). There is no paid API, no metering, and no revenue share — the only costs you'd incur are the API tokens you spend on your chosen model provider when you translate or benchmark. Champollion itself takes nothing.",
25
25
  "sources": [
26
26
  "/docs/network/sovereignty/economic-model",
27
27
  "/docs/intro"
@@ -39,7 +39,7 @@
39
39
  {
40
40
  "id": "the-cli",
41
41
  "question": "What is the champollion CLI and what does it do?",
42
- "answer": "The CLI is the deployment end of Champollion: run `npx champollion sync` and it auto-detects your locale files, format, and target languages, translates what's missing, skips what's done, validates every result through a quality gate, and writes clean output. Per How It Works (/docs/how-it-works), it tracks every source string with SHA-256 hashes so re-runs only re-translate what changed, caches results in Translation Memory to save cost, and lets you pick a different translation method per language pair. It's a full i18n pipeline — sync, watch, lint, verify, audit, XLIFF export, and more (/docs/intro).",
42
+ "answer": "The CLI is the deployment end of Champollion: run `npx champollion sync` and it auto-detects your locale files, format, and target languages, translates what's missing, skips what's done, checks every result for broken output (empty, echoed, looping or wrong-script — not wrong meaning) through a quality gate, and writes clean output. Per How It Works (/docs/how-it-works), it tracks every source string with SHA-256 hashes so re-runs only re-translate what changed, caches results in Translation Memory to save cost, and lets you pick a different translation method per language pair. It's a full i18n pipeline — sync, watch, lint, verify, audit, XLIFF export, and more (/docs/intro).",
43
43
  "sources": [
44
44
  "/docs/intro",
45
45
  "/docs/how-it-works"
@@ -135,7 +135,7 @@
135
135
  {
136
136
  "id": "submit-a-method",
137
137
  "question": "How do I submit a benchmark run or a method to the leaderboard?",
138
- "answer": "Install the harness with `pip install mt-eval`, run it against a registered corpus (for example `mt-eval run --corpus eval-amh-fra-globalvoices-test-v1 --model gemini-pro`), review the run card it produces, then publish with `mt-eval publish` — the full walkthrough is Submit a Method (/docs/network/getting-started/submit-a-method). Your run first appears as self-benchmarked; the server then re-scores your outputs against the sha-pinned corpus and, when it reproduces, promotes it to Champollion Verified. A hosted upload API and web UI are planned but not yet live, so `mt-eval publish` or a pull request to the harness repo are the working paths today.",
138
+ "answer": "Install the harness with `pip install mt-eval-harness`, run it against a registered corpus (for example `mt-eval run --corpus eval-amh-fra-globalvoices-test-v1 --model gemini-pro`), review the run card it produces, then publish with `mt-eval publish` — the full walkthrough is Submit a Method (/docs/network/getting-started/submit-a-method). Your run first appears as self-benchmarked; the server then re-scores your outputs against the sha-pinned corpus and, when it reproduces, promotes it to Champollion Verified. A hosted upload API and web UI are planned but not yet live, so `mt-eval publish` or a pull request to the harness repo are the working paths today.",
139
139
  "sources": [
140
140
  "/docs/network/getting-started/submit-a-method",
141
141
  "/docs/network/leaderboard/rules"
@@ -228,7 +228,7 @@
228
228
  {
229
229
  "id": "researcher-involved",
230
230
  "question": "I'm an ML researcher — how do I get involved?",
231
- "answer": "Build a method and benchmark it. The Network Agent Guide (/docs/network/getting-started/agent-guide) walks through installing the harness (`pip install mt-eval`), running baselines against real corpora, and implementing the simple `TranslationMethod` protocol — coached LLM, FST-gated pipeline, fine-tuned model, or anything that produces translations. You get standardized, reproducible, fingerprinted scoring against a shared corpus catalogue, and every run adds a point to a shared map of what works. Get Involved (/get-involved) points to the developer paths, including running the public benchmark queue.",
231
+ "answer": "Build a method and benchmark it. The Network Agent Guide (/docs/network/getting-started/agent-guide) walks through installing the harness (`pip install mt-eval-harness`), running baselines against real corpora, and implementing the simple `TranslationMethod` protocol — coached LLM, FST-gated pipeline, fine-tuned model, or anything that produces translations. You get standardized, reproducible, fingerprinted scoring against a shared corpus catalogue, and every run adds a point to a shared map of what works. Get Involved (/get-involved) points to the developer paths, including running the public benchmark queue.",
232
232
  "sources": [
233
233
  "/docs/network/getting-started/agent-guide",
234
234
  "/get-involved",
@@ -264,17 +264,15 @@
264
264
  ]
265
265
  },
266
266
  {
267
- "id": "ocap-aspirant-meaning",
268
- "question": "What does 'OCAP®-aspirant' mean?",
269
- "answer": "OCAP® — Ownership, Control, Access, and Possession — is a set of First Nations data-governance principles, and it is a registered trademark of the First Nations Information Governance Centre (FNIGC). Champollion's posture is OCAP®-aspirant: as Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) puts it, the design is built so communities *can* exercise ownership, control, access, and possession of their data and anything derived from it — but whether it *achieves* that is for communities to decide, which is exactly why the objections channel exists. We say OCAP®-aspirant deliberately, rather than claiming to be OCAP®-compliant or certified.",
267
+ "id": "sovereignty-aspirant-meaning",
268
+ "question": "What does 'sovereignty-aspirant' mean?",
269
+ "answer": "Champollion's posture is sovereignty-aspirant — built around Indigenous data-sovereignty principles: community ownership and control of language data. As Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) puts it, the design is built so communities *can* exercise ownership and control of their data and anything derived from it — but whether it *achieves* that is for communities to decide, which is exactly why the objections channel exists. We say sovereignty-aspirant deliberately, rather than claiming compliance with or certification under any named framework.",
270
270
  "sources": [
271
271
  "/docs/network/community/contact-objections-takedown",
272
272
  "/docs/network/sovereignty/data-sovereignty"
273
273
  ],
274
274
  "keywords": [
275
- "ocap",
276
- "ocap aspirant",
277
- "fnigc",
275
+ "sovereignty aspirant",
278
276
  "sovereignty principles",
279
277
  "ownership control access possession",
280
278
  "indigenous"
@@ -283,7 +281,7 @@
283
281
  {
284
282
  "id": "data-sovereignty",
285
283
  "question": "What is data sovereignty here — do you store or scrape my language data?",
286
- "answer": "No. Data Stewardship (/docs/network/sovereignty/data-sovereignty) states the position: language data is biodata, so the people who provide a corpus hold the keys to it and to anything measured against it. Champollion never holds the data — corpora are registered as hash-pinned metadata cards and fetched from the steward's own hosting at evaluation time; take your archive offline and evaluation simply stops. Every license and community restriction is respected by gate, not by promise (CI checks and database triggers), and the design is informed by OCAP®, the CARE Principles, Te Mana Raraunga, and Te Hiku Media's Kaitiakitanga License.",
284
+ "answer": "No. Data Stewardship (/docs/network/sovereignty/data-sovereignty) states the position: language data is biodata, so the people who provide a corpus hold the keys to it and to anything measured against it. Champollion never holds the data — corpora are registered as hash-pinned metadata cards and fetched from the steward's own hosting at evaluation time; take your archive offline and evaluation simply stops. Every license and community restriction is respected by gate, not by promise (local pre-push checks and database triggers), and the design is informed by Indigenous data-sovereignty principles — community ownership and control of language data — including the CARE Principles, Te Mana Raraunga, and Te Hiku Media's Kaitiakitanga License.",
287
285
  "sources": [
288
286
  "/docs/network/sovereignty/data-sovereignty",
289
287
  "/docs/network/sovereignty/registering-corpora"
@@ -341,7 +339,7 @@
341
339
  {
342
340
  "id": "speakers-paid",
343
341
  "question": "Do speakers get paid, and how much?",
344
- "answer": "Yes — paying speakers is treated as non-negotiable, and payment is unconditional (you're paid whether or not your ratings are used). How Speakers Get Paid (/docs/network/perspectives/how-speakers-get-paid) publishes the rates: bilingual speaker work at $50–65 CAD per hour, with corpus curation budgeted at $2,500–6,000 and a full metric-validation round at $1,475–1,920. Payment does not buy your data — you are paid for the work and remain the steward of what you build. These are the published rates for the current Plains Cree work; rates for future languages are set with the partner community and published before the work starts.",
342
+ "answer": "Yes, once the work is funded — no speaker work is funded or underway today. Paying speakers is treated as non-negotiable, and payment is unconditional (you're paid whether or not your ratings are used). How Speakers Get Paid (/docs/network/perspectives/how-speakers-get-paid) publishes the rates: bilingual speaker work at $50–65 CAD per hour, with corpus curation budgeted at $2,500–6,000 and a full metric-validation round at $1,475–1,920. Payment does not buy your data — you are paid for the work and remain the steward of what you build. These are the rates we intend to pay for Plains Cree work; rates for future languages are set with the partner community and published before the work starts.",
345
343
  "sources": [
346
344
  "/docs/network/perspectives/how-speakers-get-paid",
347
345
  "/docs/network/sovereignty/economic-model"
@@ -398,7 +396,7 @@
398
396
  {
399
397
  "id": "ai-agent-usage",
400
398
  "question": "How should an AI agent use Champollion?",
401
- "answer": "Two things are built for agents. First, the machine-readable /llms.txt index maps the whole site, the license lanes, and the public data feeds (queue, mesh, corpus registry). Second, the MCP server `@champollion/mcp-server` (on npm) is the agent-facing door — translate with Translation-Memory caching and a deterministic quality gate, browse the benchmark queue, estimate cost, run benchmarks, read the leaderboard, and query per-language metric-trust evidence, all as tools. The Agent Guides (/docs/guides/agent-guide for the CLI, /docs/network/getting-started/agent-guide for benchmarking) give step-by-step recipes. The best pattern is to set up your own agent with a workspace and your own API key — tokens aren't free.",
399
+ "answer": "Two things are built for agents. First, the machine-readable /llms.txt index maps the whole site, the license lanes, and the public data feeds (queue, mesh, corpus registry). Second, the MCP server `champollion-mcp-server` (on npm) is the agent-facing door — translate with Translation-Memory caching and a deterministic quality gate, browse the benchmark queue, estimate cost, run benchmarks, read the leaderboard, and query per-language metric-trust evidence, all as tools. The Agent Guides (/docs/guides/agent-guide for the CLI, /docs/network/getting-started/agent-guide for benchmarking) give step-by-step recipes. The best pattern is to set up your own agent with a workspace and your own API key — tokens aren't free.",
402
400
  "sources": [
403
401
  "/llms.txt",
404
402
  "/docs/guides/agent-guide",
@@ -418,7 +416,7 @@
418
416
  {
419
417
  "id": "contribute-compute",
420
418
  "question": "How do I contribute compute by running the benchmark queue?",
421
- "answer": "The leaderboard has empty squares — (language pair, model, condition) combinations nobody has measured — kept in a public queue at /queue.json. Install the harness (`pipx install mt-eval`), set one provider API key, and run the highest-value open items with `mt-eval queue --top 5` (add `--dry-run` to see the plan and spend nothing first); the full guide is Contributing Compute (/docs/network/getting-started/contributing-compute) and the Contribute page (/contribute). No account is needed — results publish as 'anonymous' unless you sign in to put your name on the board — and every run is a real, citable contribution to low-resource MT evaluation.",
419
+ "answer": "The leaderboard has empty squares — (language pair, model, condition) combinations nobody has measured — kept in a public queue at /queue.json. Install the harness (`pipx install mt-eval-harness`), set one provider API key, and run the highest-value open items with `mt-eval queue --top 5` (add `--dry-run` to see the plan and spend nothing first); the full guide is Contributing Compute (/docs/network/getting-started/contributing-compute) and the Contribute page (/contribute). No account is needed — results publish as 'anonymous' unless you sign in to put your name on the board — and every run is a real, citable contribution to low-resource MT evaluation.",
422
420
  "sources": [
423
421
  "/docs/network/getting-started/contributing-compute",
424
422
  "/contribute"
@@ -472,7 +470,7 @@
472
470
  {
473
471
  "id": "write-code-for-me",
474
472
  "question": "Can the site guide write code, debug, or run the harness for me?",
475
- "answer": "No — the site guide is an index and a docent, not an engineering service; Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) is explicit that this isn't a place to request development help or custom scripts. What it can do is point you at the tooling so your own setup does the work: the /llms.txt cookbook, the MCP server `@champollion/mcp-server`, and the Agent Guides (/docs/guides/agent-guide and /docs/network/getting-started/agent-guide). The recommended path is to set up your own agent with a workspace and your own API key — tokens aren't free, and your agent can then run the harness or the CLI for you.",
473
+ "answer": "No — the site guide is an index and a docent, not an engineering service; Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) is explicit that this isn't a place to request development help or custom scripts. What it can do is point you at the tooling so your own setup does the work: the /llms.txt cookbook, the MCP server `champollion-mcp-server`, and the Agent Guides (/docs/guides/agent-guide and /docs/network/getting-started/agent-guide). The recommended path is to set up your own agent with a workspace and your own API key — tokens aren't free, and your agent can then run the harness or the CLI for you.",
476
474
  "sources": [
477
475
  "/docs/network/community/contact-objections-takedown",
478
476
  "/docs/guides/agent-guide",
@@ -18,7 +18,8 @@ careful, honest, and humble about what this project is.
18
18
 
19
19
  Champollion is open translation infrastructure for low-resource languages: a
20
20
  translation CLI, a machine-translation evaluation network ("the Network"), and
21
- a data-sovereignty posture built around the First Nations principles of OCAP®.
21
+ a data-sovereignty posture built around Indigenous data-sovereignty
22
+ principles — community ownership and control of language data.
22
23
  It is a permanent work in progress, built to be flexible and to support MT
23
24
  developers — and especially low-resource-language speaker communities.
24
25
 
@@ -40,22 +41,19 @@ which is provided to you fresh with each question under RETRIEVED CONTEXT below.
40
41
  the retrieved context. If two sources disagree, present both — never pick a
41
42
  winner or manufacture a consensus.
42
43
 
43
- ## Sovereignty: how to talk about OCAP® and communities
44
-
45
- - OCAP® — Ownership, Control, Access, Possession — is a First Nations data
46
- governance framework and a **registered trademark of the First Nations
47
- Information Governance Centre (FNIGC)**. Always write it with the ® mark, and
48
- attribute it to FNIGC when you explain it.
49
- - The project is **OCAP®-aspirant**: designed *with the First Nations principles
50
- of OCAP® in mind*. That means the design is built so communities *can*
51
- exercise ownership, control, access, and possession of their data and anything
52
- derived from it. Whether it *achieves* sovereignty is **for communities to
53
- decide, not for us to claim.** Say this nuance plainly and without defensiveness
54
- when asked.
55
- - **Never** say the project is OCAP®-compliant, -certified, -aligned, -forward,
56
- or that it "follows OCAP®." We do the work so that others can judge it —
57
- self-certification is worth nothing. ("OCAP®-forward" was a retired earlier
58
- term; the only sanctioned form is "OCAP®-aspirant".)
44
+ ## Sovereignty: how to talk about data sovereignty and communities
45
+
46
+ - The project's posture is built around Indigenous data-sovereignty
47
+ principles — community ownership and control of language data and anything
48
+ derived from it.
49
+ - The project is **sovereignty-aspirant**: the design is built so communities
50
+ *can* exercise ownership and control of their data and anything derived from
51
+ it. Whether it *achieves* sovereignty is **for communities to decide, not for
52
+ us to claim.** Say this nuance plainly and without defensiveness when asked.
53
+ - **Never** say the project is compliant with, certified under, or aligned with
54
+ any named data-governance framework. We do the work so that others can judge
55
+ it — self-certification is worth nothing. (The only sanctioned house term is
56
+ "sovereignty-aspirant".)
59
57
  - **Never name a specific nation, community, or organization as a partner, key
60
58
  custodian, or endorser.** If asked who the custodians are, say they are
61
59
  "community key custodians (in confirmation)" and that no one is named publicly
@@ -75,7 +73,7 @@ You are a guide, not a developer, a translator, or a general assistant.
75
73
  pair-programming session — and that tokens aren't free. Point them to the
76
74
  agent-native path instead: set up their own agent with its own workspace and
77
75
  give it `champollion.dev/llms.txt` and the MCP server
78
- (`@champollion/mcp-server`), which expose the queue, language metadata, and
76
+ (`champollion-mcp-server`), which expose the queue, language metadata, and
79
77
  translation/benchmark tools as real tools an agent can drive. Offer to tour
80
78
  them to the exact docs.
81
79
  - **No translation-on-demand.** You don't translate documents or sentences for
@@ -115,7 +113,7 @@ quickly."
115
113
 
116
114
  If the visitor is itself an agent (or asks how an agent should use Champollion),
117
115
  skip the tour and point it straight at `champollion.dev/llms.txt` and the MCP
118
- server `@champollion/mcp-server`, and briefly name what those expose (benchmark
116
+ server `champollion-mcp-server`, and briefly name what those expose (benchmark
119
117
  queue, language metadata, translate/benchmark tools, the coaching "open problem"
120
118
  path). That's the token-efficient path for both of us.
121
119
 
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "_meta": {
3
- "generated": "2026-08-12T23:23:44.431Z",
3
+ "generated": "2026-09-27T08:15:46.589Z",
4
4
  "generator": "cli/scripts/build-explainers.mjs",
5
5
  "description": "Typological feature explainers: per-feature plain-English names, value code → label maps, MT-relevance notes and citations. Join from a fact via propertyIndex[\"<source>|<property>\"], or from a card typologicalProfile key via propertyIndex[\"card|<key>\"].",
6
6
  "counts": {
@@ -7832,8 +7832,8 @@
7832
7832
  "id": "derived-grambank-hasObliqueCase",
7833
7833
  "source": "grambank-1.0.3",
7834
7834
  "property": "hasObliqueCase",
7835
- "name": "Oblique case (GB071)",
7836
- "question": "Whether pronouns take case marking for core grammatical roles.",
7835
+ "name": "Oblique case (GB072)",
7836
+ "question": "Whether full nouns take case marking for oblique (non-core) roles such as location or instrument.",
7837
7837
  "values": {
7838
7838
  "true": {
7839
7839
  "label": "Yes",
@@ -7844,10 +7844,10 @@
7844
7844
  "plain": "The language does not have this feature."
7845
7845
  }
7846
7846
  },
7847
- "mt_relevance": "Pronoun case errors (I/me-type) are highly salient grammatical mistakes in generated text.",
7847
+ "mt_relevance": "Oblique case endings carry roles English expresses with prepositions (in, with, to), so each preposition must be mapped to the right ending.",
7848
7848
  "citation": {
7849
7849
  "authors": "Skirgård, Hedvig, Hannah J. Haynie, Damián E. Blasi, Hedvig Skirgård et al. (Grambank Consortium)",
7850
- "title": "Oblique case (GB071) — derived from Grambank v1.0.3",
7850
+ "title": "Oblique case (GB072) — derived from Grambank v1.0.3",
7851
7851
  "source_url": "https://grambank.clld.org"
7852
7852
  }
7853
7853
  },
@@ -7878,8 +7878,8 @@
7878
7878
  "id": "derived-grambank-marksPresentTense",
7879
7879
  "source": "grambank-1.0.3",
7880
7880
  "property": "marksPresentTense",
7881
- "name": "Marks present tense (GB084)",
7882
- "question": "Whether the grammar has a dedicated marker for present time reference.",
7881
+ "name": "Marks present tense (GB082)",
7882
+ "question": "Whether verbs carry overt morphological marking of present tense.",
7883
7883
  "values": {
7884
7884
  "true": {
7885
7885
  "label": "Yes",
@@ -7893,7 +7893,7 @@
7893
7893
  "mt_relevance": "Present-tense marking interacts with aspect: rendering it as a plain English present often asserts a habituality the source did not.",
7894
7894
  "citation": {
7895
7895
  "authors": "Skirgård, Hedvig, Hannah J. Haynie, Damián E. Blasi, Hedvig Skirgård et al. (Grambank Consortium)",
7896
- "title": "Marks present tense (GB084) — derived from Grambank v1.0.3",
7896
+ "title": "Marks present tense (GB082) — derived from Grambank v1.0.3",
7897
7897
  "source_url": "https://grambank.clld.org"
7898
7898
  }
7899
7899
  },
@@ -8127,7 +8127,7 @@
8127
8127
  "plain": "absent"
8128
8128
  }
8129
8129
  },
8130
- "mt_relevance": "Classifier systems require noun-specific function words that have no counterpart in non-classifier languages.",
8130
+ "mt_relevance": "A polite/familiar \"you\" must be chosen on every second-person reference; English marks neither, so MT has to infer the relationship between speaker and addressee.",
8131
8131
  "citation": {
8132
8132
  "authors": "Skirgård, Hedvig, Hannah J. Haynie, Damián E. Blasi, Hedvig Skirgård et al. (Grambank Consortium)",
8133
8133
  "title": "Is there a politeness distinction in 2nd person forms? (Grambank GB415)",
@@ -19258,11 +19258,11 @@
19258
19258
  "name": "Noun classifiers",
19259
19259
  "question": "Whether nouns combine with classifier words that sort them by type.",
19260
19260
  "values": {},
19261
- "mt_relevance": "Classifier systems require noun-specific function words that have no counterpart in non-classifier languages.",
19261
+ "mt_relevance": "Numeral classifiers must be chosen per noun; the wrong classifier is immediately visible to native speakers.",
19262
19262
  "citation": {
19263
19263
  "authors": "Champollion enrichment pipeline (derived)",
19264
- "title": "Noun classifiers — derived from Grambank GB415",
19265
- "source_url": "https://grambank.clld.org/parameters/GB415"
19264
+ "title": "Noun classifiers — derived from Grambank GB057 (numeral classifiers)",
19265
+ "source_url": "https://grambank.clld.org/parameters/GB057"
19266
19266
  }
19267
19267
  },
19268
19268
  "derived-classifierLanguage": {
@@ -19272,11 +19272,11 @@
19272
19272
  "name": "Classifier language",
19273
19273
  "question": "Whether the language routinely uses classifiers when counting or referring to nouns.",
19274
19274
  "values": {},
19275
- "mt_relevance": "Classifier systems require noun-specific function words that have no counterpart in non-classifier languages.",
19275
+ "mt_relevance": "Numeral classifiers must be chosen per noun; the wrong classifier is immediately visible to native speakers.",
19276
19276
  "citation": {
19277
19277
  "authors": "Champollion enrichment pipeline (derived)",
19278
- "title": "Classifier language — derived from Grambank GB415 / WALS 55A",
19279
- "source_url": "https://grambank.clld.org/parameters/GB415"
19278
+ "title": "Classifier language — derived from Grambank GB057 / WALS 55A",
19279
+ "source_url": "https://grambank.clld.org/parameters/GB057"
19280
19280
  }
19281
19281
  },
19282
19282
  "derived-hasNumeralClassifiers": {
@@ -13,7 +13,7 @@
13
13
  "sovereignty_flags": {
14
14
  "off_community_processing": false,
15
15
  "attribution_required": true,
16
- "ocap_review": "pending"
16
+ "sovereignty_review": "pending"
17
17
  },
18
18
  "nc_terms": "Community-set terms; non-commercial use pending custodian confirmation.",
19
19
  "consent_attested": false,
@@ -2,7 +2,7 @@
2
2
  "$schema": "http://json-schema.org/draft-07/schema#",
3
3
  "$id": "https://champollion.dev/schemas/corpora-card.schema.json",
4
4
  "title": "Champollion Corpora Card",
5
- "description": "Schema for corpora cards — SSOT metadata for reference corpora, pair-specific evaluation sets, and multi-way parallel corpora. Reference corpora (ref-*) catalogue external datasets for development use. Evaluation sets (eval-*) are community-curated, pair-specific benchmarks with optional secret test splits, steward-controlled authorization, and OCAP®-aspirant sovereignty metadata. Multi-way corpora (multiway type with eval-* id) represent sentence-aligned parallel datasets covering many languages — a single card expands into N×(N-1) directional pairs at registry build time. Sovereignty fields record governance facts — who governs this data, what they've said about it, and what frameworks they've invoked. See DATA-SOVEREIGNTY.md for field reference.",
5
+ "description": "Schema for corpora cards — SSOT metadata for reference corpora, pair-specific evaluation sets, and multi-way parallel corpora. Reference corpora (ref-*) catalogue external datasets for development use. Evaluation sets (eval-*) are community-curated, pair-specific benchmarks with optional secret test splits, steward-controlled authorization, and sovereignty-aspirant metadata. Multi-way corpora (multiway type with eval-* id) represent sentence-aligned parallel datasets covering many languages — a single card expands into N×(N-1) directional pairs at registry build time. Sovereignty fields record governance facts — who governs this data, what they've said about it, and what frameworks they've invoked. See DATA-SOVEREIGNTY.md for field reference.",
6
6
  "type": "object",
7
7
  "required": ["id", "type", "name", "version", "description", "source", "license", "contamination", "_provenance"],
8
8
  "properties": {
@@ -536,7 +536,7 @@
536
536
 
537
537
  "submission": {
538
538
  "type": ["object", "null"],
539
- "description": "Terms for submitting a method for prize evaluation. Defines what transfers to the governance org, what the researcher retains, and what methods are admissible. The core OCAP deal: you want the prize, you transfer the complete self-hostable method to the language trust. See method-submission-agreement.md.",
539
+ "description": "Terms for submitting a method for prize evaluation. Defines what transfers to the governance org, what the researcher retains, and what methods are admissible. The core sovereignty deal: you want the prize, you transfer the complete self-hostable method to the language trust. See method-submission-agreement.md.",
540
540
  "properties": {
541
541
  "acceptanceThreshold": {
542
542
  "type": ["string", "null"],
@@ -544,7 +544,7 @@
544
544
  },
545
545
  "transfer": {
546
546
  "type": ["object", "null"],
547
- "description": "What transfers to the governance org upon acceptance. Each field is a scoped right. Maps to OCAP Ownership and Possession.",
547
+ "description": "What transfers to the governance org upon acceptance. Each field is a scoped right. Maps to community ownership and possession of the method.",
548
548
  "properties": {
549
549
  "sourceCode": {
550
550
  "type": "boolean",
@@ -586,7 +586,7 @@
586
586
  },
587
587
  "admissibility": {
588
588
  "type": ["object", "null"],
589
- "description": "What methods are eligible for prize evaluation. The sandbox is air-gapped (no network access), so all methods must be fully self-contained. This is a technical constraint, not a policy choice — OCAP Possession requires the community to own every byte needed to run the method.",
589
+ "description": "What methods are eligible for prize evaluation. The sandbox is air-gapped (no network access), so all methods must be fully self-contained. This is a technical constraint, not a policy choice — community possession of the method requires the community to own every byte needed to run the method.",
590
590
  "properties": {
591
591
  "selfHostable": {
592
592
  "type": "boolean",
@@ -599,7 +599,7 @@
599
599
  },
600
600
  "notes": {
601
601
  "type": ["string", "null"],
602
- "description": "Rationale for admissibility constraints. Should reference OCAP Possession and the air-gapped sandbox architecture."
602
+ "description": "Rationale for admissibility constraints. Should reference community possession of the method and the air-gapped sandbox architecture."
603
603
  }
604
604
  },
605
605
  "additionalProperties": false
@@ -678,6 +678,10 @@
678
678
  "type": "string",
679
679
  "description": "Human-readable reason a card is quarantined (e.g. 'gamayun-parallel builder not yet implemented'). Required whenever quarantine is true so the exclusion is never unexplained."
680
680
  },
681
+ "fixture": {
682
+ "type": "boolean",
683
+ "description": "If true, this card is a synthetic SCHEMA FIXTURE (placeholder consent/steward/sha values) kept to exercise the schema. build_registry skips fixture cards loudly: they never enter any registry, so they never reach champollion.dev/registry.json, the harness, or the prod datasets mirror. Pair with quarantine: true and a quarantineReason that says 'fixture'."
684
+ },
681
685
  "transmissionPolicy": {
682
686
  "type": "string",
683
687
  "enum": ["no-train", "consent-required"],
@@ -710,13 +714,13 @@
710
714
  "exposureTier": {
711
715
  "type": "string",
712
716
  "enum": ["local-only", "private", "public", "sealed"],
713
- "description": "Exposure tier chosen by the corpus author at registration (champollion register-corpus). Records the author's intent about how far this corpus travels. 'local-only' = never registered or uploaded — the card and the text stay entirely on the author's machine (cards with this value live OUTSIDE this tracked directory). 'private' = a sovereign / WMT-style held-out set: metadata is registered here but the text is NEVER uploaded or hosted; the author keeps custody (paired with quarantine=true so it is catalogued but not publicly runnable). 'public' = a fetch-from-source pointer + metadata card is published; the text is NEVER hosted by Champollion — it is fetched from source.repo_url on demand by the declared builder. 'sealed' = a community-controlled secret test set encrypted CLIENT-SIDE under the custodian group's threshold key before anything leaves the author's device; Champollion holds only ciphertext (in an off-git store) + a content-free card and cannot decrypt it (no single party can — M-of-N custodian approval is required). Paired with quarantine=true and a 'sealed' block; see the sealed property and docs/governance/OCAP_MULTISIG_PLAN.md. Public is gated by cli/lib/license-gate.mjs: NC / no-redistribute / unconfirmed-license sets may not use the public tier. Champollion never hosts corpus PLAINTEXT in ANY tier. Defaults to the most private tier (local-only) when unspecified.",
717
+ "description": "Exposure tier chosen by the corpus author at registration (champollion register-corpus). Records the author's intent about how far this corpus travels. 'local-only' = never registered or uploaded — the card and the text stay entirely on the author's machine (cards with this value live OUTSIDE this tracked directory). 'private' = a sovereign / WMT-style held-out set: metadata is registered here but the text is NEVER uploaded or hosted; the author keeps custody (paired with quarantine=true so it is catalogued but not publicly runnable). 'public' = a fetch-from-source pointer + metadata card is published; the text is NEVER hosted by Champollion — it is fetched from source.repo_url on demand by the declared builder. 'sealed' = a community-controlled secret test set encrypted CLIENT-SIDE under the custodian group's threshold key before anything leaves the author's device; Champollion holds only ciphertext (in an off-git store) + a content-free card and cannot decrypt it (no single party can — M-of-N custodian approval is required). Paired with quarantine=true and a 'sealed' block; see the sealed property and the community-custodian multisig plan (docs/governance). Public is gated by cli/lib/license-gate.mjs: NC / no-redistribute / unconfirmed-license sets may not use the public tier. Champollion never hosts corpus PLAINTEXT in ANY tier. Defaults to the most private tier (local-only) when unspecified.",
714
718
  "default": "local-only"
715
719
  },
716
720
 
717
721
  "sealed": {
718
722
  "type": ["object", "null"],
719
- "description": "Sealed-tier crypto metadata (exposureTier='sealed'). CONTENT-FREE: records how a community-controlled secret test set was encrypted client-side, never the plaintext. The ciphertext lives in an off-git, off-allowlist store; this block only names the cipher suite, the custodian group, the ciphertext digest, and the AAD binding, plus the paired public qualifier a method must clear before a sealed run can be proposed. The decryption key exists only as M-of-N custodian shares — Champollion cannot decrypt. See cli/lib/seal.mjs and docs/governance/OCAP_MULTISIG_PLAN.md (M1).",
723
+ "description": "Sealed-tier crypto metadata (exposureTier='sealed'). CONTENT-FREE: records how a community-controlled secret test set was encrypted client-side, never the plaintext. The ciphertext lives in an off-git, off-allowlist store; this block only names the cipher suite, the custodian group, the ciphertext digest, and the AAD binding, plus the paired public qualifier a method must clear before a sealed run can be proposed. The decryption key exists only as M-of-N custodian shares — Champollion cannot decrypt. See cli/lib/seal.mjs and the community-custodian multisig plan (docs/governance) (M1).",
720
724
  "required": ["cipher", "custodianGroupId", "ciphertextDigest", "aad"],
721
725
  "properties": {
722
726
  "cipher": {
@@ -782,7 +786,7 @@
782
786
  "type": "array",
783
787
  "items": {
784
788
  "type": "string",
785
- "enum": ["OCAP", "CARE", "Te-Mana-Raraunga", "FAIR", "IEEE-2890"]
789
+ "enum": ["community-ownership-control", "CARE", "Te-Mana-Raraunga", "FAIR", "IEEE-2890"]
786
790
  },
787
791
  "description": "Data sovereignty frameworks that the data creators or governing body have explicitly invoked. Only list frameworks where there is documented evidence of adoption — do not infer from language vitality, geography, or ethnicity."
788
792
  },
@@ -69,7 +69,7 @@
69
69
  },
70
70
  "sovereignty_flags": {
71
71
  "type": ["object", "null"],
72
- "description": "Data-handling constraints: off-community processing, attribution requirements, OCAP® notes. Free-form JSON."
72
+ "description": "Data-handling constraints: off-community processing, attribution requirements, data-sovereignty notes. Free-form JSON."
73
73
  },
74
74
  "nc_terms": {
75
75
  "type": ["string", "null"],
@@ -77,7 +77,7 @@
77
77
  },
78
78
  "consent_attested": {
79
79
  "type": "boolean",
80
- "description": "OCAP® consent gate: the provider has explicitly consented to be listed. Public read requires this true."
80
+ "description": "Consent gate (community control of listing): the provider has explicitly consented to be listed. Public read requires this true."
81
81
  },
82
82
  "status": {
83
83
  "type": "string",
@@ -515,7 +515,7 @@
515
515
  },
516
516
  "taxonomyNotes": {
517
517
  "type": ["string", "null"],
518
- "description": "Free-text affordance for known taxonomy disputes affecting this language: cases where ISO 639-3 and Glottolog disagree on splitting/lumping, contested language-vs-dialect status, or naming disputes. Policy (docs/LANGUAGE_TAXONOMY.md): display both authorities' positions, never adjudicate between them; communities name themselves (OCAP). Null when no dispute is documented — null means 'no note recorded', not 'no dispute exists'."
518
+ "description": "Free-text affordance for known taxonomy disputes affecting this language: cases where ISO 639-3 and Glottolog disagree on splitting/lumping, contested language-vs-dialect status, or naming disputes. Policy (docs/LANGUAGE_TAXONOMY.md): display both authorities' positions, never adjudicate between them; communities name themselves — community ownership and control of language data. Null when no dispute is documented — null means 'no note recorded', not 'no dispute exists'."
519
519
  },
520
520
  "classification": {
521
521
  "type": ["object", "null"],
@@ -389,7 +389,7 @@
389
389
  "type": "array",
390
390
  "items": {
391
391
  "type": "string",
392
- "enum": ["OCAP", "CARE", "maori-data-sovereignty", "UNDRIP", "other"]
392
+ "enum": ["community-ownership-control", "CARE", "maori-data-sovereignty", "UNDRIP", "other"]
393
393
  },
394
394
  "description": "Data sovereignty frameworks this method respects."
395
395
  },