champollion 0.3.3 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. package/README.md +52 -37
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +51 -3
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +289 -88
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +649 -130
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +16 -10
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +197 -38
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +193 -106
  100. package/lib/seal.mjs +6 -5
  101. package/lib/sealed-qualifier.mjs +2 -2
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +3 -2
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/DATA-SOVEREIGNTY.md +19 -20
  123. package/shared/LANGUAGE-CARD-FIELDS.md +1 -1
  124. package/shared/cards-fallback.json +1 -1
  125. package/shared/catalogue/card-config.json +1 -1
  126. package/shared/curated-orthography-conventions.json +26 -8
  127. package/shared/docent/faq.en.json +14 -16
  128. package/shared/docent/system-prompt.md +17 -19
  129. package/shared/explainers/tc-features.json +15 -15
  130. package/shared/gettext-plural-forms.json +45 -0
  131. package/shared/human-services.json +1 -1
  132. package/shared/method-registry.json +2 -0
  133. package/shared/metric-registry.json +96 -18
  134. package/shared/schemas/champollion-plugin.schema.json +4 -0
  135. package/shared/schemas/corpora-card.schema.json +20 -10
  136. package/shared/schemas/human-services.schema.json +2 -2
  137. package/shared/schemas/language-card.schema.json +1 -1
  138. package/shared/schemas/method-card.schema.json +1 -1
  139. package/shared/schemas/method-index-record.schema.json +67 -0
  140. package/shared/schemas/method-registry.schema.json +4 -0
  141. package/shared/schemas/metric-registry.schema.json +55 -1
  142. package/shared/docent/corpus.json +0 -11333
@@ -14,7 +14,15 @@
14
14
  "level": "both",
15
15
  "in_composite": true,
16
16
  "verifier_reproducible": true,
17
- "notes": "Harness core (tester.py), no plugin. Binary predicted == reference; corpus rate = matches/total."
17
+ "notes": "Harness core (tester.py), no plugin. Binary predicted == reference; corpus rate = matches/total.",
18
+ "ranking": {
19
+ "rounding": 4,
20
+ "ci_columns": null,
21
+ "segment_level": false,
22
+ "signature": {
23
+ "kind": "harness"
24
+ }
25
+ }
18
26
  },
19
27
  "equivalent_match_rate": {
20
28
  "category": "surface",
@@ -42,7 +50,19 @@
42
50
  "level": "both",
43
51
  "in_composite": true,
44
52
  "verifier_reproducible": true,
45
- "notes": "sacrebleu (word_order=2), harness core. Normalized /100 for the composite (scoring.NORMALIZATIONS)."
53
+ "notes": "sacrebleu (word_order=2), harness core. Normalized /100 for the composite (scoring.NORMALIZATIONS).",
54
+ "ranking": {
55
+ "rounding": 2,
56
+ "ci_columns": [
57
+ "chrf_ci_lower",
58
+ "chrf_ci_upper"
59
+ ],
60
+ "segment_level": true,
61
+ "signature": {
62
+ "kind": "sacrebleu",
63
+ "key": "chrf"
64
+ }
65
+ }
46
66
  },
47
67
  "bleu": {
48
68
  "category": "surface",
@@ -56,7 +76,16 @@
56
76
  "level": "corpus",
57
77
  "in_composite": false,
58
78
  "verifier_reproducible": true,
59
- "notes": "sacrebleu, harness core. Reported for MT-literature compatibility, never composited. Run-card key is top-level 'corpus_bleu' (not scores.bleu) — see _proposed_renames."
79
+ "notes": "sacrebleu, harness core. Reported for MT-literature compatibility, never composited. Run-card key is top-level 'corpus_bleu' (not scores.bleu) — see _rename_decisions.",
80
+ "ranking": {
81
+ "rounding": 2,
82
+ "ci_columns": null,
83
+ "segment_level": true,
84
+ "signature": {
85
+ "kind": "sacrebleu",
86
+ "key": "bleu"
87
+ }
88
+ }
60
89
  },
61
90
  "ter": {
62
91
  "category": "surface",
@@ -70,7 +99,16 @@
70
99
  "level": "both",
71
100
  "in_composite": false,
72
101
  "verifier_reproducible": true,
73
- "notes": "sacrebleu corpus_ter. Excluded from composite (correlates with chrF++)."
102
+ "notes": "sacrebleu corpus_ter. Excluded from composite (correlates with chrF++).",
103
+ "ranking": {
104
+ "rounding": 2,
105
+ "ci_columns": null,
106
+ "segment_level": false,
107
+ "signature": {
108
+ "kind": "sacrebleu",
109
+ "key": "ter"
110
+ }
111
+ }
74
112
  },
75
113
  "length_ratio": {
76
114
  "category": "surface",
@@ -196,7 +234,16 @@
196
234
  "level": "both",
197
235
  "in_composite": false,
198
236
  "verifier_reproducible": true,
199
- "notes": "Harness core (metrics_comet.py); model auto-selected via the card's metricModelSupport. NEURAL — reported in the separate lane, never composited; verifier re-derives fail-closed (recompute_corpus_comet)."
237
+ "notes": "Harness core (metrics_comet.py); model auto-selected via the card's metricModelSupport. NEURAL — reported in the separate lane, never composited; verifier re-derives fail-closed (recompute_corpus_comet).",
238
+ "ranking": {
239
+ "rounding": 4,
240
+ "ci_columns": null,
241
+ "segment_level": false,
242
+ "signature": {
243
+ "kind": "model",
244
+ "key": "comet_model"
245
+ }
246
+ }
200
247
  },
201
248
  "qe_score": {
202
249
  "category": "neural",
@@ -284,7 +331,7 @@
284
331
  },
285
332
  "compliance_index": {
286
333
  "category": "compliance",
287
- "status": "implemented",
334
+ "status": "planned",
288
335
  "display_name": "Double-Pass Compliance",
289
336
  "plugin_name": "double_pass_compliance",
290
337
  "card_key": null,
@@ -294,11 +341,11 @@
294
341
  "level": "both",
295
342
  "in_composite": false,
296
343
  "verifier_reproducible": true,
297
- "notes": "DoublePassCompliancePlugin. Quality GATE (placeholder/quote/casing integrity), not a quality score; lives in plugin_metrics/report only, no run_cards column."
344
+ "notes": "DoublePassCompliancePlugin. Quality GATE (placeholder/quote/casing integrity), not a quality score; no run_cards column. Planned, not implemented: no run path constructs the plugin (it needs a card with a `rules` field, which the atlas cutover dropped from every card as uncited script-family templates, and no atlas parameter carries quote or letter-case conventions). Passed a card without `rules`, only the 60% variable-integrity term measures anything; quote and casing score a constant 1.0."
298
345
  },
299
346
  "repair_effectiveness": {
300
347
  "category": "compliance",
301
- "status": "implemented",
348
+ "status": "planned",
302
349
  "display_name": "Repair Effectiveness",
303
350
  "plugin_name": "double_pass_compliance",
304
351
  "card_key": null,
@@ -308,7 +355,7 @@
308
355
  "level": "corpus",
309
356
  "in_composite": false,
310
357
  "verifier_reproducible": true,
311
- "notes": "Fraction of compliance violations auto-repaired by post-translation hooks; same plugin as compliance_index."
358
+ "notes": "Fraction of compliance violations auto-repaired by post-translation hooks; same plugin as compliance_index, and planned for the same reason: no run path constructs it."
312
359
  },
313
360
  "spbleu": {
314
361
  "category": "comparator",
@@ -322,7 +369,16 @@
322
369
  "level": "corpus",
323
370
  "in_composite": false,
324
371
  "verifier_reproducible": true,
325
- "notes": "Comparability sidecar (FLORES/NLLB lingua-franca tokenizer). JSONB only."
372
+ "notes": "Comparability sidecar (FLORES/NLLB lingua-franca tokenizer). JSONB only.",
373
+ "ranking": {
374
+ "rounding": 2,
375
+ "ci_columns": null,
376
+ "segment_level": false,
377
+ "signature": {
378
+ "kind": "sacrebleu",
379
+ "key": "spbleu"
380
+ }
381
+ }
326
382
  },
327
383
  "chrf_plain": {
328
384
  "category": "comparator",
@@ -336,7 +392,16 @@
336
392
  "level": "corpus",
337
393
  "in_composite": false,
338
394
  "verifier_reproducible": true,
339
- "notes": "The chrF figure FLORES/WMT tables report. JSONB only."
395
+ "notes": "The chrF figure FLORES/WMT tables report. JSONB only.",
396
+ "ranking": {
397
+ "rounding": 2,
398
+ "ci_columns": null,
399
+ "segment_level": true,
400
+ "signature": {
401
+ "kind": "sacrebleu",
402
+ "key": "chrf_plain"
403
+ }
404
+ }
340
405
  },
341
406
  "fuse_score": {
342
407
  "category": "comparator",
@@ -383,7 +448,7 @@
383
448
  "composite": {
384
449
  "category": "composite",
385
450
  "status": "implemented",
386
- "display_name": "Composite Score (experimental)",
451
+ "display_name": "Legacy composite (retired)",
387
452
  "plugin_name": null,
388
453
  "card_key": null,
389
454
  "db_column": "composite_score",
@@ -392,12 +457,24 @@
392
457
  "level": "corpus",
393
458
  "in_composite": false,
394
459
  "verifier_reproducible": true,
395
- "notes": "Derived by scoring.compute_composite_score from the profile weight tables (SSOT: scoring spec §4.3). A convenience sort key, NOT a validated quality measurement."
460
+ "notes": "RETIRED by scoring standard/1 (2026-10-04): no new run is scored, ranked or labelled with it — new run cards publish composite = null. Runs are ranked by corpus chrF++ with its 95% bootstrap CI; BLEU, spBLEU, TER and COMET are shown beside it, never blended. A card published before the standard keeps its stored composite, re-derivable by scoring.compute_composite_score from the profile weight tables (scoring spec §4.3) and shown only as 'legacy composite (retired)'. It was a convenience sort key, never a validated quality measurement: an untrained model repeating one valid sentence for every input scored 0.6244 ('functional') at chrF++ 5.5.",
461
+ "ranking": {
462
+ "rounding": 4,
463
+ "ci_columns": [
464
+ "composite_ci_lower",
465
+ "composite_ci_upper"
466
+ ],
467
+ "segment_level": false,
468
+ "signature": {
469
+ "kind": "harness"
470
+ },
471
+ "retired": "scoring standard/1 (2026-10-04) retired the weighted composite: a NEW contest ranks on corpus chrF++ (the default) or another standard metric. A contest that already recorded primary_metric=composite still ranks on it, labelled legacy composite (retired)."
472
+ }
396
473
  },
397
474
  "cost_adjusted": {
398
475
  "category": "composite",
399
476
  "status": "implemented",
400
- "display_name": "Cost-Adjusted Score",
477
+ "display_name": "Legacy cost-adjusted composite (retired)",
401
478
  "plugin_name": null,
402
479
  "card_key": null,
403
480
  "db_column": null,
@@ -406,12 +483,12 @@
406
483
  "level": "corpus",
407
484
  "in_composite": false,
408
485
  "verifier_reproducible": true,
409
- "notes": "scoring.cost_adjusted_score (composite / log2(1 + cost*1000), penalty-only). JSONB only."
486
+ "notes": "RETIRED with the composite by scoring standard/1 (2026-10-04): new run cards publish cost_adjusted = null. Legacy cards: scoring.cost_adjusted_score (composite / log2(1 + cost*1000), penalty-only). JSONB only."
410
487
  },
411
488
  "quality_tier": {
412
489
  "category": "composite",
413
490
  "status": "implemented",
414
- "display_name": "Quality Tier",
491
+ "display_name": "Legacy quality tier (retired)",
415
492
  "plugin_name": null,
416
493
  "card_key": null,
417
494
  "db_column": "quality_tier",
@@ -420,7 +497,7 @@
420
497
  "level": "corpus",
421
498
  "in_composite": false,
422
499
  "verifier_reproducible": true,
423
- "notes": "Heuristic label on the composite (scoring.QUALITY_TIERS); only human review confirms usability."
500
+ "notes": "RETIRED by scoring standard/1 (2026-10-04): new run cards publish quality_tier = null, and no surface labels a new run with a tier. Legacy cards keep the heuristic label they were published with (scoring.QUALITY_TIERS on the retired composite); only human evaluation certifies quality."
424
501
  },
425
502
  "tokens_per_second": {
426
503
  "category": "efficiency",
@@ -591,7 +668,8 @@
591
668
  "notes": "clamp0((chrF++ - floor)/(100 - floor)) — removes the orthography-specific chance floor so scores are cross-language comparable; the clamp at 0 doubles as the noise rail (at-or-below-floor = indistinguishable from chance). PARTIAL: implemented today only in the connection-quality lane (arena/mt_eval_harness/connection_quality.py cchrf(); JS twins cli/website/src/utils/connectionQuality.mjs + arcStrength.mjs; constants SSOT shared/connection-quality.json cq-v1) — NOT computed by tester.py and never a leaderboard column. Floors: cli/website/src/data/cchrf-floors.json (196 languages, champollion-derived), regenerated from research/cchrf/results/atlas.json (Monte-Carlo N1-unigram floors over FLORES-200 dev monolingual text; the study + paper live in research/cchrf). Known caveat before any RANKING consumer wires this: N1 undershoots fluent-output chance by ~2.3 chrF++ measured on 24/204 languages (research/cchrf REVIEW_2026-07-11 M1); the N1-vs-N_w estimator choice is an open founder decision recorded in docs/METRICS_RESEARCH_PROGRAM_2026-07-10.md. Forbidden on language cards (card-integrity R3)."
592
669
  }
593
670
  },
594
- "_proposed_renames": [
671
+ "_proposed_renames": [],
672
+ "_rename_decisions": [
595
673
  {
596
674
  "current": "run-card key 'corpus_bleu' (top-level) vs canonical 'bleu' vs db_column 'corpus_bleu'",
597
675
  "proposal": "Move BLEU into scores as scores.bleu (spec §9 already shows it there) and keep db_column corpus_bleu; OR rename nothing and let this registry carry the mapping.",
@@ -42,6 +42,10 @@
42
42
  "format": "uri",
43
43
  "description": "Required for type 'api'. The server-side translation endpoint URL."
44
44
  },
45
+ "acceptsInstructions": {
46
+ "type": "boolean",
47
+ "description": "api plugins: whether the endpoint follows per-key instructions (the plural forms a key needs, a quality-gate retry's feedback). false — a trained NMT model that only translates text (e.g. nmt-forge serve): the CLI never asks it twice for the same text (it would answer the same) and sends what it refuses to the pair's fallback. true — requests carry an \"instructions\" object (key → text) beside \"keys\". Omitted — unknown: the CLI asks once more without feedback and says so."
48
+ },
45
49
  "config": {
46
50
  "type": "object",
47
51
  "description": "Method configuration — canonical MethodConfig shape.",
@@ -2,7 +2,7 @@
2
2
  "$schema": "http://json-schema.org/draft-07/schema#",
3
3
  "$id": "https://champollion.dev/schemas/corpora-card.schema.json",
4
4
  "title": "Champollion Corpora Card",
5
- "description": "Schema for corpora cards — SSOT metadata for reference corpora, pair-specific evaluation sets, and multi-way parallel corpora. Reference corpora (ref-*) catalogue external datasets for development use. Evaluation sets (eval-*) are community-curated, pair-specific benchmarks with optional secret test splits, steward-controlled authorization, and OCAP®-aspirant sovereignty metadata. Multi-way corpora (multiway type with eval-* id) represent sentence-aligned parallel datasets covering many languages — a single card expands into N×(N-1) directional pairs at registry build time. Sovereignty fields record governance facts — who governs this data, what they've said about it, and what frameworks they've invoked. See DATA-SOVEREIGNTY.md for field reference.",
5
+ "description": "Schema for corpora cards — SSOT metadata for reference corpora, pair-specific evaluation sets, and multi-way parallel corpora. Reference corpora (ref-*) catalogue external datasets for development use. Evaluation sets (eval-*) are community-curated, pair-specific benchmarks with optional secret test splits, steward-controlled authorization, and sovereignty-aspirant metadata. Multi-way corpora (multiway type with eval-* id) represent sentence-aligned parallel datasets covering many languages — a single card expands into N×(N-1) directional pairs at registry build time. Sovereignty fields record governance facts — who governs this data, what they've said about it, and what frameworks they've invoked. See DATA-SOVEREIGNTY.md for field reference.",
6
6
  "type": "object",
7
7
  "required": ["id", "type", "name", "version", "description", "source", "license", "contamination", "_provenance"],
8
8
  "properties": {
@@ -536,7 +536,7 @@
536
536
 
537
537
  "submission": {
538
538
  "type": ["object", "null"],
539
- "description": "Terms for submitting a method for prize evaluation. Defines what transfers to the governance org, what the researcher retains, and what methods are admissible. The core OCAP deal: you want the prize, you transfer the complete self-hostable method to the language trust. See method-submission-agreement.md.",
539
+ "description": "Terms for submitting a method for prize evaluation. Defines what transfers to the governance org, what the researcher retains, and what methods are admissible. The core sovereignty deal: you want the prize, you transfer the complete self-hostable method to the language trust. See method-submission-agreement.md.",
540
540
  "properties": {
541
541
  "acceptanceThreshold": {
542
542
  "type": ["string", "null"],
@@ -544,7 +544,7 @@
544
544
  },
545
545
  "transfer": {
546
546
  "type": ["object", "null"],
547
- "description": "What transfers to the governance org upon acceptance. Each field is a scoped right. Maps to OCAP Ownership and Possession.",
547
+ "description": "What transfers to the governance org upon acceptance. Each field is a scoped right. Maps to community ownership and possession of the method.",
548
548
  "properties": {
549
549
  "sourceCode": {
550
550
  "type": "boolean",
@@ -586,7 +586,7 @@
586
586
  },
587
587
  "admissibility": {
588
588
  "type": ["object", "null"],
589
- "description": "What methods are eligible for prize evaluation. The sandbox is air-gapped (no network access), so all methods must be fully self-contained. This is a technical constraint, not a policy choice — OCAP Possession requires the community to own every byte needed to run the method.",
589
+ "description": "What methods are eligible for prize evaluation. The sandbox is air-gapped (no network access), so all methods must be fully self-contained. This is a technical constraint, not a policy choice — community possession of the method requires the community to own every byte needed to run the method.",
590
590
  "properties": {
591
591
  "selfHostable": {
592
592
  "type": "boolean",
@@ -599,7 +599,7 @@
599
599
  },
600
600
  "notes": {
601
601
  "type": ["string", "null"],
602
- "description": "Rationale for admissibility constraints. Should reference OCAP Possession and the air-gapped sandbox architecture."
602
+ "description": "Rationale for admissibility constraints. Should reference community possession of the method and the air-gapped sandbox architecture."
603
603
  }
604
604
  },
605
605
  "additionalProperties": false
@@ -639,8 +639,8 @@
639
639
  "properties": {
640
640
  "risk": {
641
641
  "type": "string",
642
- "enum": ["NONE", "LOW", "MEDIUM", "HIGH"],
643
- "description": "NONE = private/unpublished. LOW = niche/recent. MEDIUM = public but not widely used. HIGH = known to be in major training sets."
642
+ "enum": ["NONE", "LOW", "MEDIUM", "HIGH", "UNCHECKED"],
643
+ "description": "NONE = private/unpublished. LOW = niche/recent. MEDIUM = public but not widely used. HIGH = known to be in major training sets. UNCHECKED = not graded: written by `champollion network register-corpus` when the file could not be compared with the public corpora (no corpora cards, public catalogue unreachable) and no grade was stated. Like every grade but LOW, it keeps a corpus in the relative-comparison-only lane."
644
644
  },
645
645
  "reasoning": {
646
646
  "type": "string",
@@ -678,6 +678,10 @@
678
678
  "type": "string",
679
679
  "description": "Human-readable reason a card is quarantined (e.g. 'gamayun-parallel builder not yet implemented'). Required whenever quarantine is true so the exclusion is never unexplained."
680
680
  },
681
+ "fixture": {
682
+ "type": "boolean",
683
+ "description": "If true, this card is a synthetic SCHEMA FIXTURE (placeholder consent/steward/sha values) kept to exercise the schema. build_registry skips fixture cards loudly: they never enter any registry, so they never reach champollion.dev/registry.json, the harness, or the prod datasets mirror. Pair with quarantine: true and a quarantineReason that says 'fixture'."
684
+ },
681
685
  "transmissionPolicy": {
682
686
  "type": "string",
683
687
  "enum": ["no-train", "consent-required"],
@@ -710,13 +714,19 @@
710
714
  "exposureTier": {
711
715
  "type": "string",
712
716
  "enum": ["local-only", "private", "public", "sealed"],
713
- "description": "Exposure tier chosen by the corpus author at registration (champollion register-corpus). Records the author's intent about how far this corpus travels. 'local-only' = never registered or uploaded — the card and the text stay entirely on the author's machine (cards with this value live OUTSIDE this tracked directory). 'private' = a sovereign / WMT-style held-out set: metadata is registered here but the text is NEVER uploaded or hosted; the author keeps custody (paired with quarantine=true so it is catalogued but not publicly runnable). 'public' = a fetch-from-source pointer + metadata card is published; the text is NEVER hosted by Champollion — it is fetched from source.repo_url on demand by the declared builder. 'sealed' = a community-controlled secret test set encrypted CLIENT-SIDE under the custodian group's threshold key before anything leaves the author's device; Champollion holds only ciphertext (in an off-git store) + a content-free card and cannot decrypt it (no single party can — M-of-N custodian approval is required). Paired with quarantine=true and a 'sealed' block; see the sealed property and docs/governance/OCAP_MULTISIG_PLAN.md. Public is gated by cli/lib/license-gate.mjs: NC / no-redistribute / unconfirmed-license sets may not use the public tier. Champollion never hosts corpus PLAINTEXT in ANY tier. Defaults to the most private tier (local-only) when unspecified.",
717
+ "description": "Exposure tier chosen by the corpus author at registration (champollion register-corpus). Records the author's intent about how far this corpus travels. 'local-only' = never registered or uploaded — the card and the text stay entirely on the author's machine (cards with this value live OUTSIDE this tracked directory). 'private' = a sovereign / WMT-style held-out set: metadata is registered here but the text is NEVER uploaded or hosted; the author keeps custody (paired with quarantine=true so it is catalogued but not publicly runnable). 'public' = a fetch-from-source pointer + metadata card is published; the text is NEVER hosted by Champollion — it is fetched from source.repo_url on demand by the declared builder. 'sealed' = a community-controlled secret test set encrypted CLIENT-SIDE under the custodian group's threshold key before anything leaves the author's device; Champollion holds only ciphertext (in an off-git store) + a content-free card and cannot decrypt it (no single party can — M-of-N custodian approval is required). Paired with quarantine=true and a 'sealed' block; see the sealed property and the community-custodian multisig plan (docs/governance). Public is gated by cli/lib/license-gate.mjs: NC / no-redistribute / unconfirmed-license sets may not use the public tier. Champollion never hosts corpus PLAINTEXT in ANY tier. Defaults to the most private tier (local-only) when unspecified.",
714
718
  "default": "local-only"
715
719
  },
716
720
 
721
+ "transmission": {
722
+ "type": "string",
723
+ "enum": ["local-only"],
724
+ "description": "The steward's transmission mark — SEPARATE from the licence. 'local-only' = only a model on the steward's own machine may see the text; every remote model API is refused. Written by `champollion network register-corpus` for a local-only set (and kept when a file it registers was already marked), mirroring the <file>.champollion.json sidecar the harness reads (mt_eval_harness.corpus_loader.read_steward_sidecar). It is not a licence term: license.* says what others may do with the text under its licence; this mark says where the steward lets it travel. Absent = no steward mark (the licence decides which model services may see it)."
725
+ },
726
+
717
727
  "sealed": {
718
728
  "type": ["object", "null"],
719
- "description": "Sealed-tier crypto metadata (exposureTier='sealed'). CONTENT-FREE: records how a community-controlled secret test set was encrypted client-side, never the plaintext. The ciphertext lives in an off-git, off-allowlist store; this block only names the cipher suite, the custodian group, the ciphertext digest, and the AAD binding, plus the paired public qualifier a method must clear before a sealed run can be proposed. The decryption key exists only as M-of-N custodian shares — Champollion cannot decrypt. See cli/lib/seal.mjs and docs/governance/OCAP_MULTISIG_PLAN.md (M1).",
729
+ "description": "Sealed-tier crypto metadata (exposureTier='sealed'). CONTENT-FREE: records how a community-controlled secret test set was encrypted client-side, never the plaintext. The ciphertext lives in an off-git, off-allowlist store; this block only names the cipher suite, the custodian group, the ciphertext digest, and the AAD binding, plus the paired public qualifier a method must clear before a sealed run can be proposed. The decryption key exists only as M-of-N custodian shares — Champollion cannot decrypt. See cli/lib/seal.mjs and the community-custodian multisig plan (docs/governance) (M1).",
720
730
  "required": ["cipher", "custodianGroupId", "ciphertextDigest", "aad"],
721
731
  "properties": {
722
732
  "cipher": {
@@ -782,7 +792,7 @@
782
792
  "type": "array",
783
793
  "items": {
784
794
  "type": "string",
785
- "enum": ["OCAP", "CARE", "Te-Mana-Raraunga", "FAIR", "IEEE-2890"]
795
+ "enum": ["community-ownership-control", "CARE", "Te-Mana-Raraunga", "FAIR", "IEEE-2890"]
786
796
  },
787
797
  "description": "Data sovereignty frameworks that the data creators or governing body have explicitly invoked. Only list frameworks where there is documented evidence of adoption — do not infer from language vitality, geography, or ethnicity."
788
798
  },
@@ -69,7 +69,7 @@
69
69
  },
70
70
  "sovereignty_flags": {
71
71
  "type": ["object", "null"],
72
- "description": "Data-handling constraints: off-community processing, attribution requirements, OCAP® notes. Free-form JSON."
72
+ "description": "Data-handling constraints: off-community processing, attribution requirements, data-sovereignty notes. Free-form JSON."
73
73
  },
74
74
  "nc_terms": {
75
75
  "type": ["string", "null"],
@@ -77,7 +77,7 @@
77
77
  },
78
78
  "consent_attested": {
79
79
  "type": "boolean",
80
- "description": "OCAP® consent gate: the provider has explicitly consented to be listed. Public read requires this true."
80
+ "description": "Consent gate (community control of listing): the provider has explicitly consented to be listed. Public read requires this true."
81
81
  },
82
82
  "status": {
83
83
  "type": "string",
@@ -515,7 +515,7 @@
515
515
  },
516
516
  "taxonomyNotes": {
517
517
  "type": ["string", "null"],
518
- "description": "Free-text affordance for known taxonomy disputes affecting this language: cases where ISO 639-3 and Glottolog disagree on splitting/lumping, contested language-vs-dialect status, or naming disputes. Policy (docs/LANGUAGE_TAXONOMY.md): display both authorities' positions, never adjudicate between them; communities name themselves (OCAP). Null when no dispute is documented — null means 'no note recorded', not 'no dispute exists'."
518
+ "description": "Free-text affordance for known taxonomy disputes affecting this language: cases where ISO 639-3 and Glottolog disagree on splitting/lumping, contested language-vs-dialect status, or naming disputes. Policy (docs/LANGUAGE_TAXONOMY.md): display both authorities' positions, never adjudicate between them; communities name themselves — community ownership and control of language data. Null when no dispute is documented — null means 'no note recorded', not 'no dispute exists'."
519
519
  },
520
520
  "classification": {
521
521
  "type": ["object", "null"],
@@ -389,7 +389,7 @@
389
389
  "type": "array",
390
390
  "items": {
391
391
  "type": "string",
392
- "enum": ["OCAP", "CARE", "maori-data-sovereignty", "UNDRIP", "other"]
392
+ "enum": ["community-ownership-control", "CARE", "maori-data-sovereignty", "UNDRIP", "other"]
393
393
  },
394
394
  "description": "Data sovereignty frameworks this method respects."
395
395
  },
@@ -0,0 +1,67 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "$id": "https://champollion.dev/schemas/method-index-record.schema.json",
4
+ "title": "Champollion Method Index Record",
5
+ "description": "The public index record of a contest method: a deterministic projection of its bundle manifest (arena/mt_eval_harness/method_index.py) that keeps what the method is, who owns it, under what licence and how it decodes, plus the bundle's sha256 — and drops the contest binding, the qualifier receipt and the developer email. The record's canonical bytes (sorted keys, no whitespace, UTF-8) are minted once, by the harness; a node's signed score manifest carries their sha256 as indexEntrySha256. Consumers verify that sha over the bytes they are given; they never re-serialize.",
6
+ "type": "object",
7
+ "required": ["indexRecordVersion", "lane", "method", "owner", "licence", "languagePair", "methodSha256"],
8
+ "additionalProperties": false,
9
+ "properties": {
10
+ "indexRecordVersion": { "const": 1 },
11
+ "lane": {
12
+ "type": "string",
13
+ "enum": ["declarative-model", "method-execution"],
14
+ "description": "declarative-model = Lane A (weights run by the node's trusted engine, no participant code); method-execution = Lane B (code run in a --network=none sandbox)."
15
+ },
16
+ "method": {
17
+ "type": "object",
18
+ "required": ["name", "version"],
19
+ "additionalProperties": false,
20
+ "properties": {
21
+ "name": { "type": "string", "minLength": 1 },
22
+ "version": { "type": "string", "minLength": 1 },
23
+ "class": { "type": ["string", "null"] },
24
+ "paradigm": { "type": ["string", "null"] }
25
+ }
26
+ },
27
+ "owner": {
28
+ "type": "object",
29
+ "required": ["name"],
30
+ "additionalProperties": false,
31
+ "description": "Display identity only. Never an email.",
32
+ "properties": {
33
+ "name": { "type": ["string", "null"], "not": { "type": "string", "pattern": "@" } },
34
+ "affiliation": { "type": ["string", "null"] }
35
+ }
36
+ },
37
+ "licence": { "type": "string", "minLength": 1, "description": "The weights/code licence the participant declared (SPDX id or LicenseRef-*)." },
38
+ "weightsPublic": { "type": ["boolean", "null"] },
39
+ "parameterCount": { "type": ["integer", "null"], "minimum": 1 },
40
+ "track": { "type": ["string", "null"], "enum": ["constrained", "unconstrained", null] },
41
+ "trainingData": { "type": ["string", "null"], "maxLength": 2000 },
42
+ "languagePair": {
43
+ "type": "object",
44
+ "required": ["source", "target"],
45
+ "additionalProperties": false,
46
+ "description": "The pair measured. Index entries are keyed to the variety measured and never compared across test sets.",
47
+ "properties": {
48
+ "source": { "type": "string", "minLength": 1 },
49
+ "target": { "type": "string", "minLength": 1 }
50
+ }
51
+ },
52
+ "methodSha256": { "type": "string", "pattern": "^[0-9a-f]{64}$" },
53
+ "model": {
54
+ "type": "object",
55
+ "description": "Lane A only: the decoding the node's engine applies.",
56
+ "additionalProperties": false,
57
+ "properties": {
58
+ "architecture": { "type": ["string", "null"] },
59
+ "srcLang": { "type": ["string", "null"] },
60
+ "tgtLang": { "type": ["string", "null"] },
61
+ "generation": { "type": "object" }
62
+ }
63
+ },
64
+ "requirements": { "type": "object", "description": "Lane B only: the resources the method declared." },
65
+ "imageDigest": { "type": ["string", "null"], "description": "Lane B only: the built image the node ran." }
66
+ }
67
+ }
@@ -60,6 +60,10 @@
60
60
  "description": "When true, ALL credential_env vars must be set for 'ready' — key-pair auth (e.g. AWS access-key id + secret; Lara id + secret). Default false = any one suffices (alias lists). Only meaningful alongside credential_env."
61
61
  },
62
62
  "default_base_url": { "type": "string" },
63
+ "keyless": {
64
+ "type": "boolean",
65
+ "description": "True when the default endpoint needs no credential at all — a server on the user's own machine (local) or a free public API (apertium). Its env vars only POINT ELSEWHERE, so availability surfaces report it ready with no key set. Hosted APIs with a default_base_url are NOT keyless."
66
+ },
63
67
  "homepage": { "type": "string" },
64
68
  "license": { "type": "string" },
65
69
  "commercialReady": { "type": "boolean" },
@@ -32,6 +32,21 @@
32
32
  "decision": { "type": "string", "enum": ["founder-pending", "accepted", "rejected"] }
33
33
  }
34
34
  }
35
+ },
36
+ "_rename_decisions": {
37
+ "type": "array",
38
+ "description": "Rename proposals the FOUNDER has already decided, moved out of _proposed_renames with the decision recorded verbatim (dated wording such as KEEP or DEFER, which the pending-proposal enum does not express). Advisory — not consumed by code.",
39
+ "items": {
40
+ "type": "object",
41
+ "required": ["current", "proposal", "impact", "decision"],
42
+ "additionalProperties": false,
43
+ "properties": {
44
+ "current": { "type": "string" },
45
+ "proposal": { "type": "string" },
46
+ "impact": { "type": "string" },
47
+ "decision": { "type": "string", "minLength": 1 }
48
+ }
49
+ }
35
50
  }
36
51
  },
37
52
  "$defs": {
@@ -89,7 +104,46 @@
89
104
  "type": "boolean",
90
105
  "description": "True iff the verifier can deterministically re-derive it from the sha-pinned corpus + stored entries (+ card-pinned FST / pinned neural model under the fail-closed contract). Speed/cost/style metrics are false."
91
106
  },
92
- "notes": { "type": "string" }
107
+ "notes": { "type": "string" },
108
+ "ranking": { "$ref": "#/$defs/ranking" }
109
+ }
110
+ },
111
+ "ranking": {
112
+ "type": "object",
113
+ "description": "Present iff a contest may rank on this metric (contests.metadata.primary_metric). Absent = not rankable. Adding a metric to the rankable set is a data change here plus a rankable_metrics row (migration 076 pattern), never a code change.",
114
+ "required": ["rounding", "ci_columns", "segment_level", "signature"],
115
+ "additionalProperties": false,
116
+ "properties": {
117
+ "rounding": {
118
+ "type": "integer", "minimum": 0, "maximum": 6,
119
+ "description": "Display rounding; also the precision of the point-equality tie rung."
120
+ },
121
+ "ci_columns": {
122
+ "description": "The run_cards bootstrap-CI column pair [lower, upper] for the CI-overlap tie rung, or null when the metric has no CI columns.",
123
+ "oneOf": [
124
+ { "type": "null" },
125
+ { "type": "array", "items": { "type": "string" }, "minItems": 2, "maxItems": 2 }
126
+ ]
127
+ },
128
+ "segment_level": {
129
+ "type": "boolean",
130
+ "description": "True iff the harness has a corpus function over per-segment rows (significance.py) so a paired test (approximate randomization / paired bootstrap) can compare two entries on this metric."
131
+ },
132
+ "retired": {
133
+ "type": "string",
134
+ "minLength": 1,
135
+ "description": "Present iff the metric is RETIRED as a ranking choice for NEW contests (the reason, said to whoever tries). A contest that already recorded it as metadata.primary_metric still ranks on it, labelled legacy; the block stays so the rankable_metrics table (migration 076) keeps its row and existing contests keep passing contest_lifecycle_guard()."
136
+ },
137
+ "signature": {
138
+ "type": "object",
139
+ "description": "How this metric's computation is identified, so a contest can freeze it as a promise (contests.metadata.metric_signature). sacrebleu: the run card's scores.sacrebleu_signatures[key]; model: the run card's scores[key] (the neural model id) plus the harness version; harness: the harness version alone.",
140
+ "required": ["kind"],
141
+ "additionalProperties": false,
142
+ "properties": {
143
+ "kind": { "type": "string", "enum": ["sacrebleu", "model", "harness"] },
144
+ "key": { "type": "string", "minLength": 1 }
145
+ }
146
+ }
93
147
  }
94
148
  }
95
149
  }