champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,64 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "$id": "https://champollion.dev/schemas/domain-taxonomy.schema.json",
4
+ "title": "Champollion Domain Taxonomy",
5
+ "description": "Schema for shared/domain-taxonomy.json — the closed SSOT of translation-domain codes. Enforces exactly the 17 content codes plus the single composite marker 'mixed'. The classifier (domain_classifier.py) and the corpora-card schema both reference this list; this schema is what keeps the SSOT itself well-formed and closed.",
6
+ "type": "object",
7
+ "required": ["version", "description", "domains", "composite"],
8
+ "additionalProperties": false,
9
+ "properties": {
10
+ "$schema": { "type": "string" },
11
+ "version": {
12
+ "type": "string",
13
+ "description": "Semver of the taxonomy. Bump when a code is added or its meaning changes."
14
+ },
15
+ "description": {
16
+ "type": "string",
17
+ "description": "What this SSOT is and which consumers must stay in sync with it."
18
+ },
19
+ "domains": {
20
+ "type": "object",
21
+ "description": "The closed set of 17 content domain codes. Each maps to a one-line definition. Adding or removing a key here is a taxonomy change: update domain_classifier.py and corpora-card.schema.json in lockstep.",
22
+ "minProperties": 17,
23
+ "maxProperties": 17,
24
+ "additionalProperties": false,
25
+ "required": [
26
+ "ui", "legal", "medical", "financial", "edu", "ecommerce",
27
+ "marketing", "gov", "scientific", "religious", "support",
28
+ "subtitles", "news", "literary", "conv", "tech", "humanitarian"
29
+ ],
30
+ "propertyNames": {
31
+ "enum": [
32
+ "ui", "legal", "medical", "financial", "edu", "ecommerce",
33
+ "marketing", "gov", "scientific", "religious", "support",
34
+ "subtitles", "news", "literary", "conv", "tech", "humanitarian"
35
+ ]
36
+ },
37
+ "patternProperties": {
38
+ "^.+$": { "type": "string", "minLength": 1 }
39
+ }
40
+ },
41
+ "composite": {
42
+ "type": "object",
43
+ "description": "Non-content composition markers. The only member is 'mixed', which requires a domainDistribution wherever it is used as a domain.",
44
+ "additionalProperties": false,
45
+ "required": ["mixed"],
46
+ "properties": {
47
+ "mixed": {
48
+ "type": "object",
49
+ "additionalProperties": false,
50
+ "required": ["definition", "requiresDistribution"],
51
+ "properties": {
52
+ "definition": { "type": "string", "minLength": 1 },
53
+ "requiresDistribution": { "type": "boolean", "const": true }
54
+ }
55
+ }
56
+ }
57
+ },
58
+ "notes": {
59
+ "type": "object",
60
+ "description": "Free-text caveats keyed by topic (e.g. 'religious', 'humanitarian').",
61
+ "additionalProperties": { "type": "string" }
62
+ }
63
+ }
64
+ }
@@ -0,0 +1,314 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "$id": "https://champollion.dev/schemas/external-results.schema.json",
4
+ "title": "Champollion External MT Results Catalogue (cite-only SSOT)",
5
+ "description": "Hand-curated, in-git registry of EXTERNALLY-PUBLISHED machine-translation results that Champollion CITES but has NOT independently reproduced. THREE axes are tracked PER RECORD and must NEVER be conflated: (1) `verified` — every record MUST be a single concrete datapoint (one model, one benchmark, one NAMED metric, one numeric value, an explicit lower_is_better direction, and a source_url) whose value has been checked against the cited source; a record may not be added unverified. (2) `badge` — the constant 'unverified — not reproduced by Champollion' disclaimer: the number is someone else's measurement, run by them on their data, not re-run in the Champollion harness. So `verified:true` means 'we confirmed the citation', NOT 'we reproduced the result'. (3) `signal_strength` — the OBJECTIVE strength of the corpus the number was measured on (size · example length · domain breadth · contamination). PROVENANCE (external/not-reproduced) and SIGNAL STRENGTH (grade) are SEPARATE: an external result is FLAGGED as not-reproduced but is NOT penalized on grade merely for being external; a number we run ourselves on a 12-single-word test pair is a WEAK edge, while OPUS-MT eng→fra on thousands of lengthy multi-domain examples is a STRONG edge. Metric VARIANTS (legacy-BLEU vs spBLEU vs chrF vs chrF++ vs MetricX-24 vs a specific COMET version) are NOT directly comparable; each record names its exact metric. Records may carry a `pair` (source/target ISO codes) so the datapoint surfaces as a flagged external EDGE on the /mesh map, and a `method_ref` into the top-level `methods` availability index (a catalogue of ALL methods seen in results — not just the runnable method-registry). NEVER store corpus content, gold references, or per-sentence outputs here — only the citation, the URL, the metric, and the published value. Consumed at build time by the public Index page (cli/website/src/pages/catalogue.js) and the /mesh external-edge overlay via the shared-data catalogue distiller. Mirrors the method-registry.json / human-services.json SSOT pattern.",
6
+ "type": "object",
7
+ "required": ["version", "results", "manifest"],
8
+ "additionalProperties": false,
9
+ "properties": {
10
+ "_comment": { "type": "string" },
11
+ "version": { "type": "integer", "minimum": 1 },
12
+ "results": {
13
+ "type": "array",
14
+ "description": "Cite-only external result records.",
15
+ "items": { "$ref": "#/$defs/result" }
16
+ },
17
+ "methods": {
18
+ "type": "array",
19
+ "description": "The METHOD-AVAILABILITY INDEX: a catalogue of every translation method/system that appears in `results` (and notable peers), independent of whether Champollion can run it today. Each entry records availability, weight openness, and how the method is accessed, so users can be pointed to a method now and Champollion can implement-for-them later. This is a SUPERSET of shared/method-registry.json — entries flagged `runnable_in_champollion:true` are dispatchable today; the rest are cited-only.",
20
+ "items": { "$ref": "#/$defs/method" }
21
+ },
22
+ "manifest": {
23
+ "type": "object",
24
+ "description": "Field-literacy manifest: which benchmark families Champollion RUNS, which it only CITES, and which it deliberately EXCLUDES (with reasons).",
25
+ "required": ["run", "cite", "exclude"],
26
+ "additionalProperties": false,
27
+ "properties": {
28
+ "run": {
29
+ "type": "array",
30
+ "description": "Benchmark families Champollion runs in its own harness (the runnable registry). Names mirror arena/datasets/registry.json registry_source values; counts are derived live at build, not stored here.",
31
+ "items": { "$ref": "#/$defs/manifestRun" }
32
+ },
33
+ "cite": {
34
+ "type": "array",
35
+ "description": "External result sources Champollion cites but does not reproduce. Each references the cite-only `results` records by source id or describes a source.",
36
+ "items": { "$ref": "#/$defs/manifestCite" }
37
+ },
38
+ "exclude": {
39
+ "type": "array",
40
+ "description": "Benchmark families/datasets deliberately NOT indexed, each with a plain-English reason (field literacy, not omission).",
41
+ "items": { "$ref": "#/$defs/manifestExclude" }
42
+ }
43
+ }
44
+ }
45
+ },
46
+ "$defs": {
47
+ "result": {
48
+ "type": "object",
49
+ "additionalProperties": false,
50
+ "description": "One externally-published, concretely-cited datapoint: a single (model, benchmark, metric, value) tuple with a source URL. NO aggregate/leaderboard pointers belong here — those live in manifest.cite. Every record is verified against its source and carries the not-reproduced-by-Champollion badge.",
51
+ "required": [
52
+ "id",
53
+ "model",
54
+ "source",
55
+ "benchmark",
56
+ "metric",
57
+ "value",
58
+ "lower_is_better",
59
+ "source_url",
60
+ "citation",
61
+ "verified",
62
+ "badge"
63
+ ],
64
+ "properties": {
65
+ "id": {
66
+ "type": "string",
67
+ "pattern": "^[a-z0-9][a-z0-9-]*$",
68
+ "description": "Stable kebab-case id, unique within results (e.g. 'translategemma-27b-metricx')."
69
+ },
70
+ "model": {
71
+ "type": "string",
72
+ "minLength": 1,
73
+ "description": "The exact system/model the number was measured on (e.g. 'TranslateGemma 27B', 'Gemma 3 12B'). This is the thing being scored — NOT the publication."
74
+ },
75
+ "source": {
76
+ "type": "string",
77
+ "minLength": 1,
78
+ "description": "Short name of the publication the number comes from (e.g. 'TranslateGemma Technical Report'). Used for grouping/citation context; distinct from `model`."
79
+ },
80
+ "org": {
81
+ "type": ["string", "null"],
82
+ "description": "Publishing organization or team (e.g. 'Google', 'Meta AI', 'Cohere for AI')."
83
+ },
84
+ "year": {
85
+ "type": ["integer", "null"],
86
+ "description": "Publication year."
87
+ },
88
+ "category": {
89
+ "type": "string",
90
+ "enum": ["paper", "model-report", "leaderboard", "aggregator"],
91
+ "description": "Kind of source. (Verified single-number records are typically 'model-report' or 'paper'.)"
92
+ },
93
+ "benchmark": {
94
+ "type": "string",
95
+ "minLength": 1,
96
+ "description": "The exact evaluation set the number was measured on (e.g. 'WMT24++', 'FLORES-200 devtest'). Names the test suite so absolute scores are read on the right distribution."
97
+ },
98
+ "metric": {
99
+ "type": "string",
100
+ "minLength": 1,
101
+ "description": "The exact NAMED metric and version (e.g. 'MetricX-24', 'COMET-22', 'chrF++', 'spBLEU'). Generic 'BLEU'/'COMET' without a variant is insufficient."
102
+ },
103
+ "value": {
104
+ "type": "number",
105
+ "description": "The published numeric score exactly as reported on `benchmark` with `metric` (e.g. 3.09, 84.4). No rounding/derivation; record the source's stated figure."
106
+ },
107
+ "lower_is_better": {
108
+ "type": "boolean",
109
+ "description": "Direction of the metric: true when a LOWER value is better (e.g. MetricX-24, error counts), false when HIGHER is better (e.g. COMET-22, chrF++, BLEU). Required so the value is never misread."
110
+ },
111
+ "metric_variant_flag": {
112
+ "type": ["string", "null"],
113
+ "description": "Optional cross-comparability caveat: the metric's scale and why it is NOT directly comparable to other variants (legacy-BLEU / spBLEU / chrF / chrF++ / a specific COMET or MetricX version)."
114
+ },
115
+ "headline": {
116
+ "type": ["string", "null"],
117
+ "description": "Optional one-line paraphrase of the source's own headline claim — NOT a Champollion measurement; read with the badge."
118
+ },
119
+ "langs_or_pairs": {
120
+ "type": ["string", "null"],
121
+ "description": "Optional language / pair coverage the number averages over (e.g. 'WMT24++ — 55 language pairs, averaged')."
122
+ },
123
+ "source_url": {
124
+ "type": "string",
125
+ "format": "uri",
126
+ "description": "Canonical link to the source the value was verified against (paper, report, or leaderboard page)."
127
+ },
128
+ "citation": {
129
+ "type": "string",
130
+ "minLength": 1,
131
+ "description": "Bibliographic citation (authors, title, venue/arXiv id)."
132
+ },
133
+ "verified": {
134
+ "const": true,
135
+ "description": "MUST be the boolean true. Asserts the (model, benchmark, metric, value) tuple was checked against the cited source. A record may not be added unverified. NOTE: this is citation-accuracy, NOT independent reproduction — see `badge`."
136
+ },
137
+ "badge": {
138
+ "type": "string",
139
+ "const": "unverified — not reproduced by Champollion",
140
+ "description": "Constant integrity disclaimer: the number is someone else's measurement, not re-run in the Champollion harness. Orthogonal to `verified` (which only asserts the citation is accurate)."
141
+ },
142
+ "notes": {
143
+ "type": ["string", "null"],
144
+ "description": "Optional caveats (e.g. evaluation protocol detail, why the corpus itself is excluded even though the result is cited)."
145
+ },
146
+ "pair": {
147
+ "$ref": "#/$defs/pair",
148
+ "description": "Optional directed language pair (ISO 639-3 source → target). When present, this datapoint is mappable to a flagged EXTERNAL EDGE on the /mesh map. Absent for aggregate/averaged numbers (e.g. an average over 55 WMT24++ directions), which are not single edges."
149
+ },
150
+ "signal_strength": {
151
+ "$ref": "#/$defs/signalStrength",
152
+ "description": "Optional OBJECTIVE signal-strength grade of the corpus this number was measured on. SEPARATE from provenance: an external result is not penalized on grade for being external. Interim rubric pending the canonical mesh-routing rubric (see notes/rationale); the coordinator reconciles."
153
+ },
154
+ "method_ref": {
155
+ "type": ["string", "null"],
156
+ "description": "Optional id into the top-level `methods` availability index, linking this result to the method/system it was measured on."
157
+ }
158
+ }
159
+ },
160
+ "pair": {
161
+ "type": "object",
162
+ "additionalProperties": false,
163
+ "required": ["source", "target"],
164
+ "description": "A directed language pair by ISO 639-3 code (matches arena registry language_pair).",
165
+ "properties": {
166
+ "source": { "type": "string", "pattern": "^[a-z]{3}$", "description": "ISO 639-3 source code." },
167
+ "target": { "type": "string", "pattern": "^[a-z]{3}$", "description": "ISO 639-3 target code." }
168
+ }
169
+ },
170
+ "signalStrength": {
171
+ "type": "object",
172
+ "additionalProperties": false,
173
+ "required": ["grade", "contamination", "rationale"],
174
+ "description": "The OBJECTIVE strength of the measurement's corpus — what makes an edge weak or strong regardless of who ran it. Letter grade is derived from the four factors below; `rationale` states the derivation in plain English so it is auditable and re-gradeable.",
175
+ "properties": {
176
+ "grade": {
177
+ "type": "string",
178
+ "enum": ["A", "B", "C", "D", "F"],
179
+ "description": "Signal-strength letter grade (A strongest … F weakest), reflecting corpus size, example length, domain breadth, and contamination together. NOT a quality score of the model and NOT a provenance penalty."
180
+ },
181
+ "corpus_size": {
182
+ "type": ["integer", "null"],
183
+ "description": "Number of evaluation items the number was measured on (segments / sentence pairs), or null when the source does not state it."
184
+ },
185
+ "example_length": {
186
+ "type": ["string", "null"],
187
+ "enum": ["multi-sentence", "sentence", "short", "word-level", "mixed", "unknown", null],
188
+ "description": "Typical length of the evaluated examples. Longer, fuller examples carry more signal than single words."
189
+ },
190
+ "domain_breadth": {
191
+ "type": ["string", "null"],
192
+ "enum": ["multi-domain", "broad", "single-domain", "narrow", "unknown", null],
193
+ "description": "Breadth of domains the test set covers. Broad multi-domain sets generalize better than a single narrow domain."
194
+ },
195
+ "contamination": {
196
+ "type": "string",
197
+ "enum": ["HIGH", "MEDIUM", "LOW", "UNKNOWN"],
198
+ "description": "Contamination posture of the test set (how broadly it is trained on). HIGH = relative-only / illustration; LOW = absolute-eligible. UNKNOWN fails safe (treated as not absolute-eligible)."
199
+ },
200
+ "rationale": {
201
+ "type": "string",
202
+ "minLength": 1,
203
+ "description": "Plain-English justification for the grade from the four factors (auditable; lets the coordinator re-derive against the canonical rubric)."
204
+ }
205
+ }
206
+ },
207
+ "method": {
208
+ "type": "object",
209
+ "additionalProperties": false,
210
+ "required": ["id", "name", "availability", "weights", "access", "runnable_in_champollion"],
211
+ "description": "One entry in the method-availability index: a translation method/system seen in results, with how available it is and how it is accessed.",
212
+ "properties": {
213
+ "id": {
214
+ "type": "string",
215
+ "pattern": "^[a-z0-9][a-z0-9-]*$",
216
+ "description": "Stable kebab-case id (e.g. 'opus-mt', 'nllb-200'), referenced by results[].method_ref."
217
+ },
218
+ "name": { "type": "string", "minLength": 1, "description": "Display name (e.g. 'OPUS-MT', 'NLLB-200')." },
219
+ "org": { "type": ["string", "null"], "description": "Publishing organization (e.g. 'Helsinki-NLP', 'Meta AI', 'Google')." },
220
+ "paradigm": {
221
+ "type": ["string", "null"],
222
+ "description": "Method paradigm (e.g. 'neural-mt', 'dedicated-mt', 'llm', 'rule-based')."
223
+ },
224
+ "availability": {
225
+ "type": "string",
226
+ "enum": ["open-weights", "open-source", "gated-weights", "api-only", "closed", "research-only", "unknown"],
227
+ "description": "Overall availability posture of the method/system."
228
+ },
229
+ "weights": {
230
+ "type": "string",
231
+ "enum": ["open", "gated", "closed", "none", "unknown"],
232
+ "description": "Are the model weights publicly obtainable? 'open' = freely downloadable; 'gated' = login/acceptance required; 'none' = no published weights (e.g. API-only)."
233
+ },
234
+ "access": {
235
+ "type": "array",
236
+ "minItems": 1,
237
+ "items": {
238
+ "type": "string",
239
+ "enum": ["huggingface", "download", "api", "pip", "github", "cli", "web", "unknown"]
240
+ },
241
+ "description": "How the method is accessed in practice (one or more channels)."
242
+ },
243
+ "license": { "type": ["string", "null"], "description": "Weights/use license (e.g. 'CC-BY-4.0', 'CC-BY-NC-4.0', 'Apache-2.0', 'Proprietary (ToS)')." },
244
+ "commercial_use": {
245
+ "type": ["boolean", "null"],
246
+ "description": "True when commercial use is permitted; false for non-commercial-only (NC); null when unclear."
247
+ },
248
+ "homepage": { "type": ["string", "null"], "format": "uri", "description": "Project homepage / docs." },
249
+ "weights_url": { "type": ["string", "null"], "format": "uri", "description": "Where the weights live (e.g. a Hugging Face org/model page)." },
250
+ "source_url": { "type": ["string", "null"], "format": "uri", "description": "Canonical citation/source (paper, repo, or card) for the availability facts." },
251
+ "runnable_in_champollion": {
252
+ "type": "boolean",
253
+ "description": "True when the method is dispatchable TODAY via shared/method-registry.json; false = cited-only (point users to it now, implement-for-them later)."
254
+ },
255
+ "notes": { "type": ["string", "null"], "description": "Optional caveats (e.g. 'NC license — excluded from commercial lane', 'one model per pair')." }
256
+ }
257
+ },
258
+ "manifestRun": {
259
+ "type": "object",
260
+ "additionalProperties": false,
261
+ "required": ["family", "name", "note"],
262
+ "properties": {
263
+ "family": {
264
+ "type": "string",
265
+ "description": "registry_source value in arena/datasets/registry.json (e.g. 'flores', 'tico19')."
266
+ },
267
+ "name": { "type": "string", "description": "Display name." },
268
+ "license": { "type": ["string", "null"], "description": "Predominant license of the family." },
269
+ "contamination_posture": {
270
+ "type": ["string", "null"],
271
+ "description": "How the family is treated by the scoring lane (e.g. 'HIGH — relative-only', 'LOW — absolute-eligible')."
272
+ },
273
+ "note": { "type": "string", "description": "Plain-English note on what the family is and how it is used." }
274
+ }
275
+ },
276
+ "manifestCite": {
277
+ "type": "object",
278
+ "additionalProperties": false,
279
+ "required": ["source", "note"],
280
+ "properties": {
281
+ "source": { "type": "string", "description": "Source name (matches a results[].source or describes a citing source)." },
282
+ "result_ids": {
283
+ "type": "array",
284
+ "items": { "type": "string" },
285
+ "description": "Optional ids in results[] this entry covers."
286
+ },
287
+ "note": { "type": "string", "description": "Why we cite (not reproduce) this source." }
288
+ }
289
+ },
290
+ "manifestExclude": {
291
+ "type": "object",
292
+ "additionalProperties": false,
293
+ "required": ["name", "reason"],
294
+ "properties": {
295
+ "name": { "type": "string", "description": "Excluded benchmark/dataset family." },
296
+ "reason_code": {
297
+ "type": ["string", "null"],
298
+ "enum": [
299
+ "contaminated-colonial",
300
+ "mixed-license",
301
+ "embedding-not-mt",
302
+ "monolingual-training",
303
+ "speech-out-of-scope",
304
+ "improper-subset",
305
+ "no-redistribute",
306
+ null
307
+ ],
308
+ "description": "Machine-readable exclusion reason."
309
+ },
310
+ "reason": { "type": "string", "description": "Plain-English exclusion reason (field literacy, not omission)." }
311
+ }
312
+ }
313
+ }
314
+ }
@@ -0,0 +1,90 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "$id": "https://champollion.dev/schemas/human-services.schema.json",
4
+ "title": "Champollion Human Translation Services Registry (SSOT)",
5
+ "description": "Public, opt-in registry of human translation providers keyed by language pair — the in-git SSOT for both the CLI and the website (mirrors the model-aliases.json dual-runtime pattern). It is uploaded to public.translation_services (migration 034). PII (the provider `contact` address/endpoint) is NEVER carried here — it is supplied to the uploader out-of-band. See docs/HUMAN_SERVICES_HUB.md.",
6
+ "type": "array",
7
+ "items": { "$ref": "#/$defs/service" },
8
+ "$defs": {
9
+ "service": {
10
+ "type": "object",
11
+ "additionalProperties": false,
12
+ "required": [
13
+ "service_id",
14
+ "display_name",
15
+ "source_lang",
16
+ "target_lang",
17
+ "consent_attested",
18
+ "status"
19
+ ],
20
+ "properties": {
21
+ "service_id": {
22
+ "type": "string",
23
+ "pattern": "^[a-z0-9][a-z0-9-]*$",
24
+ "description": "Stable public id (kebab-case). Globally unique; the upsert key."
25
+ },
26
+ "display_name": {
27
+ "type": "string",
28
+ "minLength": 1,
29
+ "description": "Provider's consented public name. Never name a nation/org as a custodian before confirmation — use 'community key custodians (in confirmation)'."
30
+ },
31
+ "provider_type": {
32
+ "type": "string",
33
+ "enum": ["community_org", "agency", "individual"],
34
+ "description": "Kind of provider."
35
+ },
36
+ "source_lang": {
37
+ "type": "string",
38
+ "pattern": "^[a-z]{3}$",
39
+ "description": "ISO 639-3 source language code."
40
+ },
41
+ "target_lang": {
42
+ "type": "string",
43
+ "pattern": "^[a-z]{3}$",
44
+ "description": "ISO 639-3 target language code."
45
+ },
46
+ "variety": {
47
+ "type": ["string", "null"],
48
+ "description": "Script/dialect qualifier (e.g. 'SRO', 'Latn'). Disambiguates crk Plains Cree from crm Moose Cree etc."
49
+ },
50
+ "dispatch_channel": {
51
+ "type": ["string", "null"],
52
+ "enum": ["email", "webhook", "api", null],
53
+ "description": "How a request reaches the provider (the async channel; the actual address lives in `contact`, which is PII and not stored here)."
54
+ },
55
+ "turnaround_days": {
56
+ "type": ["integer", "null"],
57
+ "minimum": 0,
58
+ "description": "Typical turnaround in days."
59
+ },
60
+ "rate_per_word": {
61
+ "type": ["number", "null"],
62
+ "minimum": 0,
63
+ "description": "Rate per source word (provider currency; settled off-platform)."
64
+ },
65
+ "domains": {
66
+ "type": ["array", "null"],
67
+ "items": { "type": "string" },
68
+ "description": "Subject domains served (e.g. ['legal', 'health'])."
69
+ },
70
+ "sovereignty_flags": {
71
+ "type": ["object", "null"],
72
+ "description": "Data-handling constraints: off-community processing, attribution requirements, OCAP® notes. Free-form JSON."
73
+ },
74
+ "nc_terms": {
75
+ "type": ["string", "null"],
76
+ "description": "Non-commercial / community-set usage terms, if any (cf. EdTeKLA/NC carve-outs)."
77
+ },
78
+ "consent_attested": {
79
+ "type": "boolean",
80
+ "description": "OCAP® consent gate: the provider has explicitly consented to be listed. Public read requires this true."
81
+ },
82
+ "status": {
83
+ "type": "string",
84
+ "enum": ["pending", "approved", "suspended"],
85
+ "description": "Admin moderation state. Public read requires 'approved'. v0 intake is admin-approval out-of-band."
86
+ }
87
+ }
88
+ }
89
+ }
90
+ }