champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://champollion.dev/schemas/licenses.schema.json",
|
|
4
|
+
"title": "Champollion license registry (shared/licenses.json)",
|
|
5
|
+
"description": "GENERATED cross-runtime SSOT (mt-eval-arena/scripts/build-shared-licenses.mjs): per-source license facts for every card-data source. `sources` is a map keyed by source id; each value is a licenseEntry whose `source` field must equal its key. The bundled mini-schema validator does not support additionalProperties-as-schema, so per-entry validation iterates sources against $defs.licenseEntry (see cli/test/shared-ssot-schemas.test.js).",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"required": [
|
|
8
|
+
"_generated",
|
|
9
|
+
"sources"
|
|
10
|
+
],
|
|
11
|
+
"additionalProperties": false,
|
|
12
|
+
"properties": {
|
|
13
|
+
"_generated": {
|
|
14
|
+
"type": "object",
|
|
15
|
+
"required": [
|
|
16
|
+
"generatedBy",
|
|
17
|
+
"generatedAt",
|
|
18
|
+
"counts"
|
|
19
|
+
],
|
|
20
|
+
"properties": {
|
|
21
|
+
"generatedBy": {
|
|
22
|
+
"type": "string",
|
|
23
|
+
"minLength": 1
|
|
24
|
+
},
|
|
25
|
+
"generatedAt": {
|
|
26
|
+
"type": "string",
|
|
27
|
+
"minLength": 1
|
|
28
|
+
},
|
|
29
|
+
"inputs": {
|
|
30
|
+
"type": "array",
|
|
31
|
+
"items": {
|
|
32
|
+
"type": "string"
|
|
33
|
+
}
|
|
34
|
+
},
|
|
35
|
+
"counts": {
|
|
36
|
+
"type": "object",
|
|
37
|
+
"required": [
|
|
38
|
+
"total"
|
|
39
|
+
],
|
|
40
|
+
"properties": {
|
|
41
|
+
"fromTcLicenses": {
|
|
42
|
+
"type": "integer",
|
|
43
|
+
"minimum": 0
|
|
44
|
+
},
|
|
45
|
+
"supplemental": {
|
|
46
|
+
"type": "integer",
|
|
47
|
+
"minimum": 0
|
|
48
|
+
},
|
|
49
|
+
"total": {
|
|
50
|
+
"type": "integer",
|
|
51
|
+
"minimum": 0
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
},
|
|
55
|
+
"note": {
|
|
56
|
+
"type": "string"
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
},
|
|
60
|
+
"sources": {
|
|
61
|
+
"type": "object",
|
|
62
|
+
"description": "Map: source id -> licenseEntry (validated per-entry by the test)."
|
|
63
|
+
}
|
|
64
|
+
},
|
|
65
|
+
"$defs": {
|
|
66
|
+
"licenseEntry": {
|
|
67
|
+
"type": "object",
|
|
68
|
+
"required": [
|
|
69
|
+
"source",
|
|
70
|
+
"license_spdx",
|
|
71
|
+
"allows_redistribution",
|
|
72
|
+
"requires_attribution",
|
|
73
|
+
"requires_sharealike",
|
|
74
|
+
"non_commercial_only"
|
|
75
|
+
],
|
|
76
|
+
"additionalProperties": false,
|
|
77
|
+
"properties": {
|
|
78
|
+
"source": {
|
|
79
|
+
"type": "string",
|
|
80
|
+
"minLength": 1
|
|
81
|
+
},
|
|
82
|
+
"license_spdx": {
|
|
83
|
+
"type": [
|
|
84
|
+
"string",
|
|
85
|
+
"null"
|
|
86
|
+
]
|
|
87
|
+
},
|
|
88
|
+
"license_url": {
|
|
89
|
+
"type": [
|
|
90
|
+
"string",
|
|
91
|
+
"null"
|
|
92
|
+
]
|
|
93
|
+
},
|
|
94
|
+
"attribution": {
|
|
95
|
+
"type": [
|
|
96
|
+
"string",
|
|
97
|
+
"null"
|
|
98
|
+
]
|
|
99
|
+
},
|
|
100
|
+
"allows_redistribution": {
|
|
101
|
+
"enum": [
|
|
102
|
+
0,
|
|
103
|
+
1
|
|
104
|
+
]
|
|
105
|
+
},
|
|
106
|
+
"requires_attribution": {
|
|
107
|
+
"enum": [
|
|
108
|
+
0,
|
|
109
|
+
1
|
|
110
|
+
]
|
|
111
|
+
},
|
|
112
|
+
"requires_sharealike": {
|
|
113
|
+
"enum": [
|
|
114
|
+
0,
|
|
115
|
+
1
|
|
116
|
+
]
|
|
117
|
+
},
|
|
118
|
+
"non_commercial_only": {
|
|
119
|
+
"enum": [
|
|
120
|
+
0,
|
|
121
|
+
1
|
|
122
|
+
]
|
|
123
|
+
},
|
|
124
|
+
"dataset_url": {
|
|
125
|
+
"type": [
|
|
126
|
+
"string",
|
|
127
|
+
"null"
|
|
128
|
+
]
|
|
129
|
+
},
|
|
130
|
+
"dataset_version": {
|
|
131
|
+
"type": [
|
|
132
|
+
"string",
|
|
133
|
+
"null"
|
|
134
|
+
]
|
|
135
|
+
},
|
|
136
|
+
"notes": {
|
|
137
|
+
"type": [
|
|
138
|
+
"string",
|
|
139
|
+
"null"
|
|
140
|
+
]
|
|
141
|
+
},
|
|
142
|
+
"registered_at": {
|
|
143
|
+
"type": [
|
|
144
|
+
"string",
|
|
145
|
+
"null"
|
|
146
|
+
]
|
|
147
|
+
},
|
|
148
|
+
"_pendingClaim": {
|
|
149
|
+
"type": "string",
|
|
150
|
+
"description": "An evidence stream disagrees with the committed license claim; parked for per-source founder adjudication (build-shared-licenses.mjs merge). The committed claim stands until accepted by name."
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
}
|
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://github.com/gamedaysuits/champollion/schemas/method-card.schema.json",
|
|
4
|
+
"title": "Champollion Method Card",
|
|
5
|
+
"description": "Describes a translation method's architecture, components, and provenance. Embedded in run_cards at publish time and displayed on the leaderboard. Independent of the plugin manifest — a method card describes HOW a method works; a plugin manifest describes HOW TO INSTALL it.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
|
|
8
|
+
"required": ["name", "class", "version"],
|
|
9
|
+
|
|
10
|
+
"properties": {
|
|
11
|
+
"name": {
|
|
12
|
+
"type": "string",
|
|
13
|
+
"description": "Human-readable method name (e.g., 'FST-Gated Coached Pipeline v7')."
|
|
14
|
+
},
|
|
15
|
+
"class": {
|
|
16
|
+
"type": "string",
|
|
17
|
+
"enum": [
|
|
18
|
+
"direct-llm",
|
|
19
|
+
"coached-llm",
|
|
20
|
+
"api-translation",
|
|
21
|
+
"rule-based",
|
|
22
|
+
"hybrid-pipeline",
|
|
23
|
+
"human-in-the-loop",
|
|
24
|
+
"ensemble",
|
|
25
|
+
"fine-tuned",
|
|
26
|
+
"dictionary-augmented",
|
|
27
|
+
"community-reviewed"
|
|
28
|
+
],
|
|
29
|
+
"description": "Primary method class. Determines how the method is categorized on the leaderboard.\n\n- direct-llm: Single LLM call with a system prompt\n- coached-llm: LLM with language-specific coaching data injected into context\n- api-translation: Commercial translation API (Google, DeepL, Microsoft, etc.)\n- rule-based: Finite-state transducers, grammar rules, morphological analyzers\n- hybrid-pipeline: Multi-stage combining 2+ method classes (e.g., LLM + FST validation)\n- human-in-the-loop: Human post-editing or validation in the pipeline\n- ensemble: Multiple models/methods with voting or selection\n- fine-tuned: Fine-tuned or adapted model (LoRA, QLoRA, full fine-tune)\n- dictionary-augmented: Translation augmented with bilingual dictionary lookup\n- community-reviewed: Translation produced by any method then reviewed by community speakers"
|
|
30
|
+
},
|
|
31
|
+
"paradigm": {
|
|
32
|
+
"type": "string",
|
|
33
|
+
"enum": [
|
|
34
|
+
"rule-based",
|
|
35
|
+
"statistical",
|
|
36
|
+
"neural-nmt",
|
|
37
|
+
"llm",
|
|
38
|
+
"hybrid",
|
|
39
|
+
"human",
|
|
40
|
+
"unknown"
|
|
41
|
+
],
|
|
42
|
+
"description": "MT paradigm — the algorithmic approach, ORTHOGONAL to `class` and to the dependency class. Makes rule-based vs statistical vs neural vs LLM comparable on the leaderboard regardless of class (e.g. Apertium = rule-based, Google/OPUS-MT = neural-nmt, a raw GPT call = llm, an FST-gated LLM = hybrid). Optional; absent is treated as 'unknown'. Enforced by the harness (config.VALID_PARADIGMS) and recorded as the run_cards.paradigm column on publish."
|
|
43
|
+
},
|
|
44
|
+
"version": {
|
|
45
|
+
"type": "string",
|
|
46
|
+
"pattern": "^\\d+\\.\\d+",
|
|
47
|
+
"description": "Semver or major.minor version of this method (e.g., '7.2', '1.0.0')."
|
|
48
|
+
},
|
|
49
|
+
"description": {
|
|
50
|
+
"type": "string",
|
|
51
|
+
"description": "Human-readable description of the method's approach, key innovations, and design rationale. This is displayed on the leaderboard detail panel."
|
|
52
|
+
},
|
|
53
|
+
"author": {
|
|
54
|
+
"type": "string",
|
|
55
|
+
"description": "Who developed and tested this method. May be a person, team, or organization."
|
|
56
|
+
},
|
|
57
|
+
"contact": {
|
|
58
|
+
"type": "string",
|
|
59
|
+
"format": "email",
|
|
60
|
+
"description": "Contact email for questions about this method."
|
|
61
|
+
},
|
|
62
|
+
"url": {
|
|
63
|
+
"type": "string",
|
|
64
|
+
"format": "uri",
|
|
65
|
+
"description": "URL for the method's documentation, paper, or repository."
|
|
66
|
+
},
|
|
67
|
+
|
|
68
|
+
"target_languages": {
|
|
69
|
+
"type": "array",
|
|
70
|
+
"items": { "type": "string" },
|
|
71
|
+
"description": "CLDR locale codes this method was designed and validated for. A method might generalize beyond these, but these are the languages with explicit testing."
|
|
72
|
+
},
|
|
73
|
+
"language_agnostic": {
|
|
74
|
+
"type": "boolean",
|
|
75
|
+
"default": false,
|
|
76
|
+
"description": "If true, the method makes no language-specific assumptions. Most LLM-based and API methods are language-agnostic. FST-based and coached methods are typically language-specific."
|
|
77
|
+
},
|
|
78
|
+
|
|
79
|
+
"pipeline": {
|
|
80
|
+
"type": "array",
|
|
81
|
+
"items": { "$ref": "#/$defs/pipelineStage" },
|
|
82
|
+
"description": "Ordered list of processing stages. For single-stage methods, this is a one-element array. For hybrid pipelines, each stage describes one component."
|
|
83
|
+
},
|
|
84
|
+
|
|
85
|
+
"model": {
|
|
86
|
+
"$ref": "#/$defs/modelSpec",
|
|
87
|
+
"description": "Primary model used. For ensemble methods, use the 'models' array instead."
|
|
88
|
+
},
|
|
89
|
+
"models": {
|
|
90
|
+
"type": "array",
|
|
91
|
+
"items": { "$ref": "#/$defs/modelSpec" },
|
|
92
|
+
"description": "All models used in ensemble or multi-stage methods."
|
|
93
|
+
},
|
|
94
|
+
|
|
95
|
+
"tools_used": {
|
|
96
|
+
"type": "array",
|
|
97
|
+
"items": { "type": "string" },
|
|
98
|
+
"description": "List of tools, libraries, APIs, or resources used (e.g., 'GiellaLT FST', 'sacrebleu', 'itwêwina dictionary', 'Google Translate API'). Displayed as tags on the leaderboard."
|
|
99
|
+
},
|
|
100
|
+
"tools_enabled": {
|
|
101
|
+
"type": "boolean",
|
|
102
|
+
"default": false,
|
|
103
|
+
"description": "Whether the LLM was given access to tool-calling (function calling) during translation."
|
|
104
|
+
},
|
|
105
|
+
|
|
106
|
+
"coaching": {
|
|
107
|
+
"$ref": "#/$defs/coachingSpec",
|
|
108
|
+
"description": "Coaching data configuration. Only applicable for coached-llm and hybrid-pipeline methods."
|
|
109
|
+
},
|
|
110
|
+
|
|
111
|
+
"fst": {
|
|
112
|
+
"$ref": "#/$defs/fstSpec",
|
|
113
|
+
"description": "Finite-state transducer configuration. Only applicable for rule-based and hybrid-pipeline methods."
|
|
114
|
+
},
|
|
115
|
+
|
|
116
|
+
"post_processing": {
|
|
117
|
+
"type": "array",
|
|
118
|
+
"items": { "$ref": "#/$defs/postProcessingStep" },
|
|
119
|
+
"description": "Post-processing steps applied after translation (e.g., script conversion, morphological repair, formatting normalization)."
|
|
120
|
+
},
|
|
121
|
+
|
|
122
|
+
"quality_control": {
|
|
123
|
+
"$ref": "#/$defs/qualityControlSpec",
|
|
124
|
+
"description": "Quality control mechanisms used in the method."
|
|
125
|
+
},
|
|
126
|
+
|
|
127
|
+
"data_sovereignty": {
|
|
128
|
+
"$ref": "#/$defs/dataSovereigntySpec",
|
|
129
|
+
"description": "Data sovereignty and ethical considerations. Especially important for Indigenous and endangered language methods."
|
|
130
|
+
},
|
|
131
|
+
|
|
132
|
+
"dependency_class": {
|
|
133
|
+
"type": "string",
|
|
134
|
+
"enum": ["S", "O", "A1", "A2", "X"],
|
|
135
|
+
"description": "Effective dependency class — the most restrictive class among the method's declared dependencies (order S < O < A1 < A2 < X).\n\n- S: self-contained — all code/data/weights ship in the method directory under licenses permitting redistribution and community transfer\n- O: open external — externally hosted open-licensed artifacts (incl. copyleft), pinned and mirrored into the submission\n- A1: substitutable LLM inference at runtime, via the sandbox LLM gateway; the model is interchangeable configuration\n- A2: non-substitutable external data/service API (e.g., an API serving unlicensed proprietary content); flagged 'external dependency' on the leaderboard, not prize-eligible until rights holder permissions exist\n- X: bundles content the submitter has no right to include; inadmissible\n\nDrives sandbox runnability and prize eligibility. Canonical definitions: arena Method Interface spec, 'Method Validity and Dependency Classes'."
|
|
136
|
+
},
|
|
137
|
+
"dependencies": {
|
|
138
|
+
"type": "array",
|
|
139
|
+
"items": { "$ref": "#/$defs/dependencySpec" },
|
|
140
|
+
"description": "Dependency manifest: every external artifact or service the method requires at install or runtime, with license and access mode. The harness derives the effective dependency_class from this list; a mismatch with the declared dependency_class is a validation error. An empty array is an affirmative declaration of Class S."
|
|
141
|
+
},
|
|
142
|
+
"open_source": {
|
|
143
|
+
"type": "boolean",
|
|
144
|
+
"description": "Whether the method's code, prompts, and coaching data are publicly available."
|
|
145
|
+
},
|
|
146
|
+
"license": {
|
|
147
|
+
"type": "string",
|
|
148
|
+
"description": "SPDX license identifier or license description (e.g., 'MIT', 'CC-BY-NC-4.0', 'community-controlled')."
|
|
149
|
+
},
|
|
150
|
+
"reproducibility": {
|
|
151
|
+
"type": "string",
|
|
152
|
+
"enum": ["full", "partial", "non-reproducible"],
|
|
153
|
+
"description": "How reproducible this method is.\n\n- full: Deterministic with fixed random seed, same code produces same output\n- partial: Same approach but LLM non-determinism means outputs vary\n- non-reproducible: Requires specific hardware, proprietary data, or human intervention"
|
|
154
|
+
},
|
|
155
|
+
|
|
156
|
+
"limitations": {
|
|
157
|
+
"type": "array",
|
|
158
|
+
"items": { "type": "string" },
|
|
159
|
+
"description": "Known limitations, failure modes, or caveats (e.g., 'Struggles with long-distance morphological agreement', 'No support for mixed-script input')."
|
|
160
|
+
},
|
|
161
|
+
"notes": {
|
|
162
|
+
"type": "string",
|
|
163
|
+
"description": "Free-form notes from the method author."
|
|
164
|
+
}
|
|
165
|
+
},
|
|
166
|
+
|
|
167
|
+
"additionalProperties": false,
|
|
168
|
+
|
|
169
|
+
"$defs": {
|
|
170
|
+
"pipelineStage": {
|
|
171
|
+
"type": "object",
|
|
172
|
+
"required": ["name", "type"],
|
|
173
|
+
"properties": {
|
|
174
|
+
"name": {
|
|
175
|
+
"type": "string",
|
|
176
|
+
"description": "Stage name (e.g., 'initial-translation', 'fst-validation', 'coached-repair')."
|
|
177
|
+
},
|
|
178
|
+
"type": {
|
|
179
|
+
"type": "string",
|
|
180
|
+
"enum": ["llm", "api", "fst", "dictionary", "post-processor", "validator", "human-review", "ensemble-selector"],
|
|
181
|
+
"description": "Component type for this stage."
|
|
182
|
+
},
|
|
183
|
+
"description": {
|
|
184
|
+
"type": "string",
|
|
185
|
+
"description": "What this stage does."
|
|
186
|
+
},
|
|
187
|
+
"model": {
|
|
188
|
+
"type": "string",
|
|
189
|
+
"description": "Model used in this stage (if applicable)."
|
|
190
|
+
},
|
|
191
|
+
"fallback": {
|
|
192
|
+
"type": "string",
|
|
193
|
+
"description": "What happens if this stage fails (e.g., 'pass-through', 'retry-with-different-model', 'reject')."
|
|
194
|
+
}
|
|
195
|
+
},
|
|
196
|
+
"additionalProperties": false
|
|
197
|
+
},
|
|
198
|
+
|
|
199
|
+
"modelSpec": {
|
|
200
|
+
"type": "object",
|
|
201
|
+
"properties": {
|
|
202
|
+
"slug": {
|
|
203
|
+
"type": "string",
|
|
204
|
+
"description": "OpenRouter model slug or equivalent identifier (e.g., 'google/gemini-2.5-flash', 'anthropic/claude-sonnet-4')."
|
|
205
|
+
},
|
|
206
|
+
"provider": {
|
|
207
|
+
"type": "string",
|
|
208
|
+
"description": "Model provider (e.g., 'openrouter', 'openai', 'anthropic', 'google', 'local')."
|
|
209
|
+
},
|
|
210
|
+
"temperature": {
|
|
211
|
+
"type": "number",
|
|
212
|
+
"minimum": 0,
|
|
213
|
+
"maximum": 2,
|
|
214
|
+
"description": "Sampling temperature used."
|
|
215
|
+
},
|
|
216
|
+
"context_window": {
|
|
217
|
+
"type": "integer",
|
|
218
|
+
"description": "Effective context window size used (may be smaller than model max)."
|
|
219
|
+
},
|
|
220
|
+
"fine_tuned": {
|
|
221
|
+
"type": "boolean",
|
|
222
|
+
"default": false,
|
|
223
|
+
"description": "Whether this is a fine-tuned variant."
|
|
224
|
+
},
|
|
225
|
+
"fine_tune_details": {
|
|
226
|
+
"type": "string",
|
|
227
|
+
"description": "Details about fine-tuning: base model, dataset size, technique (LoRA, QLoRA, full)."
|
|
228
|
+
}
|
|
229
|
+
},
|
|
230
|
+
"additionalProperties": false
|
|
231
|
+
},
|
|
232
|
+
|
|
233
|
+
"coachingSpec": {
|
|
234
|
+
"type": "object",
|
|
235
|
+
"properties": {
|
|
236
|
+
"type": {
|
|
237
|
+
"type": "string",
|
|
238
|
+
"enum": ["in-context", "few-shot", "system-prompt", "rag", "fine-tune-data"],
|
|
239
|
+
"description": "How coaching data is injected into the translation process."
|
|
240
|
+
},
|
|
241
|
+
"sources": {
|
|
242
|
+
"type": "array",
|
|
243
|
+
"items": { "type": "string" },
|
|
244
|
+
"description": "Where coaching data comes from (e.g., 'community elders', 'published grammars', 'morphological paradigms')."
|
|
245
|
+
},
|
|
246
|
+
"example_count": {
|
|
247
|
+
"type": "integer",
|
|
248
|
+
"description": "Number of translation examples in the coaching data."
|
|
249
|
+
},
|
|
250
|
+
"includes_grammar": {
|
|
251
|
+
"type": "boolean",
|
|
252
|
+
"description": "Whether coaching data includes explicit grammar rules or paradigm tables."
|
|
253
|
+
},
|
|
254
|
+
"includes_glossary": {
|
|
255
|
+
"type": "boolean",
|
|
256
|
+
"description": "Whether coaching data includes a bilingual glossary."
|
|
257
|
+
},
|
|
258
|
+
"community_approved": {
|
|
259
|
+
"type": "boolean",
|
|
260
|
+
"description": "Whether the coaching data has been reviewed and approved by community language authorities."
|
|
261
|
+
}
|
|
262
|
+
},
|
|
263
|
+
"additionalProperties": false
|
|
264
|
+
},
|
|
265
|
+
|
|
266
|
+
"fstSpec": {
|
|
267
|
+
"type": "object",
|
|
268
|
+
"properties": {
|
|
269
|
+
"analyzer": {
|
|
270
|
+
"type": "string",
|
|
271
|
+
"description": "FST analyzer identifier (e.g., 'giellalt/lang-crk', 'apertium-crk')."
|
|
272
|
+
},
|
|
273
|
+
"version": {
|
|
274
|
+
"type": "string",
|
|
275
|
+
"description": "Version or commit of the FST analyzer."
|
|
276
|
+
},
|
|
277
|
+
"role": {
|
|
278
|
+
"type": "string",
|
|
279
|
+
"enum": ["validator", "generator", "gate", "post-processor"],
|
|
280
|
+
"description": "How the FST is used in the pipeline.\n\n- validator: Checks morphological validity of output (accept/reject)\n- generator: Generates morphologically correct forms\n- gate: Blocks invalid output and triggers retry\n- post-processor: Repairs morphological errors in output"
|
|
281
|
+
},
|
|
282
|
+
"coverage": {
|
|
283
|
+
"type": "string",
|
|
284
|
+
"description": "Estimated lexical coverage of the FST analyzer (e.g., '~85% of everyday vocabulary')."
|
|
285
|
+
}
|
|
286
|
+
},
|
|
287
|
+
"additionalProperties": false
|
|
288
|
+
},
|
|
289
|
+
|
|
290
|
+
"postProcessingStep": {
|
|
291
|
+
"type": "object",
|
|
292
|
+
"required": ["name"],
|
|
293
|
+
"properties": {
|
|
294
|
+
"name": {
|
|
295
|
+
"type": "string",
|
|
296
|
+
"description": "Step name (e.g., 'syllabics-conversion', 'placeholder-restoration', 'whitespace-normalization')."
|
|
297
|
+
},
|
|
298
|
+
"type": {
|
|
299
|
+
"type": "string",
|
|
300
|
+
"enum": ["script-conversion", "morphological-repair", "format-normalization", "variable-restoration", "custom"],
|
|
301
|
+
"description": "Step category."
|
|
302
|
+
},
|
|
303
|
+
"description": {
|
|
304
|
+
"type": "string",
|
|
305
|
+
"description": "What this step does."
|
|
306
|
+
}
|
|
307
|
+
},
|
|
308
|
+
"additionalProperties": false
|
|
309
|
+
},
|
|
310
|
+
|
|
311
|
+
"qualityControlSpec": {
|
|
312
|
+
"type": "object",
|
|
313
|
+
"properties": {
|
|
314
|
+
"retries_on_failure": {
|
|
315
|
+
"type": "boolean",
|
|
316
|
+
"description": "Whether the method retries failed translations with different parameters."
|
|
317
|
+
},
|
|
318
|
+
"max_retries": {
|
|
319
|
+
"type": "integer",
|
|
320
|
+
"description": "Maximum number of retries per entry."
|
|
321
|
+
},
|
|
322
|
+
"validation_layers": {
|
|
323
|
+
"type": "array",
|
|
324
|
+
"items": { "type": "string" },
|
|
325
|
+
"description": "Validation checks applied (e.g., 'script-check', 'length-ratio', 'placeholder-integrity', 'fst-morphology', 'source-echo-detection')."
|
|
326
|
+
},
|
|
327
|
+
"fallback_strategy": {
|
|
328
|
+
"type": "string",
|
|
329
|
+
"description": "What happens when all retries fail (e.g., 'return-best-attempt', 'return-empty', 'flag-for-review')."
|
|
330
|
+
}
|
|
331
|
+
},
|
|
332
|
+
"additionalProperties": false
|
|
333
|
+
},
|
|
334
|
+
|
|
335
|
+
"dependencySpec": {
|
|
336
|
+
"type": "object",
|
|
337
|
+
"required": ["id", "kind", "license", "access", "redistributable", "transferable"],
|
|
338
|
+
"properties": {
|
|
339
|
+
"id": {
|
|
340
|
+
"type": "string",
|
|
341
|
+
"description": "Stable identifier for the dependency (e.g., 'giellalt-lang-crk-fst', 'llm-inference')."
|
|
342
|
+
},
|
|
343
|
+
"kind": {
|
|
344
|
+
"type": "string",
|
|
345
|
+
"enum": ["data", "model", "software", "service"],
|
|
346
|
+
"description": "What the dependency is: a dataset/dictionary (data), model weights or hosted inference (model), a library/tool (software), or a remote API (service)."
|
|
347
|
+
},
|
|
348
|
+
"license": {
|
|
349
|
+
"type": "string",
|
|
350
|
+
"description": "SPDX license identifier, 'proprietary', or 'none'. 'none' means no public license exists — treated as all-rights-reserved."
|
|
351
|
+
},
|
|
352
|
+
"access": {
|
|
353
|
+
"type": "string",
|
|
354
|
+
"enum": ["bundled", "mirrored", "gateway", "external-api"],
|
|
355
|
+
"description": "How the method obtains the dependency.\n\n- bundled: ships inside the method directory\n- mirrored: fetched at install time from an external host, pinned, and vendored into the submission\n- gateway: runtime LLM inference routed through the sandbox evaluation gateway\n- external-api: any other runtime network call"
|
|
356
|
+
},
|
|
357
|
+
"source": {
|
|
358
|
+
"type": "string",
|
|
359
|
+
"description": "Canonical URL or 'provider:slug' identifier for the artifact or endpoint."
|
|
360
|
+
},
|
|
361
|
+
"pin": {
|
|
362
|
+
"type": "string",
|
|
363
|
+
"description": "Version, commit, or content hash pinning the exact artifact. Required for access mode 'mirrored'."
|
|
364
|
+
},
|
|
365
|
+
"substitutable": {
|
|
366
|
+
"type": "boolean",
|
|
367
|
+
"description": "For 'gateway'/'external-api' access: whether any compatible endpoint can serve this dependency (true distinguishes Class A1 from A2). A method depending on provider-side state (provider-hosted fine-tunes, file stores) is not substitutable."
|
|
368
|
+
},
|
|
369
|
+
"redistributable": {
|
|
370
|
+
"type": "boolean",
|
|
371
|
+
"description": "Whether the license permits redistributing the artifact (e.g., mirroring it into a sandbox submission)."
|
|
372
|
+
},
|
|
373
|
+
"transferable": {
|
|
374
|
+
"type": "boolean",
|
|
375
|
+
"description": "Whether the artifact, or rights sufficient for community use/modification/redeployment, can convey to a language community's governance organization under prize transfer terms."
|
|
376
|
+
},
|
|
377
|
+
"notes": {
|
|
378
|
+
"type": "string",
|
|
379
|
+
"description": "Free-form context (e.g., rights-holder status, pending permission asks)."
|
|
380
|
+
}
|
|
381
|
+
},
|
|
382
|
+
"additionalProperties": false
|
|
383
|
+
},
|
|
384
|
+
|
|
385
|
+
"dataSovereigntySpec": {
|
|
386
|
+
"type": "object",
|
|
387
|
+
"properties": {
|
|
388
|
+
"principles": {
|
|
389
|
+
"type": "array",
|
|
390
|
+
"items": {
|
|
391
|
+
"type": "string",
|
|
392
|
+
"enum": ["OCAP", "CARE", "maori-data-sovereignty", "UNDRIP", "other"]
|
|
393
|
+
},
|
|
394
|
+
"description": "Data sovereignty frameworks this method respects."
|
|
395
|
+
},
|
|
396
|
+
"community_consent": {
|
|
397
|
+
"type": "boolean",
|
|
398
|
+
"description": "Whether the relevant language community has consented to this method's use of their language data."
|
|
399
|
+
},
|
|
400
|
+
"data_stays_local": {
|
|
401
|
+
"type": "boolean",
|
|
402
|
+
"description": "Whether all language-specific data remains on-premise (not sent to cloud APIs)."
|
|
403
|
+
},
|
|
404
|
+
"notes": {
|
|
405
|
+
"type": "string",
|
|
406
|
+
"description": "Additional sovereignty context."
|
|
407
|
+
}
|
|
408
|
+
},
|
|
409
|
+
"additionalProperties": false
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"$id": "https://champollion.dev/schemas/method-registry.schema.json",
|
|
4
|
+
"title": "Champollion Method Registry (SSOT)",
|
|
5
|
+
"description": "Declarative registry of translation methods (MT engines) and LLM providers, shared by the Python arena harness and the JS CLI. See shared/method-registry.json.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"required": ["version", "entries"],
|
|
8
|
+
"additionalProperties": true,
|
|
9
|
+
"properties": {
|
|
10
|
+
"_comment": { "type": "string" },
|
|
11
|
+
"version": { "type": "integer", "minimum": 1 },
|
|
12
|
+
"entries": {
|
|
13
|
+
"type": "object",
|
|
14
|
+
"minProperties": 1,
|
|
15
|
+
"additionalProperties": { "$ref": "#/$defs/entry" },
|
|
16
|
+
"propertyNames": {
|
|
17
|
+
"pattern": "^[a-z0-9][a-z0-9-]*$",
|
|
18
|
+
"description": "Canonical method/provider name (kebab-case). Must match the registry key in both runtimes."
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
},
|
|
22
|
+
"$defs": {
|
|
23
|
+
"entry": {
|
|
24
|
+
"type": "object",
|
|
25
|
+
"required": ["kind", "method_class", "paradigm"],
|
|
26
|
+
"additionalProperties": false,
|
|
27
|
+
"properties": {
|
|
28
|
+
"kind": {
|
|
29
|
+
"type": "string",
|
|
30
|
+
"enum": ["mt-api", "llm-provider", "local-model", "plugin"],
|
|
31
|
+
"description": "How the runtime dispatches it: mt-api (HttpMTMethod subclass), llm-provider (LLMProvider), local-model (local inference, optional deps), plugin (method_loader directory)."
|
|
32
|
+
},
|
|
33
|
+
"cli_name": {
|
|
34
|
+
"type": "string",
|
|
35
|
+
"description": "The CLI's METHOD_REGISTRY key when it differs from the canonical name (e.g. the 'openrouter' provider is 'llm' in the CLI). Absent means the CLI uses the same name."
|
|
36
|
+
},
|
|
37
|
+
"method_class": {
|
|
38
|
+
"type": "string",
|
|
39
|
+
"enum": ["raw-llm", "coached-llm", "pipeline", "custom-plugin", "api", "human"],
|
|
40
|
+
"description": "Must match config.VALID_METHOD_CLASSES (the harness enum that governs DB columns)."
|
|
41
|
+
},
|
|
42
|
+
"paradigm": {
|
|
43
|
+
"type": "string",
|
|
44
|
+
"enum": ["rule-based", "statistical", "neural-nmt", "llm", "hybrid", "human", "unknown"],
|
|
45
|
+
"description": "Must match config.VALID_PARADIGMS."
|
|
46
|
+
},
|
|
47
|
+
"env": {
|
|
48
|
+
"type": "array",
|
|
49
|
+
"items": { "type": "string" },
|
|
50
|
+
"description": "Environment variable name(s) the adapter reads, in resolution order. Empty for keyless engines."
|
|
51
|
+
},
|
|
52
|
+
"credential_env": {
|
|
53
|
+
"type": "array",
|
|
54
|
+
"items": { "type": "string" },
|
|
55
|
+
"minItems": 1,
|
|
56
|
+
"description": "The subset of `env` that are CREDENTIALS. Availability/readiness surfaces (recommend, doctor) judge these vars only — non-credential config vars (regions, endpoints) stay in `env` for the adapter but must never make a method read 'ready' (e.g. AWS_REGION alone is not AWS auth). Absent = every `env` var is a credential. Any one present suffices unless credential_env_all."
|
|
57
|
+
},
|
|
58
|
+
"credential_env_all": {
|
|
59
|
+
"type": "boolean",
|
|
60
|
+
"description": "When true, ALL credential_env vars must be set for 'ready' — key-pair auth (e.g. AWS access-key id + secret; Lara id + secret). Default false = any one suffices (alias lists). Only meaningful alongside credential_env."
|
|
61
|
+
},
|
|
62
|
+
"default_base_url": { "type": "string" },
|
|
63
|
+
"homepage": { "type": "string" },
|
|
64
|
+
"license": { "type": "string" },
|
|
65
|
+
"commercialReady": { "type": "boolean" },
|
|
66
|
+
"cost_note": { "type": "string" },
|
|
67
|
+
"max_batch": { "type": "integer", "minimum": 1 },
|
|
68
|
+
"locale_map": {
|
|
69
|
+
"type": "object",
|
|
70
|
+
"additionalProperties": { "type": "string" },
|
|
71
|
+
"description": "ISO/BCP-47 code -> provider-specific code (e.g. Google he->iw)."
|
|
72
|
+
},
|
|
73
|
+
"optional_extra": {
|
|
74
|
+
"type": "string",
|
|
75
|
+
"description": "For local-model/aws engines: the pip extra that must be installed (e.g. 'local-models', 'aws')."
|
|
76
|
+
},
|
|
77
|
+
"runtimes": {
|
|
78
|
+
"type": "array",
|
|
79
|
+
"items": { "type": "string", "enum": ["harness", "cli"] },
|
|
80
|
+
"description": "Which runtimes implement this entry. Absent = both. Lets an engine land in one runtime first (e.g. Amazon is harness-only until the CLI SigV4 adapter exists). Parity tests only require an entry in a runtime it lists."
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|