champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"$id": "https://champollion.dev/schemas/metric-registry.schema.json",
|
|
4
|
+
"title": "Champollion Metric Identity Registry (SSOT)",
|
|
5
|
+
"description": "Declarative registry of evaluation-metric identity — the one place that maps each canonical metric id to its Python plugin name, language-card evalMetrics key, and denormalized run_cards DB column. Shared by the Python arena harness and the JS CLI/website. See shared/metric-registry.json and cli/website/docs/network/specifications/scoring.md §2.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"required": ["version", "entries"],
|
|
8
|
+
"additionalProperties": true,
|
|
9
|
+
"properties": {
|
|
10
|
+
"_comment": { "type": "string" },
|
|
11
|
+
"version": { "type": "integer", "minimum": 1 },
|
|
12
|
+
"entries": {
|
|
13
|
+
"type": "object",
|
|
14
|
+
"minProperties": 1,
|
|
15
|
+
"additionalProperties": { "$ref": "#/$defs/entry" },
|
|
16
|
+
"propertyNames": {
|
|
17
|
+
"pattern": "^[a-z0-9][a-z0-9_]*$",
|
|
18
|
+
"description": "canonical_id: the run-card scores key (snake_case). Also the primary lookup key across all runtimes."
|
|
19
|
+
}
|
|
20
|
+
},
|
|
21
|
+
"_proposed_renames": {
|
|
22
|
+
"type": "array",
|
|
23
|
+
"description": "Rename proposals the FOUNDER decides (published run cards embed these names). Advisory — not consumed by code.",
|
|
24
|
+
"items": {
|
|
25
|
+
"type": "object",
|
|
26
|
+
"required": ["current", "proposal", "impact", "decision"],
|
|
27
|
+
"additionalProperties": false,
|
|
28
|
+
"properties": {
|
|
29
|
+
"current": { "type": "string" },
|
|
30
|
+
"proposal": { "type": "string" },
|
|
31
|
+
"impact": { "type": "string" },
|
|
32
|
+
"decision": { "type": "string", "enum": ["founder-pending", "accepted", "rejected"] }
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
},
|
|
37
|
+
"$defs": {
|
|
38
|
+
"entry": {
|
|
39
|
+
"type": "object",
|
|
40
|
+
"required": [
|
|
41
|
+
"category", "status", "display_name", "plugin_name", "card_key",
|
|
42
|
+
"db_column", "scale", "direction", "level", "in_composite",
|
|
43
|
+
"verifier_reproducible"
|
|
44
|
+
],
|
|
45
|
+
"additionalProperties": false,
|
|
46
|
+
"properties": {
|
|
47
|
+
"category": {
|
|
48
|
+
"type": "string",
|
|
49
|
+
"enum": ["surface", "structural", "semantic", "neural", "behavioral", "compliance", "comparator", "composite", "efficiency"],
|
|
50
|
+
"description": "The scoring-spec §2 category (plus composite/efficiency for §4/§6/§7 derived figures)."
|
|
51
|
+
},
|
|
52
|
+
"status": {
|
|
53
|
+
"type": "string",
|
|
54
|
+
"enum": ["implemented", "partial", "planned", "proposed"],
|
|
55
|
+
"description": "Scoring spec §3 status tier: implemented (values in run cards today), partial (language-specific proxy only), planned (specified, not built), proposed (under discussion)."
|
|
56
|
+
},
|
|
57
|
+
"display_name": { "type": "string", "minLength": 1 },
|
|
58
|
+
"plugin_name": {
|
|
59
|
+
"type": ["string", "null"],
|
|
60
|
+
"description": "The Python MetricPlugin `name` attribute that computes this metric, or null for harness-core metrics with no plugin (e.g. exact_match_rate, chrf_plus_plus) and derived figures."
|
|
61
|
+
},
|
|
62
|
+
"card_key": {
|
|
63
|
+
"type": ["string", "null"],
|
|
64
|
+
"description": "The language-card evalMetrics key that declares the plugin (e.g. 'lyss-eq'), or null when the metric is not card-declared (harness-core or always-loaded behavioral plugins)."
|
|
65
|
+
},
|
|
66
|
+
"db_column": {
|
|
67
|
+
"type": ["string", "null"],
|
|
68
|
+
"description": "The denormalized run_cards column (arena/DATABASE_SCHEMA.md), or null when the metric lives only in the run_card JSONB / plugin_metrics."
|
|
69
|
+
},
|
|
70
|
+
"scale": {
|
|
71
|
+
"type": "string",
|
|
72
|
+
"description": "Human-readable value range (e.g. '0.0-1.0', '0-100', '0-25', '0-inf', 'USD', or a label enumeration)."
|
|
73
|
+
},
|
|
74
|
+
"direction": {
|
|
75
|
+
"type": "string",
|
|
76
|
+
"enum": ["higher", "lower", "neutral"],
|
|
77
|
+
"description": "Whether higher or lower is better (neutral = diagnostic/informational, e.g. length_ratio, coverage, quality_tier label)."
|
|
78
|
+
},
|
|
79
|
+
"level": {
|
|
80
|
+
"type": "string",
|
|
81
|
+
"enum": ["entry", "corpus", "both"],
|
|
82
|
+
"description": "Whether the metric is computed per-entry, corpus-level, or both."
|
|
83
|
+
},
|
|
84
|
+
"in_composite": {
|
|
85
|
+
"type": "boolean",
|
|
86
|
+
"description": "True iff the metric currently carries weight in at least one active profile in scoring.PROFILE_REGISTRY when available (declared-but-INACTIVE metrics like orthographic_accuracy are false)."
|
|
87
|
+
},
|
|
88
|
+
"verifier_reproducible": {
|
|
89
|
+
"type": "boolean",
|
|
90
|
+
"description": "True iff the verifier can deterministically re-derive it from the sha-pinned corpus + stored entries (+ card-pinned FST / pinned neural model under the fail-closed contract). Speed/cost/style metrics are false."
|
|
91
|
+
},
|
|
92
|
+
"notes": { "type": "string" }
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
}
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"$id": "https://champollion.dev/schemas/metric-reliability.schema.json",
|
|
4
|
+
"title": "Metric Reliability Index (WMT meta-evaluation, machine-derived)",
|
|
5
|
+
"description": "Correlations between automatic MT metrics and WMT Metrics-task human judgments (DA/MQM/ESA, wmt19-wmt25), per (test set, language pair, human lane, level, metric), rolled up per TARGET-language family (master plan B3). Every value is champollion-derived from pinned upstream data (google-research/mt-metrics-eval + google/wmt-mqm-human-evaluation); no upstream text, judgment or score is redistributed. RELIABILITY evidence, not quality scores. Non-commercial hold until the founder clears the upstream data license (license_lane). Regenerated by arena/scripts/import_wmt_metaeval.py; tracked in the monorepo, NOT bundled into the npm package or PyPI wheel.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"required": ["version", "provider", "sources", "provenance", "license_lane", "score_orientation", "correlation_definitions", "human_lanes", "metrics", "languages", "families", "cells", "excluded", "counts"],
|
|
8
|
+
"additionalProperties": false,
|
|
9
|
+
"properties": {
|
|
10
|
+
"_comment": { "type": "string" },
|
|
11
|
+
"version": { "type": "integer", "minimum": 1 },
|
|
12
|
+
"provider": { "enum": ["wmt-metrics-metaeval"] },
|
|
13
|
+
"sources": {
|
|
14
|
+
"type": "array",
|
|
15
|
+
"minItems": 2,
|
|
16
|
+
"items": {
|
|
17
|
+
"type": "object",
|
|
18
|
+
"required": ["name", "repo", "commit", "retrieved", "license", "credit"],
|
|
19
|
+
"additionalProperties": true,
|
|
20
|
+
"properties": {
|
|
21
|
+
"name": { "type": "string" },
|
|
22
|
+
"repo": { "type": "string", "format": "uri" },
|
|
23
|
+
"commit": { "type": "string", "pattern": "^[0-9a-f]{40}$", "description": "Upstream commit SHA pinned at probe time." },
|
|
24
|
+
"data_url": { "type": "string", "format": "uri" },
|
|
25
|
+
"data_sha256": { "type": ["string", "null"], "pattern": "^[0-9a-f]{64}$", "description": "sha256 of the downloaded data tarball — REQUIRED for the mt-metrics-eval source because the tarball is not immutable." },
|
|
26
|
+
"data_etag": { "type": ["string", "null"] },
|
|
27
|
+
"data_last_modified": { "type": ["string", "null"] },
|
|
28
|
+
"data_bytes": { "type": ["integer", "null"] },
|
|
29
|
+
"retrieved": { "type": "string", "pattern": "^\\d{4}-\\d{2}-\\d{2}$" },
|
|
30
|
+
"license": { "type": "string", "minLength": 5 },
|
|
31
|
+
"credit": { "type": "string", "minLength": 10, "description": "Human credit for the upstream project and annotators (always-cite-work principle)." },
|
|
32
|
+
"note": { "type": "string" }
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
},
|
|
36
|
+
"provenance": { "type": "string", "pattern": "champollion-derived", "description": "Must carry champollion-derived naming the upstreams: the correlations are our computation; the judgments and scores are theirs." },
|
|
37
|
+
"license_lane": {
|
|
38
|
+
"type": "object",
|
|
39
|
+
"required": ["commercial_ok", "note"],
|
|
40
|
+
"properties": {
|
|
41
|
+
"commercial_ok": { "enum": [false], "description": "Pinned false until the founder license review clears the upstream data — the bundled human-judgment/test-set text has no explicit upstream license statement." },
|
|
42
|
+
"note": { "type": "string", "minLength": 20 }
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
"score_orientation": { "type": "string", "minLength": 20 },
|
|
46
|
+
"correlation_definitions": {
|
|
47
|
+
"type": "object",
|
|
48
|
+
"required": ["sys", "seg"],
|
|
49
|
+
"properties": {
|
|
50
|
+
"sys": { "type": "string", "minLength": 20 },
|
|
51
|
+
"seg": { "type": "string", "minLength": 20 }
|
|
52
|
+
}
|
|
53
|
+
},
|
|
54
|
+
"human_lanes": {
|
|
55
|
+
"type": "object",
|
|
56
|
+
"required": ["preference", "note"],
|
|
57
|
+
"properties": {
|
|
58
|
+
"preference": { "type": "array", "minItems": 1, "items": { "type": "string" } },
|
|
59
|
+
"note": { "type": "string" }
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
"metrics": {
|
|
63
|
+
"type": "object",
|
|
64
|
+
"description": "Keyed by shared/metric-registry.json canonical id. Every cells[].metric MUST appear here (cross-checked by cli/test/metric-reliability.test.js).",
|
|
65
|
+
"additionalProperties": {
|
|
66
|
+
"type": "object",
|
|
67
|
+
"required": ["registry_id", "upstream_candidates", "upstream_names_used", "note"],
|
|
68
|
+
"properties": {
|
|
69
|
+
"registry_id": { "type": "string" },
|
|
70
|
+
"upstream_candidates": { "type": "array", "items": { "type": "string" } },
|
|
71
|
+
"upstream_names_used": { "type": "array", "items": { "type": "string" } },
|
|
72
|
+
"note": { "type": "string" }
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
},
|
|
76
|
+
"languages": {
|
|
77
|
+
"type": "object",
|
|
78
|
+
"description": "Every src/tgt code used in cells -> iso639-3 + family/genus from cli/data/champollion.db.",
|
|
79
|
+
"additionalProperties": {
|
|
80
|
+
"type": "object",
|
|
81
|
+
"required": ["iso639_3", "family"],
|
|
82
|
+
"properties": {
|
|
83
|
+
"iso639_3": { "type": "string", "minLength": 2 },
|
|
84
|
+
"family": { "type": ["string", "null"] },
|
|
85
|
+
"genus": { "type": ["string", "null"] }
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
},
|
|
89
|
+
"family_rollup_note": { "type": "string" },
|
|
90
|
+
"families": {
|
|
91
|
+
"type": "object",
|
|
92
|
+
"description": "TARGET-language family -> per-metric per-level weighted-mean correlations over preferred cells only.",
|
|
93
|
+
"additionalProperties": {
|
|
94
|
+
"type": "object",
|
|
95
|
+
"required": ["n_pairs", "metrics"],
|
|
96
|
+
"properties": {
|
|
97
|
+
"n_pairs": { "type": "integer", "minimum": 1 },
|
|
98
|
+
"metrics": {
|
|
99
|
+
"type": "object",
|
|
100
|
+
"additionalProperties": {
|
|
101
|
+
"type": "object",
|
|
102
|
+
"additionalProperties": {
|
|
103
|
+
"type": "object",
|
|
104
|
+
"required": ["n_cells", "n_pairs", "weight", "pairs"],
|
|
105
|
+
"properties": {
|
|
106
|
+
"n_cells": { "type": "integer", "minimum": 1 },
|
|
107
|
+
"n_pairs": { "type": "integer", "minimum": 1 },
|
|
108
|
+
"weight": { "type": "integer", "minimum": 1 },
|
|
109
|
+
"pairs": { "type": "array", "items": { "type": "string" } },
|
|
110
|
+
"pearson_weighted_mean": { "type": "number", "minimum": -1, "maximum": 1 },
|
|
111
|
+
"spearman_weighted_mean": { "type": "number", "minimum": -1, "maximum": 1 },
|
|
112
|
+
"kendall_tau_b_weighted_mean": { "type": "number", "minimum": -1, "maximum": 1 },
|
|
113
|
+
"pairwise_accuracy_weighted_mean": { "type": "number", "minimum": 0, "maximum": 1 }
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
},
|
|
121
|
+
"cells": {
|
|
122
|
+
"type": "array",
|
|
123
|
+
"items": {
|
|
124
|
+
"type": "object",
|
|
125
|
+
"required": ["testset", "pair", "pair_verbatim", "src", "tgt", "human", "level", "preferred", "metric", "upstream_metric", "upstream_ref", "computed", "n_sys"],
|
|
126
|
+
"additionalProperties": false,
|
|
127
|
+
"properties": {
|
|
128
|
+
"testset": { "type": "string" },
|
|
129
|
+
"pair": { "type": "string", "pattern": "^[a-z]{2,3}-[a-z]{2,3}$" },
|
|
130
|
+
"pair_verbatim": { "type": "string" },
|
|
131
|
+
"src": { "type": "string" },
|
|
132
|
+
"tgt": { "type": "string" },
|
|
133
|
+
"human": { "type": "string" },
|
|
134
|
+
"level": { "enum": ["sys", "seg"] },
|
|
135
|
+
"preferred": { "type": "boolean" },
|
|
136
|
+
"metric": { "type": "string" },
|
|
137
|
+
"upstream_metric": { "type": "string" },
|
|
138
|
+
"upstream_ref": { "type": "string" },
|
|
139
|
+
"computed": { "type": "boolean", "description": "true = Champollion computed the metric scores (sacrebleu chrF++ lane); false = scores verbatim from the WMT submission/baseline files." },
|
|
140
|
+
"n_sys": { "type": "integer", "minimum": 2 },
|
|
141
|
+
"n_points": { "type": "integer", "minimum": 10 },
|
|
142
|
+
"pearson": { "type": ["number", "null"], "minimum": -1, "maximum": 1 },
|
|
143
|
+
"spearman": { "type": ["number", "null"], "minimum": -1, "maximum": 1 },
|
|
144
|
+
"kendall_tau_b": { "type": ["number", "null"], "minimum": -1, "maximum": 1 },
|
|
145
|
+
"pairwise_accuracy": { "type": ["number", "null"], "minimum": 0, "maximum": 1 }
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
},
|
|
149
|
+
"excluded": {
|
|
150
|
+
"type": "array",
|
|
151
|
+
"description": "Fail-honest ledger: everything skipped, with a reason (label-honesty rule A4).",
|
|
152
|
+
"items": {
|
|
153
|
+
"type": "object",
|
|
154
|
+
"required": ["reason"],
|
|
155
|
+
"properties": {
|
|
156
|
+
"testset": { "type": "string" },
|
|
157
|
+
"pair": { "type": "string" },
|
|
158
|
+
"human": { "type": "string" },
|
|
159
|
+
"level": { "type": "string" },
|
|
160
|
+
"metric": { "type": "string" },
|
|
161
|
+
"reason": { "type": "string", "minLength": 10 }
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
},
|
|
165
|
+
"counts": {
|
|
166
|
+
"type": "object",
|
|
167
|
+
"required": ["cells", "preferred_cells", "testsets", "pairs", "families", "excluded"],
|
|
168
|
+
"properties": {
|
|
169
|
+
"cells": { "type": "integer", "minimum": 0 },
|
|
170
|
+
"preferred_cells": { "type": "integer", "minimum": 0 },
|
|
171
|
+
"testsets": { "type": "integer", "minimum": 0 },
|
|
172
|
+
"pairs": { "type": "integer", "minimum": 0 },
|
|
173
|
+
"families": { "type": "integer", "minimum": 0 },
|
|
174
|
+
"excluded": { "type": "integer", "minimum": 0 }
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://champollion.dev/schemas/model-aliases.schema.json",
|
|
4
|
+
"title": "Champollion model aliases (shared/model-aliases.json)",
|
|
5
|
+
"description": "Cross-runtime SSOT mapping short convenience aliases (e.g. 'gemini-flash') to full provider-qualified model ids (e.g. 'google/gemini-3.5-flash'). Keys other than the underscore-prefixed metadata fields are alias names; their values are model ids. The bundled mini-schema validator does not support additionalProperties-as-schema, so per-entry validation iterates the entries against $defs.aliasName / $defs.modelId (see cli/test/shared-ssot-schemas.test.js).",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"properties": {
|
|
8
|
+
"_comment": {
|
|
9
|
+
"type": "string",
|
|
10
|
+
"description": "Human note; not an alias."
|
|
11
|
+
}
|
|
12
|
+
},
|
|
13
|
+
"$defs": {
|
|
14
|
+
"aliasName": {
|
|
15
|
+
"type": "string",
|
|
16
|
+
"description": "Short alias key: lowercase alphanumerics and hyphens.",
|
|
17
|
+
"pattern": "^[a-z0-9][a-z0-9-]*$",
|
|
18
|
+
"minLength": 2
|
|
19
|
+
},
|
|
20
|
+
"modelId": {
|
|
21
|
+
"type": "string",
|
|
22
|
+
"description": "Full OpenRouter-style model id: <provider>/<model>.",
|
|
23
|
+
"pattern": "^[a-z0-9-]+/[A-Za-z0-9._-]+$",
|
|
24
|
+
"minLength": 3
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"$id": "https://champollion.dev/schemas/source-snapshot.schema.json",
|
|
4
|
+
"title": "Source snapshot",
|
|
5
|
+
"description": "The provenance record a fetcher writes into cli/data/<source>/SNAPSHOT.json. It is what makes 'regenerate the cards from source' a checkable claim rather than an aspiration: it names the upstream, pins ONE immutable release, and checksums the bytes taken from it. A source with no snapshot cannot be rebuilt, and facts derived from it carry no release.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"required": ["source", "upstream", "files", "verified", "fetchedBy", "fetchedAt"],
|
|
8
|
+
"additionalProperties": false,
|
|
9
|
+
"properties": {
|
|
10
|
+
"$schema": { "type": "string" },
|
|
11
|
+
"source": {
|
|
12
|
+
"type": "string",
|
|
13
|
+
"description": "The source name used in facts.source. Must match exactly, or the pin attaches to nothing.",
|
|
14
|
+
"minLength": 1
|
|
15
|
+
},
|
|
16
|
+
"upstream": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"description": "The URL the fetcher read. A landing page for a versioned deposit is fine; an undated 'latest' file URL is not a pin and must be reflected by pin:null."
|
|
19
|
+
},
|
|
20
|
+
"license": {
|
|
21
|
+
"type": ["string", "null"],
|
|
22
|
+
"description": "The licence the UPSTREAM declares, verbatim (SPDX id where the upstream gives one). Never our reading of it — that lives in the licence register, which this evidences."
|
|
23
|
+
},
|
|
24
|
+
"licenseUrl": { "type": ["string", "null"] },
|
|
25
|
+
"citation": {
|
|
26
|
+
"type": ["string", "null"],
|
|
27
|
+
"description": "How the upstream asks to be cited."
|
|
28
|
+
},
|
|
29
|
+
"pin": {
|
|
30
|
+
"type": ["object", "null"],
|
|
31
|
+
"description": "The immutable release identifier. NULL means this source could not be pinned — which is a real state and must not be faked. Downstream, no pin means no source_release row, which means every fact from it is flagged not-yet-regenerable.",
|
|
32
|
+
"required": ["kind", "value"],
|
|
33
|
+
"additionalProperties": false,
|
|
34
|
+
"properties": {
|
|
35
|
+
"kind": {
|
|
36
|
+
"enum": ["doi", "commit", "release", "version"],
|
|
37
|
+
"description": "doi = a versioned DOI (strongest). commit = a git SHA. release = a publisher's dated/named release. version = a declared version string with nothing stronger behind it."
|
|
38
|
+
},
|
|
39
|
+
"value": { "type": "string", "minLength": 1 },
|
|
40
|
+
"doi": { "type": ["string", "null"] },
|
|
41
|
+
"date": {
|
|
42
|
+
"type": ["string", "null"],
|
|
43
|
+
"description": "When the PUBLISHER released it. Not when we fetched it — that is fetchedAt, and conflating the two misdates our own provenance."
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
},
|
|
47
|
+
"files": {
|
|
48
|
+
"type": "array",
|
|
49
|
+
"minItems": 1,
|
|
50
|
+
"description": "Every file this snapshot vouches for. Anything in the directory not listed here is unaccounted for.",
|
|
51
|
+
"items": {
|
|
52
|
+
"type": "object",
|
|
53
|
+
"required": ["path", "bytes", "sha256"],
|
|
54
|
+
"additionalProperties": false,
|
|
55
|
+
"properties": {
|
|
56
|
+
"path": { "type": "string", "description": "Relative to the source directory." },
|
|
57
|
+
"bytes": { "type": "integer", "minimum": 0 },
|
|
58
|
+
"sha256": {
|
|
59
|
+
"type": "string",
|
|
60
|
+
"pattern": "^[0-9a-f]{64}$",
|
|
61
|
+
"description": "Computed by us on retrieval. Proves the file has not changed since; does NOT by itself prove it is what the publisher shipped."
|
|
62
|
+
},
|
|
63
|
+
"url": { "type": ["string", "null"] },
|
|
64
|
+
"upstreamChecksum": {
|
|
65
|
+
"type": ["string", "null"],
|
|
66
|
+
"description": "The publisher's own checksum, '<algo>:<hex>'. Null where the publisher provides none (SIL, for instance)."
|
|
67
|
+
},
|
|
68
|
+
"upstreamVerified": {
|
|
69
|
+
"type": "boolean",
|
|
70
|
+
"description": "True ONLY when upstreamChecksum was present AND matched the bytes on disk. This is the difference between provenance and assumption: a matching publisher checksum is proof the local file IS the named release; a plausible filename is not."
|
|
71
|
+
},
|
|
72
|
+
"derivedFrom": {
|
|
73
|
+
"type": ["string", "null"],
|
|
74
|
+
"description": "For a file extracted from a verified archive: the archive it came out of. This is how a CSV inherits a DOI's authority legibly — the CSV's own hash is ours, but it came from bytes the publisher vouched for. Without this the chain would have to be assumed."
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
},
|
|
79
|
+
"verified": {
|
|
80
|
+
"type": "boolean",
|
|
81
|
+
"description": "Whether this snapshot's pin is trustworthy. Must be false whenever pin is null — a snapshot cannot be verified against no particular release."
|
|
82
|
+
},
|
|
83
|
+
"fetchedBy": { "type": "string", "description": "Path of the fetcher that wrote this." },
|
|
84
|
+
"fetchedAt": { "type": "string", "description": "ISO 8601. When WE retrieved it." },
|
|
85
|
+
"notes": { "type": ["string", "null"] }
|
|
86
|
+
},
|
|
87
|
+
"allOf": [
|
|
88
|
+
{
|
|
89
|
+
"if": { "properties": { "verified": { "const": true } }, "required": ["verified"] },
|
|
90
|
+
"then": {
|
|
91
|
+
"properties": { "pin": { "type": "object" } },
|
|
92
|
+
"required": ["pin"]
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
]
|
|
96
|
+
}
|