modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
decision/engine.py
ADDED
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
"""Run the slice-1 decision stages against one offline snapshot."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
from collections.abc import Callable, Mapping
|
|
7
|
+
from dataclasses import replace
|
|
8
|
+
|
|
9
|
+
from decision.computed import with_computed
|
|
10
|
+
from decision.contract import (
|
|
11
|
+
DEFAULT_TASK_TOKENS,
|
|
12
|
+
Decision,
|
|
13
|
+
Estimate,
|
|
14
|
+
FacetLookup,
|
|
15
|
+
InventoryProfile,
|
|
16
|
+
MayQualify,
|
|
17
|
+
OfferingRef,
|
|
18
|
+
Result,
|
|
19
|
+
Spec,
|
|
20
|
+
spec_hash,
|
|
21
|
+
)
|
|
22
|
+
from decision.filter import FilterResult, apply
|
|
23
|
+
from decision.optimise import EvidenceSelector, optimise
|
|
24
|
+
from decision.relax import fewest, smallest_changes
|
|
25
|
+
from decision.resolve import Resolved, resolve
|
|
26
|
+
from decision.snapshot import ExplanationIndex
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _objective_names(spec: Spec) -> list[str]:
|
|
30
|
+
objective = spec.optimize
|
|
31
|
+
return (
|
|
32
|
+
[objective.max]
|
|
33
|
+
if objective.max
|
|
34
|
+
else [objective.min]
|
|
35
|
+
if objective.min
|
|
36
|
+
else [step.facet for step in objective.lexicographic]
|
|
37
|
+
if objective.lexicographic
|
|
38
|
+
else list(objective.weights or objective.pareto)
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def offering_ref(snapshot, cid: str) -> OfferingRef:
|
|
43
|
+
if snapshot.kind(cid) == "model":
|
|
44
|
+
return OfferingRef(model=cid)
|
|
45
|
+
return OfferingRef(
|
|
46
|
+
model=snapshot.model_of(cid),
|
|
47
|
+
**{
|
|
48
|
+
key: snapshot.fact(cid, "offering." + key).value
|
|
49
|
+
for key in ("provider", "region", "tier")
|
|
50
|
+
},
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def split_missing(ordered):
|
|
55
|
+
"""Rank only complete rows; a missing objective value is a capability unknown.
|
|
56
|
+
|
|
57
|
+
Principle 1 of the design: unknown means may qualify, never ranked last
|
|
58
|
+
and never dropped. Returns the ranked stage result and, per candidate
|
|
59
|
+
without a value, the objective facets it is unknown on.
|
|
60
|
+
"""
|
|
61
|
+
missing = set(ordered.missing)
|
|
62
|
+
ranked = tuple(row for row in ordered.results if row.candidate_id not in missing)
|
|
63
|
+
return replace(ordered, results=ranked), {
|
|
64
|
+
cid: list(ordered.unknown.get(cid, ())) for cid in ordered.missing}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def run_optimise(snapshot, filtered, spec, selectors, domains):
|
|
68
|
+
penalties = {}
|
|
69
|
+
for penalty in filtered.penalties:
|
|
70
|
+
for cid in penalty.failing + penalty.unknown:
|
|
71
|
+
penalties.setdefault(cid, {})[penalty.condition] = penalty.penalty
|
|
72
|
+
return optimise(
|
|
73
|
+
snapshot,
|
|
74
|
+
filtered.feasible,
|
|
75
|
+
spec.optimize,
|
|
76
|
+
penalties=penalties,
|
|
77
|
+
evidence_selectors=selectors,
|
|
78
|
+
domains=domains,
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def validate(
|
|
83
|
+
spec: Spec,
|
|
84
|
+
snapshot: ExplanationIndex,
|
|
85
|
+
*,
|
|
86
|
+
facets: FacetLookup | None = None,
|
|
87
|
+
profiles: Mapping[str, InventoryProfile] | None = None,
|
|
88
|
+
) -> Resolved:
|
|
89
|
+
"""Run the decision stages through resolve, stopping before filtering."""
|
|
90
|
+
if snapshot is None:
|
|
91
|
+
raise ValueError("a loaded decision snapshot is required")
|
|
92
|
+
snapshot = with_computed(snapshot, spec.task_tokens or DEFAULT_TASK_TOKENS)
|
|
93
|
+
if spec.explain in ("summary", "full"):
|
|
94
|
+
snapshot.require_explanation_records()
|
|
95
|
+
return resolve(spec, facets=facets, profiles=profiles)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def decide(
|
|
99
|
+
spec: Spec,
|
|
100
|
+
snapshot: ExplanationIndex,
|
|
101
|
+
*,
|
|
102
|
+
facets: FacetLookup | None = None,
|
|
103
|
+
profiles: Mapping[str, InventoryProfile] | None = None,
|
|
104
|
+
evidence_selectors: Mapping[str, EvidenceSelector] | None = None,
|
|
105
|
+
_filter_trace: Callable[[FilterResult], None] | None = None,
|
|
106
|
+
) -> Decision:
|
|
107
|
+
"""Return a reproducible decision. Explanation work is skipped at ``none``."""
|
|
108
|
+
resolved = validate(spec, snapshot, facets=facets, profiles=profiles)
|
|
109
|
+
# Computed facets (offering.cost_per_task) depend on the spec, so the
|
|
110
|
+
# remaining stages use the same per-decision view validation prepared for.
|
|
111
|
+
snapshot = with_computed(snapshot, spec.task_tokens or DEFAULT_TASK_TOKENS)
|
|
112
|
+
domains = frozenset(snapshot.domain_ids())
|
|
113
|
+
requested = frozenset(spec.capabilities or {})
|
|
114
|
+
selectors = dict(evidence_selectors or {})
|
|
115
|
+
names = _objective_names(spec)
|
|
116
|
+
for signed in names:
|
|
117
|
+
name = signed.removeprefix("-")
|
|
118
|
+
if resolved.facets(name).subject == "evidence" and name not in domains:
|
|
119
|
+
selectors.setdefault(
|
|
120
|
+
name,
|
|
121
|
+
EvidenceSelector.from_qualifiers(
|
|
122
|
+
name, resolved.objective_qualifiers.get(name), domains=requested
|
|
123
|
+
),
|
|
124
|
+
)
|
|
125
|
+
filtered = apply(resolved, snapshot)
|
|
126
|
+
if _filter_trace is not None:
|
|
127
|
+
_filter_trace(filtered)
|
|
128
|
+
ordered, objective_unknown = split_missing(
|
|
129
|
+
run_optimise(snapshot, filtered, spec, selectors, domains)
|
|
130
|
+
)
|
|
131
|
+
digest = spec_hash(spec)
|
|
132
|
+
objective_domains = [name.removeprefix("-") for name in names
|
|
133
|
+
if name.removeprefix("-") in domains]
|
|
134
|
+
shown_domains = sorted(requested | set(objective_domains))
|
|
135
|
+
proxy_only_domains = {
|
|
136
|
+
domain
|
|
137
|
+
for domain in shown_domains
|
|
138
|
+
if {
|
|
139
|
+
directness
|
|
140
|
+
for item in snapshot.capability_items.values()
|
|
141
|
+
for tagged_domain, directness in item.get("domains", ())
|
|
142
|
+
if tagged_domain == domain
|
|
143
|
+
}
|
|
144
|
+
== {"proxy"}
|
|
145
|
+
}
|
|
146
|
+
probability_domain = (
|
|
147
|
+
objective_domains[0] if len(objective_domains) == 1 and len(names) == 1 else None
|
|
148
|
+
)
|
|
149
|
+
probabilities = {}
|
|
150
|
+
model_estimates = {}
|
|
151
|
+
if probability_domain is not None:
|
|
152
|
+
from decision.capability import CapabilityEstimate, deterministic_probabilities
|
|
153
|
+
|
|
154
|
+
for row in ordered.results:
|
|
155
|
+
model_id = snapshot.model_of(row.candidate_id)
|
|
156
|
+
stored = snapshot.capability_estimate(row.candidate_id, probability_domain)
|
|
157
|
+
if stored is not None:
|
|
158
|
+
model_estimates.setdefault(
|
|
159
|
+
model_id,
|
|
160
|
+
CapabilityEstimate(stored.value, stored.low, stored.high, stored.sd),
|
|
161
|
+
)
|
|
162
|
+
probabilities = deterministic_probabilities(
|
|
163
|
+
model_estimates,
|
|
164
|
+
seed_material=f"{snapshot.snapshot_id}:{digest}:{probability_domain}",
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
results = []
|
|
168
|
+
for i, row in enumerate(ordered.results[: spec.limit]):
|
|
169
|
+
model_id = snapshot.model_of(row.candidate_id)
|
|
170
|
+
stored_estimates = [
|
|
171
|
+
(domain, snapshot.capability_estimate(row.candidate_id, domain))
|
|
172
|
+
for domain in shown_domains
|
|
173
|
+
]
|
|
174
|
+
estimates = [
|
|
175
|
+
Estimate(domain=domain, value=estimate.value, interval=(estimate.low, estimate.high))
|
|
176
|
+
for domain, estimate in stored_estimates
|
|
177
|
+
if estimate is not None
|
|
178
|
+
]
|
|
179
|
+
warnings = list(row.warnings)
|
|
180
|
+
if row.candidate_id in filtered.deprecated:
|
|
181
|
+
warnings.append("deprecated")
|
|
182
|
+
if any(estimate.domain in proxy_only_domains for estimate in estimates):
|
|
183
|
+
warnings.append("proxy_evidence_only")
|
|
184
|
+
current = model_estimates.get(model_id)
|
|
185
|
+
if current is not None and any(
|
|
186
|
+
other_id != model_id
|
|
187
|
+
and max(current.low, other.low) <= min(current.high, other.high)
|
|
188
|
+
for other_id, other in model_estimates.items()
|
|
189
|
+
):
|
|
190
|
+
warnings.append("not_separable")
|
|
191
|
+
p_best, top3 = probabilities.get(model_id, (None, None))
|
|
192
|
+
results.append(Result(
|
|
193
|
+
rank=i + 1,
|
|
194
|
+
offering=offering_ref(snapshot, row.candidate_id),
|
|
195
|
+
estimates=estimates or None,
|
|
196
|
+
p_best=p_best,
|
|
197
|
+
top3_stability=top3,
|
|
198
|
+
soft_penalty=row.soft_penalty,
|
|
199
|
+
warnings=warnings,
|
|
200
|
+
))
|
|
201
|
+
relax, relax_to = [], []
|
|
202
|
+
if ordered.status == "no_feasible":
|
|
203
|
+
# Never the class or a requested domain: that would change the question.
|
|
204
|
+
if not filtered.feasible:
|
|
205
|
+
relax = fewest(resolved, snapshot, requested)
|
|
206
|
+
relax_to = smallest_changes(resolved, snapshot, requested)
|
|
207
|
+
if not relax:
|
|
208
|
+
relax = [ordered.reason or "no candidates in the snapshot"]
|
|
209
|
+
decision = Decision(
|
|
210
|
+
decision_id="dec_"
|
|
211
|
+
+ hashlib.sha256((digest + snapshot.snapshot_id).encode()).hexdigest()[:24],
|
|
212
|
+
spec_hash=digest,
|
|
213
|
+
snapshot=snapshot.snapshot_id,
|
|
214
|
+
explain=spec.explain,
|
|
215
|
+
status="partial"
|
|
216
|
+
if ordered.status == "answered" and filtered.may_qualify
|
|
217
|
+
else ordered.status,
|
|
218
|
+
results=results,
|
|
219
|
+
relax=relax,
|
|
220
|
+
relax_to=relax_to,
|
|
221
|
+
may_qualify=[
|
|
222
|
+
MayQualify(
|
|
223
|
+
model=snapshot.model_of(cid),
|
|
224
|
+
offering=offering_ref(snapshot, cid),
|
|
225
|
+
unknown=unknown,
|
|
226
|
+
)
|
|
227
|
+
for cid, unknown in sorted(
|
|
228
|
+
[(m.candidate, list(m.unknown)) for m in filtered.may_qualify]
|
|
229
|
+
+ list(objective_unknown.items())
|
|
230
|
+
)
|
|
231
|
+
],
|
|
232
|
+
out_of_lineup=getattr(snapshot, "out_of_lineup", 0),
|
|
233
|
+
)
|
|
234
|
+
if spec.explain != "none":
|
|
235
|
+
from decision.explain import explain
|
|
236
|
+
|
|
237
|
+
explain(decision, resolved, snapshot, filtered, ordered, selectors, domains)
|
|
238
|
+
return decision
|
decision/excluded.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""The excluded-source rules shared by the catalogue guard and snapshot build."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from urllib.parse import urlsplit
|
|
8
|
+
|
|
9
|
+
# These publishers and their owned measurements were removed by MODEL-117.
|
|
10
|
+
REMOVED_HOSTS = ("artificialanalysis.ai", "zapier.com")
|
|
11
|
+
REMOVED_TEXT = re.compile(
|
|
12
|
+
r"artificial[\s-]?analysis|zapier|automationbench|gdpval-aa|aa-lcr|"
|
|
13
|
+
r"aa[\s-]intelligence[\s-]index|omniscience",
|
|
14
|
+
re.IGNORECASE,
|
|
15
|
+
)
|
|
16
|
+
REMOVED_ID = re.compile(r"^(aa_|artificial_?analysis|automationbench)|_aa$|_aa_")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class ExcludedSources:
|
|
21
|
+
hosts: tuple[str, ...] = REMOVED_HOSTS
|
|
22
|
+
text: re.Pattern[str] = REMOVED_TEXT
|
|
23
|
+
ids: re.Pattern[str] = REMOVED_ID
|
|
24
|
+
|
|
25
|
+
def url(self, url: object) -> bool:
|
|
26
|
+
host = (urlsplit(str(url or "").strip()).hostname or "").lower().rstrip(".")
|
|
27
|
+
return any(host == item or host.endswith("." + item) for item in self.hosts)
|
|
28
|
+
|
|
29
|
+
def benchmark(self, benchmark_id: object) -> bool:
|
|
30
|
+
return bool(self.ids.search(str(benchmark_id)))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def excluded_sources() -> ExcludedSources:
|
|
34
|
+
return ExcludedSources()
|