modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/engine.py ADDED
@@ -0,0 +1,238 @@
1
+ """Run the slice-1 decision stages against one offline snapshot."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ from collections.abc import Callable, Mapping
7
+ from dataclasses import replace
8
+
9
+ from decision.computed import with_computed
10
+ from decision.contract import (
11
+ DEFAULT_TASK_TOKENS,
12
+ Decision,
13
+ Estimate,
14
+ FacetLookup,
15
+ InventoryProfile,
16
+ MayQualify,
17
+ OfferingRef,
18
+ Result,
19
+ Spec,
20
+ spec_hash,
21
+ )
22
+ from decision.filter import FilterResult, apply
23
+ from decision.optimise import EvidenceSelector, optimise
24
+ from decision.relax import fewest, smallest_changes
25
+ from decision.resolve import Resolved, resolve
26
+ from decision.snapshot import ExplanationIndex
27
+
28
+
29
+ def _objective_names(spec: Spec) -> list[str]:
30
+ objective = spec.optimize
31
+ return (
32
+ [objective.max]
33
+ if objective.max
34
+ else [objective.min]
35
+ if objective.min
36
+ else [step.facet for step in objective.lexicographic]
37
+ if objective.lexicographic
38
+ else list(objective.weights or objective.pareto)
39
+ )
40
+
41
+
42
+ def offering_ref(snapshot, cid: str) -> OfferingRef:
43
+ if snapshot.kind(cid) == "model":
44
+ return OfferingRef(model=cid)
45
+ return OfferingRef(
46
+ model=snapshot.model_of(cid),
47
+ **{
48
+ key: snapshot.fact(cid, "offering." + key).value
49
+ for key in ("provider", "region", "tier")
50
+ },
51
+ )
52
+
53
+
54
+ def split_missing(ordered):
55
+ """Rank only complete rows; a missing objective value is a capability unknown.
56
+
57
+ Principle 1 of the design: unknown means may qualify, never ranked last
58
+ and never dropped. Returns the ranked stage result and, per candidate
59
+ without a value, the objective facets it is unknown on.
60
+ """
61
+ missing = set(ordered.missing)
62
+ ranked = tuple(row for row in ordered.results if row.candidate_id not in missing)
63
+ return replace(ordered, results=ranked), {
64
+ cid: list(ordered.unknown.get(cid, ())) for cid in ordered.missing}
65
+
66
+
67
+ def run_optimise(snapshot, filtered, spec, selectors, domains):
68
+ penalties = {}
69
+ for penalty in filtered.penalties:
70
+ for cid in penalty.failing + penalty.unknown:
71
+ penalties.setdefault(cid, {})[penalty.condition] = penalty.penalty
72
+ return optimise(
73
+ snapshot,
74
+ filtered.feasible,
75
+ spec.optimize,
76
+ penalties=penalties,
77
+ evidence_selectors=selectors,
78
+ domains=domains,
79
+ )
80
+
81
+
82
+ def validate(
83
+ spec: Spec,
84
+ snapshot: ExplanationIndex,
85
+ *,
86
+ facets: FacetLookup | None = None,
87
+ profiles: Mapping[str, InventoryProfile] | None = None,
88
+ ) -> Resolved:
89
+ """Run the decision stages through resolve, stopping before filtering."""
90
+ if snapshot is None:
91
+ raise ValueError("a loaded decision snapshot is required")
92
+ snapshot = with_computed(snapshot, spec.task_tokens or DEFAULT_TASK_TOKENS)
93
+ if spec.explain in ("summary", "full"):
94
+ snapshot.require_explanation_records()
95
+ return resolve(spec, facets=facets, profiles=profiles)
96
+
97
+
98
+ def decide(
99
+ spec: Spec,
100
+ snapshot: ExplanationIndex,
101
+ *,
102
+ facets: FacetLookup | None = None,
103
+ profiles: Mapping[str, InventoryProfile] | None = None,
104
+ evidence_selectors: Mapping[str, EvidenceSelector] | None = None,
105
+ _filter_trace: Callable[[FilterResult], None] | None = None,
106
+ ) -> Decision:
107
+ """Return a reproducible decision. Explanation work is skipped at ``none``."""
108
+ resolved = validate(spec, snapshot, facets=facets, profiles=profiles)
109
+ # Computed facets (offering.cost_per_task) depend on the spec, so the
110
+ # remaining stages use the same per-decision view validation prepared for.
111
+ snapshot = with_computed(snapshot, spec.task_tokens or DEFAULT_TASK_TOKENS)
112
+ domains = frozenset(snapshot.domain_ids())
113
+ requested = frozenset(spec.capabilities or {})
114
+ selectors = dict(evidence_selectors or {})
115
+ names = _objective_names(spec)
116
+ for signed in names:
117
+ name = signed.removeprefix("-")
118
+ if resolved.facets(name).subject == "evidence" and name not in domains:
119
+ selectors.setdefault(
120
+ name,
121
+ EvidenceSelector.from_qualifiers(
122
+ name, resolved.objective_qualifiers.get(name), domains=requested
123
+ ),
124
+ )
125
+ filtered = apply(resolved, snapshot)
126
+ if _filter_trace is not None:
127
+ _filter_trace(filtered)
128
+ ordered, objective_unknown = split_missing(
129
+ run_optimise(snapshot, filtered, spec, selectors, domains)
130
+ )
131
+ digest = spec_hash(spec)
132
+ objective_domains = [name.removeprefix("-") for name in names
133
+ if name.removeprefix("-") in domains]
134
+ shown_domains = sorted(requested | set(objective_domains))
135
+ proxy_only_domains = {
136
+ domain
137
+ for domain in shown_domains
138
+ if {
139
+ directness
140
+ for item in snapshot.capability_items.values()
141
+ for tagged_domain, directness in item.get("domains", ())
142
+ if tagged_domain == domain
143
+ }
144
+ == {"proxy"}
145
+ }
146
+ probability_domain = (
147
+ objective_domains[0] if len(objective_domains) == 1 and len(names) == 1 else None
148
+ )
149
+ probabilities = {}
150
+ model_estimates = {}
151
+ if probability_domain is not None:
152
+ from decision.capability import CapabilityEstimate, deterministic_probabilities
153
+
154
+ for row in ordered.results:
155
+ model_id = snapshot.model_of(row.candidate_id)
156
+ stored = snapshot.capability_estimate(row.candidate_id, probability_domain)
157
+ if stored is not None:
158
+ model_estimates.setdefault(
159
+ model_id,
160
+ CapabilityEstimate(stored.value, stored.low, stored.high, stored.sd),
161
+ )
162
+ probabilities = deterministic_probabilities(
163
+ model_estimates,
164
+ seed_material=f"{snapshot.snapshot_id}:{digest}:{probability_domain}",
165
+ )
166
+
167
+ results = []
168
+ for i, row in enumerate(ordered.results[: spec.limit]):
169
+ model_id = snapshot.model_of(row.candidate_id)
170
+ stored_estimates = [
171
+ (domain, snapshot.capability_estimate(row.candidate_id, domain))
172
+ for domain in shown_domains
173
+ ]
174
+ estimates = [
175
+ Estimate(domain=domain, value=estimate.value, interval=(estimate.low, estimate.high))
176
+ for domain, estimate in stored_estimates
177
+ if estimate is not None
178
+ ]
179
+ warnings = list(row.warnings)
180
+ if row.candidate_id in filtered.deprecated:
181
+ warnings.append("deprecated")
182
+ if any(estimate.domain in proxy_only_domains for estimate in estimates):
183
+ warnings.append("proxy_evidence_only")
184
+ current = model_estimates.get(model_id)
185
+ if current is not None and any(
186
+ other_id != model_id
187
+ and max(current.low, other.low) <= min(current.high, other.high)
188
+ for other_id, other in model_estimates.items()
189
+ ):
190
+ warnings.append("not_separable")
191
+ p_best, top3 = probabilities.get(model_id, (None, None))
192
+ results.append(Result(
193
+ rank=i + 1,
194
+ offering=offering_ref(snapshot, row.candidate_id),
195
+ estimates=estimates or None,
196
+ p_best=p_best,
197
+ top3_stability=top3,
198
+ soft_penalty=row.soft_penalty,
199
+ warnings=warnings,
200
+ ))
201
+ relax, relax_to = [], []
202
+ if ordered.status == "no_feasible":
203
+ # Never the class or a requested domain: that would change the question.
204
+ if not filtered.feasible:
205
+ relax = fewest(resolved, snapshot, requested)
206
+ relax_to = smallest_changes(resolved, snapshot, requested)
207
+ if not relax:
208
+ relax = [ordered.reason or "no candidates in the snapshot"]
209
+ decision = Decision(
210
+ decision_id="dec_"
211
+ + hashlib.sha256((digest + snapshot.snapshot_id).encode()).hexdigest()[:24],
212
+ spec_hash=digest,
213
+ snapshot=snapshot.snapshot_id,
214
+ explain=spec.explain,
215
+ status="partial"
216
+ if ordered.status == "answered" and filtered.may_qualify
217
+ else ordered.status,
218
+ results=results,
219
+ relax=relax,
220
+ relax_to=relax_to,
221
+ may_qualify=[
222
+ MayQualify(
223
+ model=snapshot.model_of(cid),
224
+ offering=offering_ref(snapshot, cid),
225
+ unknown=unknown,
226
+ )
227
+ for cid, unknown in sorted(
228
+ [(m.candidate, list(m.unknown)) for m in filtered.may_qualify]
229
+ + list(objective_unknown.items())
230
+ )
231
+ ],
232
+ out_of_lineup=getattr(snapshot, "out_of_lineup", 0),
233
+ )
234
+ if spec.explain != "none":
235
+ from decision.explain import explain
236
+
237
+ explain(decision, resolved, snapshot, filtered, ordered, selectors, domains)
238
+ return decision
decision/excluded.py ADDED
@@ -0,0 +1,34 @@
1
+ """The excluded-source rules shared by the catalogue guard and snapshot build."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from dataclasses import dataclass
7
+ from urllib.parse import urlsplit
8
+
9
+ # These publishers and their owned measurements were removed by MODEL-117.
10
+ REMOVED_HOSTS = ("artificialanalysis.ai", "zapier.com")
11
+ REMOVED_TEXT = re.compile(
12
+ r"artificial[\s-]?analysis|zapier|automationbench|gdpval-aa|aa-lcr|"
13
+ r"aa[\s-]intelligence[\s-]index|omniscience",
14
+ re.IGNORECASE,
15
+ )
16
+ REMOVED_ID = re.compile(r"^(aa_|artificial_?analysis|automationbench)|_aa$|_aa_")
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class ExcludedSources:
21
+ hosts: tuple[str, ...] = REMOVED_HOSTS
22
+ text: re.Pattern[str] = REMOVED_TEXT
23
+ ids: re.Pattern[str] = REMOVED_ID
24
+
25
+ def url(self, url: object) -> bool:
26
+ host = (urlsplit(str(url or "").strip()).hostname or "").lower().rstrip(".")
27
+ return any(host == item or host.endswith("." + item) for item in self.hosts)
28
+
29
+ def benchmark(self, benchmark_id: object) -> bool:
30
+ return bool(self.ids.search(str(benchmark_id)))
31
+
32
+
33
+ def excluded_sources() -> ExcludedSources:
34
+ return ExcludedSources()