modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
decision/optimise.py
ADDED
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
"""Slice-1 optimisation over filtered candidates, without capability blending.
|
|
2
|
+
|
|
3
|
+
The resolver supplies evidence selectors and domain IDs because Objective v1
|
|
4
|
+
contains facet IDs only. This module returns stage values, not a Decision:
|
|
5
|
+
MODEL-145 resolves offering identities and registered source IDs for transport.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Mapping, Sequence
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from datetime import date
|
|
12
|
+
from math import isclose, isfinite
|
|
13
|
+
from typing import TYPE_CHECKING, Literal
|
|
14
|
+
|
|
15
|
+
from decision.contract import EvidenceQualifiers, Objective, Tolerance
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
from decision.snapshot import CapabilityEstimateValue, EvidenceValue, SnapshotIndex
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class EvidenceSelector:
|
|
23
|
+
"""A resolver's explicit benchmark selector; omitted subcategory means aggregate."""
|
|
24
|
+
|
|
25
|
+
benchmark_id: str
|
|
26
|
+
version: str | None = None
|
|
27
|
+
subcategory: str | None = None
|
|
28
|
+
measured_by: frozenset[str] | None = None
|
|
29
|
+
effort: str | None = None
|
|
30
|
+
harness: str | None = None
|
|
31
|
+
after: date | None = None
|
|
32
|
+
direct: bool = False
|
|
33
|
+
#: The capabilities asked about; ``direct`` is relative to them.
|
|
34
|
+
domains: frozenset[str] = frozenset()
|
|
35
|
+
|
|
36
|
+
@classmethod
|
|
37
|
+
def from_qualifiers(
|
|
38
|
+
cls, benchmark_id: str, qualifiers: EvidenceQualifiers | None,
|
|
39
|
+
domains: frozenset[str] = frozenset(),
|
|
40
|
+
) -> EvidenceSelector:
|
|
41
|
+
qualifiers = qualifiers or EvidenceQualifiers()
|
|
42
|
+
measured = {
|
|
43
|
+
"independent": frozenset({
|
|
44
|
+
"benchmark_author",
|
|
45
|
+
"independent",
|
|
46
|
+
"independent_evaluator",
|
|
47
|
+
"modelspec",
|
|
48
|
+
"outcome_protocol",
|
|
49
|
+
}),
|
|
50
|
+
"provider_self_report": frozenset({"provider_self_report"}),
|
|
51
|
+
"any": None,
|
|
52
|
+
None: None,
|
|
53
|
+
}[qualifiers.measured_by]
|
|
54
|
+
return cls(
|
|
55
|
+
benchmark_id=benchmark_id,
|
|
56
|
+
measured_by=measured,
|
|
57
|
+
effort=qualifiers.effort,
|
|
58
|
+
harness=qualifiers.harness,
|
|
59
|
+
after=qualifiers.measured_after,
|
|
60
|
+
direct=qualifiers.direct,
|
|
61
|
+
domains=frozenset(domains),
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass(frozen=True)
|
|
66
|
+
class Normalisation:
|
|
67
|
+
minimum: float | None
|
|
68
|
+
maximum: float | None
|
|
69
|
+
direction: Literal["max", "min"]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True)
|
|
73
|
+
class DimensionContribution:
|
|
74
|
+
dimension: str
|
|
75
|
+
raw_value: float | None
|
|
76
|
+
value: float | None
|
|
77
|
+
weight: float
|
|
78
|
+
normalisation: Normalisation
|
|
79
|
+
sources: tuple[str, ...] = ()
|
|
80
|
+
evidence: tuple[EvidenceValue, ...] = ()
|
|
81
|
+
estimate: CapabilityEstimateValue | None = None
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass(frozen=True)
|
|
85
|
+
class OptimisedResult:
|
|
86
|
+
candidate_id: str
|
|
87
|
+
contributions: tuple[DimensionContribution, ...]
|
|
88
|
+
score: float | None
|
|
89
|
+
soft_penalty: float
|
|
90
|
+
penalties: tuple[tuple[str, float], ...]
|
|
91
|
+
warnings: tuple[str, ...] = ()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass(frozen=True)
|
|
95
|
+
class WeightTippingPoint:
|
|
96
|
+
"""Infimum of positive single-weight changes; cross in the stated direction.
|
|
97
|
+
|
|
98
|
+
At the threshold scores tie, so candidate ID may retain the incumbent.
|
|
99
|
+
Other weights and feasible-set normalisation remain fixed.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
dimension: str
|
|
103
|
+
threshold: float
|
|
104
|
+
direction: Literal["increase", "decrease"]
|
|
105
|
+
change: float
|
|
106
|
+
new_top: str
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass(frozen=True)
|
|
110
|
+
class Optimisation:
|
|
111
|
+
"""Ordered stage results, with the nearest weight crossing first.
|
|
112
|
+
|
|
113
|
+
For Pareto, dominance maps each excluded complete candidate to every candidate
|
|
114
|
+
that is at least as good in all penalty-adjusted dimensions and better in one.
|
|
115
|
+
Missing candidates cannot establish Pareto membership and are listed separately.
|
|
116
|
+
"""
|
|
117
|
+
|
|
118
|
+
status: Literal["answered", "partial", "no_feasible"]
|
|
119
|
+
results: tuple[OptimisedResult, ...]
|
|
120
|
+
reason: str | None = None
|
|
121
|
+
dominance: dict[str, tuple[str, ...]] = field(default_factory=dict)
|
|
122
|
+
missing: tuple[str, ...] = ()
|
|
123
|
+
tipping_points: tuple[WeightTippingPoint, ...] = ()
|
|
124
|
+
#: Each missing candidate's objective facets without a value, unsigned.
|
|
125
|
+
unknown: dict[str, tuple[str, ...]] = field(default_factory=dict)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _number(value: object) -> float | None:
|
|
129
|
+
if isinstance(value, bool) or not isinstance(value, int | float):
|
|
130
|
+
return None
|
|
131
|
+
return float(value) if isfinite(value) else None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _dimensions(objective: Objective) -> list[tuple[str, float, Tolerance | None]]:
|
|
135
|
+
if objective.weights is not None:
|
|
136
|
+
return [(key, weight, None) for key, weight in sorted(objective.weights.items())]
|
|
137
|
+
if objective.pareto is not None:
|
|
138
|
+
return [(key, 1, None) for key in sorted(objective.pareto)]
|
|
139
|
+
if objective.lexicographic is not None:
|
|
140
|
+
return [(step.max or f"-{step.min}", 1, step.within)
|
|
141
|
+
for step in objective.lexicographic]
|
|
142
|
+
return [(objective.max or f"-{objective.min}", 1, None)]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _adjusted(row: OptimisedResult, i: int) -> float:
|
|
146
|
+
value = row.contributions[i].value
|
|
147
|
+
assert value is not None
|
|
148
|
+
return value - row.soft_penalty
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _lex_order(rows: list[OptimisedResult],
|
|
152
|
+
dimensions: list[tuple[str, float, Tolerance | None]],
|
|
153
|
+
depth: int = 0) -> list[OptimisedResult]:
|
|
154
|
+
if depth == len(dimensions):
|
|
155
|
+
return sorted(rows, key=lambda row: row.candidate_id)
|
|
156
|
+
remaining = sorted(rows, key=lambda row: (-_adjusted(row, depth), row.candidate_id))
|
|
157
|
+
ordered = []
|
|
158
|
+
tolerance = dimensions[depth][2]
|
|
159
|
+
while remaining:
|
|
160
|
+
best = remaining[0]
|
|
161
|
+
width = 0.0
|
|
162
|
+
norm = best.contributions[depth].normalisation
|
|
163
|
+
span = norm.maximum - norm.minimum
|
|
164
|
+
if tolerance is not None and span:
|
|
165
|
+
raw_best = best.contributions[depth].raw_value
|
|
166
|
+
width = (tolerance.absolute if tolerance.absolute is not None
|
|
167
|
+
else abs(raw_best) * tolerance.relative) / span
|
|
168
|
+
cutoff = _adjusted(best, depth) - width
|
|
169
|
+
group = [row for row in remaining if _adjusted(row, depth) >= cutoff
|
|
170
|
+
or (tolerance is not None
|
|
171
|
+
and isclose(_adjusted(row, depth), cutoff, rel_tol=0, abs_tol=1e-15))]
|
|
172
|
+
ordered.extend(_lex_order(group, dimensions, depth + 1))
|
|
173
|
+
remaining = remaining[len(group):]
|
|
174
|
+
return ordered
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _tipping_points(rows: list[OptimisedResult]) -> tuple[WeightTippingPoint, ...]:
|
|
178
|
+
winner = rows[0]
|
|
179
|
+
points = []
|
|
180
|
+
for i, contribution in enumerate(winner.contributions):
|
|
181
|
+
for direction, sign in (("increase", 1), ("decrease", -1)):
|
|
182
|
+
crossings = []
|
|
183
|
+
for challenger in rows[1:]:
|
|
184
|
+
slope = challenger.contributions[i].value - contribution.value
|
|
185
|
+
if sign * slope <= 0:
|
|
186
|
+
continue
|
|
187
|
+
delta = (winner.score - challenger.score) / slope
|
|
188
|
+
threshold = contribution.weight + delta
|
|
189
|
+
if threshold <= 0 or not isfinite(threshold):
|
|
190
|
+
continue
|
|
191
|
+
crossings.append((abs(delta), -sign * slope, challenger.candidate_id, threshold))
|
|
192
|
+
if crossings:
|
|
193
|
+
change, _, cid, threshold = min(crossings)
|
|
194
|
+
points.append(WeightTippingPoint(contribution.dimension, threshold,
|
|
195
|
+
direction, change, cid))
|
|
196
|
+
return tuple(sorted(points, key=lambda p: (p.change, p.dimension, p.direction, p.new_top)))
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _read(snapshot: SnapshotIndex, cid: str, facet: str,
|
|
200
|
+
selector: EvidenceSelector | None, domains: set[str] | frozenset[str]
|
|
201
|
+
) -> tuple[
|
|
202
|
+
float | None,
|
|
203
|
+
tuple[str, ...],
|
|
204
|
+
tuple[EvidenceValue, ...],
|
|
205
|
+
CapabilityEstimateValue | None,
|
|
206
|
+
]:
|
|
207
|
+
if facet in domains and selector is None:
|
|
208
|
+
estimate = snapshot.capability_estimate(cid, facet)
|
|
209
|
+
return (None, (), (), None) if estimate is None else (estimate.value, (), (), estimate)
|
|
210
|
+
if selector is None:
|
|
211
|
+
fact = snapshot.fact(cid, facet)
|
|
212
|
+
value = _number(fact.value) if fact.state == "known" else None
|
|
213
|
+
return value, fact.sources, (), None
|
|
214
|
+
if selector.direct and not snapshot.direct_for(selector.benchmark_id, selector.domains):
|
|
215
|
+
return None, (), (), None
|
|
216
|
+
evidence = snapshot.evidence(
|
|
217
|
+
cid, selector.benchmark_id,
|
|
218
|
+
measured_by=set(selector.measured_by) if selector.measured_by is not None else None,
|
|
219
|
+
effort=selector.effort, harness=selector.harness, after=selector.after)
|
|
220
|
+
matches = [e for e in evidence if e.verified and _number(e.value) is not None
|
|
221
|
+
and (selector.version is None or e.version == selector.version)
|
|
222
|
+
and e.subcategory == selector.subcategory]
|
|
223
|
+
# Multiple measurements need a resolver decision, not an implicit max or average.
|
|
224
|
+
if len(matches) != 1:
|
|
225
|
+
return None, (), (), None
|
|
226
|
+
item = matches[0]
|
|
227
|
+
return float(item.value), tuple(item.source_ids), (item,), None
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def optimise(snapshot: SnapshotIndex, candidates: Sequence[str], objective: Objective, *,
|
|
231
|
+
penalties: Mapping[str, Mapping[str, float]] | None = None,
|
|
232
|
+
evidence_selectors: Mapping[str, EvidenceSelector] | None = None,
|
|
233
|
+
domains: set[str] | frozenset[str] = frozenset()) -> Optimisation:
|
|
234
|
+
"""Order complete candidates first; ID breaks exact ties without adding a bonus.
|
|
235
|
+
|
|
236
|
+
All dimensions use feasible-set min-max values. Constant dimensions contribute
|
|
237
|
+
zero. Weights are used as supplied, not rescaled to sum to one. Penalties are
|
|
238
|
+
subtracted once from a scalar total, or from each lexicographic/Pareto dimension.
|
|
239
|
+
Tolerances are in raw units, converted to normalised units before grouping.
|
|
240
|
+
The resolver must supply all domain objective IDs in ``domains`` and bind
|
|
241
|
+
benchmark objectives through ``evidence_selectors``. Neither is inferred from
|
|
242
|
+
spelling. Penalties map candidate IDs to condition IDs and incurred amounts.
|
|
243
|
+
"""
|
|
244
|
+
dimensions = _dimensions(objective)
|
|
245
|
+
selectors = evidence_selectors or {}
|
|
246
|
+
estimate_lookup = getattr(snapshot, "capability_estimate", None)
|
|
247
|
+
for signed, _, _ in dimensions:
|
|
248
|
+
facet = signed.removeprefix("-")
|
|
249
|
+
if facet not in domains or facet in selectors:
|
|
250
|
+
continue
|
|
251
|
+
if estimate_lookup is None or not any(
|
|
252
|
+
estimate_lookup(cid, facet) is not None for cid in candidates
|
|
253
|
+
):
|
|
254
|
+
return Optimisation(
|
|
255
|
+
"no_feasible", (),
|
|
256
|
+
"specify a benchmark or wait for the capability model (MODEL-129)",
|
|
257
|
+
)
|
|
258
|
+
cids = sorted(set(candidates))
|
|
259
|
+
contributions: dict[str, list[DimensionContribution]] = {cid: [] for cid in cids}
|
|
260
|
+
for signed, weight, _ in dimensions:
|
|
261
|
+
if not isfinite(weight):
|
|
262
|
+
raise ValueError("weights must be finite")
|
|
263
|
+
facet = signed.removeprefix("-")
|
|
264
|
+
readings = {
|
|
265
|
+
cid: _read(snapshot, cid, facet, selectors.get(facet), domains) for cid in cids
|
|
266
|
+
}
|
|
267
|
+
raw = {cid: reading[0] for cid, reading in readings.items()}
|
|
268
|
+
known = [v for v in raw.values() if v is not None]
|
|
269
|
+
low, high = (min(known), max(known)) if known else (None, None)
|
|
270
|
+
norm = Normalisation(low, high, "min" if signed.startswith("-") else "max")
|
|
271
|
+
for cid, value in raw.items():
|
|
272
|
+
normalised = None
|
|
273
|
+
if value is not None:
|
|
274
|
+
normalised = 0.0 if high == low else (value - low) / (high - low)
|
|
275
|
+
if norm.direction == "min" and high != low:
|
|
276
|
+
normalised = 1 - normalised
|
|
277
|
+
contributions[cid].append(DimensionContribution(
|
|
278
|
+
signed, value, normalised, weight, norm, readings[cid][1], readings[cid][2],
|
|
279
|
+
readings[cid][3]))
|
|
280
|
+
complete, missing = [], []
|
|
281
|
+
for cid in cids:
|
|
282
|
+
parts = tuple(sorted((penalties or {}).get(cid, {}).items()))
|
|
283
|
+
if any(not isfinite(p) or p < 0 for _, p in parts):
|
|
284
|
+
raise ValueError("penalties must be finite and nonnegative")
|
|
285
|
+
penalty = sum(p for _, p in parts)
|
|
286
|
+
values = contributions[cid]
|
|
287
|
+
unknown = any(c.value is None for c in values)
|
|
288
|
+
scalar = objective.lexicographic is None and objective.pareto is None
|
|
289
|
+
score = sum(c.value * c.weight for c in values) - penalty if not unknown else None
|
|
290
|
+
row = OptimisedResult(cid, tuple(values), score if scalar else None, penalty, parts,
|
|
291
|
+
("missing_objective_value",) if unknown else ())
|
|
292
|
+
(missing if unknown else complete).append(row)
|
|
293
|
+
missing_ids = tuple(row.candidate_id for row in missing)
|
|
294
|
+
unknown = {row.candidate_id: tuple(c.dimension.removeprefix("-")
|
|
295
|
+
for c in row.contributions if c.value is None)
|
|
296
|
+
for row in missing}
|
|
297
|
+
if not complete:
|
|
298
|
+
return Optimisation("no_feasible", (), "no complete objective values", missing=missing_ids,
|
|
299
|
+
unknown=unknown)
|
|
300
|
+
dominance = {}
|
|
301
|
+
if objective.pareto is not None:
|
|
302
|
+
for row in complete:
|
|
303
|
+
dominators = []
|
|
304
|
+
for other in complete:
|
|
305
|
+
differences = [_adjusted(other, i) - _adjusted(row, i)
|
|
306
|
+
for i in range(len(dimensions))]
|
|
307
|
+
if all(d >= 0 for d in differences) and any(d > 0 for d in differences):
|
|
308
|
+
dominators.append(other.candidate_id)
|
|
309
|
+
if dominators:
|
|
310
|
+
dominance[row.candidate_id] = tuple(dominators)
|
|
311
|
+
ordered = [row for row in complete if row.candidate_id not in dominance]
|
|
312
|
+
elif objective.lexicographic is not None:
|
|
313
|
+
ordered = _lex_order(complete, dimensions) + missing
|
|
314
|
+
else:
|
|
315
|
+
ordered = sorted(complete, key=lambda row: (-row.score, row.candidate_id)) + missing
|
|
316
|
+
points = _tipping_points([row for row in ordered if row.score is not None]) \
|
|
317
|
+
if objective.weights is not None else ()
|
|
318
|
+
return Optimisation("partial" if missing else "answered", tuple(ordered),
|
|
319
|
+
dominance=dominance, missing=missing_ids, tipping_points=points,
|
|
320
|
+
unknown=unknown)
|