modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
decision/capability.py
ADDED
|
@@ -0,0 +1,872 @@
|
|
|
1
|
+
"""Learn domain capability estimates from verified benchmark evidence.
|
|
2
|
+
|
|
3
|
+
The fit is a hierarchical bifactor item-response model. Each model has a
|
|
4
|
+
general factor and one deviation per observed domain. Each benchmark learns a
|
|
5
|
+
positive discrimination and an intercept. Percentage measurements use a
|
|
6
|
+
logistic link with the benchmark's registered random baseline. Other numeric
|
|
7
|
+
measurements use a standardised linear link.
|
|
8
|
+
|
|
9
|
+
The module has no benchmark IDs. Benchmark membership and directness come from
|
|
10
|
+
the registry tags carried into the snapshot builder. Missing cells add no
|
|
11
|
+
likelihood term. Old evidence loses precision, provider self-reports get a
|
|
12
|
+
learned offset, and a learned proxy loading keeps proxy tags weaker than direct
|
|
13
|
+
tags. The build stores posterior means, intervals, and explanation drivers.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import hashlib
|
|
19
|
+
import math
|
|
20
|
+
import random
|
|
21
|
+
from collections import defaultdict
|
|
22
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
23
|
+
from dataclasses import dataclass, field
|
|
24
|
+
from datetime import date
|
|
25
|
+
from typing import Literal
|
|
26
|
+
|
|
27
|
+
Direction = Literal["higher_is_better", "lower_is_better"]
|
|
28
|
+
Directness = Literal["direct", "proxy"]
|
|
29
|
+
|
|
30
|
+
_INTERVAL_Z = 1.2815515655446004 # central 80 percent interval
|
|
31
|
+
_STATIC_HALF_LIFE_DAYS = 365.0
|
|
32
|
+
_MIN_RECENCY = 0.25
|
|
33
|
+
_RIDGE_GENERAL = 1.0
|
|
34
|
+
_RIDGE_DOMAIN = 4.0
|
|
35
|
+
_RIDGE_ITEM = 0.25
|
|
36
|
+
# Two observations are the smallest sample that can establish an ordering and
|
|
37
|
+
# a non-zero within-item spread. Keeping those items lets sparse domains rank
|
|
38
|
+
# their directly measured frontier; the prior still makes their intervals
|
|
39
|
+
# appropriately wide.
|
|
40
|
+
_MIN_ITEM_MODELS = 2
|
|
41
|
+
_DOMAIN_PRIOR_PRECISION = 0.25
|
|
42
|
+
_PROJECTION_PROXY_LOADING = 0.35
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class BenchmarkSpec:
|
|
47
|
+
random_baseline: float | None = None
|
|
48
|
+
sample_size: int | None = None
|
|
49
|
+
direction: Direction = "higher_is_better"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(frozen=True)
|
|
53
|
+
class CapabilityObservation:
|
|
54
|
+
model_id: str
|
|
55
|
+
benchmark_id: str
|
|
56
|
+
value: float
|
|
57
|
+
unit: str | None
|
|
58
|
+
measured_by: str
|
|
59
|
+
date: date
|
|
60
|
+
record_id: str
|
|
61
|
+
version: str | None
|
|
62
|
+
domains: tuple[tuple[str, Directness], ...]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass(frozen=True)
|
|
66
|
+
class CapabilityEstimate:
|
|
67
|
+
value: float
|
|
68
|
+
low: float
|
|
69
|
+
high: float
|
|
70
|
+
sd: float
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass(frozen=True)
|
|
74
|
+
class EstimateDriver:
|
|
75
|
+
record_id: str
|
|
76
|
+
benchmark_id: str
|
|
77
|
+
version: str | None
|
|
78
|
+
loading: float
|
|
79
|
+
weight: float
|
|
80
|
+
recency_weight: float
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass(frozen=True)
|
|
84
|
+
class DirectnessFit:
|
|
85
|
+
direct_loading: float = 1.0
|
|
86
|
+
proxy_loading: float = 0.35
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@dataclass(frozen=True)
|
|
90
|
+
class ItemFit:
|
|
91
|
+
id: str
|
|
92
|
+
benchmark_id: str
|
|
93
|
+
version: str | None
|
|
94
|
+
link: Literal["logistic", "linear"]
|
|
95
|
+
direction: Direction
|
|
96
|
+
domains: tuple[tuple[str, Directness], ...]
|
|
97
|
+
discrimination: float
|
|
98
|
+
intercept: float
|
|
99
|
+
residual_sd: float
|
|
100
|
+
random_baseline: float
|
|
101
|
+
center: float
|
|
102
|
+
scale: float
|
|
103
|
+
sample_size: int
|
|
104
|
+
models: int
|
|
105
|
+
|
|
106
|
+
def information(self, ability: float) -> float:
|
|
107
|
+
"""Fisher information on the raw measurement scale."""
|
|
108
|
+
if self.link == "logistic":
|
|
109
|
+
probability = self.random_baseline + (1 - self.random_baseline) * _sigmoid(
|
|
110
|
+
self.discrimination * ability + self.intercept
|
|
111
|
+
)
|
|
112
|
+
slope = self.discrimination * max(
|
|
113
|
+
1e-9,
|
|
114
|
+
(probability - self.random_baseline)
|
|
115
|
+
* (1 - probability)
|
|
116
|
+
/ max(1e-9, 1 - self.random_baseline),
|
|
117
|
+
)
|
|
118
|
+
variance = self.residual_sd**2 + max(
|
|
119
|
+
1e-6, probability * (1 - probability) / self.sample_size
|
|
120
|
+
)
|
|
121
|
+
return slope * slope / variance
|
|
122
|
+
return self.discrimination**2 / max(1e-9, self.residual_sd**2)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@dataclass(frozen=True)
|
|
126
|
+
class BacktestResult:
|
|
127
|
+
cells: int
|
|
128
|
+
model_rmse: float
|
|
129
|
+
naive_rmse: float
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
@dataclass
|
|
133
|
+
class _ModelFit:
|
|
134
|
+
model_id: str
|
|
135
|
+
dimensions: tuple[str, ...]
|
|
136
|
+
values: list[float]
|
|
137
|
+
covariance: list[list[float]] = field(default_factory=list)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@dataclass
|
|
141
|
+
class _Prepared:
|
|
142
|
+
observation: CapabilityObservation
|
|
143
|
+
item_id: str
|
|
144
|
+
target: float
|
|
145
|
+
base_weight: float
|
|
146
|
+
recency_weight: float
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@dataclass
|
|
150
|
+
class CapabilityFit:
|
|
151
|
+
as_of: date
|
|
152
|
+
items: dict[str, ItemFit]
|
|
153
|
+
models: dict[str, _ModelFit]
|
|
154
|
+
domain_sd: dict[str, float]
|
|
155
|
+
source_offsets: dict[str, float]
|
|
156
|
+
directness: DirectnessFit
|
|
157
|
+
drivers: dict[tuple[str, str], tuple[EstimateDriver, ...]]
|
|
158
|
+
domain_estimates: dict[tuple[str, str], CapabilityEstimate] = field(default_factory=dict)
|
|
159
|
+
|
|
160
|
+
def estimate(self, model_id: str, domain: str) -> CapabilityEstimate | None:
|
|
161
|
+
domain_estimate = self.domain_estimates.get((model_id, domain))
|
|
162
|
+
if domain_estimate is not None:
|
|
163
|
+
return domain_estimate
|
|
164
|
+
model = self.models.get(model_id)
|
|
165
|
+
if model is None or not model.covariance or domain not in model.dimensions:
|
|
166
|
+
return None
|
|
167
|
+
weights = [1.0 if dimension == "g" else 1.0 if dimension == domain else 0.0
|
|
168
|
+
for dimension in model.dimensions]
|
|
169
|
+
mean = sum(weight * value for weight, value in zip(weights, model.values))
|
|
170
|
+
variance = sum(
|
|
171
|
+
weights[i] * model.covariance[i][j] * weights[j]
|
|
172
|
+
for i in range(len(weights))
|
|
173
|
+
for j in range(len(weights))
|
|
174
|
+
)
|
|
175
|
+
sd = math.sqrt(max(variance, 1e-9))
|
|
176
|
+
return CapabilityEstimate(mean, mean - _INTERVAL_Z * sd, mean + _INTERVAL_Z * sd, sd)
|
|
177
|
+
|
|
178
|
+
def explain(self, model_id: str, domain: str, *, limit: int = 8) -> tuple[EstimateDriver, ...]:
|
|
179
|
+
return self.drivers.get((model_id, domain), ())[:limit]
|
|
180
|
+
|
|
181
|
+
def predict(self, model_id: str, benchmark_id: str, version: str | None = None) -> float | None:
|
|
182
|
+
item = _find_item(self.items, benchmark_id, version)
|
|
183
|
+
model = self.models.get(model_id)
|
|
184
|
+
if item is None or model is None:
|
|
185
|
+
return None
|
|
186
|
+
theta = _theta(model, item.domains, self.directness.proxy_loading)
|
|
187
|
+
linear = item.discrimination * theta + item.intercept
|
|
188
|
+
if item.link == "logistic":
|
|
189
|
+
p = item.random_baseline + (1 - item.random_baseline) * _sigmoid(linear)
|
|
190
|
+
value = 100 * p
|
|
191
|
+
else:
|
|
192
|
+
value = item.center + item.scale * linear
|
|
193
|
+
if item.direction == "lower_is_better":
|
|
194
|
+
return 100 - value if item.link == "logistic" else -value
|
|
195
|
+
return value
|
|
196
|
+
|
|
197
|
+
def to_payload(self, model_ids: Iterable[str] | None = None) -> dict[str, object]:
|
|
198
|
+
keep = set(self.models) if model_ids is None else set(model_ids)
|
|
199
|
+
domains = sorted(self.domain_sd)
|
|
200
|
+
estimates: dict[str, dict[str, list[float]]] = {}
|
|
201
|
+
driver_rows: dict[str, dict[str, list[list[object]]]] = {}
|
|
202
|
+
for model_id in sorted(keep & set(self.models)):
|
|
203
|
+
by_domain: dict[str, list[float]] = {}
|
|
204
|
+
by_driver: dict[str, list[list[object]]] = {}
|
|
205
|
+
for domain in domains:
|
|
206
|
+
estimate = self.estimate(model_id, domain)
|
|
207
|
+
if estimate is None:
|
|
208
|
+
continue
|
|
209
|
+
by_domain[domain] = [_round(estimate.value), _round(estimate.low),
|
|
210
|
+
_round(estimate.high), _round(estimate.sd)]
|
|
211
|
+
rows = self.explain(model_id, domain)
|
|
212
|
+
if rows:
|
|
213
|
+
by_driver[domain] = [
|
|
214
|
+
[row.record_id, row.benchmark_id, row.version,
|
|
215
|
+
_round(row.loading), _round(row.weight), _round(row.recency_weight)]
|
|
216
|
+
for row in rows
|
|
217
|
+
]
|
|
218
|
+
if by_domain:
|
|
219
|
+
estimates[model_id] = by_domain
|
|
220
|
+
if by_driver:
|
|
221
|
+
driver_rows[model_id] = by_driver
|
|
222
|
+
return {
|
|
223
|
+
"method": "hierarchical-bifactor-irt",
|
|
224
|
+
"as_of": self.as_of.isoformat(),
|
|
225
|
+
"directness": {
|
|
226
|
+
"direct_loading": _round(self.directness.direct_loading),
|
|
227
|
+
"proxy_loading": _round(self.directness.proxy_loading),
|
|
228
|
+
},
|
|
229
|
+
"source_offsets": {key: _round(value) for key, value in sorted(
|
|
230
|
+
self.source_offsets.items()
|
|
231
|
+
)},
|
|
232
|
+
"domain_sd": {key: _round(value) for key, value in sorted(self.domain_sd.items())},
|
|
233
|
+
"items": {
|
|
234
|
+
key: {
|
|
235
|
+
"benchmark": item.benchmark_id,
|
|
236
|
+
"version": item.version,
|
|
237
|
+
"link": item.link,
|
|
238
|
+
"direction": item.direction,
|
|
239
|
+
"domains": [list(tag) for tag in item.domains],
|
|
240
|
+
"discrimination": _round(item.discrimination),
|
|
241
|
+
"intercept": _round(item.intercept),
|
|
242
|
+
"residual_sd": _round(item.residual_sd),
|
|
243
|
+
"random_baseline": _round(item.random_baseline),
|
|
244
|
+
"center": _round(item.center),
|
|
245
|
+
"scale": _round(item.scale),
|
|
246
|
+
"sample_size": item.sample_size,
|
|
247
|
+
"models": item.models,
|
|
248
|
+
}
|
|
249
|
+
for key, item in sorted(self.items.items())
|
|
250
|
+
},
|
|
251
|
+
"estimates": estimates,
|
|
252
|
+
"drivers": driver_rows,
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _round(value: float) -> float:
|
|
257
|
+
return round(float(value), 12)
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _sigmoid(value: float) -> float:
|
|
261
|
+
if value >= 0:
|
|
262
|
+
return 1 / (1 + math.exp(-min(value, 700)))
|
|
263
|
+
exp = math.exp(max(value, -700))
|
|
264
|
+
return exp / (1 + exp)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _logit(value: float) -> float:
|
|
268
|
+
value = min(max(value, 1e-4), 1 - 1e-4)
|
|
269
|
+
return math.log(value / (1 - value))
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _solve(matrix: list[list[float]], vector: list[float]) -> list[float]:
|
|
273
|
+
"""Solve a small positive-definite system with deterministic elimination."""
|
|
274
|
+
size = len(vector)
|
|
275
|
+
augmented = [list(row) + [vector[index]] for index, row in enumerate(matrix)]
|
|
276
|
+
for pivot in range(size):
|
|
277
|
+
row = max(range(pivot, size), key=lambda index: abs(augmented[index][pivot]))
|
|
278
|
+
augmented[pivot], augmented[row] = augmented[row], augmented[pivot]
|
|
279
|
+
divisor = augmented[pivot][pivot]
|
|
280
|
+
if abs(divisor) < 1e-12:
|
|
281
|
+
divisor = 1e-12
|
|
282
|
+
for column in range(pivot, size + 1):
|
|
283
|
+
augmented[pivot][column] /= divisor
|
|
284
|
+
for index in range(size):
|
|
285
|
+
if index == pivot:
|
|
286
|
+
continue
|
|
287
|
+
factor = augmented[index][pivot]
|
|
288
|
+
for column in range(pivot, size + 1):
|
|
289
|
+
augmented[index][column] -= factor * augmented[pivot][column]
|
|
290
|
+
return [augmented[index][-1] for index in range(size)]
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _inverse(matrix: list[list[float]]) -> list[list[float]]:
|
|
294
|
+
size = len(matrix)
|
|
295
|
+
columns = [
|
|
296
|
+
_solve(matrix, [1.0 if row == column else 0.0 for row in range(size)])
|
|
297
|
+
for column in range(size)
|
|
298
|
+
]
|
|
299
|
+
return [[columns[column][row] for column in range(size)] for row in range(size)]
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _recency(observed: date, as_of: date) -> float:
|
|
303
|
+
age = max(0, (as_of - observed).days)
|
|
304
|
+
return float(max(_MIN_RECENCY, 0.5 ** (age / _STATIC_HALF_LIFE_DAYS)))
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _item_ids(observations: Sequence[CapabilityObservation]) -> dict[int, str]:
|
|
308
|
+
versions: dict[str, set[str | None]] = defaultdict(set)
|
|
309
|
+
for row in observations:
|
|
310
|
+
versions[row.benchmark_id].add(row.version)
|
|
311
|
+
return {
|
|
312
|
+
index: row.benchmark_id
|
|
313
|
+
if len(versions[row.benchmark_id]) == 1
|
|
314
|
+
else f"{row.benchmark_id}@{row.version or 'unversioned'}"
|
|
315
|
+
for index, row in enumerate(observations)
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _loading(tags: Sequence[tuple[str, Directness]], proxy: float) -> dict[str, float]:
|
|
320
|
+
values = {
|
|
321
|
+
domain: 1.0 if directness == "direct" else proxy
|
|
322
|
+
for domain, directness in tags
|
|
323
|
+
}
|
|
324
|
+
total = sum(values.values()) or 1.0
|
|
325
|
+
return {"g": 1.0, **{domain: value / total for domain, value in values.items()}}
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def _theta(model: _ModelFit, tags: Sequence[tuple[str, Directness]], proxy: float) -> float:
|
|
329
|
+
coefficients = _loading(tags, proxy)
|
|
330
|
+
by_dimension = dict(zip(model.dimensions, model.values))
|
|
331
|
+
return sum(coefficient * by_dimension.get(dimension, 0.0)
|
|
332
|
+
for dimension, coefficient in coefficients.items())
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def _find_item(items: Mapping[str, ItemFit], benchmark: str,
|
|
336
|
+
version: str | None) -> ItemFit | None:
|
|
337
|
+
found = [item for item in items.values()
|
|
338
|
+
if item.benchmark_id == benchmark and (version is None or item.version == version)]
|
|
339
|
+
return found[0] if len(found) == 1 else None
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _prepare(
|
|
343
|
+
observations: Sequence[CapabilityObservation],
|
|
344
|
+
specs: Mapping[str, BenchmarkSpec],
|
|
345
|
+
as_of: date,
|
|
346
|
+
) -> tuple[dict[str, ItemFit], list[_Prepared]]:
|
|
347
|
+
ordered = sorted(observations, key=lambda row: (
|
|
348
|
+
row.benchmark_id, row.version or "", row.model_id, row.measured_by, row.record_id
|
|
349
|
+
))
|
|
350
|
+
ids = _item_ids(ordered)
|
|
351
|
+
grouped: dict[str, list[CapabilityObservation]] = defaultdict(list)
|
|
352
|
+
for index, row in enumerate(ordered):
|
|
353
|
+
if row.domains and math.isfinite(row.value):
|
|
354
|
+
grouped[ids[index]].append(row)
|
|
355
|
+
items: dict[str, ItemFit] = {}
|
|
356
|
+
prepared: list[_Prepared] = []
|
|
357
|
+
for item_id, rows in sorted(grouped.items()):
|
|
358
|
+
models = len({row.model_id for row in rows})
|
|
359
|
+
if models < _MIN_ITEM_MODELS:
|
|
360
|
+
continue
|
|
361
|
+
first = rows[0]
|
|
362
|
+
spec = specs.get(first.benchmark_id, BenchmarkSpec())
|
|
363
|
+
percent = (first.unit or "").casefold() in {"percent", "%", "percentage"}
|
|
364
|
+
baseline = 0.0
|
|
365
|
+
if (percent and spec.direction == "higher_is_better"
|
|
366
|
+
and spec.random_baseline is not None):
|
|
367
|
+
registered = spec.random_baseline
|
|
368
|
+
baseline = min(0.5, max(0.0, registered if registered <= 1 else registered / 100))
|
|
369
|
+
sample_size = min(3000, max(100, spec.sample_size or 300))
|
|
370
|
+
raw = [
|
|
371
|
+
(100 - row.value if percent else -row.value)
|
|
372
|
+
if spec.direction == "lower_is_better" else row.value
|
|
373
|
+
for row in rows
|
|
374
|
+
]
|
|
375
|
+
if percent:
|
|
376
|
+
center, scale = 0.0, 1.0
|
|
377
|
+
targets = [_logit((value / 100 - baseline) / max(1e-9, 1 - baseline))
|
|
378
|
+
for value in raw]
|
|
379
|
+
link: Literal["logistic", "linear"] = "logistic"
|
|
380
|
+
else:
|
|
381
|
+
center = sum(raw) / len(raw)
|
|
382
|
+
variance = sum((value - center) ** 2 for value in raw) / max(1, len(raw) - 1)
|
|
383
|
+
scale = math.sqrt(variance) or 1.0
|
|
384
|
+
targets = [(value - center) / scale for value in raw]
|
|
385
|
+
link = "linear"
|
|
386
|
+
intercept = sum(targets) / len(targets)
|
|
387
|
+
item = ItemFit(
|
|
388
|
+
id=item_id,
|
|
389
|
+
benchmark_id=first.benchmark_id,
|
|
390
|
+
version=first.version,
|
|
391
|
+
link=link,
|
|
392
|
+
direction=spec.direction,
|
|
393
|
+
domains=tuple(sorted(first.domains)),
|
|
394
|
+
discrimination=1.0,
|
|
395
|
+
intercept=intercept,
|
|
396
|
+
residual_sd=0.35,
|
|
397
|
+
random_baseline=baseline,
|
|
398
|
+
center=center,
|
|
399
|
+
scale=scale,
|
|
400
|
+
sample_size=sample_size,
|
|
401
|
+
models=models,
|
|
402
|
+
)
|
|
403
|
+
items[item_id] = item
|
|
404
|
+
for row, target in zip(rows, targets):
|
|
405
|
+
recency = _recency(row.date, as_of)
|
|
406
|
+
if link == "logistic":
|
|
407
|
+
oriented = 100 - row.value if spec.direction == "lower_is_better" else row.value
|
|
408
|
+
probability = min(max(oriented / 100, 1e-4), 1 - 1e-4)
|
|
409
|
+
slope = max(1e-4, (probability - baseline) * (1 - probability)
|
|
410
|
+
/ max(1e-9, 1 - baseline))
|
|
411
|
+
raw_variance = 0.04**2 + probability * (1 - probability) / sample_size
|
|
412
|
+
base_weight = recency * slope * slope / raw_variance
|
|
413
|
+
else:
|
|
414
|
+
base_weight = recency
|
|
415
|
+
prepared.append(_Prepared(row, item_id, target, base_weight, recency))
|
|
416
|
+
return items, prepared
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _domain_estimates(
|
|
420
|
+
items: Mapping[str, ItemFit],
|
|
421
|
+
rows: Sequence[_Prepared],
|
|
422
|
+
) -> tuple[
|
|
423
|
+
dict[tuple[str, str], CapabilityEstimate],
|
|
424
|
+
dict[tuple[str, str], tuple[EstimateDriver, ...]],
|
|
425
|
+
]:
|
|
426
|
+
"""Project every domain from its own tagged measurements.
|
|
427
|
+
|
|
428
|
+
The bifactor fit may use one item for a direct domain and, at a lower
|
|
429
|
+
loading, for a proxy domain. Its other factors must not then order another
|
|
430
|
+
domain. This projection uses only rows tagged to the requested domain.
|
|
431
|
+
Positive within-item scores make it monotone in each observed value, while
|
|
432
|
+
proxy loadings add less precision and therefore leave broader intervals.
|
|
433
|
+
"""
|
|
434
|
+
domains = {domain for item in items.values() for domain, _ in item.domains}
|
|
435
|
+
estimates: dict[tuple[str, str], CapabilityEstimate] = {}
|
|
436
|
+
drivers: dict[tuple[str, str], tuple[EstimateDriver, ...]] = {}
|
|
437
|
+
|
|
438
|
+
for domain in sorted(domains):
|
|
439
|
+
domain_rows = [
|
|
440
|
+
row
|
|
441
|
+
for row in rows
|
|
442
|
+
if domain in dict(items[row.item_id].domains)
|
|
443
|
+
]
|
|
444
|
+
by_item: dict[str, list[_Prepared]] = defaultdict(list)
|
|
445
|
+
for row in domain_rows:
|
|
446
|
+
by_item[row.item_id].append(row)
|
|
447
|
+
|
|
448
|
+
scores: dict[str, list[tuple[_Prepared, float, float]]] = defaultdict(list)
|
|
449
|
+
for item_id, item_rows in sorted(by_item.items()):
|
|
450
|
+
targets = [row.target for row in item_rows]
|
|
451
|
+
center = sum(targets) / len(targets)
|
|
452
|
+
spread = math.sqrt(
|
|
453
|
+
sum((value - center) ** 2 for value in targets)
|
|
454
|
+
/ max(1, len(targets) - 1)
|
|
455
|
+
) or 1.0
|
|
456
|
+
for row, value in zip(item_rows, targets):
|
|
457
|
+
z_score = (value - center) / spread
|
|
458
|
+
directness_loading = (
|
|
459
|
+
1.0
|
|
460
|
+
if dict(items[item_id].domains)[domain] == "direct"
|
|
461
|
+
else _PROJECTION_PROXY_LOADING
|
|
462
|
+
)
|
|
463
|
+
precision = directness_loading**2 * row.recency_weight
|
|
464
|
+
scores[row.observation.model_id].append((row, z_score, precision))
|
|
465
|
+
|
|
466
|
+
for model_id, model_rows in sorted(scores.items()):
|
|
467
|
+
precision = _DOMAIN_PRIOR_PRECISION + sum(
|
|
468
|
+
weight for _, _, weight in model_rows
|
|
469
|
+
)
|
|
470
|
+
mean = sum(weight * value for _, value, weight in model_rows) / precision
|
|
471
|
+
sd = math.sqrt(1 / precision)
|
|
472
|
+
estimates[(model_id, domain)] = CapabilityEstimate(
|
|
473
|
+
mean,
|
|
474
|
+
mean - _INTERVAL_Z * sd,
|
|
475
|
+
mean + _INTERVAL_Z * sd,
|
|
476
|
+
sd,
|
|
477
|
+
)
|
|
478
|
+
total = sum(weight for _, _, weight in model_rows) or 1.0
|
|
479
|
+
driver_rows = [
|
|
480
|
+
EstimateDriver(
|
|
481
|
+
row.observation.record_id,
|
|
482
|
+
row.observation.benchmark_id,
|
|
483
|
+
row.observation.version,
|
|
484
|
+
1.0
|
|
485
|
+
if dict(items[row.item_id].domains)[domain] == "direct"
|
|
486
|
+
else _PROJECTION_PROXY_LOADING,
|
|
487
|
+
weight / total,
|
|
488
|
+
row.recency_weight,
|
|
489
|
+
)
|
|
490
|
+
for row, _, weight in model_rows
|
|
491
|
+
]
|
|
492
|
+
driver_rows.sort(key=lambda driver: (-driver.weight, driver.record_id))
|
|
493
|
+
drivers[(model_id, domain)] = tuple(driver_rows)
|
|
494
|
+
|
|
495
|
+
return estimates, drivers
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def fit_capabilities(
|
|
499
|
+
observations: Iterable[CapabilityObservation],
|
|
500
|
+
benchmark_specs: Mapping[str, BenchmarkSpec],
|
|
501
|
+
*,
|
|
502
|
+
as_of: date,
|
|
503
|
+
sweeps: int = 60,
|
|
504
|
+
) -> CapabilityFit:
|
|
505
|
+
"""Fit all admitted evidence. Input order does not change the result."""
|
|
506
|
+
rows_in = tuple(observations)
|
|
507
|
+
items, rows = _prepare(rows_in, benchmark_specs, as_of)
|
|
508
|
+
by_model: dict[str, list[_Prepared]] = defaultdict(list)
|
|
509
|
+
by_item: dict[str, list[_Prepared]] = defaultdict(list)
|
|
510
|
+
domains = sorted({domain for row in rows for domain, _ in row.observation.domains})
|
|
511
|
+
for row in rows:
|
|
512
|
+
by_model[row.observation.model_id].append(row)
|
|
513
|
+
by_item[row.item_id].append(row)
|
|
514
|
+
models = {
|
|
515
|
+
model_id: _ModelFit(
|
|
516
|
+
model_id,
|
|
517
|
+
("g", *sorted({domain for row in model_rows
|
|
518
|
+
for domain, _ in row.observation.domains})),
|
|
519
|
+
[0.0] * (1 + len({domain for row in model_rows
|
|
520
|
+
for domain, _ in row.observation.domains})),
|
|
521
|
+
)
|
|
522
|
+
for model_id, model_rows in sorted(by_model.items())
|
|
523
|
+
}
|
|
524
|
+
source_offsets = {kind: 0.0 for kind in sorted({row.observation.measured_by for row in rows})}
|
|
525
|
+
proxy = 0.35
|
|
526
|
+
|
|
527
|
+
for _ in range(sweeps):
|
|
528
|
+
before = {model_id: tuple(model.values) for model_id, model in models.items()}
|
|
529
|
+
for model_id, model in models.items():
|
|
530
|
+
index = {dimension: position for position, dimension in enumerate(model.dimensions)}
|
|
531
|
+
size = len(index)
|
|
532
|
+
matrix = [[0.0] * size for _ in range(size)]
|
|
533
|
+
vector = [0.0] * size
|
|
534
|
+
for position, dimension in enumerate(model.dimensions):
|
|
535
|
+
matrix[position][position] = (_RIDGE_GENERAL if dimension == "g"
|
|
536
|
+
else _RIDGE_DOMAIN)
|
|
537
|
+
for row in by_model[model_id]:
|
|
538
|
+
item = items[row.item_id]
|
|
539
|
+
coefficients = _loading(item.domains, proxy)
|
|
540
|
+
design = [item.discrimination * coefficients.get(dimension, 0.0)
|
|
541
|
+
for dimension in model.dimensions]
|
|
542
|
+
target = (row.target - item.intercept
|
|
543
|
+
- source_offsets.get(row.observation.measured_by, 0.0))
|
|
544
|
+
weight = row.base_weight / max(0.02, item.residual_sd**2)
|
|
545
|
+
for i in range(size):
|
|
546
|
+
vector[i] += weight * design[i] * target
|
|
547
|
+
for j in range(size):
|
|
548
|
+
matrix[i][j] += weight * design[i] * design[j]
|
|
549
|
+
model.values = _solve(matrix, vector)
|
|
550
|
+
|
|
551
|
+
next_items = {}
|
|
552
|
+
for item_id, item in items.items():
|
|
553
|
+
item_rows = by_item[item_id]
|
|
554
|
+
xs = [_theta(models[row.observation.model_id], item.domains, proxy)
|
|
555
|
+
for row in item_rows]
|
|
556
|
+
ys = [row.target - source_offsets.get(row.observation.measured_by, 0.0)
|
|
557
|
+
for row in item_rows]
|
|
558
|
+
weights = [row.base_weight for row in item_rows]
|
|
559
|
+
sw = sum(weights) + _RIDGE_ITEM
|
|
560
|
+
sx = sum(weight * x for weight, x in zip(weights, xs))
|
|
561
|
+
sy = sum(weight * y for weight, y in zip(weights, ys))
|
|
562
|
+
sxx = sum(weight * x * x for weight, x in zip(weights, xs)) + _RIDGE_ITEM
|
|
563
|
+
sxy = sum(weight * x * y for weight, x, y in zip(weights, xs, ys)) + _RIDGE_ITEM
|
|
564
|
+
determinant = sw * sxx - sx * sx
|
|
565
|
+
if determinant > 1e-9:
|
|
566
|
+
discrimination = max(0.05, min(6.0, (sw * sxy - sx * sy) / determinant))
|
|
567
|
+
intercept = (sy - discrimination * sx) / sw
|
|
568
|
+
else:
|
|
569
|
+
discrimination, intercept = item.discrimination, item.intercept
|
|
570
|
+
residuals = [y - (intercept + discrimination * x) for x, y in zip(xs, ys)]
|
|
571
|
+
residual_sd = math.sqrt(
|
|
572
|
+
(sum(weight * residual * residual for weight, residual in zip(weights, residuals))
|
|
573
|
+
+ 10 * 0.35**2)
|
|
574
|
+
/ (sum(weights) + 10)
|
|
575
|
+
)
|
|
576
|
+
next_items[item_id] = ItemFit(
|
|
577
|
+
**{**item.__dict__, "discrimination": discrimination,
|
|
578
|
+
"intercept": intercept, "residual_sd": max(0.05, residual_sd)}
|
|
579
|
+
)
|
|
580
|
+
items = next_items
|
|
581
|
+
|
|
582
|
+
for kind in source_offsets:
|
|
583
|
+
if kind in {"independent", "independent_evaluator", "benchmark_author", "modelspec",
|
|
584
|
+
"outcome_protocol"}:
|
|
585
|
+
source_offsets[kind] = 0.0
|
|
586
|
+
continue
|
|
587
|
+
kind_rows = [row for row in rows if row.observation.measured_by == kind]
|
|
588
|
+
residuals = []
|
|
589
|
+
weights = []
|
|
590
|
+
for row in kind_rows:
|
|
591
|
+
item = items[row.item_id]
|
|
592
|
+
predicted = item.intercept + item.discrimination * _theta(
|
|
593
|
+
models[row.observation.model_id], item.domains, proxy
|
|
594
|
+
)
|
|
595
|
+
residuals.append(row.target - predicted)
|
|
596
|
+
weights.append(row.base_weight)
|
|
597
|
+
source_offsets[kind] = (
|
|
598
|
+
sum(weight * residual for weight, residual in zip(weights, residuals))
|
|
599
|
+
/ (sum(weights) + 4.0)
|
|
600
|
+
) if residuals else 0.0
|
|
601
|
+
|
|
602
|
+
proxy_rows = [row for row in rows
|
|
603
|
+
if any(directness == "proxy" for _, directness in row.observation.domains)]
|
|
604
|
+
if proxy_rows:
|
|
605
|
+
candidates = [0.1 + 0.05 * index for index in range(17)]
|
|
606
|
+
proxy = min(candidates, key=lambda candidate: sum(
|
|
607
|
+
row.base_weight * (
|
|
608
|
+
row.target
|
|
609
|
+
- items[row.item_id].intercept
|
|
610
|
+
- source_offsets.get(row.observation.measured_by, 0.0)
|
|
611
|
+
- items[row.item_id].discrimination * _theta(
|
|
612
|
+
models[row.observation.model_id], items[row.item_id].domains, candidate
|
|
613
|
+
)
|
|
614
|
+
) ** 2
|
|
615
|
+
for row in proxy_rows
|
|
616
|
+
) + 0.2 * (candidate - 0.35) ** 2)
|
|
617
|
+
moved = max((abs(value - prior[position])
|
|
618
|
+
for model_id, model in models.items()
|
|
619
|
+
for position, value in enumerate(model.values)
|
|
620
|
+
for prior in [before[model_id]]), default=0.0)
|
|
621
|
+
if moved < 1e-7:
|
|
622
|
+
break
|
|
623
|
+
|
|
624
|
+
general = [model.values[0] for model in models.values() if len(by_model[model.model_id]) >= 2]
|
|
625
|
+
mean = sum(general) / len(general) if general else 0.0
|
|
626
|
+
sd = math.sqrt(sum((value - mean) ** 2 for value in general) / len(general)) if general else 1.0
|
|
627
|
+
sd = sd or 1.0
|
|
628
|
+
for model in models.values():
|
|
629
|
+
model.values = [(value - mean if index == 0 else value) / sd
|
|
630
|
+
for index, value in enumerate(model.values)]
|
|
631
|
+
items = {
|
|
632
|
+
key: ItemFit(**{**item.__dict__,
|
|
633
|
+
"discrimination": item.discrimination * sd,
|
|
634
|
+
"intercept": item.intercept + item.discrimination * mean})
|
|
635
|
+
for key, item in items.items()
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
domain_sd = {}
|
|
639
|
+
for domain in domains:
|
|
640
|
+
values = [model.values[model.dimensions.index(domain)]
|
|
641
|
+
for model in models.values() if domain in model.dimensions]
|
|
642
|
+
spread = math.sqrt(sum(value * value for value in values) / len(values)) if values else 0.5
|
|
643
|
+
domain_sd[domain] = min(1.2, max(0.2, spread))
|
|
644
|
+
|
|
645
|
+
for model_id, model in models.items():
|
|
646
|
+
index = {dimension: position for position, dimension in enumerate(model.dimensions)}
|
|
647
|
+
size = len(index)
|
|
648
|
+
matrix = [[0.0] * size for _ in range(size)]
|
|
649
|
+
for position, dimension in enumerate(model.dimensions):
|
|
650
|
+
matrix[position][position] = (_RIDGE_GENERAL if dimension == "g"
|
|
651
|
+
else 1 / domain_sd.get(dimension, 0.5) ** 2)
|
|
652
|
+
for row in by_model[model_id]:
|
|
653
|
+
item = items[row.item_id]
|
|
654
|
+
coefficients = _loading(item.domains, proxy)
|
|
655
|
+
design = [item.discrimination * coefficients.get(dimension, 0.0)
|
|
656
|
+
for dimension in model.dimensions]
|
|
657
|
+
weight = row.base_weight / max(0.02, item.residual_sd**2)
|
|
658
|
+
for i in range(size):
|
|
659
|
+
for j in range(size):
|
|
660
|
+
matrix[i][j] += weight * design[i] * design[j]
|
|
661
|
+
model.covariance = _inverse(matrix)
|
|
662
|
+
|
|
663
|
+
drivers: dict[tuple[str, str], tuple[EstimateDriver, ...]] = {}
|
|
664
|
+
for model_id, model_rows in by_model.items():
|
|
665
|
+
for domain in domains:
|
|
666
|
+
driver_candidates: list[EstimateDriver] = []
|
|
667
|
+
for row in model_rows:
|
|
668
|
+
item = items[row.item_id]
|
|
669
|
+
coefficients = _loading(item.domains, proxy)
|
|
670
|
+
domain_loading = coefficients.get(domain, 0.0)
|
|
671
|
+
loading = item.discrimination * domain_loading
|
|
672
|
+
if loading <= 0:
|
|
673
|
+
continue
|
|
674
|
+
weight = row.base_weight * loading * loading / max(0.02, item.residual_sd**2)
|
|
675
|
+
driver_candidates.append(EstimateDriver(
|
|
676
|
+
row.observation.record_id,
|
|
677
|
+
row.observation.benchmark_id,
|
|
678
|
+
row.observation.version,
|
|
679
|
+
loading,
|
|
680
|
+
weight,
|
|
681
|
+
row.recency_weight,
|
|
682
|
+
))
|
|
683
|
+
total = sum(driver.weight for driver in driver_candidates) or 1.0
|
|
684
|
+
normalised = [EstimateDriver(
|
|
685
|
+
driver.record_id, driver.benchmark_id, driver.version, driver.loading,
|
|
686
|
+
driver.weight / total, driver.recency_weight,
|
|
687
|
+
) for driver in driver_candidates]
|
|
688
|
+
normalised.sort(key=lambda driver: (-driver.weight, driver.record_id))
|
|
689
|
+
drivers[(model_id, domain)] = tuple(normalised)
|
|
690
|
+
|
|
691
|
+
domain_estimates, domain_drivers = _domain_estimates(items, rows)
|
|
692
|
+
drivers.update(domain_drivers)
|
|
693
|
+
|
|
694
|
+
return CapabilityFit(
|
|
695
|
+
as_of=as_of,
|
|
696
|
+
items=items,
|
|
697
|
+
models=models,
|
|
698
|
+
domain_sd=domain_sd,
|
|
699
|
+
source_offsets=source_offsets,
|
|
700
|
+
directness=DirectnessFit(proxy_loading=proxy),
|
|
701
|
+
drivers=drivers,
|
|
702
|
+
domain_estimates=domain_estimates,
|
|
703
|
+
)
|
|
704
|
+
|
|
705
|
+
|
|
706
|
+
def _heldout_cells(
|
|
707
|
+
observations: Sequence[CapabilityObservation], holdout: float, seed: int,
|
|
708
|
+
) -> tuple[list[CapabilityObservation], list[CapabilityObservation]]:
|
|
709
|
+
cells: dict[tuple[str, str, str | None], list[CapabilityObservation]] = defaultdict(list)
|
|
710
|
+
for row in observations:
|
|
711
|
+
cells[(row.model_id, row.benchmark_id, row.version)].append(row)
|
|
712
|
+
keys = sorted(cells)
|
|
713
|
+
random.Random(seed).shuffle(keys)
|
|
714
|
+
per_model: dict[str, int] = defaultdict(int)
|
|
715
|
+
per_item: dict[tuple[str, str | None], int] = defaultdict(int)
|
|
716
|
+
for model_id, benchmark, version in keys:
|
|
717
|
+
per_model[model_id] += 1
|
|
718
|
+
per_item[(benchmark, version)] += 1
|
|
719
|
+
selected: set[tuple[str, str, str | None]] = set()
|
|
720
|
+
for key in keys:
|
|
721
|
+
if len(selected) >= int(len(keys) * holdout):
|
|
722
|
+
break
|
|
723
|
+
model_id, benchmark, version = key
|
|
724
|
+
if per_model[model_id] <= 2 or per_item[(benchmark, version)] <= 4:
|
|
725
|
+
continue
|
|
726
|
+
selected.add(key)
|
|
727
|
+
per_model[model_id] -= 1
|
|
728
|
+
per_item[(benchmark, version)] -= 1
|
|
729
|
+
train = [row for key, values in cells.items() if key not in selected for row in values]
|
|
730
|
+
held = [row for key, values in cells.items() if key in selected for row in values]
|
|
731
|
+
return train, held
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def backtest_capabilities(
|
|
735
|
+
observations: Sequence[CapabilityObservation],
|
|
736
|
+
benchmark_specs: Mapping[str, BenchmarkSpec],
|
|
737
|
+
*,
|
|
738
|
+
as_of: date,
|
|
739
|
+
seeds: Sequence[int] = (7, 19, 37, 53, 71),
|
|
740
|
+
holdout: float = 0.15,
|
|
741
|
+
) -> BacktestResult:
|
|
742
|
+
"""Hold out model-benchmark cells and compare with the benchmark mean."""
|
|
743
|
+
model_errors: list[float] = []
|
|
744
|
+
naive_errors: list[float] = []
|
|
745
|
+
for seed in seeds:
|
|
746
|
+
train, held = _heldout_cells(observations, holdout, seed)
|
|
747
|
+
fitted, naive = _prediction_errors(train, held, benchmark_specs, as_of)
|
|
748
|
+
model_errors.extend(fitted)
|
|
749
|
+
naive_errors.extend(naive)
|
|
750
|
+
if not model_errors:
|
|
751
|
+
return BacktestResult(0, math.inf, math.inf)
|
|
752
|
+
return BacktestResult(
|
|
753
|
+
len(model_errors),
|
|
754
|
+
math.sqrt(sum(model_errors) / len(model_errors)),
|
|
755
|
+
math.sqrt(sum(naive_errors) / len(naive_errors)),
|
|
756
|
+
)
|
|
757
|
+
|
|
758
|
+
|
|
759
|
+
def _prediction_errors(
|
|
760
|
+
train: Sequence[CapabilityObservation],
|
|
761
|
+
held: Sequence[CapabilityObservation],
|
|
762
|
+
benchmark_specs: Mapping[str, BenchmarkSpec],
|
|
763
|
+
as_of: date,
|
|
764
|
+
) -> tuple[list[float], list[float]]:
|
|
765
|
+
fit = fit_capabilities(train, benchmark_specs, as_of=as_of)
|
|
766
|
+
means: dict[tuple[str, str | None], float] = {}
|
|
767
|
+
for benchmark, version in sorted({(row.benchmark_id, row.version) for row in train}):
|
|
768
|
+
values = [
|
|
769
|
+
row.value
|
|
770
|
+
for row in train
|
|
771
|
+
if row.benchmark_id == benchmark and row.version == version
|
|
772
|
+
]
|
|
773
|
+
means[(benchmark, version)] = sum(values) / len(values)
|
|
774
|
+
model_errors: list[float] = []
|
|
775
|
+
naive_errors: list[float] = []
|
|
776
|
+
for row in held:
|
|
777
|
+
predicted = fit.predict(row.model_id, row.benchmark_id, row.version)
|
|
778
|
+
key = (row.benchmark_id, row.version)
|
|
779
|
+
if predicted is None or key not in means:
|
|
780
|
+
continue
|
|
781
|
+
peers = [
|
|
782
|
+
other.value
|
|
783
|
+
for other in train
|
|
784
|
+
if other.benchmark_id == row.benchmark_id and other.version == row.version
|
|
785
|
+
]
|
|
786
|
+
scale = max(
|
|
787
|
+
5.0,
|
|
788
|
+
math.sqrt(
|
|
789
|
+
sum((value - means[key]) ** 2 for value in peers) / max(1, len(peers) - 1)
|
|
790
|
+
),
|
|
791
|
+
)
|
|
792
|
+
model_errors.append(((predicted - row.value) / scale) ** 2)
|
|
793
|
+
naive_errors.append(((means[key] - row.value) / scale) ** 2)
|
|
794
|
+
return model_errors, naive_errors
|
|
795
|
+
|
|
796
|
+
|
|
797
|
+
def backtest_newest_capabilities(
|
|
798
|
+
observations: Sequence[CapabilityObservation],
|
|
799
|
+
benchmark_specs: Mapping[str, BenchmarkSpec],
|
|
800
|
+
*,
|
|
801
|
+
as_of: date,
|
|
802
|
+
) -> BacktestResult:
|
|
803
|
+
"""Hold out each model's newest eligible score and compare with an item mean."""
|
|
804
|
+
cells: dict[tuple[str, str, str | None], list[CapabilityObservation]] = defaultdict(list)
|
|
805
|
+
for row in observations:
|
|
806
|
+
cells[(row.model_id, row.benchmark_id, row.version)].append(row)
|
|
807
|
+
item_counts: dict[tuple[str, str | None], int] = defaultdict(int)
|
|
808
|
+
model_cells: dict[str, list[tuple[str, str, str | None]]] = defaultdict(list)
|
|
809
|
+
for key in sorted(cells):
|
|
810
|
+
model_cells[key[0]].append(key)
|
|
811
|
+
item_counts[(key[1], key[2])] += 1
|
|
812
|
+
selected: set[tuple[str, str, str | None]] = set()
|
|
813
|
+
for model_id, keys in sorted(model_cells.items()):
|
|
814
|
+
if len(keys) <= 2:
|
|
815
|
+
continue
|
|
816
|
+
newest = sorted(
|
|
817
|
+
keys,
|
|
818
|
+
key=lambda key: (
|
|
819
|
+
max(row.date for row in cells[key]),
|
|
820
|
+
key[1],
|
|
821
|
+
key[2] or "",
|
|
822
|
+
),
|
|
823
|
+
reverse=True,
|
|
824
|
+
)
|
|
825
|
+
for key in newest:
|
|
826
|
+
item = (key[1], key[2])
|
|
827
|
+
if item_counts[item] > 4:
|
|
828
|
+
selected.add(key)
|
|
829
|
+
item_counts[item] -= 1
|
|
830
|
+
break
|
|
831
|
+
train = [row for key, values in cells.items() if key not in selected for row in values]
|
|
832
|
+
held = [row for key, values in cells.items() if key in selected for row in values]
|
|
833
|
+
model_errors, naive_errors = _prediction_errors(train, held, benchmark_specs, as_of)
|
|
834
|
+
if not model_errors:
|
|
835
|
+
return BacktestResult(0, math.inf, math.inf)
|
|
836
|
+
return BacktestResult(
|
|
837
|
+
len(model_errors),
|
|
838
|
+
math.sqrt(sum(model_errors) / len(model_errors)),
|
|
839
|
+
math.sqrt(sum(naive_errors) / len(naive_errors)),
|
|
840
|
+
)
|
|
841
|
+
|
|
842
|
+
|
|
843
|
+
def deterministic_probabilities(
|
|
844
|
+
estimates: Mapping[str, CapabilityEstimate],
|
|
845
|
+
*,
|
|
846
|
+
seed_material: str,
|
|
847
|
+
samples: int = 256,
|
|
848
|
+
) -> dict[str, tuple[float, float]]:
|
|
849
|
+
"""Return P(best) and top-three stability with a snapshot-derived seed.
|
|
850
|
+
|
|
851
|
+
The page presents these probabilities to whole-percent precision. A 256
|
|
852
|
+
draw deterministic sample keeps that honesty while bounding the dominant
|
|
853
|
+
CPU loop on a cold Python Worker request.
|
|
854
|
+
"""
|
|
855
|
+
ordered = sorted(estimates)
|
|
856
|
+
if not ordered:
|
|
857
|
+
return {}
|
|
858
|
+
seed = int.from_bytes(hashlib.sha256(seed_material.encode()).digest()[:8], "big")
|
|
859
|
+
rng = random.Random(seed)
|
|
860
|
+
best = {model_id: 0 for model_id in ordered}
|
|
861
|
+
top3 = {model_id: 0 for model_id in ordered}
|
|
862
|
+
for _ in range(samples):
|
|
863
|
+
drawn = sorted(
|
|
864
|
+
((rng.gauss(estimates[model_id].value, estimates[model_id].sd), model_id)
|
|
865
|
+
for model_id in ordered),
|
|
866
|
+
key=lambda pair: (-pair[0], pair[1]),
|
|
867
|
+
)
|
|
868
|
+
best[drawn[0][1]] += 1
|
|
869
|
+
for _, model_id in drawn[:3]:
|
|
870
|
+
top3[model_id] += 1
|
|
871
|
+
return {model_id: (best[model_id] / samples, top3[model_id] / samples)
|
|
872
|
+
for model_id in ordered}
|