modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/capability.py ADDED
@@ -0,0 +1,872 @@
1
+ """Learn domain capability estimates from verified benchmark evidence.
2
+
3
+ The fit is a hierarchical bifactor item-response model. Each model has a
4
+ general factor and one deviation per observed domain. Each benchmark learns a
5
+ positive discrimination and an intercept. Percentage measurements use a
6
+ logistic link with the benchmark's registered random baseline. Other numeric
7
+ measurements use a standardised linear link.
8
+
9
+ The module has no benchmark IDs. Benchmark membership and directness come from
10
+ the registry tags carried into the snapshot builder. Missing cells add no
11
+ likelihood term. Old evidence loses precision, provider self-reports get a
12
+ learned offset, and a learned proxy loading keeps proxy tags weaker than direct
13
+ tags. The build stores posterior means, intervals, and explanation drivers.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import hashlib
19
+ import math
20
+ import random
21
+ from collections import defaultdict
22
+ from collections.abc import Iterable, Mapping, Sequence
23
+ from dataclasses import dataclass, field
24
+ from datetime import date
25
+ from typing import Literal
26
+
27
+ Direction = Literal["higher_is_better", "lower_is_better"]
28
+ Directness = Literal["direct", "proxy"]
29
+
30
+ _INTERVAL_Z = 1.2815515655446004 # central 80 percent interval
31
+ _STATIC_HALF_LIFE_DAYS = 365.0
32
+ _MIN_RECENCY = 0.25
33
+ _RIDGE_GENERAL = 1.0
34
+ _RIDGE_DOMAIN = 4.0
35
+ _RIDGE_ITEM = 0.25
36
+ # Two observations are the smallest sample that can establish an ordering and
37
+ # a non-zero within-item spread. Keeping those items lets sparse domains rank
38
+ # their directly measured frontier; the prior still makes their intervals
39
+ # appropriately wide.
40
+ _MIN_ITEM_MODELS = 2
41
+ _DOMAIN_PRIOR_PRECISION = 0.25
42
+ _PROJECTION_PROXY_LOADING = 0.35
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class BenchmarkSpec:
47
+ random_baseline: float | None = None
48
+ sample_size: int | None = None
49
+ direction: Direction = "higher_is_better"
50
+
51
+
52
+ @dataclass(frozen=True)
53
+ class CapabilityObservation:
54
+ model_id: str
55
+ benchmark_id: str
56
+ value: float
57
+ unit: str | None
58
+ measured_by: str
59
+ date: date
60
+ record_id: str
61
+ version: str | None
62
+ domains: tuple[tuple[str, Directness], ...]
63
+
64
+
65
+ @dataclass(frozen=True)
66
+ class CapabilityEstimate:
67
+ value: float
68
+ low: float
69
+ high: float
70
+ sd: float
71
+
72
+
73
+ @dataclass(frozen=True)
74
+ class EstimateDriver:
75
+ record_id: str
76
+ benchmark_id: str
77
+ version: str | None
78
+ loading: float
79
+ weight: float
80
+ recency_weight: float
81
+
82
+
83
+ @dataclass(frozen=True)
84
+ class DirectnessFit:
85
+ direct_loading: float = 1.0
86
+ proxy_loading: float = 0.35
87
+
88
+
89
+ @dataclass(frozen=True)
90
+ class ItemFit:
91
+ id: str
92
+ benchmark_id: str
93
+ version: str | None
94
+ link: Literal["logistic", "linear"]
95
+ direction: Direction
96
+ domains: tuple[tuple[str, Directness], ...]
97
+ discrimination: float
98
+ intercept: float
99
+ residual_sd: float
100
+ random_baseline: float
101
+ center: float
102
+ scale: float
103
+ sample_size: int
104
+ models: int
105
+
106
+ def information(self, ability: float) -> float:
107
+ """Fisher information on the raw measurement scale."""
108
+ if self.link == "logistic":
109
+ probability = self.random_baseline + (1 - self.random_baseline) * _sigmoid(
110
+ self.discrimination * ability + self.intercept
111
+ )
112
+ slope = self.discrimination * max(
113
+ 1e-9,
114
+ (probability - self.random_baseline)
115
+ * (1 - probability)
116
+ / max(1e-9, 1 - self.random_baseline),
117
+ )
118
+ variance = self.residual_sd**2 + max(
119
+ 1e-6, probability * (1 - probability) / self.sample_size
120
+ )
121
+ return slope * slope / variance
122
+ return self.discrimination**2 / max(1e-9, self.residual_sd**2)
123
+
124
+
125
+ @dataclass(frozen=True)
126
+ class BacktestResult:
127
+ cells: int
128
+ model_rmse: float
129
+ naive_rmse: float
130
+
131
+
132
+ @dataclass
133
+ class _ModelFit:
134
+ model_id: str
135
+ dimensions: tuple[str, ...]
136
+ values: list[float]
137
+ covariance: list[list[float]] = field(default_factory=list)
138
+
139
+
140
+ @dataclass
141
+ class _Prepared:
142
+ observation: CapabilityObservation
143
+ item_id: str
144
+ target: float
145
+ base_weight: float
146
+ recency_weight: float
147
+
148
+
149
+ @dataclass
150
+ class CapabilityFit:
151
+ as_of: date
152
+ items: dict[str, ItemFit]
153
+ models: dict[str, _ModelFit]
154
+ domain_sd: dict[str, float]
155
+ source_offsets: dict[str, float]
156
+ directness: DirectnessFit
157
+ drivers: dict[tuple[str, str], tuple[EstimateDriver, ...]]
158
+ domain_estimates: dict[tuple[str, str], CapabilityEstimate] = field(default_factory=dict)
159
+
160
+ def estimate(self, model_id: str, domain: str) -> CapabilityEstimate | None:
161
+ domain_estimate = self.domain_estimates.get((model_id, domain))
162
+ if domain_estimate is not None:
163
+ return domain_estimate
164
+ model = self.models.get(model_id)
165
+ if model is None or not model.covariance or domain not in model.dimensions:
166
+ return None
167
+ weights = [1.0 if dimension == "g" else 1.0 if dimension == domain else 0.0
168
+ for dimension in model.dimensions]
169
+ mean = sum(weight * value for weight, value in zip(weights, model.values))
170
+ variance = sum(
171
+ weights[i] * model.covariance[i][j] * weights[j]
172
+ for i in range(len(weights))
173
+ for j in range(len(weights))
174
+ )
175
+ sd = math.sqrt(max(variance, 1e-9))
176
+ return CapabilityEstimate(mean, mean - _INTERVAL_Z * sd, mean + _INTERVAL_Z * sd, sd)
177
+
178
+ def explain(self, model_id: str, domain: str, *, limit: int = 8) -> tuple[EstimateDriver, ...]:
179
+ return self.drivers.get((model_id, domain), ())[:limit]
180
+
181
+ def predict(self, model_id: str, benchmark_id: str, version: str | None = None) -> float | None:
182
+ item = _find_item(self.items, benchmark_id, version)
183
+ model = self.models.get(model_id)
184
+ if item is None or model is None:
185
+ return None
186
+ theta = _theta(model, item.domains, self.directness.proxy_loading)
187
+ linear = item.discrimination * theta + item.intercept
188
+ if item.link == "logistic":
189
+ p = item.random_baseline + (1 - item.random_baseline) * _sigmoid(linear)
190
+ value = 100 * p
191
+ else:
192
+ value = item.center + item.scale * linear
193
+ if item.direction == "lower_is_better":
194
+ return 100 - value if item.link == "logistic" else -value
195
+ return value
196
+
197
+ def to_payload(self, model_ids: Iterable[str] | None = None) -> dict[str, object]:
198
+ keep = set(self.models) if model_ids is None else set(model_ids)
199
+ domains = sorted(self.domain_sd)
200
+ estimates: dict[str, dict[str, list[float]]] = {}
201
+ driver_rows: dict[str, dict[str, list[list[object]]]] = {}
202
+ for model_id in sorted(keep & set(self.models)):
203
+ by_domain: dict[str, list[float]] = {}
204
+ by_driver: dict[str, list[list[object]]] = {}
205
+ for domain in domains:
206
+ estimate = self.estimate(model_id, domain)
207
+ if estimate is None:
208
+ continue
209
+ by_domain[domain] = [_round(estimate.value), _round(estimate.low),
210
+ _round(estimate.high), _round(estimate.sd)]
211
+ rows = self.explain(model_id, domain)
212
+ if rows:
213
+ by_driver[domain] = [
214
+ [row.record_id, row.benchmark_id, row.version,
215
+ _round(row.loading), _round(row.weight), _round(row.recency_weight)]
216
+ for row in rows
217
+ ]
218
+ if by_domain:
219
+ estimates[model_id] = by_domain
220
+ if by_driver:
221
+ driver_rows[model_id] = by_driver
222
+ return {
223
+ "method": "hierarchical-bifactor-irt",
224
+ "as_of": self.as_of.isoformat(),
225
+ "directness": {
226
+ "direct_loading": _round(self.directness.direct_loading),
227
+ "proxy_loading": _round(self.directness.proxy_loading),
228
+ },
229
+ "source_offsets": {key: _round(value) for key, value in sorted(
230
+ self.source_offsets.items()
231
+ )},
232
+ "domain_sd": {key: _round(value) for key, value in sorted(self.domain_sd.items())},
233
+ "items": {
234
+ key: {
235
+ "benchmark": item.benchmark_id,
236
+ "version": item.version,
237
+ "link": item.link,
238
+ "direction": item.direction,
239
+ "domains": [list(tag) for tag in item.domains],
240
+ "discrimination": _round(item.discrimination),
241
+ "intercept": _round(item.intercept),
242
+ "residual_sd": _round(item.residual_sd),
243
+ "random_baseline": _round(item.random_baseline),
244
+ "center": _round(item.center),
245
+ "scale": _round(item.scale),
246
+ "sample_size": item.sample_size,
247
+ "models": item.models,
248
+ }
249
+ for key, item in sorted(self.items.items())
250
+ },
251
+ "estimates": estimates,
252
+ "drivers": driver_rows,
253
+ }
254
+
255
+
256
+ def _round(value: float) -> float:
257
+ return round(float(value), 12)
258
+
259
+
260
+ def _sigmoid(value: float) -> float:
261
+ if value >= 0:
262
+ return 1 / (1 + math.exp(-min(value, 700)))
263
+ exp = math.exp(max(value, -700))
264
+ return exp / (1 + exp)
265
+
266
+
267
+ def _logit(value: float) -> float:
268
+ value = min(max(value, 1e-4), 1 - 1e-4)
269
+ return math.log(value / (1 - value))
270
+
271
+
272
+ def _solve(matrix: list[list[float]], vector: list[float]) -> list[float]:
273
+ """Solve a small positive-definite system with deterministic elimination."""
274
+ size = len(vector)
275
+ augmented = [list(row) + [vector[index]] for index, row in enumerate(matrix)]
276
+ for pivot in range(size):
277
+ row = max(range(pivot, size), key=lambda index: abs(augmented[index][pivot]))
278
+ augmented[pivot], augmented[row] = augmented[row], augmented[pivot]
279
+ divisor = augmented[pivot][pivot]
280
+ if abs(divisor) < 1e-12:
281
+ divisor = 1e-12
282
+ for column in range(pivot, size + 1):
283
+ augmented[pivot][column] /= divisor
284
+ for index in range(size):
285
+ if index == pivot:
286
+ continue
287
+ factor = augmented[index][pivot]
288
+ for column in range(pivot, size + 1):
289
+ augmented[index][column] -= factor * augmented[pivot][column]
290
+ return [augmented[index][-1] for index in range(size)]
291
+
292
+
293
+ def _inverse(matrix: list[list[float]]) -> list[list[float]]:
294
+ size = len(matrix)
295
+ columns = [
296
+ _solve(matrix, [1.0 if row == column else 0.0 for row in range(size)])
297
+ for column in range(size)
298
+ ]
299
+ return [[columns[column][row] for column in range(size)] for row in range(size)]
300
+
301
+
302
+ def _recency(observed: date, as_of: date) -> float:
303
+ age = max(0, (as_of - observed).days)
304
+ return float(max(_MIN_RECENCY, 0.5 ** (age / _STATIC_HALF_LIFE_DAYS)))
305
+
306
+
307
+ def _item_ids(observations: Sequence[CapabilityObservation]) -> dict[int, str]:
308
+ versions: dict[str, set[str | None]] = defaultdict(set)
309
+ for row in observations:
310
+ versions[row.benchmark_id].add(row.version)
311
+ return {
312
+ index: row.benchmark_id
313
+ if len(versions[row.benchmark_id]) == 1
314
+ else f"{row.benchmark_id}@{row.version or 'unversioned'}"
315
+ for index, row in enumerate(observations)
316
+ }
317
+
318
+
319
+ def _loading(tags: Sequence[tuple[str, Directness]], proxy: float) -> dict[str, float]:
320
+ values = {
321
+ domain: 1.0 if directness == "direct" else proxy
322
+ for domain, directness in tags
323
+ }
324
+ total = sum(values.values()) or 1.0
325
+ return {"g": 1.0, **{domain: value / total for domain, value in values.items()}}
326
+
327
+
328
+ def _theta(model: _ModelFit, tags: Sequence[tuple[str, Directness]], proxy: float) -> float:
329
+ coefficients = _loading(tags, proxy)
330
+ by_dimension = dict(zip(model.dimensions, model.values))
331
+ return sum(coefficient * by_dimension.get(dimension, 0.0)
332
+ for dimension, coefficient in coefficients.items())
333
+
334
+
335
+ def _find_item(items: Mapping[str, ItemFit], benchmark: str,
336
+ version: str | None) -> ItemFit | None:
337
+ found = [item for item in items.values()
338
+ if item.benchmark_id == benchmark and (version is None or item.version == version)]
339
+ return found[0] if len(found) == 1 else None
340
+
341
+
342
+ def _prepare(
343
+ observations: Sequence[CapabilityObservation],
344
+ specs: Mapping[str, BenchmarkSpec],
345
+ as_of: date,
346
+ ) -> tuple[dict[str, ItemFit], list[_Prepared]]:
347
+ ordered = sorted(observations, key=lambda row: (
348
+ row.benchmark_id, row.version or "", row.model_id, row.measured_by, row.record_id
349
+ ))
350
+ ids = _item_ids(ordered)
351
+ grouped: dict[str, list[CapabilityObservation]] = defaultdict(list)
352
+ for index, row in enumerate(ordered):
353
+ if row.domains and math.isfinite(row.value):
354
+ grouped[ids[index]].append(row)
355
+ items: dict[str, ItemFit] = {}
356
+ prepared: list[_Prepared] = []
357
+ for item_id, rows in sorted(grouped.items()):
358
+ models = len({row.model_id for row in rows})
359
+ if models < _MIN_ITEM_MODELS:
360
+ continue
361
+ first = rows[0]
362
+ spec = specs.get(first.benchmark_id, BenchmarkSpec())
363
+ percent = (first.unit or "").casefold() in {"percent", "%", "percentage"}
364
+ baseline = 0.0
365
+ if (percent and spec.direction == "higher_is_better"
366
+ and spec.random_baseline is not None):
367
+ registered = spec.random_baseline
368
+ baseline = min(0.5, max(0.0, registered if registered <= 1 else registered / 100))
369
+ sample_size = min(3000, max(100, spec.sample_size or 300))
370
+ raw = [
371
+ (100 - row.value if percent else -row.value)
372
+ if spec.direction == "lower_is_better" else row.value
373
+ for row in rows
374
+ ]
375
+ if percent:
376
+ center, scale = 0.0, 1.0
377
+ targets = [_logit((value / 100 - baseline) / max(1e-9, 1 - baseline))
378
+ for value in raw]
379
+ link: Literal["logistic", "linear"] = "logistic"
380
+ else:
381
+ center = sum(raw) / len(raw)
382
+ variance = sum((value - center) ** 2 for value in raw) / max(1, len(raw) - 1)
383
+ scale = math.sqrt(variance) or 1.0
384
+ targets = [(value - center) / scale for value in raw]
385
+ link = "linear"
386
+ intercept = sum(targets) / len(targets)
387
+ item = ItemFit(
388
+ id=item_id,
389
+ benchmark_id=first.benchmark_id,
390
+ version=first.version,
391
+ link=link,
392
+ direction=spec.direction,
393
+ domains=tuple(sorted(first.domains)),
394
+ discrimination=1.0,
395
+ intercept=intercept,
396
+ residual_sd=0.35,
397
+ random_baseline=baseline,
398
+ center=center,
399
+ scale=scale,
400
+ sample_size=sample_size,
401
+ models=models,
402
+ )
403
+ items[item_id] = item
404
+ for row, target in zip(rows, targets):
405
+ recency = _recency(row.date, as_of)
406
+ if link == "logistic":
407
+ oriented = 100 - row.value if spec.direction == "lower_is_better" else row.value
408
+ probability = min(max(oriented / 100, 1e-4), 1 - 1e-4)
409
+ slope = max(1e-4, (probability - baseline) * (1 - probability)
410
+ / max(1e-9, 1 - baseline))
411
+ raw_variance = 0.04**2 + probability * (1 - probability) / sample_size
412
+ base_weight = recency * slope * slope / raw_variance
413
+ else:
414
+ base_weight = recency
415
+ prepared.append(_Prepared(row, item_id, target, base_weight, recency))
416
+ return items, prepared
417
+
418
+
419
+ def _domain_estimates(
420
+ items: Mapping[str, ItemFit],
421
+ rows: Sequence[_Prepared],
422
+ ) -> tuple[
423
+ dict[tuple[str, str], CapabilityEstimate],
424
+ dict[tuple[str, str], tuple[EstimateDriver, ...]],
425
+ ]:
426
+ """Project every domain from its own tagged measurements.
427
+
428
+ The bifactor fit may use one item for a direct domain and, at a lower
429
+ loading, for a proxy domain. Its other factors must not then order another
430
+ domain. This projection uses only rows tagged to the requested domain.
431
+ Positive within-item scores make it monotone in each observed value, while
432
+ proxy loadings add less precision and therefore leave broader intervals.
433
+ """
434
+ domains = {domain for item in items.values() for domain, _ in item.domains}
435
+ estimates: dict[tuple[str, str], CapabilityEstimate] = {}
436
+ drivers: dict[tuple[str, str], tuple[EstimateDriver, ...]] = {}
437
+
438
+ for domain in sorted(domains):
439
+ domain_rows = [
440
+ row
441
+ for row in rows
442
+ if domain in dict(items[row.item_id].domains)
443
+ ]
444
+ by_item: dict[str, list[_Prepared]] = defaultdict(list)
445
+ for row in domain_rows:
446
+ by_item[row.item_id].append(row)
447
+
448
+ scores: dict[str, list[tuple[_Prepared, float, float]]] = defaultdict(list)
449
+ for item_id, item_rows in sorted(by_item.items()):
450
+ targets = [row.target for row in item_rows]
451
+ center = sum(targets) / len(targets)
452
+ spread = math.sqrt(
453
+ sum((value - center) ** 2 for value in targets)
454
+ / max(1, len(targets) - 1)
455
+ ) or 1.0
456
+ for row, value in zip(item_rows, targets):
457
+ z_score = (value - center) / spread
458
+ directness_loading = (
459
+ 1.0
460
+ if dict(items[item_id].domains)[domain] == "direct"
461
+ else _PROJECTION_PROXY_LOADING
462
+ )
463
+ precision = directness_loading**2 * row.recency_weight
464
+ scores[row.observation.model_id].append((row, z_score, precision))
465
+
466
+ for model_id, model_rows in sorted(scores.items()):
467
+ precision = _DOMAIN_PRIOR_PRECISION + sum(
468
+ weight for _, _, weight in model_rows
469
+ )
470
+ mean = sum(weight * value for _, value, weight in model_rows) / precision
471
+ sd = math.sqrt(1 / precision)
472
+ estimates[(model_id, domain)] = CapabilityEstimate(
473
+ mean,
474
+ mean - _INTERVAL_Z * sd,
475
+ mean + _INTERVAL_Z * sd,
476
+ sd,
477
+ )
478
+ total = sum(weight for _, _, weight in model_rows) or 1.0
479
+ driver_rows = [
480
+ EstimateDriver(
481
+ row.observation.record_id,
482
+ row.observation.benchmark_id,
483
+ row.observation.version,
484
+ 1.0
485
+ if dict(items[row.item_id].domains)[domain] == "direct"
486
+ else _PROJECTION_PROXY_LOADING,
487
+ weight / total,
488
+ row.recency_weight,
489
+ )
490
+ for row, _, weight in model_rows
491
+ ]
492
+ driver_rows.sort(key=lambda driver: (-driver.weight, driver.record_id))
493
+ drivers[(model_id, domain)] = tuple(driver_rows)
494
+
495
+ return estimates, drivers
496
+
497
+
498
+ def fit_capabilities(
499
+ observations: Iterable[CapabilityObservation],
500
+ benchmark_specs: Mapping[str, BenchmarkSpec],
501
+ *,
502
+ as_of: date,
503
+ sweeps: int = 60,
504
+ ) -> CapabilityFit:
505
+ """Fit all admitted evidence. Input order does not change the result."""
506
+ rows_in = tuple(observations)
507
+ items, rows = _prepare(rows_in, benchmark_specs, as_of)
508
+ by_model: dict[str, list[_Prepared]] = defaultdict(list)
509
+ by_item: dict[str, list[_Prepared]] = defaultdict(list)
510
+ domains = sorted({domain for row in rows for domain, _ in row.observation.domains})
511
+ for row in rows:
512
+ by_model[row.observation.model_id].append(row)
513
+ by_item[row.item_id].append(row)
514
+ models = {
515
+ model_id: _ModelFit(
516
+ model_id,
517
+ ("g", *sorted({domain for row in model_rows
518
+ for domain, _ in row.observation.domains})),
519
+ [0.0] * (1 + len({domain for row in model_rows
520
+ for domain, _ in row.observation.domains})),
521
+ )
522
+ for model_id, model_rows in sorted(by_model.items())
523
+ }
524
+ source_offsets = {kind: 0.0 for kind in sorted({row.observation.measured_by for row in rows})}
525
+ proxy = 0.35
526
+
527
+ for _ in range(sweeps):
528
+ before = {model_id: tuple(model.values) for model_id, model in models.items()}
529
+ for model_id, model in models.items():
530
+ index = {dimension: position for position, dimension in enumerate(model.dimensions)}
531
+ size = len(index)
532
+ matrix = [[0.0] * size for _ in range(size)]
533
+ vector = [0.0] * size
534
+ for position, dimension in enumerate(model.dimensions):
535
+ matrix[position][position] = (_RIDGE_GENERAL if dimension == "g"
536
+ else _RIDGE_DOMAIN)
537
+ for row in by_model[model_id]:
538
+ item = items[row.item_id]
539
+ coefficients = _loading(item.domains, proxy)
540
+ design = [item.discrimination * coefficients.get(dimension, 0.0)
541
+ for dimension in model.dimensions]
542
+ target = (row.target - item.intercept
543
+ - source_offsets.get(row.observation.measured_by, 0.0))
544
+ weight = row.base_weight / max(0.02, item.residual_sd**2)
545
+ for i in range(size):
546
+ vector[i] += weight * design[i] * target
547
+ for j in range(size):
548
+ matrix[i][j] += weight * design[i] * design[j]
549
+ model.values = _solve(matrix, vector)
550
+
551
+ next_items = {}
552
+ for item_id, item in items.items():
553
+ item_rows = by_item[item_id]
554
+ xs = [_theta(models[row.observation.model_id], item.domains, proxy)
555
+ for row in item_rows]
556
+ ys = [row.target - source_offsets.get(row.observation.measured_by, 0.0)
557
+ for row in item_rows]
558
+ weights = [row.base_weight for row in item_rows]
559
+ sw = sum(weights) + _RIDGE_ITEM
560
+ sx = sum(weight * x for weight, x in zip(weights, xs))
561
+ sy = sum(weight * y for weight, y in zip(weights, ys))
562
+ sxx = sum(weight * x * x for weight, x in zip(weights, xs)) + _RIDGE_ITEM
563
+ sxy = sum(weight * x * y for weight, x, y in zip(weights, xs, ys)) + _RIDGE_ITEM
564
+ determinant = sw * sxx - sx * sx
565
+ if determinant > 1e-9:
566
+ discrimination = max(0.05, min(6.0, (sw * sxy - sx * sy) / determinant))
567
+ intercept = (sy - discrimination * sx) / sw
568
+ else:
569
+ discrimination, intercept = item.discrimination, item.intercept
570
+ residuals = [y - (intercept + discrimination * x) for x, y in zip(xs, ys)]
571
+ residual_sd = math.sqrt(
572
+ (sum(weight * residual * residual for weight, residual in zip(weights, residuals))
573
+ + 10 * 0.35**2)
574
+ / (sum(weights) + 10)
575
+ )
576
+ next_items[item_id] = ItemFit(
577
+ **{**item.__dict__, "discrimination": discrimination,
578
+ "intercept": intercept, "residual_sd": max(0.05, residual_sd)}
579
+ )
580
+ items = next_items
581
+
582
+ for kind in source_offsets:
583
+ if kind in {"independent", "independent_evaluator", "benchmark_author", "modelspec",
584
+ "outcome_protocol"}:
585
+ source_offsets[kind] = 0.0
586
+ continue
587
+ kind_rows = [row for row in rows if row.observation.measured_by == kind]
588
+ residuals = []
589
+ weights = []
590
+ for row in kind_rows:
591
+ item = items[row.item_id]
592
+ predicted = item.intercept + item.discrimination * _theta(
593
+ models[row.observation.model_id], item.domains, proxy
594
+ )
595
+ residuals.append(row.target - predicted)
596
+ weights.append(row.base_weight)
597
+ source_offsets[kind] = (
598
+ sum(weight * residual for weight, residual in zip(weights, residuals))
599
+ / (sum(weights) + 4.0)
600
+ ) if residuals else 0.0
601
+
602
+ proxy_rows = [row for row in rows
603
+ if any(directness == "proxy" for _, directness in row.observation.domains)]
604
+ if proxy_rows:
605
+ candidates = [0.1 + 0.05 * index for index in range(17)]
606
+ proxy = min(candidates, key=lambda candidate: sum(
607
+ row.base_weight * (
608
+ row.target
609
+ - items[row.item_id].intercept
610
+ - source_offsets.get(row.observation.measured_by, 0.0)
611
+ - items[row.item_id].discrimination * _theta(
612
+ models[row.observation.model_id], items[row.item_id].domains, candidate
613
+ )
614
+ ) ** 2
615
+ for row in proxy_rows
616
+ ) + 0.2 * (candidate - 0.35) ** 2)
617
+ moved = max((abs(value - prior[position])
618
+ for model_id, model in models.items()
619
+ for position, value in enumerate(model.values)
620
+ for prior in [before[model_id]]), default=0.0)
621
+ if moved < 1e-7:
622
+ break
623
+
624
+ general = [model.values[0] for model in models.values() if len(by_model[model.model_id]) >= 2]
625
+ mean = sum(general) / len(general) if general else 0.0
626
+ sd = math.sqrt(sum((value - mean) ** 2 for value in general) / len(general)) if general else 1.0
627
+ sd = sd or 1.0
628
+ for model in models.values():
629
+ model.values = [(value - mean if index == 0 else value) / sd
630
+ for index, value in enumerate(model.values)]
631
+ items = {
632
+ key: ItemFit(**{**item.__dict__,
633
+ "discrimination": item.discrimination * sd,
634
+ "intercept": item.intercept + item.discrimination * mean})
635
+ for key, item in items.items()
636
+ }
637
+
638
+ domain_sd = {}
639
+ for domain in domains:
640
+ values = [model.values[model.dimensions.index(domain)]
641
+ for model in models.values() if domain in model.dimensions]
642
+ spread = math.sqrt(sum(value * value for value in values) / len(values)) if values else 0.5
643
+ domain_sd[domain] = min(1.2, max(0.2, spread))
644
+
645
+ for model_id, model in models.items():
646
+ index = {dimension: position for position, dimension in enumerate(model.dimensions)}
647
+ size = len(index)
648
+ matrix = [[0.0] * size for _ in range(size)]
649
+ for position, dimension in enumerate(model.dimensions):
650
+ matrix[position][position] = (_RIDGE_GENERAL if dimension == "g"
651
+ else 1 / domain_sd.get(dimension, 0.5) ** 2)
652
+ for row in by_model[model_id]:
653
+ item = items[row.item_id]
654
+ coefficients = _loading(item.domains, proxy)
655
+ design = [item.discrimination * coefficients.get(dimension, 0.0)
656
+ for dimension in model.dimensions]
657
+ weight = row.base_weight / max(0.02, item.residual_sd**2)
658
+ for i in range(size):
659
+ for j in range(size):
660
+ matrix[i][j] += weight * design[i] * design[j]
661
+ model.covariance = _inverse(matrix)
662
+
663
+ drivers: dict[tuple[str, str], tuple[EstimateDriver, ...]] = {}
664
+ for model_id, model_rows in by_model.items():
665
+ for domain in domains:
666
+ driver_candidates: list[EstimateDriver] = []
667
+ for row in model_rows:
668
+ item = items[row.item_id]
669
+ coefficients = _loading(item.domains, proxy)
670
+ domain_loading = coefficients.get(domain, 0.0)
671
+ loading = item.discrimination * domain_loading
672
+ if loading <= 0:
673
+ continue
674
+ weight = row.base_weight * loading * loading / max(0.02, item.residual_sd**2)
675
+ driver_candidates.append(EstimateDriver(
676
+ row.observation.record_id,
677
+ row.observation.benchmark_id,
678
+ row.observation.version,
679
+ loading,
680
+ weight,
681
+ row.recency_weight,
682
+ ))
683
+ total = sum(driver.weight for driver in driver_candidates) or 1.0
684
+ normalised = [EstimateDriver(
685
+ driver.record_id, driver.benchmark_id, driver.version, driver.loading,
686
+ driver.weight / total, driver.recency_weight,
687
+ ) for driver in driver_candidates]
688
+ normalised.sort(key=lambda driver: (-driver.weight, driver.record_id))
689
+ drivers[(model_id, domain)] = tuple(normalised)
690
+
691
+ domain_estimates, domain_drivers = _domain_estimates(items, rows)
692
+ drivers.update(domain_drivers)
693
+
694
+ return CapabilityFit(
695
+ as_of=as_of,
696
+ items=items,
697
+ models=models,
698
+ domain_sd=domain_sd,
699
+ source_offsets=source_offsets,
700
+ directness=DirectnessFit(proxy_loading=proxy),
701
+ drivers=drivers,
702
+ domain_estimates=domain_estimates,
703
+ )
704
+
705
+
706
+ def _heldout_cells(
707
+ observations: Sequence[CapabilityObservation], holdout: float, seed: int,
708
+ ) -> tuple[list[CapabilityObservation], list[CapabilityObservation]]:
709
+ cells: dict[tuple[str, str, str | None], list[CapabilityObservation]] = defaultdict(list)
710
+ for row in observations:
711
+ cells[(row.model_id, row.benchmark_id, row.version)].append(row)
712
+ keys = sorted(cells)
713
+ random.Random(seed).shuffle(keys)
714
+ per_model: dict[str, int] = defaultdict(int)
715
+ per_item: dict[tuple[str, str | None], int] = defaultdict(int)
716
+ for model_id, benchmark, version in keys:
717
+ per_model[model_id] += 1
718
+ per_item[(benchmark, version)] += 1
719
+ selected: set[tuple[str, str, str | None]] = set()
720
+ for key in keys:
721
+ if len(selected) >= int(len(keys) * holdout):
722
+ break
723
+ model_id, benchmark, version = key
724
+ if per_model[model_id] <= 2 or per_item[(benchmark, version)] <= 4:
725
+ continue
726
+ selected.add(key)
727
+ per_model[model_id] -= 1
728
+ per_item[(benchmark, version)] -= 1
729
+ train = [row for key, values in cells.items() if key not in selected for row in values]
730
+ held = [row for key, values in cells.items() if key in selected for row in values]
731
+ return train, held
732
+
733
+
734
+ def backtest_capabilities(
735
+ observations: Sequence[CapabilityObservation],
736
+ benchmark_specs: Mapping[str, BenchmarkSpec],
737
+ *,
738
+ as_of: date,
739
+ seeds: Sequence[int] = (7, 19, 37, 53, 71),
740
+ holdout: float = 0.15,
741
+ ) -> BacktestResult:
742
+ """Hold out model-benchmark cells and compare with the benchmark mean."""
743
+ model_errors: list[float] = []
744
+ naive_errors: list[float] = []
745
+ for seed in seeds:
746
+ train, held = _heldout_cells(observations, holdout, seed)
747
+ fitted, naive = _prediction_errors(train, held, benchmark_specs, as_of)
748
+ model_errors.extend(fitted)
749
+ naive_errors.extend(naive)
750
+ if not model_errors:
751
+ return BacktestResult(0, math.inf, math.inf)
752
+ return BacktestResult(
753
+ len(model_errors),
754
+ math.sqrt(sum(model_errors) / len(model_errors)),
755
+ math.sqrt(sum(naive_errors) / len(naive_errors)),
756
+ )
757
+
758
+
759
+ def _prediction_errors(
760
+ train: Sequence[CapabilityObservation],
761
+ held: Sequence[CapabilityObservation],
762
+ benchmark_specs: Mapping[str, BenchmarkSpec],
763
+ as_of: date,
764
+ ) -> tuple[list[float], list[float]]:
765
+ fit = fit_capabilities(train, benchmark_specs, as_of=as_of)
766
+ means: dict[tuple[str, str | None], float] = {}
767
+ for benchmark, version in sorted({(row.benchmark_id, row.version) for row in train}):
768
+ values = [
769
+ row.value
770
+ for row in train
771
+ if row.benchmark_id == benchmark and row.version == version
772
+ ]
773
+ means[(benchmark, version)] = sum(values) / len(values)
774
+ model_errors: list[float] = []
775
+ naive_errors: list[float] = []
776
+ for row in held:
777
+ predicted = fit.predict(row.model_id, row.benchmark_id, row.version)
778
+ key = (row.benchmark_id, row.version)
779
+ if predicted is None or key not in means:
780
+ continue
781
+ peers = [
782
+ other.value
783
+ for other in train
784
+ if other.benchmark_id == row.benchmark_id and other.version == row.version
785
+ ]
786
+ scale = max(
787
+ 5.0,
788
+ math.sqrt(
789
+ sum((value - means[key]) ** 2 for value in peers) / max(1, len(peers) - 1)
790
+ ),
791
+ )
792
+ model_errors.append(((predicted - row.value) / scale) ** 2)
793
+ naive_errors.append(((means[key] - row.value) / scale) ** 2)
794
+ return model_errors, naive_errors
795
+
796
+
797
+ def backtest_newest_capabilities(
798
+ observations: Sequence[CapabilityObservation],
799
+ benchmark_specs: Mapping[str, BenchmarkSpec],
800
+ *,
801
+ as_of: date,
802
+ ) -> BacktestResult:
803
+ """Hold out each model's newest eligible score and compare with an item mean."""
804
+ cells: dict[tuple[str, str, str | None], list[CapabilityObservation]] = defaultdict(list)
805
+ for row in observations:
806
+ cells[(row.model_id, row.benchmark_id, row.version)].append(row)
807
+ item_counts: dict[tuple[str, str | None], int] = defaultdict(int)
808
+ model_cells: dict[str, list[tuple[str, str, str | None]]] = defaultdict(list)
809
+ for key in sorted(cells):
810
+ model_cells[key[0]].append(key)
811
+ item_counts[(key[1], key[2])] += 1
812
+ selected: set[tuple[str, str, str | None]] = set()
813
+ for model_id, keys in sorted(model_cells.items()):
814
+ if len(keys) <= 2:
815
+ continue
816
+ newest = sorted(
817
+ keys,
818
+ key=lambda key: (
819
+ max(row.date for row in cells[key]),
820
+ key[1],
821
+ key[2] or "",
822
+ ),
823
+ reverse=True,
824
+ )
825
+ for key in newest:
826
+ item = (key[1], key[2])
827
+ if item_counts[item] > 4:
828
+ selected.add(key)
829
+ item_counts[item] -= 1
830
+ break
831
+ train = [row for key, values in cells.items() if key not in selected for row in values]
832
+ held = [row for key, values in cells.items() if key in selected for row in values]
833
+ model_errors, naive_errors = _prediction_errors(train, held, benchmark_specs, as_of)
834
+ if not model_errors:
835
+ return BacktestResult(0, math.inf, math.inf)
836
+ return BacktestResult(
837
+ len(model_errors),
838
+ math.sqrt(sum(model_errors) / len(model_errors)),
839
+ math.sqrt(sum(naive_errors) / len(naive_errors)),
840
+ )
841
+
842
+
843
+ def deterministic_probabilities(
844
+ estimates: Mapping[str, CapabilityEstimate],
845
+ *,
846
+ seed_material: str,
847
+ samples: int = 256,
848
+ ) -> dict[str, tuple[float, float]]:
849
+ """Return P(best) and top-three stability with a snapshot-derived seed.
850
+
851
+ The page presents these probabilities to whole-percent precision. A 256
852
+ draw deterministic sample keeps that honesty while bounding the dominant
853
+ CPU loop on a cold Python Worker request.
854
+ """
855
+ ordered = sorted(estimates)
856
+ if not ordered:
857
+ return {}
858
+ seed = int.from_bytes(hashlib.sha256(seed_material.encode()).digest()[:8], "big")
859
+ rng = random.Random(seed)
860
+ best = {model_id: 0 for model_id in ordered}
861
+ top3 = {model_id: 0 for model_id in ordered}
862
+ for _ in range(samples):
863
+ drawn = sorted(
864
+ ((rng.gauss(estimates[model_id].value, estimates[model_id].sd), model_id)
865
+ for model_id in ordered),
866
+ key=lambda pair: (-pair[0], pair[1]),
867
+ )
868
+ best[drawn[0][1]] += 1
869
+ for _, model_id in drawn[:3]:
870
+ top3[model_id] += 1
871
+ return {model_id: (best[model_id] / samples, top3[model_id] / samples)
872
+ for model_id in ordered}