modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/explain.py ADDED
@@ -0,0 +1,908 @@
1
+ """Explain decisions using retained snapshot records, without capability blending."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from decision.contract import Contribution, DomainEvidence, EvidenceItem
6
+ from decision.registry import UNREGISTERED
7
+ from decision.registry import facet as registry_facet
8
+
9
+
10
+ class ExplanationError(ValueError):
11
+ """A displayed measurement cannot be traced to a verified snapshot record."""
12
+
13
+
14
+ #: Facets a top candidate always shows when known, beside those the spec names:
15
+ #: what the decide page displays for a candidate (contract 1.4).
16
+ DISPLAY_FACETS = (
17
+ "licence.commercial_use",
18
+ "model.class",
19
+ "model.context_window",
20
+ "model.lifecycle",
21
+ "model.release_date",
22
+ "model.weights_openness",
23
+ "offering.cost_per_task",
24
+ "offering.data.retention",
25
+ "offering.price.input",
26
+ "offering.price.output",
27
+ "offering.speed.throughput",
28
+ "offering.speed.time_to_first_token",
29
+ "origin.lab_jurisdiction",
30
+ )
31
+
32
+ #: The sections whose numbers a decision presents, and so the only ones
33
+ #: ``number_origins`` covers. A shown fact in ``top`` carries its own record
34
+ #: and source IDs, so it needs no origin; ``top``'s contributions repeat the
35
+ #: optimiser's arithmetic, already on their records.
36
+ PRESENTED = ("results", "top/*/evidence", "near_misses", "constraint_costs",
37
+ "tipping_points", "eliminated/models")
38
+
39
+ #: Numbers with nothing to trace: the spec's own weights echoed back, the sum
40
+ #: of its soft penalties, ranks and counts. Funnel counts and ``out_of_lineup``
41
+ #: sit outside the presented sections for the same reason.
42
+ UNTRACED = frozenset({"weight", "soft_penalty", "rank", "admits"})
43
+
44
+
45
+ def facet_unit(facet_id):
46
+ return registry_facet(facet_id).unit
47
+
48
+
49
+ def named_facets(resolved):
50
+ """Every facet the spec's conditions (profile rules included) and objective name."""
51
+ from decision.contract import AllOf, AnyOf, Known, NotOf
52
+
53
+ names = set()
54
+
55
+ def walk(cond):
56
+ if isinstance(cond, AnyOf | AllOf):
57
+ for child in cond.any if isinstance(cond, AnyOf) else cond.all:
58
+ walk(child)
59
+ elif isinstance(cond, NotOf):
60
+ walk(cond.not_)
61
+ else:
62
+ names.add(cond.known if isinstance(cond, Known) else cond.facet)
63
+
64
+ for cond in resolved.conditions:
65
+ walk(cond)
66
+ objective = resolved.spec.optimize
67
+ terms = [objective.max, objective.min, *(step.facet for step in objective.lexicographic or ()),
68
+ *(objective.weights or ()), *(objective.pareto or ())]
69
+ names.update(term.removeprefix("-") for term in terms if term)
70
+ return names
71
+
72
+
73
+ def benchmark_domains(snapshot, benchmark):
74
+ return [domain for domain, _ in snapshot.benchmark_domain_tags().get(benchmark, ())]
75
+
76
+
77
+ def computed(snapshot, cid, facet_id):
78
+ """The computed value behind a fact (MODEL-153), or ``None`` for a stored one."""
79
+ lookup = getattr(snapshot, "computed", None)
80
+ return None if lookup is None else lookup(cid, facet_id)
81
+
82
+
83
+ def fact_provenance(snapshot, cid, facet_id):
84
+ """Records, unit and formula behind a non-evidence fact. A stored fact has
85
+ one checked record; a computed one has the records it was computed from."""
86
+ found = computed(snapshot, cid, facet_id)
87
+ if found is not None:
88
+ for rid in found.records:
89
+ checked_record(snapshot, rid)
90
+ return list(found.records), facet_unit(facet_id), found.formula
91
+ fact = snapshot.fact(cid, facet_id)
92
+ checked_record(snapshot, fact.record_id)
93
+ return [fact.record_id], facet_unit(facet_id), None
94
+
95
+
96
+ def checked_record(snapshot, rid):
97
+ if not rid:
98
+ raise ExplanationError("snapshot lacks retained provenance; rebuild it before explaining")
99
+ record = snapshot.record(rid)
100
+ if record["verification"]["outcome"] != "verified":
101
+ raise ExplanationError(f"{rid}: not verified")
102
+ verification = record["verification"]
103
+ collector, verifier = verification.get("collector", {}), verification.get("verifier", {})
104
+ if (
105
+ not collector.get("agent")
106
+ or not verifier.get("agent")
107
+ or collector["agent"] == verifier["agent"]
108
+ or not collector.get("method")
109
+ or not verifier.get("method")
110
+ or collector["method"] == verifier["method"]
111
+ ):
112
+ raise ExplanationError(f"{rid}: verification must use a different agent and method")
113
+ if not record.get("sources"):
114
+ raise ExplanationError(f"{rid}: no source")
115
+ from urllib.parse import urlsplit
116
+
117
+ for source in record["sources"]:
118
+ url = snapshot.source_url(source["source_id"])
119
+ parsed = urlsplit(url)
120
+ if parsed.scheme not in ("http", "https") or not parsed.netloc:
121
+ raise ExplanationError(f"{rid}: source must be an http(s) URL")
122
+ return record
123
+
124
+
125
+ def evidence_item(
126
+ snapshot,
127
+ row,
128
+ domain,
129
+ *,
130
+ loading=None,
131
+ estimate_weight=None,
132
+ recency_weight=None,
133
+ ):
134
+ record = checked_record(snapshot, row.record_id)
135
+ if not row.verified or row.date is None or row.directness is None:
136
+ raise ExplanationError(f"{row.record_id}: missing evidence date or domain directness")
137
+ if record.get("score") != row.value:
138
+ raise ExplanationError(f"{row.record_id}: evidence differs from retained record")
139
+ measured = {"independent_evaluator": "independent"}.get(row.measured_by, row.measured_by)
140
+ date_type = {"evaluated": "observed"}.get(row.date_type, row.date_type)
141
+ unregistered = row.harness == UNREGISTERED
142
+ return EvidenceItem(
143
+ requested_domain=domain,
144
+ record_id=row.record_id,
145
+ benchmark=row.benchmark_id,
146
+ version=row.version,
147
+ sub_category=row.subcategory,
148
+ value=row.value,
149
+ unit=row.unit,
150
+ measured_by=measured,
151
+ effort=row.effort,
152
+ harness=None if unregistered else row.harness,
153
+ harness_unregistered=unregistered,
154
+ date=row.date,
155
+ date_type=date_type,
156
+ source=snapshot.source_url(row.source_ids[0]),
157
+ source_snapshot=row.source_snapshot,
158
+ directness=row.directness,
159
+ n=record.get("n"),
160
+ loading=loading,
161
+ estimate_weight=estimate_weight,
162
+ recency_weight=recency_weight,
163
+ )
164
+
165
+
166
+ def domain_evidence(snapshot, cid, domains, benchmarks=None):
167
+ """Verified evidence per domain. An offering answers its model's evidence,
168
+ so a row the offering and its model both return is listed once."""
169
+ subjects = [cid]
170
+ if snapshot.model_of(cid) != cid:
171
+ subjects.append(snapshot.model_of(cid))
172
+ groups = []
173
+ for domain in sorted(domains):
174
+ rows = {}
175
+ for subject in subjects:
176
+ for row in snapshot.evidence_for_domain(subject, domain):
177
+ if row.verified and (benchmarks is None or row.benchmark_id in benchmarks):
178
+ rows.setdefault(row.record_id, row)
179
+ groups.append(DomainEvidence(
180
+ domain=domain, items=[evidence_item(snapshot, row, domain) for row in rows.values()]))
181
+ return groups
182
+
183
+
184
+ def estimate_evidence(snapshot, cid, domain):
185
+ """The tagged measurements that drove one stored capability estimate."""
186
+ from dataclasses import replace
187
+
188
+ items = []
189
+ model_id = snapshot.model_of(cid)
190
+ for driver in snapshot.capability_drivers(cid, domain):
191
+ lookup = getattr(snapshot, "evidence_record", None)
192
+ row = lookup(model_id, driver.record_id) if lookup is not None else next((
193
+ row
194
+ for candidate in snapshot.candidates()
195
+ if snapshot.model_of(candidate) == model_id
196
+ for row in snapshot.evidence(candidate, driver.benchmark_id)
197
+ if row.record_id == driver.record_id
198
+ ), None)
199
+ if row is None:
200
+ raise ExplanationError(
201
+ f"{driver.record_id}: capability driver is not retained evidence"
202
+ )
203
+ directness = dict(
204
+ snapshot.benchmark_domain_tags().get(driver.benchmark_id, ())
205
+ ).get(domain)
206
+ if directness is None:
207
+ raise ExplanationError(
208
+ f"{driver.record_id}: capability driver is not tagged for {domain}"
209
+ )
210
+ items.append(evidence_item(
211
+ snapshot,
212
+ replace(row, directness=directness),
213
+ domain,
214
+ loading=driver.loading,
215
+ estimate_weight=driver.weight,
216
+ recency_weight=driver.recency_weight,
217
+ ))
218
+ return items
219
+
220
+
221
+ def part_provenance(snapshot, cid, part):
222
+ """Records, unit and formula behind one objective dimension's raw value."""
223
+ if part.evidence:
224
+ records = sorted({e.record_id for e in part.evidence})
225
+ for rid in records:
226
+ checked_record(snapshot, rid)
227
+ return records, part.evidence[0].unit, None
228
+ if part.estimate is not None:
229
+ domain = part.dimension.removeprefix("-")
230
+ records = [driver.record_id for driver in snapshot.capability_drivers(
231
+ cid, domain
232
+ )]
233
+ for rid in records:
234
+ checked_record(snapshot, rid)
235
+ directness = {
236
+ kind
237
+ for item in snapshot.capability_items.values()
238
+ for tagged_domain, kind in item.get("domains", ())
239
+ if tagged_domain == domain
240
+ }
241
+ formula = (
242
+ "proxy-only monotone domain evidence estimate"
243
+ if directness == {"proxy"}
244
+ else "monotone domain evidence estimate"
245
+ )
246
+ return records, "latent capability", formula
247
+ if part.raw_value is not None:
248
+ return fact_provenance(snapshot, cid, part.dimension.removeprefix("-"))
249
+ return [], None, None
250
+
251
+
252
+ def _items(groups, ids):
253
+ """The evidence items with these record IDs, each once."""
254
+ found = {}
255
+ for group in groups:
256
+ for item in group.items:
257
+ if item.record_id in ids:
258
+ found.setdefault(item.record_id, item)
259
+ return list(found.values())
260
+
261
+
262
+ def contributions(snapshot, cid, parts, evidence):
263
+ out = []
264
+ for part in parts:
265
+ records, unit, formula = part_provenance(snapshot, cid, part)
266
+ items = []
267
+ if part.estimate is not None:
268
+ items = estimate_evidence(snapshot, cid, part.dimension.removeprefix("-"))
269
+ if part.evidence:
270
+ items = _items(evidence, set(records))
271
+ # A benchmark objective may have no requested domain. Do not invent
272
+ # directness: read it from the domains that tag the benchmark.
273
+ if not items:
274
+ domains = {d for e in part.evidence for d in benchmark_domains(
275
+ snapshot, e.benchmark_id)}
276
+ items = _items(domain_evidence(snapshot, cid, domains), set(records))
277
+ norm = part.normalisation
278
+ out.append(
279
+ Contribution(
280
+ dimension=part.dimension,
281
+ weight=part.weight,
282
+ value=part.value,
283
+ raw_value=part.raw_value,
284
+ unit=unit,
285
+ records=records,
286
+ normalisation=f"feasible min-max; {norm.direction}; "
287
+ f"minimum={norm.minimum} {unit or 'unit not recorded'}; "
288
+ f"maximum={norm.maximum} {unit or 'unit not recorded'}; "
289
+ "constant dimensions contribute zero dimensionless",
290
+ evidence=items,
291
+ formula=formula,
292
+ )
293
+ )
294
+ return out
295
+
296
+
297
+ def explain(decision, resolved, snapshot, filtered, ordered, selectors, domains):
298
+ requested = set(resolved.spec.capabilities or {})
299
+ for result, row in zip(decision.results, ordered.results):
300
+ result.evidence = domain_evidence(snapshot, row.candidate_id, requested)
301
+ result.contributions = contributions(
302
+ snapshot, row.candidate_id, row.contributions, result.evidence
303
+ )
304
+ decision.eliminated.funnel = [step.as_contract() for step in filtered.funnel]
305
+ _alternatives(decision, resolved, snapshot, filtered, ordered, selectors, domains)
306
+ from decision.contract import TippingPoint
307
+
308
+ decision.tipping_points = [
309
+ TippingPoint(
310
+ description=(
311
+ f"{point.direction} weight past threshold; other weights and normalisation fixed"
312
+ ),
313
+ dimension=point.dimension,
314
+ threshold=point.threshold,
315
+ new_top=snapshot.model_of(point.new_top),
316
+ )
317
+ for point in ordered.tipping_points
318
+ ]
319
+ if decision.explain == "full":
320
+ _full(decision, snapshot, ordered, requested, named_facets(resolved))
321
+ decision.chart = contribution_chart(decision)
322
+ decision.number_origins = list(number_origins(decision, snapshot))
323
+ decision.sources = cited_sources(decision, snapshot)
324
+
325
+
326
+ def _distance(reason, condition):
327
+ from decision.contract import Compare, Window
328
+
329
+ value, threshold = reason.value, reason.threshold
330
+ if isinstance(value, (tuple, list)) and isinstance(condition, Window):
331
+ numeric = [v for v in value if isinstance(v, (int, float)) and not isinstance(v, bool)]
332
+ return min((max(threshold[0] - v, v - threshold[1], 0) for v in numeric), default=None)
333
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
334
+ return None
335
+ if isinstance(condition, Compare) and condition.op in ("<", "<=", ">", ">=", "="):
336
+ if isinstance(threshold, (int, float)) and not isinstance(threshold, bool):
337
+ # Strict comparisons report distance to the boundary, not an invented epsilon.
338
+ return abs(value - threshold)
339
+ if isinstance(condition, Window):
340
+ return max(threshold[0] - value, value - threshold[1], 0)
341
+ return None
342
+
343
+
344
+ def _alternatives(decision, resolved, snapshot, filtered, ordered, selectors, domains):
345
+ from dataclasses import replace
346
+
347
+ from decision.contract import (
348
+ ConstraintCost,
349
+ ModelElimination,
350
+ ModelEliminationGroup,
351
+ NearMiss,
352
+ OfferingElimination,
353
+ render_condition,
354
+ )
355
+ from decision.engine import offering_ref, run_optimise
356
+ from decision.filter import apply
357
+
358
+ current = {
359
+ part.dimension: part
360
+ for row in ordered.results
361
+ for part in row.contributions
362
+ if part.raw_value is not None
363
+ }
364
+
365
+ # Best observed raw value per dimension, independent of the weighted winner.
366
+ def best(rows, dimension):
367
+ values = [
368
+ p.raw_value
369
+ for r in rows
370
+ for p in r.contributions
371
+ if p.dimension == dimension and p.raw_value is not None
372
+ ]
373
+ if not values:
374
+ return None
375
+ return min(values) if dimension.startswith("-") else max(values)
376
+
377
+ candidate_near_misses = {}
378
+ for i, condition in enumerate(resolved.conditions):
379
+ if condition.soft is not None:
380
+ continue
381
+ text = render_condition(condition)
382
+ single = apply(replace(resolved, conditions=(condition,)), snapshot)
383
+ relaxed = apply(
384
+ replace(resolved, conditions=resolved.conditions[:i] + resolved.conditions[i + 1 :]),
385
+ snapshot,
386
+ )
387
+ alternative = run_optimise(snapshot, relaxed, resolved.spec, selectors, domains)
388
+ gains, units, records, bests = {}, {}, set(), {}
389
+ for dimension in current:
390
+ before, after = best(ordered.results, dimension), best(alternative.results, dimension)
391
+ if before is not None and after is not None:
392
+ gains[dimension] = after - before
393
+ bests[dimension] = (before, after)
394
+ # A gain rests on the two values it subtracts: the records behind those.
395
+ for side, rows in enumerate((ordered.results, alternative.results)):
396
+ for row in rows:
397
+ for part in row.contributions:
398
+ if part.dimension in bests and part.raw_value == bests[part.dimension][side]:
399
+ rids, units[part.dimension], _ = part_provenance(
400
+ snapshot, row.candidate_id, part)
401
+ records.update(rids)
402
+ decision.constraint_costs.append(
403
+ ConstraintCost(
404
+ condition=text,
405
+ admits=len(set(relaxed.feasible) - set(filtered.feasible)),
406
+ gain=gains,
407
+ units=units,
408
+ records=sorted(records),
409
+ )
410
+ )
411
+ for reason in single.eliminated:
412
+ values = list(reason.value) if isinstance(reason.value, (tuple, list)) else []
413
+ value = None if values else reason.value
414
+ ref = offering_ref(snapshot, reason.candidate)
415
+ records = []
416
+ unit = None
417
+ formula = None
418
+ if reason.facet and reason.value is not None:
419
+ fact = snapshot.fact(reason.candidate, reason.facet)
420
+ if computed(snapshot, reason.candidate, reason.facet) is not None:
421
+ records, unit, formula = fact_provenance(
422
+ snapshot, reason.candidate, reason.facet)
423
+ elif fact.record_id:
424
+ checked_record(snapshot, fact.record_id)
425
+ records = [fact.record_id]
426
+ unit = facet_unit(reason.facet)
427
+ else:
428
+ for row in snapshot.evidence(reason.candidate, reason.facet):
429
+ if row.value in (values or [reason.value]):
430
+ checked_record(snapshot, row.record_id)
431
+ records.append(row.record_id)
432
+ unit = row.unit
433
+ if decision.explain == "full":
434
+ decision.eliminated.models.append(
435
+ ModelElimination(
436
+ model=ref.model,
437
+ offering=ref,
438
+ condition=text + (": unverified: may qualify" if reason.unverified else ""),
439
+ value=value,
440
+ values=values,
441
+ unit=unit,
442
+ records=records,
443
+ formula=formula,
444
+ )
445
+ )
446
+ if (
447
+ reason.candidate in relaxed.feasible
448
+ and not reason.unverified
449
+ and reason.candidate not in filtered.feasible
450
+ ):
451
+ if not (ref.provider is None and (reason.facet or "").startswith("offering.")):
452
+ candidate_near_misses[reason.candidate] = NearMiss(
453
+ offering=ref,
454
+ condition=text,
455
+ facet=reason.facet,
456
+ value=value,
457
+ values=values,
458
+ distance=_distance(reason, condition),
459
+ unit=unit,
460
+ records=records,
461
+ formula=formula,
462
+ )
463
+ if decision.explain == "full":
464
+ already = {m.offering.model_dump_json() for m in decision.eliminated.models}
465
+ for reason in filtered.eliminated:
466
+ ref = offering_ref(snapshot, reason.candidate)
467
+ if ref.model_dump_json() not in already:
468
+ decision.eliminated.models.append(
469
+ ModelElimination(
470
+ model=ref.model,
471
+ offering=ref,
472
+ condition=reason.condition,
473
+ value=reason.value,
474
+ )
475
+ )
476
+ for cid, dominators in ordered.dominance.items():
477
+ ref = offering_ref(snapshot, cid)
478
+ decision.eliminated.models.append(
479
+ ModelElimination(
480
+ model=ref.model, offering=ref, condition="dominated by " + ", ".join(dominators)
481
+ )
482
+ )
483
+ for row in ordered.results[len(decision.results) :]:
484
+ ref = offering_ref(snapshot, row.candidate_id)
485
+ decision.eliminated.models.append(
486
+ ModelElimination(
487
+ model=ref.model, offering=ref, condition="outside requested result limit"
488
+ )
489
+ )
490
+
491
+ grouped = {}
492
+ for row in decision.eliminated.models:
493
+ group = grouped.setdefault(row.model, {"model": None, "offerings": {}})
494
+ if row.offering is None or row.offering.provider is None:
495
+ group["model"] = group["model"] or row
496
+ else:
497
+ key = row.offering.model_dump_json()
498
+ group["offerings"].setdefault(key, row)
499
+ decision.eliminated.model_groups = [
500
+ ModelEliminationGroup(
501
+ model=model,
502
+ model_elimination=rows["model"],
503
+ offerings=[
504
+ OfferingElimination(**row.model_dump(exclude={"model"}))
505
+ for row in rows["offerings"].values()
506
+ ],
507
+ )
508
+ for model, rows in sorted(grouped.items())
509
+ ]
510
+
511
+ # A near miss is one model, represented by its best offering. Optimise all
512
+ # offerings of that model without the hard conditions, then retain it only
513
+ # when that best offering failed exactly one condition.
514
+ by_model = {}
515
+ for cid in candidate_near_misses:
516
+ by_model.setdefault(snapshot.model_of(cid), []).append(cid)
517
+ for model in sorted(by_model):
518
+ if any(snapshot.model_of(cid) == model for cid in filtered.feasible):
519
+ continue
520
+ offerings = [
521
+ cid for cid in snapshot.candidates()
522
+ if snapshot.model_of(cid) == model and snapshot.kind(cid) == "offering"
523
+ ]
524
+ candidates = offerings or [model]
525
+ unconstrained = replace(filtered, feasible=tuple(candidates), may_qualify=(), eliminated=())
526
+ best = run_optimise(snapshot, unconstrained, resolved.spec, selectors, domains).results
527
+ if best and best[0].candidate_id in candidate_near_misses:
528
+ decision.near_misses.append(candidate_near_misses[best[0].candidate_id])
529
+
530
+
531
+ def _full(decision, snapshot, ordered, requested, named):
532
+ """Up to 20 optimised candidates with the facets the spec names, the display
533
+ set, and evidence on the benchmarks the spec names: not every value held."""
534
+ from decision.computed import COMPUTED_FACETS
535
+ from decision.contract import CandidateValues, ShownFact
536
+ from decision.engine import offering_ref
537
+
538
+ shown = named | set(DISPLAY_FACETS)
539
+ stored = [facet for facet in snapshot.facet_ids() if facet in shown]
540
+ benchmarks = named & set(snapshot.benchmark_ids())
541
+ ranked = len(decision.results)
542
+ for position, row in enumerate(ordered.results[:20]):
543
+ cid = row.candidate_id
544
+ facts = []
545
+ for facet in stored:
546
+ fact = snapshot.fact(cid, facet)
547
+ if fact.state != "known":
548
+ continue
549
+ unit = None
550
+ if fact.record_id:
551
+ unit = facet_unit(facet)
552
+ facts.append(ShownFact(
553
+ facet=facet, value=fact.value, unit=unit, record_id=fact.record_id,
554
+ source_ids=source_ids(snapshot, [fact.record_id] if fact.record_id else [])))
555
+ for facet in COMPUTED_FACETS:
556
+ found = computed(snapshot, cid, facet) if facet in shown else None
557
+ if found is not None:
558
+ records, unit, formula = fact_provenance(snapshot, cid, facet)
559
+ facts.append(ShownFact(facet=facet, value=found.value, unit=unit,
560
+ records=records, formula=formula,
561
+ source_ids=source_ids(snapshot, records)))
562
+ # Group each named benchmark under the requested domains that tag it,
563
+ # or, when none does, under the first domain that does.
564
+ domains = set(requested)
565
+ for benchmark in benchmarks:
566
+ tagged = benchmark_domains(snapshot, benchmark)
567
+ if tagged and not requested & set(tagged):
568
+ domains.add(tagged[0])
569
+ evidence = [group for group in domain_evidence(snapshot, cid, domains, benchmarks)
570
+ if group.items]
571
+ decision.top.append(
572
+ CandidateValues(
573
+ offering=offering_ref(snapshot, cid),
574
+ facts=facts,
575
+ evidence=evidence,
576
+ # A ranked candidate's contributions are in ``results``, not repeated.
577
+ contributions=[] if position < ranked else contributions(
578
+ snapshot, cid, row.contributions, evidence),
579
+ )
580
+ )
581
+
582
+
583
+ def top_contributions(decision):
584
+ """Each top candidate with its contributions, from ``results`` when ranked."""
585
+ ranked = {r.offering.model_dump_json(): r.contributions for r in decision.results}
586
+ for row in decision.top:
587
+ yield row, row.contributions or ranked.get(row.offering.model_dump_json(), [])
588
+
589
+
590
+ def contribution_chart(decision):
591
+ """Faceted raw-value bars: dimensions with different units never share a scale."""
592
+ from html import escape
593
+
594
+ rows = [
595
+ (r.offering.model, c)
596
+ for r, parts in top_contributions(decision)
597
+ for c in parts
598
+ if c.raw_value is not None
599
+ ]
600
+ parts = [
601
+ f'<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 800 {max(60, len(rows) * 55)}" '
602
+ 'role="img" aria-label="Objective contributions in original units">'
603
+ ]
604
+ maxima = {
605
+ c.dimension: max(
606
+ abs(other.raw_value) for _, other in rows if other.dimension == c.dimension
607
+ )
608
+ for _, c in rows
609
+ }
610
+ for i, (model, c) in enumerate(rows):
611
+ width = 300 * abs(c.raw_value) / (maxima[c.dimension] or 1)
612
+ reported = (
613
+ " · Lab-reported"
614
+ if any(e.measured_by == "provider_self_report" for e in c.evidence)
615
+ else ""
616
+ )
617
+ label = escape(
618
+ f"{model}{reported} · {c.dimension}: {c.raw_value:g} {c.unit or 'unit not recorded'}"
619
+ )
620
+ parts += [
621
+ f'<text x="8" y="{i * 55 + 18}">{label}</text>',
622
+ f'<rect x="8" y="{i * 55 + 27}" width="{width:g}" height="12" />',
623
+ ]
624
+ return "".join(parts) + "</svg>"
625
+
626
+
627
+ def numbers(value, path=""):
628
+ """Walk numeric JSON leaves. IDs, dates and SVG geometry are not measurements."""
629
+ if isinstance(value, bool):
630
+ return
631
+ if isinstance(value, (int, float)):
632
+ yield path, value
633
+ elif isinstance(value, dict):
634
+ for key, child in value.items():
635
+ if key != "number_origins":
636
+ yield from numbers(child, path + "/" + key)
637
+ elif isinstance(value, list):
638
+ for i, child in enumerate(value):
639
+ yield from numbers(child, path + "/" + str(i))
640
+
641
+
642
+ def presented_values(data):
643
+ """The numbers ``number_origins`` covers: those in the presented sections."""
644
+ def within(node, path, keys):
645
+ if not keys:
646
+ yield from ((p, v) for p, v in numbers(node, path)
647
+ if p.rsplit("/", 1)[-1] not in UNTRACED)
648
+ elif keys[0] == "*":
649
+ for i, child in enumerate(node):
650
+ yield from within(child, f"{path}/{i}", keys[1:])
651
+ else:
652
+ yield from within(node.get(keys[0], []), f"{path}/{keys[0]}", keys[1:])
653
+
654
+ for section in PRESENTED:
655
+ yield from within(data, "", section.split("/"))
656
+
657
+
658
+ def _node(data, path):
659
+ parent, ancestors = data, []
660
+ nodes = path.strip("/").split("/")
661
+ for node in nodes[:-1]:
662
+ ancestors.append(parent)
663
+ parent = parent[int(node)] if isinstance(parent, list) else parent[node]
664
+ key = nodes[-1]
665
+ if isinstance(parent, list):
666
+ return ancestors[-1], nodes[-2], int(key)
667
+ return parent, key, None
668
+
669
+
670
+ def number_origins(decision, snapshot):
671
+ """Distinguish measurements from spec inputs and reproducible calculations.
672
+
673
+ Covers the presented numbers only (``presented_values``). An origin names
674
+ its records and their sources by ID; ``cited_sources`` lists each source
675
+ once.
676
+ """
677
+ from decision.contract import NumberOrigin
678
+
679
+ data = decision.model_dump(mode="json")
680
+ for path, value in presented_values(data):
681
+ parent, key, list_index = _node(data, path)
682
+ records = parent.get("records", []) if isinstance(parent, dict) else []
683
+ if isinstance(parent, dict) and parent.get("record_id"):
684
+ records = [parent["record_id"]]
685
+ raw = (
686
+ key == "raw_value"
687
+ or key in ("value", "values")
688
+ and ("record_id" in parent or "condition" in parent or "facet" in parent)
689
+ )
690
+ if raw and isinstance(parent, dict) and parent.get("formula"):
691
+ # Computed per decision from the records listed (MODEL-153), so it is
692
+ # checked against the formula, not against one record's value.
693
+ basis = "computed per decision: " + parent["formula"]
694
+ elif raw and records:
695
+ basis = "snapshot measurement"
696
+
697
+ def original(rid):
698
+ record = checked_record(snapshot, rid)
699
+ raw = record.get("score", record.get("value"))
700
+ return raw[list_index] if list_index is not None and isinstance(raw, list) else raw
701
+
702
+ if not any(value == original(rid) for rid in records):
703
+ raise ExplanationError(f"{path}: value differs from retained record")
704
+ elif key == "distance":
705
+ basis = "distance from snapshot value to spec condition boundary, in original units"
706
+ elif "/gain/" in path:
707
+ basis = "best relaxed raw objective value minus best current raw value"
708
+ elif key == "threshold":
709
+ basis = "optimise weight crossing with other weights and normalisation fixed"
710
+ elif key == "value" and "dimension" in parent:
711
+ basis = "feasible-set min-max normalisation of snapshot measurements"
712
+ elif key == "n":
713
+ basis = "sample count from snapshot evidence"
714
+ else:
715
+ basis = "count or ordinal from snapshot candidates after filtering and optimisation"
716
+ records = []
717
+ records = sorted(set(records))
718
+ yield NumberOrigin(
719
+ path=path, basis=basis, records=records, source_ids=source_ids(snapshot, records))
720
+
721
+
722
+ def source_ids(snapshot, records):
723
+ """The registered sources behind checked records, by ID."""
724
+ return sorted({source["source_id"] for rid in records
725
+ for source in checked_record(snapshot, rid)["sources"]})
726
+
727
+
728
+ def cited_sources(decision, snapshot):
729
+ """Each source the origins and shown facts cite, once: URL, and the latest
730
+ date a record in this decision citing it was verified."""
731
+ from decision.contract import CitedSource
732
+
733
+ records = {rid for origin in decision.number_origins for rid in origin.records}
734
+ for row in decision.top:
735
+ for fact in row.facts:
736
+ records.update(fact.records)
737
+ if fact.record_id:
738
+ records.add(fact.record_id)
739
+ verified = {}
740
+ for rid in records:
741
+ record = snapshot.record(rid)
742
+ day = record["verification"].get("date")
743
+ for source in record["sources"]:
744
+ sid = source["source_id"]
745
+ verified.setdefault(sid, None)
746
+ if day and (verified[sid] is None or day > verified[sid]):
747
+ verified[sid] = day
748
+ return [CitedSource(id=sid, url=snapshot.source_url(sid), date=verified[sid])
749
+ for sid in sorted(verified)]
750
+
751
+
752
+ def render_html(decision, snapshot):
753
+ """A self-contained report. Source links are its only external resources."""
754
+ from html import escape
755
+
756
+ def esc(value):
757
+ return escape(str(value))
758
+
759
+ def quantity(value, unit):
760
+ return "unknown" if value is None else f"{value:g} {esc(unit or 'unit not recorded')}"
761
+
762
+ def links(records):
763
+ urls = sorted(
764
+ {
765
+ snapshot.source_url(s["source_id"])
766
+ for rid in records
767
+ for s in checked_record(snapshot, rid)["sources"]
768
+ }
769
+ )
770
+ return " ".join(f'<a href="{esc(url)}">Source</a>' for url in urls)
771
+
772
+ def evidence(groups):
773
+ out = []
774
+ for group in groups:
775
+ out.append(f"<h3>{esc(group.domain)}</h3>")
776
+ if not group.items:
777
+ out.append("<p>No verified evidence for this domain.</p>")
778
+ for item in group.items:
779
+ label = (
780
+ "Lab-reported"
781
+ if item.measured_by == "provider_self_report"
782
+ else item.measured_by
783
+ )
784
+ out.append(
785
+ f"<p><strong>{esc(label)}</strong> · {esc(item.benchmark)} "
786
+ f"· version {esc(item.version or 'not recorded')} "
787
+ f"· {esc(item.sub_category or 'aggregate')} "
788
+ f"· {quantity(item.value, item.unit)} "
789
+ f"· {esc(item.directness)} · {esc(item.date)} ({esc(item.date_type)}) "
790
+ f"· effort {esc(item.effort or 'not recorded')} "
791
+ f"· harness {esc(item.harness or 'not recorded')} "
792
+ f'<a href="{esc(item.source)}">Source</a></p>'
793
+ )
794
+ return "".join(out)
795
+
796
+ def parts(contributions):
797
+ out = []
798
+ for c in contributions:
799
+ out.append(
800
+ f"<p>{esc(c.dimension)}: {quantity(c.raw_value, c.unit)}; "
801
+ f"normalised value {quantity(c.value, 'dimensionless')}; "
802
+ f"weight {quantity(c.weight, 'dimensionless')}. "
803
+ + (f"Computed: {esc(c.formula)}. " if c.formula else "")
804
+ + f"{esc(c.normalisation)} {links(c.records)}</p>"
805
+ )
806
+ groups = {}
807
+ for item in c.evidence:
808
+ groups.setdefault(item.requested_domain, []).append(item)
809
+ out.append(
810
+ evidence(
811
+ [DomainEvidence(domain=domain, items=items) for domain, items in groups.items()]
812
+ )
813
+ )
814
+ return "".join(out)
815
+
816
+ out = [
817
+ '<!doctype html><html lang="en"><meta charset="utf-8">',
818
+ '<meta name="viewport" content="width=device-width, initial-scale=1">',
819
+ "<title>ModelSpec decision</title><style>",
820
+ ":root{color-scheme:light dark;--bg:#fafafa;--fg:#17212b;--accent:#176b91}",
821
+ "@media (prefers-color-scheme: dark){:root{--bg:#101820;--fg:#e7edf2;--accent:#75c9ed}}",
822
+ "body{background:var(--bg);color:var(--fg);font:16px system-ui;max-width:1000px;",
823
+ "margin:2rem auto;padding:0 1rem;line-height:1.6}a{color:var(--accent)}",
824
+ "section{border-top:1px solid #888;padding:1rem 0}svg{width:100%}",
825
+ "svg text{fill:var(--fg);font:14px system-ui}svg rect{fill:var(--accent)}",
826
+ "</style><body><h1>ModelSpec decision</h1>",
827
+ f"<p>{esc(decision.decision_id)} · {esc(decision.snapshot)} · {esc(decision.status)}</p>",
828
+ "<p>Evidence is unblended. Capability estimates and probabilities are not available.</p>",
829
+ ]
830
+ for r in decision.results:
831
+ out.append(f"<section><h2>{esc(r.offering.model)}</h2>")
832
+ if r.offering.provider:
833
+ out.append(
834
+ f"<p>{esc(r.offering.provider)} · {esc(r.offering.region)} "
835
+ f"· {esc(r.offering.tier)}</p>"
836
+ )
837
+ out.append(parts(r.contributions) + evidence(r.evidence) + "</section>")
838
+ if decision.explain == "full":
839
+ # Rebuild SVG from typed values so an imported chart cannot inject HTML.
840
+ out.append(
841
+ "<section><h2>Contribution chart</h2>" + contribution_chart(decision) + "</section>"
842
+ )
843
+ out.append("<section><h2>Funnel</h2>")
844
+ for step in decision.eliminated.funnel:
845
+ out.append(
846
+ f"<p>{esc(step.condition)}: {step.before} candidates before, "
847
+ f"{step.after} candidates after, {step.may_qualify} candidates may qualify.</p>"
848
+ )
849
+ out.append("</section><section><h2>Constraint costs</h2>")
850
+ for cost in decision.constraint_costs:
851
+ out.append(f"<p>Relax {esc(cost.condition)}: admits {cost.admits} candidates.</p>")
852
+ for dimension, gain in cost.gain.items():
853
+ out.append(
854
+ f"<p>{esc(dimension)}: relaxed minus current = "
855
+ f"{quantity(gain, cost.units.get(dimension))}. {links(cost.records)}</p>"
856
+ )
857
+ if not cost.gain:
858
+ out.append("<p>No comparable objective values.</p>")
859
+ out.append("</section><section><h2>Near misses</h2>")
860
+ for miss in decision.near_misses:
861
+ value = (
862
+ quantity(miss.value, miss.unit)
863
+ if isinstance(miss.value, (int, float))
864
+ else esc(miss.value)
865
+ )
866
+ out.append(
867
+ f"<p>{esc(miss.offering.model)} · {esc(miss.condition)}: "
868
+ f"{value}; "
869
+ f"distance to boundary {quantity(miss.distance, miss.unit)}. "
870
+ + (f"Computed: {esc(miss.formula)}. " if miss.formula else "")
871
+ + f"{links(miss.records)}</p>"
872
+ )
873
+ out.append("</section><section><h2>Why others did not win</h2>")
874
+ for reason in decision.eliminated.models:
875
+ out.append(f"<p>{esc(reason.model)}: {esc(reason.condition)}. {links(reason.records)}</p>")
876
+ for maybe in decision.may_qualify:
877
+ out.append(
878
+ f"<p>{esc(maybe.model)} may qualify; unknown: {esc(', '.join(maybe.unknown))}</p>"
879
+ )
880
+ for reason in decision.relax:
881
+ out.append(f"<p>{esc(reason)}</p>")
882
+ out.append("</section><section><h2>Tipping points</h2>")
883
+ for point in decision.tipping_points:
884
+ out.append(
885
+ f"<p>{esc(point.dimension)}: {quantity(point.threshold, 'dimensionless weight')}; "
886
+ f"{esc(point.description)}; new top {esc(point.new_top)}.</p>"
887
+ )
888
+ out.append("</section>")
889
+ if decision.top:
890
+ out.append("<section><h2>Top candidates</h2>")
891
+ for row, shown in top_contributions(decision):
892
+ out.append(f"<h3>{esc(row.offering.model)}</h3>")
893
+ for fact in row.facts:
894
+ value = (
895
+ quantity(fact.value, fact.unit)
896
+ if isinstance(fact.value, (int, float)) and not isinstance(fact.value, bool)
897
+ else esc(fact.value)
898
+ )
899
+ provenance = (
900
+ links([fact.record_id]) if fact.record_id
901
+ else f"Computed: {esc(fact.formula)}. {links(fact.records)}" if fact.formula
902
+ else "Identity"
903
+ )
904
+ out.append(f"<p>{esc(fact.facet)}: {value}. {provenance}</p>")
905
+ out.append(parts(shown) + evidence(row.evidence))
906
+ out.append("</section>")
907
+ out.append("</body></html>")
908
+ return "".join(out)