modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
decision/explain.py
ADDED
|
@@ -0,0 +1,908 @@
|
|
|
1
|
+
"""Explain decisions using retained snapshot records, without capability blending."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from decision.contract import Contribution, DomainEvidence, EvidenceItem
|
|
6
|
+
from decision.registry import UNREGISTERED
|
|
7
|
+
from decision.registry import facet as registry_facet
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ExplanationError(ValueError):
|
|
11
|
+
"""A displayed measurement cannot be traced to a verified snapshot record."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
#: Facets a top candidate always shows when known, beside those the spec names:
|
|
15
|
+
#: what the decide page displays for a candidate (contract 1.4).
|
|
16
|
+
DISPLAY_FACETS = (
|
|
17
|
+
"licence.commercial_use",
|
|
18
|
+
"model.class",
|
|
19
|
+
"model.context_window",
|
|
20
|
+
"model.lifecycle",
|
|
21
|
+
"model.release_date",
|
|
22
|
+
"model.weights_openness",
|
|
23
|
+
"offering.cost_per_task",
|
|
24
|
+
"offering.data.retention",
|
|
25
|
+
"offering.price.input",
|
|
26
|
+
"offering.price.output",
|
|
27
|
+
"offering.speed.throughput",
|
|
28
|
+
"offering.speed.time_to_first_token",
|
|
29
|
+
"origin.lab_jurisdiction",
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
#: The sections whose numbers a decision presents, and so the only ones
|
|
33
|
+
#: ``number_origins`` covers. A shown fact in ``top`` carries its own record
|
|
34
|
+
#: and source IDs, so it needs no origin; ``top``'s contributions repeat the
|
|
35
|
+
#: optimiser's arithmetic, already on their records.
|
|
36
|
+
PRESENTED = ("results", "top/*/evidence", "near_misses", "constraint_costs",
|
|
37
|
+
"tipping_points", "eliminated/models")
|
|
38
|
+
|
|
39
|
+
#: Numbers with nothing to trace: the spec's own weights echoed back, the sum
|
|
40
|
+
#: of its soft penalties, ranks and counts. Funnel counts and ``out_of_lineup``
|
|
41
|
+
#: sit outside the presented sections for the same reason.
|
|
42
|
+
UNTRACED = frozenset({"weight", "soft_penalty", "rank", "admits"})
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def facet_unit(facet_id):
|
|
46
|
+
return registry_facet(facet_id).unit
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def named_facets(resolved):
|
|
50
|
+
"""Every facet the spec's conditions (profile rules included) and objective name."""
|
|
51
|
+
from decision.contract import AllOf, AnyOf, Known, NotOf
|
|
52
|
+
|
|
53
|
+
names = set()
|
|
54
|
+
|
|
55
|
+
def walk(cond):
|
|
56
|
+
if isinstance(cond, AnyOf | AllOf):
|
|
57
|
+
for child in cond.any if isinstance(cond, AnyOf) else cond.all:
|
|
58
|
+
walk(child)
|
|
59
|
+
elif isinstance(cond, NotOf):
|
|
60
|
+
walk(cond.not_)
|
|
61
|
+
else:
|
|
62
|
+
names.add(cond.known if isinstance(cond, Known) else cond.facet)
|
|
63
|
+
|
|
64
|
+
for cond in resolved.conditions:
|
|
65
|
+
walk(cond)
|
|
66
|
+
objective = resolved.spec.optimize
|
|
67
|
+
terms = [objective.max, objective.min, *(step.facet for step in objective.lexicographic or ()),
|
|
68
|
+
*(objective.weights or ()), *(objective.pareto or ())]
|
|
69
|
+
names.update(term.removeprefix("-") for term in terms if term)
|
|
70
|
+
return names
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def benchmark_domains(snapshot, benchmark):
|
|
74
|
+
return [domain for domain, _ in snapshot.benchmark_domain_tags().get(benchmark, ())]
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def computed(snapshot, cid, facet_id):
|
|
78
|
+
"""The computed value behind a fact (MODEL-153), or ``None`` for a stored one."""
|
|
79
|
+
lookup = getattr(snapshot, "computed", None)
|
|
80
|
+
return None if lookup is None else lookup(cid, facet_id)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def fact_provenance(snapshot, cid, facet_id):
|
|
84
|
+
"""Records, unit and formula behind a non-evidence fact. A stored fact has
|
|
85
|
+
one checked record; a computed one has the records it was computed from."""
|
|
86
|
+
found = computed(snapshot, cid, facet_id)
|
|
87
|
+
if found is not None:
|
|
88
|
+
for rid in found.records:
|
|
89
|
+
checked_record(snapshot, rid)
|
|
90
|
+
return list(found.records), facet_unit(facet_id), found.formula
|
|
91
|
+
fact = snapshot.fact(cid, facet_id)
|
|
92
|
+
checked_record(snapshot, fact.record_id)
|
|
93
|
+
return [fact.record_id], facet_unit(facet_id), None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def checked_record(snapshot, rid):
|
|
97
|
+
if not rid:
|
|
98
|
+
raise ExplanationError("snapshot lacks retained provenance; rebuild it before explaining")
|
|
99
|
+
record = snapshot.record(rid)
|
|
100
|
+
if record["verification"]["outcome"] != "verified":
|
|
101
|
+
raise ExplanationError(f"{rid}: not verified")
|
|
102
|
+
verification = record["verification"]
|
|
103
|
+
collector, verifier = verification.get("collector", {}), verification.get("verifier", {})
|
|
104
|
+
if (
|
|
105
|
+
not collector.get("agent")
|
|
106
|
+
or not verifier.get("agent")
|
|
107
|
+
or collector["agent"] == verifier["agent"]
|
|
108
|
+
or not collector.get("method")
|
|
109
|
+
or not verifier.get("method")
|
|
110
|
+
or collector["method"] == verifier["method"]
|
|
111
|
+
):
|
|
112
|
+
raise ExplanationError(f"{rid}: verification must use a different agent and method")
|
|
113
|
+
if not record.get("sources"):
|
|
114
|
+
raise ExplanationError(f"{rid}: no source")
|
|
115
|
+
from urllib.parse import urlsplit
|
|
116
|
+
|
|
117
|
+
for source in record["sources"]:
|
|
118
|
+
url = snapshot.source_url(source["source_id"])
|
|
119
|
+
parsed = urlsplit(url)
|
|
120
|
+
if parsed.scheme not in ("http", "https") or not parsed.netloc:
|
|
121
|
+
raise ExplanationError(f"{rid}: source must be an http(s) URL")
|
|
122
|
+
return record
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def evidence_item(
|
|
126
|
+
snapshot,
|
|
127
|
+
row,
|
|
128
|
+
domain,
|
|
129
|
+
*,
|
|
130
|
+
loading=None,
|
|
131
|
+
estimate_weight=None,
|
|
132
|
+
recency_weight=None,
|
|
133
|
+
):
|
|
134
|
+
record = checked_record(snapshot, row.record_id)
|
|
135
|
+
if not row.verified or row.date is None or row.directness is None:
|
|
136
|
+
raise ExplanationError(f"{row.record_id}: missing evidence date or domain directness")
|
|
137
|
+
if record.get("score") != row.value:
|
|
138
|
+
raise ExplanationError(f"{row.record_id}: evidence differs from retained record")
|
|
139
|
+
measured = {"independent_evaluator": "independent"}.get(row.measured_by, row.measured_by)
|
|
140
|
+
date_type = {"evaluated": "observed"}.get(row.date_type, row.date_type)
|
|
141
|
+
unregistered = row.harness == UNREGISTERED
|
|
142
|
+
return EvidenceItem(
|
|
143
|
+
requested_domain=domain,
|
|
144
|
+
record_id=row.record_id,
|
|
145
|
+
benchmark=row.benchmark_id,
|
|
146
|
+
version=row.version,
|
|
147
|
+
sub_category=row.subcategory,
|
|
148
|
+
value=row.value,
|
|
149
|
+
unit=row.unit,
|
|
150
|
+
measured_by=measured,
|
|
151
|
+
effort=row.effort,
|
|
152
|
+
harness=None if unregistered else row.harness,
|
|
153
|
+
harness_unregistered=unregistered,
|
|
154
|
+
date=row.date,
|
|
155
|
+
date_type=date_type,
|
|
156
|
+
source=snapshot.source_url(row.source_ids[0]),
|
|
157
|
+
source_snapshot=row.source_snapshot,
|
|
158
|
+
directness=row.directness,
|
|
159
|
+
n=record.get("n"),
|
|
160
|
+
loading=loading,
|
|
161
|
+
estimate_weight=estimate_weight,
|
|
162
|
+
recency_weight=recency_weight,
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def domain_evidence(snapshot, cid, domains, benchmarks=None):
|
|
167
|
+
"""Verified evidence per domain. An offering answers its model's evidence,
|
|
168
|
+
so a row the offering and its model both return is listed once."""
|
|
169
|
+
subjects = [cid]
|
|
170
|
+
if snapshot.model_of(cid) != cid:
|
|
171
|
+
subjects.append(snapshot.model_of(cid))
|
|
172
|
+
groups = []
|
|
173
|
+
for domain in sorted(domains):
|
|
174
|
+
rows = {}
|
|
175
|
+
for subject in subjects:
|
|
176
|
+
for row in snapshot.evidence_for_domain(subject, domain):
|
|
177
|
+
if row.verified and (benchmarks is None or row.benchmark_id in benchmarks):
|
|
178
|
+
rows.setdefault(row.record_id, row)
|
|
179
|
+
groups.append(DomainEvidence(
|
|
180
|
+
domain=domain, items=[evidence_item(snapshot, row, domain) for row in rows.values()]))
|
|
181
|
+
return groups
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def estimate_evidence(snapshot, cid, domain):
|
|
185
|
+
"""The tagged measurements that drove one stored capability estimate."""
|
|
186
|
+
from dataclasses import replace
|
|
187
|
+
|
|
188
|
+
items = []
|
|
189
|
+
model_id = snapshot.model_of(cid)
|
|
190
|
+
for driver in snapshot.capability_drivers(cid, domain):
|
|
191
|
+
lookup = getattr(snapshot, "evidence_record", None)
|
|
192
|
+
row = lookup(model_id, driver.record_id) if lookup is not None else next((
|
|
193
|
+
row
|
|
194
|
+
for candidate in snapshot.candidates()
|
|
195
|
+
if snapshot.model_of(candidate) == model_id
|
|
196
|
+
for row in snapshot.evidence(candidate, driver.benchmark_id)
|
|
197
|
+
if row.record_id == driver.record_id
|
|
198
|
+
), None)
|
|
199
|
+
if row is None:
|
|
200
|
+
raise ExplanationError(
|
|
201
|
+
f"{driver.record_id}: capability driver is not retained evidence"
|
|
202
|
+
)
|
|
203
|
+
directness = dict(
|
|
204
|
+
snapshot.benchmark_domain_tags().get(driver.benchmark_id, ())
|
|
205
|
+
).get(domain)
|
|
206
|
+
if directness is None:
|
|
207
|
+
raise ExplanationError(
|
|
208
|
+
f"{driver.record_id}: capability driver is not tagged for {domain}"
|
|
209
|
+
)
|
|
210
|
+
items.append(evidence_item(
|
|
211
|
+
snapshot,
|
|
212
|
+
replace(row, directness=directness),
|
|
213
|
+
domain,
|
|
214
|
+
loading=driver.loading,
|
|
215
|
+
estimate_weight=driver.weight,
|
|
216
|
+
recency_weight=driver.recency_weight,
|
|
217
|
+
))
|
|
218
|
+
return items
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def part_provenance(snapshot, cid, part):
|
|
222
|
+
"""Records, unit and formula behind one objective dimension's raw value."""
|
|
223
|
+
if part.evidence:
|
|
224
|
+
records = sorted({e.record_id for e in part.evidence})
|
|
225
|
+
for rid in records:
|
|
226
|
+
checked_record(snapshot, rid)
|
|
227
|
+
return records, part.evidence[0].unit, None
|
|
228
|
+
if part.estimate is not None:
|
|
229
|
+
domain = part.dimension.removeprefix("-")
|
|
230
|
+
records = [driver.record_id for driver in snapshot.capability_drivers(
|
|
231
|
+
cid, domain
|
|
232
|
+
)]
|
|
233
|
+
for rid in records:
|
|
234
|
+
checked_record(snapshot, rid)
|
|
235
|
+
directness = {
|
|
236
|
+
kind
|
|
237
|
+
for item in snapshot.capability_items.values()
|
|
238
|
+
for tagged_domain, kind in item.get("domains", ())
|
|
239
|
+
if tagged_domain == domain
|
|
240
|
+
}
|
|
241
|
+
formula = (
|
|
242
|
+
"proxy-only monotone domain evidence estimate"
|
|
243
|
+
if directness == {"proxy"}
|
|
244
|
+
else "monotone domain evidence estimate"
|
|
245
|
+
)
|
|
246
|
+
return records, "latent capability", formula
|
|
247
|
+
if part.raw_value is not None:
|
|
248
|
+
return fact_provenance(snapshot, cid, part.dimension.removeprefix("-"))
|
|
249
|
+
return [], None, None
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _items(groups, ids):
|
|
253
|
+
"""The evidence items with these record IDs, each once."""
|
|
254
|
+
found = {}
|
|
255
|
+
for group in groups:
|
|
256
|
+
for item in group.items:
|
|
257
|
+
if item.record_id in ids:
|
|
258
|
+
found.setdefault(item.record_id, item)
|
|
259
|
+
return list(found.values())
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def contributions(snapshot, cid, parts, evidence):
|
|
263
|
+
out = []
|
|
264
|
+
for part in parts:
|
|
265
|
+
records, unit, formula = part_provenance(snapshot, cid, part)
|
|
266
|
+
items = []
|
|
267
|
+
if part.estimate is not None:
|
|
268
|
+
items = estimate_evidence(snapshot, cid, part.dimension.removeprefix("-"))
|
|
269
|
+
if part.evidence:
|
|
270
|
+
items = _items(evidence, set(records))
|
|
271
|
+
# A benchmark objective may have no requested domain. Do not invent
|
|
272
|
+
# directness: read it from the domains that tag the benchmark.
|
|
273
|
+
if not items:
|
|
274
|
+
domains = {d for e in part.evidence for d in benchmark_domains(
|
|
275
|
+
snapshot, e.benchmark_id)}
|
|
276
|
+
items = _items(domain_evidence(snapshot, cid, domains), set(records))
|
|
277
|
+
norm = part.normalisation
|
|
278
|
+
out.append(
|
|
279
|
+
Contribution(
|
|
280
|
+
dimension=part.dimension,
|
|
281
|
+
weight=part.weight,
|
|
282
|
+
value=part.value,
|
|
283
|
+
raw_value=part.raw_value,
|
|
284
|
+
unit=unit,
|
|
285
|
+
records=records,
|
|
286
|
+
normalisation=f"feasible min-max; {norm.direction}; "
|
|
287
|
+
f"minimum={norm.minimum} {unit or 'unit not recorded'}; "
|
|
288
|
+
f"maximum={norm.maximum} {unit or 'unit not recorded'}; "
|
|
289
|
+
"constant dimensions contribute zero dimensionless",
|
|
290
|
+
evidence=items,
|
|
291
|
+
formula=formula,
|
|
292
|
+
)
|
|
293
|
+
)
|
|
294
|
+
return out
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def explain(decision, resolved, snapshot, filtered, ordered, selectors, domains):
|
|
298
|
+
requested = set(resolved.spec.capabilities or {})
|
|
299
|
+
for result, row in zip(decision.results, ordered.results):
|
|
300
|
+
result.evidence = domain_evidence(snapshot, row.candidate_id, requested)
|
|
301
|
+
result.contributions = contributions(
|
|
302
|
+
snapshot, row.candidate_id, row.contributions, result.evidence
|
|
303
|
+
)
|
|
304
|
+
decision.eliminated.funnel = [step.as_contract() for step in filtered.funnel]
|
|
305
|
+
_alternatives(decision, resolved, snapshot, filtered, ordered, selectors, domains)
|
|
306
|
+
from decision.contract import TippingPoint
|
|
307
|
+
|
|
308
|
+
decision.tipping_points = [
|
|
309
|
+
TippingPoint(
|
|
310
|
+
description=(
|
|
311
|
+
f"{point.direction} weight past threshold; other weights and normalisation fixed"
|
|
312
|
+
),
|
|
313
|
+
dimension=point.dimension,
|
|
314
|
+
threshold=point.threshold,
|
|
315
|
+
new_top=snapshot.model_of(point.new_top),
|
|
316
|
+
)
|
|
317
|
+
for point in ordered.tipping_points
|
|
318
|
+
]
|
|
319
|
+
if decision.explain == "full":
|
|
320
|
+
_full(decision, snapshot, ordered, requested, named_facets(resolved))
|
|
321
|
+
decision.chart = contribution_chart(decision)
|
|
322
|
+
decision.number_origins = list(number_origins(decision, snapshot))
|
|
323
|
+
decision.sources = cited_sources(decision, snapshot)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _distance(reason, condition):
|
|
327
|
+
from decision.contract import Compare, Window
|
|
328
|
+
|
|
329
|
+
value, threshold = reason.value, reason.threshold
|
|
330
|
+
if isinstance(value, (tuple, list)) and isinstance(condition, Window):
|
|
331
|
+
numeric = [v for v in value if isinstance(v, (int, float)) and not isinstance(v, bool)]
|
|
332
|
+
return min((max(threshold[0] - v, v - threshold[1], 0) for v in numeric), default=None)
|
|
333
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
334
|
+
return None
|
|
335
|
+
if isinstance(condition, Compare) and condition.op in ("<", "<=", ">", ">=", "="):
|
|
336
|
+
if isinstance(threshold, (int, float)) and not isinstance(threshold, bool):
|
|
337
|
+
# Strict comparisons report distance to the boundary, not an invented epsilon.
|
|
338
|
+
return abs(value - threshold)
|
|
339
|
+
if isinstance(condition, Window):
|
|
340
|
+
return max(threshold[0] - value, value - threshold[1], 0)
|
|
341
|
+
return None
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _alternatives(decision, resolved, snapshot, filtered, ordered, selectors, domains):
|
|
345
|
+
from dataclasses import replace
|
|
346
|
+
|
|
347
|
+
from decision.contract import (
|
|
348
|
+
ConstraintCost,
|
|
349
|
+
ModelElimination,
|
|
350
|
+
ModelEliminationGroup,
|
|
351
|
+
NearMiss,
|
|
352
|
+
OfferingElimination,
|
|
353
|
+
render_condition,
|
|
354
|
+
)
|
|
355
|
+
from decision.engine import offering_ref, run_optimise
|
|
356
|
+
from decision.filter import apply
|
|
357
|
+
|
|
358
|
+
current = {
|
|
359
|
+
part.dimension: part
|
|
360
|
+
for row in ordered.results
|
|
361
|
+
for part in row.contributions
|
|
362
|
+
if part.raw_value is not None
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
# Best observed raw value per dimension, independent of the weighted winner.
|
|
366
|
+
def best(rows, dimension):
|
|
367
|
+
values = [
|
|
368
|
+
p.raw_value
|
|
369
|
+
for r in rows
|
|
370
|
+
for p in r.contributions
|
|
371
|
+
if p.dimension == dimension and p.raw_value is not None
|
|
372
|
+
]
|
|
373
|
+
if not values:
|
|
374
|
+
return None
|
|
375
|
+
return min(values) if dimension.startswith("-") else max(values)
|
|
376
|
+
|
|
377
|
+
candidate_near_misses = {}
|
|
378
|
+
for i, condition in enumerate(resolved.conditions):
|
|
379
|
+
if condition.soft is not None:
|
|
380
|
+
continue
|
|
381
|
+
text = render_condition(condition)
|
|
382
|
+
single = apply(replace(resolved, conditions=(condition,)), snapshot)
|
|
383
|
+
relaxed = apply(
|
|
384
|
+
replace(resolved, conditions=resolved.conditions[:i] + resolved.conditions[i + 1 :]),
|
|
385
|
+
snapshot,
|
|
386
|
+
)
|
|
387
|
+
alternative = run_optimise(snapshot, relaxed, resolved.spec, selectors, domains)
|
|
388
|
+
gains, units, records, bests = {}, {}, set(), {}
|
|
389
|
+
for dimension in current:
|
|
390
|
+
before, after = best(ordered.results, dimension), best(alternative.results, dimension)
|
|
391
|
+
if before is not None and after is not None:
|
|
392
|
+
gains[dimension] = after - before
|
|
393
|
+
bests[dimension] = (before, after)
|
|
394
|
+
# A gain rests on the two values it subtracts: the records behind those.
|
|
395
|
+
for side, rows in enumerate((ordered.results, alternative.results)):
|
|
396
|
+
for row in rows:
|
|
397
|
+
for part in row.contributions:
|
|
398
|
+
if part.dimension in bests and part.raw_value == bests[part.dimension][side]:
|
|
399
|
+
rids, units[part.dimension], _ = part_provenance(
|
|
400
|
+
snapshot, row.candidate_id, part)
|
|
401
|
+
records.update(rids)
|
|
402
|
+
decision.constraint_costs.append(
|
|
403
|
+
ConstraintCost(
|
|
404
|
+
condition=text,
|
|
405
|
+
admits=len(set(relaxed.feasible) - set(filtered.feasible)),
|
|
406
|
+
gain=gains,
|
|
407
|
+
units=units,
|
|
408
|
+
records=sorted(records),
|
|
409
|
+
)
|
|
410
|
+
)
|
|
411
|
+
for reason in single.eliminated:
|
|
412
|
+
values = list(reason.value) if isinstance(reason.value, (tuple, list)) else []
|
|
413
|
+
value = None if values else reason.value
|
|
414
|
+
ref = offering_ref(snapshot, reason.candidate)
|
|
415
|
+
records = []
|
|
416
|
+
unit = None
|
|
417
|
+
formula = None
|
|
418
|
+
if reason.facet and reason.value is not None:
|
|
419
|
+
fact = snapshot.fact(reason.candidate, reason.facet)
|
|
420
|
+
if computed(snapshot, reason.candidate, reason.facet) is not None:
|
|
421
|
+
records, unit, formula = fact_provenance(
|
|
422
|
+
snapshot, reason.candidate, reason.facet)
|
|
423
|
+
elif fact.record_id:
|
|
424
|
+
checked_record(snapshot, fact.record_id)
|
|
425
|
+
records = [fact.record_id]
|
|
426
|
+
unit = facet_unit(reason.facet)
|
|
427
|
+
else:
|
|
428
|
+
for row in snapshot.evidence(reason.candidate, reason.facet):
|
|
429
|
+
if row.value in (values or [reason.value]):
|
|
430
|
+
checked_record(snapshot, row.record_id)
|
|
431
|
+
records.append(row.record_id)
|
|
432
|
+
unit = row.unit
|
|
433
|
+
if decision.explain == "full":
|
|
434
|
+
decision.eliminated.models.append(
|
|
435
|
+
ModelElimination(
|
|
436
|
+
model=ref.model,
|
|
437
|
+
offering=ref,
|
|
438
|
+
condition=text + (": unverified: may qualify" if reason.unverified else ""),
|
|
439
|
+
value=value,
|
|
440
|
+
values=values,
|
|
441
|
+
unit=unit,
|
|
442
|
+
records=records,
|
|
443
|
+
formula=formula,
|
|
444
|
+
)
|
|
445
|
+
)
|
|
446
|
+
if (
|
|
447
|
+
reason.candidate in relaxed.feasible
|
|
448
|
+
and not reason.unverified
|
|
449
|
+
and reason.candidate not in filtered.feasible
|
|
450
|
+
):
|
|
451
|
+
if not (ref.provider is None and (reason.facet or "").startswith("offering.")):
|
|
452
|
+
candidate_near_misses[reason.candidate] = NearMiss(
|
|
453
|
+
offering=ref,
|
|
454
|
+
condition=text,
|
|
455
|
+
facet=reason.facet,
|
|
456
|
+
value=value,
|
|
457
|
+
values=values,
|
|
458
|
+
distance=_distance(reason, condition),
|
|
459
|
+
unit=unit,
|
|
460
|
+
records=records,
|
|
461
|
+
formula=formula,
|
|
462
|
+
)
|
|
463
|
+
if decision.explain == "full":
|
|
464
|
+
already = {m.offering.model_dump_json() for m in decision.eliminated.models}
|
|
465
|
+
for reason in filtered.eliminated:
|
|
466
|
+
ref = offering_ref(snapshot, reason.candidate)
|
|
467
|
+
if ref.model_dump_json() not in already:
|
|
468
|
+
decision.eliminated.models.append(
|
|
469
|
+
ModelElimination(
|
|
470
|
+
model=ref.model,
|
|
471
|
+
offering=ref,
|
|
472
|
+
condition=reason.condition,
|
|
473
|
+
value=reason.value,
|
|
474
|
+
)
|
|
475
|
+
)
|
|
476
|
+
for cid, dominators in ordered.dominance.items():
|
|
477
|
+
ref = offering_ref(snapshot, cid)
|
|
478
|
+
decision.eliminated.models.append(
|
|
479
|
+
ModelElimination(
|
|
480
|
+
model=ref.model, offering=ref, condition="dominated by " + ", ".join(dominators)
|
|
481
|
+
)
|
|
482
|
+
)
|
|
483
|
+
for row in ordered.results[len(decision.results) :]:
|
|
484
|
+
ref = offering_ref(snapshot, row.candidate_id)
|
|
485
|
+
decision.eliminated.models.append(
|
|
486
|
+
ModelElimination(
|
|
487
|
+
model=ref.model, offering=ref, condition="outside requested result limit"
|
|
488
|
+
)
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
grouped = {}
|
|
492
|
+
for row in decision.eliminated.models:
|
|
493
|
+
group = grouped.setdefault(row.model, {"model": None, "offerings": {}})
|
|
494
|
+
if row.offering is None or row.offering.provider is None:
|
|
495
|
+
group["model"] = group["model"] or row
|
|
496
|
+
else:
|
|
497
|
+
key = row.offering.model_dump_json()
|
|
498
|
+
group["offerings"].setdefault(key, row)
|
|
499
|
+
decision.eliminated.model_groups = [
|
|
500
|
+
ModelEliminationGroup(
|
|
501
|
+
model=model,
|
|
502
|
+
model_elimination=rows["model"],
|
|
503
|
+
offerings=[
|
|
504
|
+
OfferingElimination(**row.model_dump(exclude={"model"}))
|
|
505
|
+
for row in rows["offerings"].values()
|
|
506
|
+
],
|
|
507
|
+
)
|
|
508
|
+
for model, rows in sorted(grouped.items())
|
|
509
|
+
]
|
|
510
|
+
|
|
511
|
+
# A near miss is one model, represented by its best offering. Optimise all
|
|
512
|
+
# offerings of that model without the hard conditions, then retain it only
|
|
513
|
+
# when that best offering failed exactly one condition.
|
|
514
|
+
by_model = {}
|
|
515
|
+
for cid in candidate_near_misses:
|
|
516
|
+
by_model.setdefault(snapshot.model_of(cid), []).append(cid)
|
|
517
|
+
for model in sorted(by_model):
|
|
518
|
+
if any(snapshot.model_of(cid) == model for cid in filtered.feasible):
|
|
519
|
+
continue
|
|
520
|
+
offerings = [
|
|
521
|
+
cid for cid in snapshot.candidates()
|
|
522
|
+
if snapshot.model_of(cid) == model and snapshot.kind(cid) == "offering"
|
|
523
|
+
]
|
|
524
|
+
candidates = offerings or [model]
|
|
525
|
+
unconstrained = replace(filtered, feasible=tuple(candidates), may_qualify=(), eliminated=())
|
|
526
|
+
best = run_optimise(snapshot, unconstrained, resolved.spec, selectors, domains).results
|
|
527
|
+
if best and best[0].candidate_id in candidate_near_misses:
|
|
528
|
+
decision.near_misses.append(candidate_near_misses[best[0].candidate_id])
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def _full(decision, snapshot, ordered, requested, named):
|
|
532
|
+
"""Up to 20 optimised candidates with the facets the spec names, the display
|
|
533
|
+
set, and evidence on the benchmarks the spec names: not every value held."""
|
|
534
|
+
from decision.computed import COMPUTED_FACETS
|
|
535
|
+
from decision.contract import CandidateValues, ShownFact
|
|
536
|
+
from decision.engine import offering_ref
|
|
537
|
+
|
|
538
|
+
shown = named | set(DISPLAY_FACETS)
|
|
539
|
+
stored = [facet for facet in snapshot.facet_ids() if facet in shown]
|
|
540
|
+
benchmarks = named & set(snapshot.benchmark_ids())
|
|
541
|
+
ranked = len(decision.results)
|
|
542
|
+
for position, row in enumerate(ordered.results[:20]):
|
|
543
|
+
cid = row.candidate_id
|
|
544
|
+
facts = []
|
|
545
|
+
for facet in stored:
|
|
546
|
+
fact = snapshot.fact(cid, facet)
|
|
547
|
+
if fact.state != "known":
|
|
548
|
+
continue
|
|
549
|
+
unit = None
|
|
550
|
+
if fact.record_id:
|
|
551
|
+
unit = facet_unit(facet)
|
|
552
|
+
facts.append(ShownFact(
|
|
553
|
+
facet=facet, value=fact.value, unit=unit, record_id=fact.record_id,
|
|
554
|
+
source_ids=source_ids(snapshot, [fact.record_id] if fact.record_id else [])))
|
|
555
|
+
for facet in COMPUTED_FACETS:
|
|
556
|
+
found = computed(snapshot, cid, facet) if facet in shown else None
|
|
557
|
+
if found is not None:
|
|
558
|
+
records, unit, formula = fact_provenance(snapshot, cid, facet)
|
|
559
|
+
facts.append(ShownFact(facet=facet, value=found.value, unit=unit,
|
|
560
|
+
records=records, formula=formula,
|
|
561
|
+
source_ids=source_ids(snapshot, records)))
|
|
562
|
+
# Group each named benchmark under the requested domains that tag it,
|
|
563
|
+
# or, when none does, under the first domain that does.
|
|
564
|
+
domains = set(requested)
|
|
565
|
+
for benchmark in benchmarks:
|
|
566
|
+
tagged = benchmark_domains(snapshot, benchmark)
|
|
567
|
+
if tagged and not requested & set(tagged):
|
|
568
|
+
domains.add(tagged[0])
|
|
569
|
+
evidence = [group for group in domain_evidence(snapshot, cid, domains, benchmarks)
|
|
570
|
+
if group.items]
|
|
571
|
+
decision.top.append(
|
|
572
|
+
CandidateValues(
|
|
573
|
+
offering=offering_ref(snapshot, cid),
|
|
574
|
+
facts=facts,
|
|
575
|
+
evidence=evidence,
|
|
576
|
+
# A ranked candidate's contributions are in ``results``, not repeated.
|
|
577
|
+
contributions=[] if position < ranked else contributions(
|
|
578
|
+
snapshot, cid, row.contributions, evidence),
|
|
579
|
+
)
|
|
580
|
+
)
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def top_contributions(decision):
|
|
584
|
+
"""Each top candidate with its contributions, from ``results`` when ranked."""
|
|
585
|
+
ranked = {r.offering.model_dump_json(): r.contributions for r in decision.results}
|
|
586
|
+
for row in decision.top:
|
|
587
|
+
yield row, row.contributions or ranked.get(row.offering.model_dump_json(), [])
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def contribution_chart(decision):
|
|
591
|
+
"""Faceted raw-value bars: dimensions with different units never share a scale."""
|
|
592
|
+
from html import escape
|
|
593
|
+
|
|
594
|
+
rows = [
|
|
595
|
+
(r.offering.model, c)
|
|
596
|
+
for r, parts in top_contributions(decision)
|
|
597
|
+
for c in parts
|
|
598
|
+
if c.raw_value is not None
|
|
599
|
+
]
|
|
600
|
+
parts = [
|
|
601
|
+
f'<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 800 {max(60, len(rows) * 55)}" '
|
|
602
|
+
'role="img" aria-label="Objective contributions in original units">'
|
|
603
|
+
]
|
|
604
|
+
maxima = {
|
|
605
|
+
c.dimension: max(
|
|
606
|
+
abs(other.raw_value) for _, other in rows if other.dimension == c.dimension
|
|
607
|
+
)
|
|
608
|
+
for _, c in rows
|
|
609
|
+
}
|
|
610
|
+
for i, (model, c) in enumerate(rows):
|
|
611
|
+
width = 300 * abs(c.raw_value) / (maxima[c.dimension] or 1)
|
|
612
|
+
reported = (
|
|
613
|
+
" · Lab-reported"
|
|
614
|
+
if any(e.measured_by == "provider_self_report" for e in c.evidence)
|
|
615
|
+
else ""
|
|
616
|
+
)
|
|
617
|
+
label = escape(
|
|
618
|
+
f"{model}{reported} · {c.dimension}: {c.raw_value:g} {c.unit or 'unit not recorded'}"
|
|
619
|
+
)
|
|
620
|
+
parts += [
|
|
621
|
+
f'<text x="8" y="{i * 55 + 18}">{label}</text>',
|
|
622
|
+
f'<rect x="8" y="{i * 55 + 27}" width="{width:g}" height="12" />',
|
|
623
|
+
]
|
|
624
|
+
return "".join(parts) + "</svg>"
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def numbers(value, path=""):
|
|
628
|
+
"""Walk numeric JSON leaves. IDs, dates and SVG geometry are not measurements."""
|
|
629
|
+
if isinstance(value, bool):
|
|
630
|
+
return
|
|
631
|
+
if isinstance(value, (int, float)):
|
|
632
|
+
yield path, value
|
|
633
|
+
elif isinstance(value, dict):
|
|
634
|
+
for key, child in value.items():
|
|
635
|
+
if key != "number_origins":
|
|
636
|
+
yield from numbers(child, path + "/" + key)
|
|
637
|
+
elif isinstance(value, list):
|
|
638
|
+
for i, child in enumerate(value):
|
|
639
|
+
yield from numbers(child, path + "/" + str(i))
|
|
640
|
+
|
|
641
|
+
|
|
642
|
+
def presented_values(data):
|
|
643
|
+
"""The numbers ``number_origins`` covers: those in the presented sections."""
|
|
644
|
+
def within(node, path, keys):
|
|
645
|
+
if not keys:
|
|
646
|
+
yield from ((p, v) for p, v in numbers(node, path)
|
|
647
|
+
if p.rsplit("/", 1)[-1] not in UNTRACED)
|
|
648
|
+
elif keys[0] == "*":
|
|
649
|
+
for i, child in enumerate(node):
|
|
650
|
+
yield from within(child, f"{path}/{i}", keys[1:])
|
|
651
|
+
else:
|
|
652
|
+
yield from within(node.get(keys[0], []), f"{path}/{keys[0]}", keys[1:])
|
|
653
|
+
|
|
654
|
+
for section in PRESENTED:
|
|
655
|
+
yield from within(data, "", section.split("/"))
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
def _node(data, path):
|
|
659
|
+
parent, ancestors = data, []
|
|
660
|
+
nodes = path.strip("/").split("/")
|
|
661
|
+
for node in nodes[:-1]:
|
|
662
|
+
ancestors.append(parent)
|
|
663
|
+
parent = parent[int(node)] if isinstance(parent, list) else parent[node]
|
|
664
|
+
key = nodes[-1]
|
|
665
|
+
if isinstance(parent, list):
|
|
666
|
+
return ancestors[-1], nodes[-2], int(key)
|
|
667
|
+
return parent, key, None
|
|
668
|
+
|
|
669
|
+
|
|
670
|
+
def number_origins(decision, snapshot):
|
|
671
|
+
"""Distinguish measurements from spec inputs and reproducible calculations.
|
|
672
|
+
|
|
673
|
+
Covers the presented numbers only (``presented_values``). An origin names
|
|
674
|
+
its records and their sources by ID; ``cited_sources`` lists each source
|
|
675
|
+
once.
|
|
676
|
+
"""
|
|
677
|
+
from decision.contract import NumberOrigin
|
|
678
|
+
|
|
679
|
+
data = decision.model_dump(mode="json")
|
|
680
|
+
for path, value in presented_values(data):
|
|
681
|
+
parent, key, list_index = _node(data, path)
|
|
682
|
+
records = parent.get("records", []) if isinstance(parent, dict) else []
|
|
683
|
+
if isinstance(parent, dict) and parent.get("record_id"):
|
|
684
|
+
records = [parent["record_id"]]
|
|
685
|
+
raw = (
|
|
686
|
+
key == "raw_value"
|
|
687
|
+
or key in ("value", "values")
|
|
688
|
+
and ("record_id" in parent or "condition" in parent or "facet" in parent)
|
|
689
|
+
)
|
|
690
|
+
if raw and isinstance(parent, dict) and parent.get("formula"):
|
|
691
|
+
# Computed per decision from the records listed (MODEL-153), so it is
|
|
692
|
+
# checked against the formula, not against one record's value.
|
|
693
|
+
basis = "computed per decision: " + parent["formula"]
|
|
694
|
+
elif raw and records:
|
|
695
|
+
basis = "snapshot measurement"
|
|
696
|
+
|
|
697
|
+
def original(rid):
|
|
698
|
+
record = checked_record(snapshot, rid)
|
|
699
|
+
raw = record.get("score", record.get("value"))
|
|
700
|
+
return raw[list_index] if list_index is not None and isinstance(raw, list) else raw
|
|
701
|
+
|
|
702
|
+
if not any(value == original(rid) for rid in records):
|
|
703
|
+
raise ExplanationError(f"{path}: value differs from retained record")
|
|
704
|
+
elif key == "distance":
|
|
705
|
+
basis = "distance from snapshot value to spec condition boundary, in original units"
|
|
706
|
+
elif "/gain/" in path:
|
|
707
|
+
basis = "best relaxed raw objective value minus best current raw value"
|
|
708
|
+
elif key == "threshold":
|
|
709
|
+
basis = "optimise weight crossing with other weights and normalisation fixed"
|
|
710
|
+
elif key == "value" and "dimension" in parent:
|
|
711
|
+
basis = "feasible-set min-max normalisation of snapshot measurements"
|
|
712
|
+
elif key == "n":
|
|
713
|
+
basis = "sample count from snapshot evidence"
|
|
714
|
+
else:
|
|
715
|
+
basis = "count or ordinal from snapshot candidates after filtering and optimisation"
|
|
716
|
+
records = []
|
|
717
|
+
records = sorted(set(records))
|
|
718
|
+
yield NumberOrigin(
|
|
719
|
+
path=path, basis=basis, records=records, source_ids=source_ids(snapshot, records))
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
def source_ids(snapshot, records):
|
|
723
|
+
"""The registered sources behind checked records, by ID."""
|
|
724
|
+
return sorted({source["source_id"] for rid in records
|
|
725
|
+
for source in checked_record(snapshot, rid)["sources"]})
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
def cited_sources(decision, snapshot):
|
|
729
|
+
"""Each source the origins and shown facts cite, once: URL, and the latest
|
|
730
|
+
date a record in this decision citing it was verified."""
|
|
731
|
+
from decision.contract import CitedSource
|
|
732
|
+
|
|
733
|
+
records = {rid for origin in decision.number_origins for rid in origin.records}
|
|
734
|
+
for row in decision.top:
|
|
735
|
+
for fact in row.facts:
|
|
736
|
+
records.update(fact.records)
|
|
737
|
+
if fact.record_id:
|
|
738
|
+
records.add(fact.record_id)
|
|
739
|
+
verified = {}
|
|
740
|
+
for rid in records:
|
|
741
|
+
record = snapshot.record(rid)
|
|
742
|
+
day = record["verification"].get("date")
|
|
743
|
+
for source in record["sources"]:
|
|
744
|
+
sid = source["source_id"]
|
|
745
|
+
verified.setdefault(sid, None)
|
|
746
|
+
if day and (verified[sid] is None or day > verified[sid]):
|
|
747
|
+
verified[sid] = day
|
|
748
|
+
return [CitedSource(id=sid, url=snapshot.source_url(sid), date=verified[sid])
|
|
749
|
+
for sid in sorted(verified)]
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
def render_html(decision, snapshot):
|
|
753
|
+
"""A self-contained report. Source links are its only external resources."""
|
|
754
|
+
from html import escape
|
|
755
|
+
|
|
756
|
+
def esc(value):
|
|
757
|
+
return escape(str(value))
|
|
758
|
+
|
|
759
|
+
def quantity(value, unit):
|
|
760
|
+
return "unknown" if value is None else f"{value:g} {esc(unit or 'unit not recorded')}"
|
|
761
|
+
|
|
762
|
+
def links(records):
|
|
763
|
+
urls = sorted(
|
|
764
|
+
{
|
|
765
|
+
snapshot.source_url(s["source_id"])
|
|
766
|
+
for rid in records
|
|
767
|
+
for s in checked_record(snapshot, rid)["sources"]
|
|
768
|
+
}
|
|
769
|
+
)
|
|
770
|
+
return " ".join(f'<a href="{esc(url)}">Source</a>' for url in urls)
|
|
771
|
+
|
|
772
|
+
def evidence(groups):
|
|
773
|
+
out = []
|
|
774
|
+
for group in groups:
|
|
775
|
+
out.append(f"<h3>{esc(group.domain)}</h3>")
|
|
776
|
+
if not group.items:
|
|
777
|
+
out.append("<p>No verified evidence for this domain.</p>")
|
|
778
|
+
for item in group.items:
|
|
779
|
+
label = (
|
|
780
|
+
"Lab-reported"
|
|
781
|
+
if item.measured_by == "provider_self_report"
|
|
782
|
+
else item.measured_by
|
|
783
|
+
)
|
|
784
|
+
out.append(
|
|
785
|
+
f"<p><strong>{esc(label)}</strong> · {esc(item.benchmark)} "
|
|
786
|
+
f"· version {esc(item.version or 'not recorded')} "
|
|
787
|
+
f"· {esc(item.sub_category or 'aggregate')} "
|
|
788
|
+
f"· {quantity(item.value, item.unit)} "
|
|
789
|
+
f"· {esc(item.directness)} · {esc(item.date)} ({esc(item.date_type)}) "
|
|
790
|
+
f"· effort {esc(item.effort or 'not recorded')} "
|
|
791
|
+
f"· harness {esc(item.harness or 'not recorded')} "
|
|
792
|
+
f'<a href="{esc(item.source)}">Source</a></p>'
|
|
793
|
+
)
|
|
794
|
+
return "".join(out)
|
|
795
|
+
|
|
796
|
+
def parts(contributions):
|
|
797
|
+
out = []
|
|
798
|
+
for c in contributions:
|
|
799
|
+
out.append(
|
|
800
|
+
f"<p>{esc(c.dimension)}: {quantity(c.raw_value, c.unit)}; "
|
|
801
|
+
f"normalised value {quantity(c.value, 'dimensionless')}; "
|
|
802
|
+
f"weight {quantity(c.weight, 'dimensionless')}. "
|
|
803
|
+
+ (f"Computed: {esc(c.formula)}. " if c.formula else "")
|
|
804
|
+
+ f"{esc(c.normalisation)} {links(c.records)}</p>"
|
|
805
|
+
)
|
|
806
|
+
groups = {}
|
|
807
|
+
for item in c.evidence:
|
|
808
|
+
groups.setdefault(item.requested_domain, []).append(item)
|
|
809
|
+
out.append(
|
|
810
|
+
evidence(
|
|
811
|
+
[DomainEvidence(domain=domain, items=items) for domain, items in groups.items()]
|
|
812
|
+
)
|
|
813
|
+
)
|
|
814
|
+
return "".join(out)
|
|
815
|
+
|
|
816
|
+
out = [
|
|
817
|
+
'<!doctype html><html lang="en"><meta charset="utf-8">',
|
|
818
|
+
'<meta name="viewport" content="width=device-width, initial-scale=1">',
|
|
819
|
+
"<title>ModelSpec decision</title><style>",
|
|
820
|
+
":root{color-scheme:light dark;--bg:#fafafa;--fg:#17212b;--accent:#176b91}",
|
|
821
|
+
"@media (prefers-color-scheme: dark){:root{--bg:#101820;--fg:#e7edf2;--accent:#75c9ed}}",
|
|
822
|
+
"body{background:var(--bg);color:var(--fg);font:16px system-ui;max-width:1000px;",
|
|
823
|
+
"margin:2rem auto;padding:0 1rem;line-height:1.6}a{color:var(--accent)}",
|
|
824
|
+
"section{border-top:1px solid #888;padding:1rem 0}svg{width:100%}",
|
|
825
|
+
"svg text{fill:var(--fg);font:14px system-ui}svg rect{fill:var(--accent)}",
|
|
826
|
+
"</style><body><h1>ModelSpec decision</h1>",
|
|
827
|
+
f"<p>{esc(decision.decision_id)} · {esc(decision.snapshot)} · {esc(decision.status)}</p>",
|
|
828
|
+
"<p>Evidence is unblended. Capability estimates and probabilities are not available.</p>",
|
|
829
|
+
]
|
|
830
|
+
for r in decision.results:
|
|
831
|
+
out.append(f"<section><h2>{esc(r.offering.model)}</h2>")
|
|
832
|
+
if r.offering.provider:
|
|
833
|
+
out.append(
|
|
834
|
+
f"<p>{esc(r.offering.provider)} · {esc(r.offering.region)} "
|
|
835
|
+
f"· {esc(r.offering.tier)}</p>"
|
|
836
|
+
)
|
|
837
|
+
out.append(parts(r.contributions) + evidence(r.evidence) + "</section>")
|
|
838
|
+
if decision.explain == "full":
|
|
839
|
+
# Rebuild SVG from typed values so an imported chart cannot inject HTML.
|
|
840
|
+
out.append(
|
|
841
|
+
"<section><h2>Contribution chart</h2>" + contribution_chart(decision) + "</section>"
|
|
842
|
+
)
|
|
843
|
+
out.append("<section><h2>Funnel</h2>")
|
|
844
|
+
for step in decision.eliminated.funnel:
|
|
845
|
+
out.append(
|
|
846
|
+
f"<p>{esc(step.condition)}: {step.before} candidates before, "
|
|
847
|
+
f"{step.after} candidates after, {step.may_qualify} candidates may qualify.</p>"
|
|
848
|
+
)
|
|
849
|
+
out.append("</section><section><h2>Constraint costs</h2>")
|
|
850
|
+
for cost in decision.constraint_costs:
|
|
851
|
+
out.append(f"<p>Relax {esc(cost.condition)}: admits {cost.admits} candidates.</p>")
|
|
852
|
+
for dimension, gain in cost.gain.items():
|
|
853
|
+
out.append(
|
|
854
|
+
f"<p>{esc(dimension)}: relaxed minus current = "
|
|
855
|
+
f"{quantity(gain, cost.units.get(dimension))}. {links(cost.records)}</p>"
|
|
856
|
+
)
|
|
857
|
+
if not cost.gain:
|
|
858
|
+
out.append("<p>No comparable objective values.</p>")
|
|
859
|
+
out.append("</section><section><h2>Near misses</h2>")
|
|
860
|
+
for miss in decision.near_misses:
|
|
861
|
+
value = (
|
|
862
|
+
quantity(miss.value, miss.unit)
|
|
863
|
+
if isinstance(miss.value, (int, float))
|
|
864
|
+
else esc(miss.value)
|
|
865
|
+
)
|
|
866
|
+
out.append(
|
|
867
|
+
f"<p>{esc(miss.offering.model)} · {esc(miss.condition)}: "
|
|
868
|
+
f"{value}; "
|
|
869
|
+
f"distance to boundary {quantity(miss.distance, miss.unit)}. "
|
|
870
|
+
+ (f"Computed: {esc(miss.formula)}. " if miss.formula else "")
|
|
871
|
+
+ f"{links(miss.records)}</p>"
|
|
872
|
+
)
|
|
873
|
+
out.append("</section><section><h2>Why others did not win</h2>")
|
|
874
|
+
for reason in decision.eliminated.models:
|
|
875
|
+
out.append(f"<p>{esc(reason.model)}: {esc(reason.condition)}. {links(reason.records)}</p>")
|
|
876
|
+
for maybe in decision.may_qualify:
|
|
877
|
+
out.append(
|
|
878
|
+
f"<p>{esc(maybe.model)} may qualify; unknown: {esc(', '.join(maybe.unknown))}</p>"
|
|
879
|
+
)
|
|
880
|
+
for reason in decision.relax:
|
|
881
|
+
out.append(f"<p>{esc(reason)}</p>")
|
|
882
|
+
out.append("</section><section><h2>Tipping points</h2>")
|
|
883
|
+
for point in decision.tipping_points:
|
|
884
|
+
out.append(
|
|
885
|
+
f"<p>{esc(point.dimension)}: {quantity(point.threshold, 'dimensionless weight')}; "
|
|
886
|
+
f"{esc(point.description)}; new top {esc(point.new_top)}.</p>"
|
|
887
|
+
)
|
|
888
|
+
out.append("</section>")
|
|
889
|
+
if decision.top:
|
|
890
|
+
out.append("<section><h2>Top candidates</h2>")
|
|
891
|
+
for row, shown in top_contributions(decision):
|
|
892
|
+
out.append(f"<h3>{esc(row.offering.model)}</h3>")
|
|
893
|
+
for fact in row.facts:
|
|
894
|
+
value = (
|
|
895
|
+
quantity(fact.value, fact.unit)
|
|
896
|
+
if isinstance(fact.value, (int, float)) and not isinstance(fact.value, bool)
|
|
897
|
+
else esc(fact.value)
|
|
898
|
+
)
|
|
899
|
+
provenance = (
|
|
900
|
+
links([fact.record_id]) if fact.record_id
|
|
901
|
+
else f"Computed: {esc(fact.formula)}. {links(fact.records)}" if fact.formula
|
|
902
|
+
else "Identity"
|
|
903
|
+
)
|
|
904
|
+
out.append(f"<p>{esc(fact.facet)}: {value}. {provenance}</p>")
|
|
905
|
+
out.append(parts(shown) + evidence(row.evidence))
|
|
906
|
+
out.append("</section>")
|
|
907
|
+
out.append("</body></html>")
|
|
908
|
+
return "".join(out)
|