modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/contract.py ADDED
@@ -0,0 +1,1575 @@
1
+ """The decision contract, v1 (MODEL-135).
2
+
3
+ A **spec** asks for a decision: conditions on facets, one objective, how much
4
+ explanation to return. A **decision** answers one spec against one snapshot.
5
+ The public document is ``docs/decision-contract.md``; the JSON Schema in
6
+ ``docs/decision-contract.schema.json`` is generated from these types, and
7
+ ``tests/test_decision_contract.py`` keeps the three in agreement.
8
+
9
+ Conditions have two spellings that parse to the same type: the YAML form (a
10
+ mapping) and the compact string form (``swe_bench_pro >= 55 @independent``).
11
+ ``render_condition`` writes the canonical compact form back out.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import hashlib
17
+ import json
18
+ import re
19
+ import typing
20
+ from collections.abc import Callable, Mapping
21
+ from dataclasses import dataclass
22
+ from datetime import date
23
+ from typing import Annotated, Any, Literal
24
+
25
+ import yaml
26
+ from pydantic import (
27
+ AfterValidator,
28
+ BaseModel,
29
+ BeforeValidator,
30
+ ConfigDict,
31
+ Discriminator,
32
+ Field,
33
+ Tag,
34
+ ValidationError,
35
+ WithJsonSchema,
36
+ field_validator,
37
+ model_validator,
38
+ )
39
+
40
+ CONTRACT_VERSION = "1.7"
41
+
42
+ # ── identifiers ────────────────────────────────────────────────────────────
43
+
44
+ FACET_PATTERN = r"^[a-z][a-z0-9_-]*(\.[a-z0-9_-]+)*$"
45
+ SIGNED_FACET_PATTERN = r"^-?[a-z][a-z0-9_-]*(\.[a-z0-9_-]+)*$"
46
+ MODEL_PATTERN = r"^[a-z0-9][a-z0-9._-]*/[a-z0-9][a-z0-9._-]*$"
47
+ HARNESS_PATTERN = r"^[a-z0-9][a-z0-9-]*@[0-9]+\.[0-9]+$"
48
+ EFFORT_PATTERN = r"^[a-z0-9_-]+$"
49
+ SNAPSHOT_PATTERN = r"^snap_[A-Za-z0-9:._-]+$"
50
+ PROFILE_PATTERN = r"^profile:[a-z0-9][a-z0-9-]*$"
51
+ SAVE_AS_PATTERN = r"^[a-z0-9][a-z0-9-]{0,63}$"
52
+ DECISION_ID_PATTERN = r"^dec_[0-9A-Za-z]{8,}$"
53
+ SPEC_HASH_PATTERN = r"^sha256:[0-9a-f]{64}$"
54
+ CODE_PATTERN = r"^[a-z0-9_]+$"
55
+ URL_PATTERN = r"^https?://\S+$"
56
+
57
+
58
+ def _matching(pattern: str, what: str) -> Any:
59
+ """A string type that fails with a message a person can act on."""
60
+ compiled = re.compile(pattern)
61
+
62
+ def check(value: str) -> str:
63
+ if not compiled.fullmatch(value):
64
+ raise ValueError(f"{what} (pattern {pattern}); got {value!r}")
65
+ return value
66
+
67
+ return Annotated[str, AfterValidator(check),
68
+ WithJsonSchema({"type": "string", "pattern": pattern})]
69
+
70
+
71
+ FacetId = _matching(FACET_PATTERN, "a facet ID is lowercase, dotted")
72
+ SignedFacetId = _matching(SIGNED_FACET_PATTERN,
73
+ "a facet ID is lowercase, dotted, with an optional leading - to minimise")
74
+ ModelId = _matching(MODEL_PATTERN, "a model ID is lab/model")
75
+ HarnessId = _matching(HARNESS_PATTERN, "a harness ID is name@major.minor, e.g. claude-code@2.1")
76
+ Effort = _matching(EFFORT_PATTERN, "an effort is a lowercase word")
77
+ SnapshotId = _matching(SNAPSHOT_PATTERN,
78
+ "a decision cites a snapshot ID like snap_2026-09-24T06:00Z")
79
+ ProfileId = _matching(PROFILE_PATTERN, "a profile ID is profile:<name>")
80
+ SaveAs = _matching(SAVE_AS_PATTERN, "save_as is a lowercase slug")
81
+ DecisionId = _matching(DECISION_ID_PATTERN, "a decision ID is dec_<id>")
82
+ SpecHash = _matching(SPEC_HASH_PATTERN, "a spec hash is sha256:<64 hex>")
83
+ Code = _matching(CODE_PATTERN, "a warning is a lowercase code")
84
+ Url = _matching(URL_PATTERN, "a source is an http(s) URL")
85
+
86
+
87
+ def _snapshot_ref(value: str) -> str:
88
+ if value != "latest" and not re.fullmatch(SNAPSHOT_PATTERN, value):
89
+ raise ValueError(f"must be 'latest' or a snapshot ID like snap_2026-09-24T06:00Z; "
90
+ f"got {value!r}")
91
+ return value
92
+
93
+
94
+ SnapshotRef = Annotated[
95
+ str, AfterValidator(_snapshot_ref),
96
+ WithJsonSchema({"anyOf": [{"const": "latest"},
97
+ {"type": "string", "pattern": SNAPSHOT_PATTERN}]}),
98
+ ]
99
+
100
+ # ── closed vocabularies ────────────────────────────────────────────────────
101
+
102
+ Op = Literal["=", "!=", "<", "<=", ">", ">="]
103
+ ORDERED_OPS = frozenset({"<", "<=", ">", ">="})
104
+ UnknownPolicy = Literal["list", "fail", "pass"]
105
+ MeasuredByQualifier = Literal["independent", "provider_self_report", "any"]
106
+ MeasuredBy = Literal["benchmark_author", "independent", "provider_self_report", "modelspec",
107
+ "outcome_protocol"]
108
+ Explain = Literal["none", "summary", "full"]
109
+ Status = Literal["answered", "partial", "no_feasible"]
110
+ DateType = Literal["observed", "published"]
111
+ Directness = Literal["direct", "proxy"]
112
+ CapabilityLevel = Literal["required", "preferred"]
113
+ # The outcome protocol's task types (DPF integration spec §9.3).
114
+ TaskType = Literal["new_feature", "bug_fix", "refactor", "test_writing", "docs", "migration",
115
+ "performance", "security_fix", "review", "analysis", "data_transform",
116
+ "config_infra"]
117
+
118
+ # Compact qualifier keywords that take no argument, and what they set.
119
+ _FLAG_QUALIFIERS: dict[str, tuple[str, Any]] = {
120
+ "@independent": ("measured_by", "independent"),
121
+ "@provider_self_report": ("measured_by", "provider_self_report"),
122
+ "@any": ("measured_by", "any"),
123
+ "@default_effort": ("effort", "default"),
124
+ "@max_effort": ("effort", "max"),
125
+ "@direct": ("direct", True),
126
+ }
127
+ _ARG_QUALIFIERS = {"@effort": "effort", "@harness": "harness"}
128
+ QUALIFIER_KEYWORDS = (*_FLAG_QUALIFIERS, "@effort(x)", "@harness(x)", "measured_after")
129
+
130
+ # Value types a facet can have that carry no order, so <, >, windows and max/min
131
+ # on them are refused. Any other value type from the registry is taken as ordered.
132
+ UNORDERED_VALUE_TYPES = frozenset({"bool", "boolean", "enum", "string", "str", "text", "set",
133
+ "list"})
134
+
135
+
136
+ def _iso_date(value: Any) -> Any:
137
+ if isinstance(value, str) and re.fullmatch(r"\d{4}-\d{2}-\d{2}", value):
138
+ try:
139
+ return date.fromisoformat(value)
140
+ except ValueError:
141
+ raise ValueError(f"{value} is not a real date") from None
142
+ return value
143
+
144
+
145
+ Scalar = Annotated[bool | int | float | date | str, BeforeValidator(_iso_date)]
146
+ #: For a field named ``date``, whose default would shadow the type in its class.
147
+ Day = date
148
+
149
+
150
+ class _Strict(BaseModel):
151
+ model_config = ConfigDict(extra="forbid", populate_by_name=True)
152
+
153
+
154
+ # ── conditions ─────────────────────────────────────────────────────────────
155
+
156
+
157
+ class Soft(_Strict):
158
+ """A soft condition: violating it costs ``penalty`` of the objective instead of eliminating."""
159
+
160
+ penalty: float = Field(gt=0, le=1)
161
+
162
+
163
+ class ModelRef(_Strict):
164
+ """The right-hand side of a relative condition: ``coding >= model(openai/gpt-6-sol)``."""
165
+
166
+ model: ModelId
167
+
168
+
169
+ class EvidenceQualifiers(_Strict):
170
+ """Which evidence may satisfy a condition on an evidence facet."""
171
+
172
+ measured_by: MeasuredByQualifier | None = None
173
+ effort: Effort | None = None
174
+ harness: HarnessId | None = None
175
+ measured_after: date | None = None
176
+ direct: bool = False
177
+
178
+ def is_empty(self) -> bool:
179
+ return self == EvidenceQualifiers()
180
+
181
+
182
+ def _drop_empty_qualifiers(value: Any) -> Any:
183
+ return None if isinstance(value, EvidenceQualifiers) and value.is_empty() else value
184
+
185
+
186
+ Qualifiers = Annotated[EvidenceQualifiers | None, AfterValidator(_drop_empty_qualifiers)]
187
+
188
+
189
+ class Compare(_Strict):
190
+ """``facet op value``; the value may be ``model(<id>)`` for a relative condition."""
191
+
192
+ facet: FacetId
193
+ op: Op
194
+ value: ModelRef | Scalar
195
+ qualifiers: Qualifiers = None
196
+ soft: Soft | None = None
197
+ unknown: UnknownPolicy | None = None
198
+
199
+
200
+ class Window(_Strict):
201
+ """``facet in [low, high]``, both ends inclusive."""
202
+
203
+ facet: FacetId
204
+ between: tuple[Scalar, Scalar]
205
+ qualifiers: Qualifiers = None
206
+ soft: Soft | None = None
207
+ unknown: UnknownPolicy | None = None
208
+
209
+ @field_validator("between")
210
+ @classmethod
211
+ def _ordered(cls, value: tuple[Any, Any]) -> tuple[Any, Any]:
212
+ low, high = value
213
+ if _kind(low) != _kind(high) or _kind(low) in ("bool", "str"):
214
+ raise ValueError("window ends must both be numbers or both be dates")
215
+ if low > high:
216
+ raise ValueError(f"window low bound {low} is above its high bound {high}")
217
+ return value
218
+
219
+
220
+ class InSet(_Strict):
221
+ """``facet in {a, b}`` or ``facet not in {a, b}``. Values are kept sorted and unique."""
222
+
223
+ facet: FacetId
224
+ in_: list[Scalar] | None = Field(default=None, alias="in")
225
+ not_in: list[Scalar] | None = None
226
+ soft: Soft | None = None
227
+ unknown: UnknownPolicy | None = None
228
+
229
+ @field_validator("in_", "not_in")
230
+ @classmethod
231
+ def _canonical(cls, value: list[Any] | None) -> list[Any] | None:
232
+ if value is None:
233
+ return None
234
+ if not value:
235
+ raise ValueError("the set is empty")
236
+ unique = {json.dumps(v, default=str, sort_keys=True): v for v in value}
237
+ return [unique[k] for k in sorted(unique)]
238
+
239
+ @model_validator(mode="after")
240
+ def _one_side(self) -> InSet:
241
+ if (self.in_ is None) == (self.not_in is None):
242
+ raise ValueError("a set condition takes exactly one of in, not_in")
243
+ return self
244
+
245
+
246
+ class Known(_Strict):
247
+ """``known(facet)``: passes when the facet is known. It is never unknown itself."""
248
+
249
+ known: FacetId
250
+ soft: Soft | None = None
251
+
252
+
253
+ def _at_least_two(value: list[Any]) -> list[Any]:
254
+ if len(value) < 2:
255
+ raise ValueError("a group needs at least two conditions")
256
+ return value
257
+
258
+
259
+ class AnyOf(_Strict):
260
+ any: Annotated[list[Condition], AfterValidator(_at_least_two)]
261
+ soft: Soft | None = None
262
+ unknown: UnknownPolicy | None = None
263
+
264
+
265
+ class AllOf(_Strict):
266
+ all: Annotated[list[Condition], AfterValidator(_at_least_two)]
267
+ soft: Soft | None = None
268
+ unknown: UnknownPolicy | None = None
269
+
270
+
271
+ class NotOf(_Strict):
272
+ not_: Condition = Field(alias="not")
273
+ soft: Soft | None = None
274
+ unknown: UnknownPolicy | None = None
275
+
276
+
277
+ CONDITION_TYPES = (Compare, Window, InSet, Known, AnyOf, AllOf, NotOf)
278
+ LEAF_TYPES = (Compare, Window, InSet, Known)
279
+
280
+
281
+ class ConditionError(ValueError):
282
+ """An invalid condition: which one, on which field, and why."""
283
+
284
+ def __init__(self, condition: str, field: str | None, reason: str, path: str = "") -> None:
285
+ super().__init__(f"{condition!r}: {reason}")
286
+ self.condition = condition
287
+ self.field = field
288
+ self.reason = reason
289
+ self.path = path
290
+
291
+
292
+ def _coerce_condition(raw: Any) -> Any:
293
+ if isinstance(raw, CONDITION_TYPES):
294
+ return raw
295
+ return parse_condition(raw)
296
+
297
+
298
+ Condition = Annotated[
299
+ Compare | Window | InSet | Known | AnyOf | AllOf | NotOf,
300
+ BeforeValidator(_coerce_condition,
301
+ json_schema_input_type=str | Compare | Window | InSet | Known | AnyOf | AllOf
302
+ | NotOf),
303
+ ]
304
+
305
+ for _group in (AnyOf, AllOf, NotOf):
306
+ _group.model_rebuild()
307
+
308
+
309
+ # ── parsing conditions ─────────────────────────────────────────────────────
310
+
311
+
312
+ def parse_condition(raw: str | Mapping[str, Any], _path: str = "") -> Any:
313
+ """Parse one condition in either form. Raises ``ConditionError``."""
314
+ if isinstance(raw, str):
315
+ return _parse_compact(raw, _path)
316
+ if isinstance(raw, Mapping):
317
+ return _from_mapping(raw, _path, text=None)
318
+ raise ConditionError(repr(raw), None, "a condition is a string or a mapping", _path)
319
+
320
+
321
+ _STRUCTURAL_KEYS = {"facet", "known", "any", "all", "not"}
322
+
323
+
324
+ def _from_mapping(raw: Mapping[str, Any], path: str, text: str | None) -> Any:
325
+ text = text if text is not None else _describe(raw)
326
+ keys = set(raw)
327
+ if len(keys) == 1 and not keys & _STRUCTURAL_KEYS:
328
+ [(key, value)] = raw.items()
329
+ raise ConditionError(
330
+ f"{key}: {value}", None,
331
+ "YAML split this condition at ': '; quote it, or write soft(0.2) and unknown(fail) "
332
+ "in the compact form", path)
333
+ for group, cls in (("any", AnyOf), ("all", AllOf)):
334
+ if group in keys:
335
+ children = raw[group]
336
+ if not isinstance(children, list):
337
+ raise ConditionError(text, None, f"{group} takes a list of conditions", path)
338
+ built = [parse_condition(child, _join(path, f"{group}[{i}]"))
339
+ for i, child in enumerate(children)]
340
+ return _build(cls, {**raw, group: built}, text, None, path)
341
+ if "not" in keys:
342
+ child = parse_condition(raw["not"], _join(path, "not"))
343
+ return _build(NotOf, {**raw, "not": child}, text, None, path)
344
+ if "known" in keys:
345
+ field = raw["known"] if isinstance(raw["known"], str) else None
346
+ if "unknown" in keys:
347
+ raise ConditionError(text, field,
348
+ "known() is never unknown, so it takes no unknown policy", path)
349
+ return _build(Known, raw, text, field, path)
350
+ if "facet" in keys:
351
+ field = raw["facet"] if isinstance(raw["facet"], str) else None
352
+ if "op" in keys or "value" in keys:
353
+ return _build(Compare, raw, text, field, path)
354
+ if "between" in keys:
355
+ return _build(Window, raw, text, field, path)
356
+ if "in" in keys or "not_in" in keys:
357
+ return _build(InSet, raw, text, field, path)
358
+ raise ConditionError(text, field,
359
+ "a facet condition needs op and value, between, in, or not_in", path)
360
+ raise ConditionError(text, None,
361
+ "not a condition: expected facet, known, any, all or not", path)
362
+
363
+
364
+ def _build(cls: type[BaseModel], data: Mapping[str, Any], text: str, field: str | None,
365
+ path: str) -> Any:
366
+ try:
367
+ return cls.model_validate(data)
368
+ except ValidationError as exc:
369
+ errors = exc.errors()
370
+ # A failed union reports every member; the value_error is the one that says why.
371
+ first = next((e for e in errors if e["type"] == "value_error"), errors[0])
372
+ nested = first.get("ctx", {}).get("error")
373
+ if isinstance(nested, ConditionError):
374
+ raise nested from None
375
+ where = _loc(first["loc"])
376
+ reason = _message(first)
377
+ raise ConditionError(text, field, f"{where}: {reason}" if where else reason,
378
+ path) from None
379
+
380
+
381
+ def _describe(raw: Any) -> str:
382
+ return json.dumps(raw, default=str, sort_keys=True)
383
+
384
+
385
+ def _join(path: str, part: str) -> str:
386
+ return f"{path}.{part}" if path else part
387
+
388
+
389
+ # The compact grammar, tokenised. A word may contain ':' (aws-bedrock:us-east-1)
390
+ # and '@' (claude-code@2.1); a trailing ':' is split off so `unknown: fail` reads.
391
+ _TOKEN = re.compile(r"""
392
+ (?P<space>\s+)
393
+ | (?P<str>"(?:[^"\\]|\\.)*")
394
+ | (?P<op>==|<=|>=|!=|=|<|>)
395
+ | (?P<punct>[\[\]{}(),;])
396
+ | (?P<qual>@[a-z_]+)
397
+ | (?P<word>-?[A-Za-z0-9_][A-Za-z0-9_.:/+@-]*)
398
+ """, re.X)
399
+ _INT = re.compile(r"-?\d+")
400
+ _FLOAT = re.compile(r"-?(\d+\.\d*|\.\d+|\d+(\.\d*)?[eE][+-]?\d+)")
401
+ _DATE = re.compile(r"\d{4}-\d{2}-\d{2}")
402
+
403
+
404
+ @dataclass
405
+ class _Tok:
406
+ kind: str
407
+ text: str
408
+ start: int
409
+ end: int
410
+
411
+
412
+ def _tokenise(text: str) -> list[_Tok]:
413
+ tokens: list[_Tok] = []
414
+ pos = 0
415
+ while pos < len(text):
416
+ m = _TOKEN.match(text, pos)
417
+ if not m:
418
+ raise _SyntaxError(None, f"unexpected character {text[pos]!r} at column {pos + 1}")
419
+ kind = m.lastgroup or ""
420
+ value = m.group()
421
+ if kind == "word" and value.endswith(":") and len(value) > 1:
422
+ tokens.append(_Tok("word", value[:-1], m.start(), m.end() - 1))
423
+ tokens.append(_Tok("punct", ":", m.end() - 1, m.end()))
424
+ elif kind != "space":
425
+ tokens.append(_Tok(kind, value, m.start(), m.end()))
426
+ pos = m.end()
427
+ return tokens
428
+
429
+
430
+ class _SyntaxError(Exception):
431
+ def __init__(self, field: str | None, reason: str) -> None:
432
+ self.field = field
433
+ self.reason = reason
434
+
435
+
436
+ class _Reader:
437
+ def __init__(self, text: str) -> None:
438
+ self.text = text
439
+ self.tokens = _tokenise(text)
440
+ self.i = 0
441
+ self.field: str | None = None
442
+
443
+ def peek(self, offset: int = 0) -> _Tok | None:
444
+ j = self.i + offset
445
+ return self.tokens[j] if j < len(self.tokens) else None
446
+
447
+ def take(self) -> _Tok | None:
448
+ tok = self.peek()
449
+ if tok is not None:
450
+ self.i += 1
451
+ return tok
452
+
453
+ def expect(self, text: str, context: str) -> _Tok:
454
+ tok = self.take()
455
+ if tok is None or tok.text != text:
456
+ got = "the end" if tok is None else repr(tok.text)
457
+ raise _SyntaxError(self.field, f"expected {text!r} {context}, got {got}")
458
+ return tok
459
+
460
+ def done(self) -> bool:
461
+ return self.i >= len(self.tokens)
462
+
463
+
464
+ def _classify(word: str) -> Any:
465
+ if word == "true":
466
+ return True
467
+ if word == "false":
468
+ return False
469
+ if _INT.fullmatch(word):
470
+ return int(word)
471
+ if _FLOAT.fullmatch(word):
472
+ return float(word)
473
+ if _DATE.fullmatch(word):
474
+ try:
475
+ return date.fromisoformat(word)
476
+ except ValueError:
477
+ return word
478
+ return word
479
+
480
+
481
+ def _parse_compact(text: str, path: str) -> Any:
482
+ stripped = text.strip()
483
+ try:
484
+ reader = _Reader(stripped)
485
+ head = reader.peek()
486
+ nxt = reader.peek(1)
487
+ if head is None:
488
+ raise _SyntaxError(None, "empty condition")
489
+ if head.kind == "word" and head.text in ("any", "all", "not") and nxt and nxt.text == "(":
490
+ return _compact_group(reader, stripped, path)
491
+ raw = _compact_leaf(reader)
492
+ _compact_modifiers(reader, raw)
493
+ except _SyntaxError as bad:
494
+ raise ConditionError(text, bad.field, bad.reason, path) from None
495
+ return _from_mapping(raw, path, text=text)
496
+
497
+
498
+ def _compact_group(reader: _Reader, text: str, path: str) -> Any:
499
+ name = reader.take().text # type: ignore[union-attr]
500
+ reader.take() # "("
501
+ children: list[str] = []
502
+ depth = 0
503
+ start = reader.peek().start if reader.peek() else len(text)
504
+ while True:
505
+ tok = reader.take()
506
+ if tok is None:
507
+ raise _SyntaxError(None, f"{name}( is not closed")
508
+ if tok.text in "([{" and tok.kind == "punct":
509
+ depth += 1
510
+ elif tok.text in ")]}" and tok.kind == "punct":
511
+ if depth == 0 and tok.text == ")":
512
+ children.append(text[start:tok.start])
513
+ break
514
+ depth -= 1
515
+ elif tok.text == ";" and depth == 0:
516
+ children.append(text[start:tok.start])
517
+ nxt = reader.peek()
518
+ start = nxt.start if nxt else len(text)
519
+ children = [child.strip() for child in children]
520
+ modifiers: dict[str, Any] = {}
521
+ _compact_modifiers(reader, modifiers)
522
+ if name == "not":
523
+ if len(children) != 1 or not children[0]:
524
+ raise _SyntaxError(None, "not( takes exactly one condition")
525
+ child = parse_condition(children[0], _join(path, "not"))
526
+ return _build(NotOf, {"not": child, **modifiers}, text, None, path)
527
+ if any(not child for child in children):
528
+ raise _SyntaxError(None, f"{name}( has an empty condition")
529
+ built = [parse_condition(child, _join(path, f"{name}[{i}]"))
530
+ for i, child in enumerate(children)]
531
+ return _build(AnyOf if name == "any" else AllOf, {name: built, **modifiers}, text, None, path)
532
+
533
+
534
+ def _compact_leaf(reader: _Reader) -> dict[str, Any]:
535
+ head = reader.take()
536
+ assert head is not None
537
+ if head.kind == "word" and head.text == "known" and reader.peek() and \
538
+ reader.peek().text == "(": # type: ignore[union-attr]
539
+ reader.take()
540
+ facet = reader.take()
541
+ if facet is None or facet.kind != "word":
542
+ raise _SyntaxError(None, "known( needs a facet")
543
+ reader.field = facet.text
544
+ reader.expect(")", "after known(<facet>")
545
+ return {"known": facet.text}
546
+ if head.kind != "word":
547
+ raise _SyntaxError(None, f"a condition starts with a facet, got {head.text!r}")
548
+ reader.field = head.text
549
+ tok = reader.take()
550
+ if tok is None:
551
+ raise _SyntaxError(head.text, "missing operator after the facet")
552
+ if tok.kind == "op":
553
+ if tok.text == "==":
554
+ raise _SyntaxError(head.text, "unknown operator '=='; use =")
555
+ return {"facet": head.text, "op": tok.text, "value": _compact_value(reader, tok.text)}
556
+ negate = False
557
+ if tok.text == "not":
558
+ negate = True
559
+ tok = reader.take()
560
+ if tok is None or tok.text != "in":
561
+ got = "the end" if tok is None else repr(tok.text)
562
+ raise _SyntaxError(head.text, f"expected an operator (= != < <= > >=) or in, got {got}")
563
+ opener = reader.take()
564
+ if opener is not None and opener.text == "[":
565
+ if negate:
566
+ raise _SyntaxError(head.text, "not in takes a set {…}, not a window […]")
567
+ low = _compact_value(reader, "[")
568
+ reader.expect(",", "between the window's ends")
569
+ high = _compact_value(reader, ",")
570
+ reader.expect("]", "to close the window")
571
+ return {"facet": head.text, "between": [low, high]}
572
+ if opener is not None and opener.text == "{":
573
+ values: list[Any] = []
574
+ if reader.peek() and reader.peek().text == "}": # type: ignore[union-attr]
575
+ reader.take()
576
+ else:
577
+ while True:
578
+ values.append(_compact_value(reader, "{"))
579
+ sep = reader.take()
580
+ if sep is not None and sep.text == "}":
581
+ break
582
+ if sep is None or sep.text != ",":
583
+ raise _SyntaxError(head.text, "expected , or } in the set")
584
+ return {"facet": head.text, ("not_in" if negate else "in"): values}
585
+ raise _SyntaxError(head.text, "in takes a window [low, high] or a set {a, b}")
586
+
587
+
588
+ def _compact_value(reader: _Reader, after: str) -> Any:
589
+ tok = reader.take()
590
+ if tok is None or tok.kind not in ("word", "str"):
591
+ raise _SyntaxError(reader.field, f"missing value after {after}")
592
+ if tok.kind == "str":
593
+ return json.loads(tok.text)
594
+ if tok.text == "model" and reader.peek() and reader.peek().text == "(": # type: ignore[union-attr]
595
+ reader.take()
596
+ parts = []
597
+ while reader.peek() is not None and reader.peek().text != ")": # type: ignore[union-attr]
598
+ parts.append(reader.take().text) # type: ignore[union-attr]
599
+ reader.expect(")", "to close model(")
600
+ return {"model": " ".join(parts)}
601
+ return _classify(tok.text)
602
+
603
+
604
+ def _compact_modifiers(reader: _Reader, raw: dict[str, Any]) -> None:
605
+ qualifiers: dict[str, Any] = {}
606
+ field = reader.field
607
+
608
+ def put(key: str, value: Any, spelled: str) -> None:
609
+ if key in qualifiers and qualifiers[key] != value:
610
+ raise _SyntaxError(
611
+ field, f"{key} given twice ({spelled} conflicts with an earlier qualifier)")
612
+ qualifiers[key] = value
613
+
614
+ while not reader.done():
615
+ tok = reader.take()
616
+ assert tok is not None
617
+ if tok.kind == "qual":
618
+ if tok.text in _FLAG_QUALIFIERS:
619
+ key, value = _FLAG_QUALIFIERS[tok.text]
620
+ put(key, value, tok.text)
621
+ elif tok.text in _ARG_QUALIFIERS:
622
+ reader.expect("(", f"after {tok.text}")
623
+ arg = reader.take()
624
+ if arg is None or arg.kind != "word":
625
+ raise _SyntaxError(field, f"{tok.text}( needs a value")
626
+ reader.expect(")", f"to close {tok.text}(")
627
+ put(_ARG_QUALIFIERS[tok.text], arg.text, tok.text)
628
+ else:
629
+ known = ", ".join(QUALIFIER_KEYWORDS)
630
+ raise _SyntaxError(field, f"unknown qualifier {tok.text}; expected one of {known}")
631
+ elif tok.text == "measured_after":
632
+ when = reader.take()
633
+ value = _classify(when.text) if when is not None else None
634
+ if not isinstance(value, date):
635
+ got = "nothing" if when is None else repr(when.text)
636
+ raise _SyntaxError(field, f"measured_after needs a date (YYYY-MM-DD), got {got}")
637
+ put("measured_after", value.isoformat(), "measured_after")
638
+ elif tok.text in ("soft", "unknown"):
639
+ raw[tok.text] = _compact_call(reader, tok.text)
640
+ else:
641
+ raise _SyntaxError(field, f"unexpected {tok.text!r}; a condition ends with qualifiers, "
642
+ "measured_after, soft(…) or unknown(…)")
643
+ if qualifiers:
644
+ raw["qualifiers"] = qualifiers
645
+
646
+
647
+ def _compact_call(reader: _Reader, name: str) -> Any:
648
+ """``soft(0.2)``, ``soft(penalty: 0.2)``, ``unknown(fail)`` or ``unknown: fail``."""
649
+ field = reader.field
650
+ tok = reader.take()
651
+ if name == "unknown" and tok is not None and tok.text == ":":
652
+ value = reader.take()
653
+ if value is None:
654
+ raise _SyntaxError(field, "unknown: needs list, fail or pass")
655
+ return value.text
656
+ if tok is None or tok.text != "(":
657
+ raise _SyntaxError(field, f"{name} is written {name}(…)")
658
+ arg = reader.take()
659
+ if name == "soft" and arg is not None and arg.text == "penalty":
660
+ reader.expect(":", "after soft(penalty")
661
+ arg = reader.take()
662
+ if arg is None or arg.kind != "word":
663
+ raise _SyntaxError(field, f"{name}( needs a value")
664
+ reader.expect(")", f"to close {name}(")
665
+ return {"penalty": _classify(arg.text)} if name == "soft" else arg.text
666
+
667
+
668
+ # ── rendering conditions ───────────────────────────────────────────────────
669
+
670
+ _BARE = re.compile(r"[A-Za-z0-9_][A-Za-z0-9_.:/+@-]*")
671
+ _RESERVED_WORDS = {"in", "not", "measured_after", "soft", "unknown"}
672
+
673
+
674
+ def _render_value(value: Any) -> str:
675
+ if isinstance(value, ModelRef):
676
+ return f"model({value.model})"
677
+ if isinstance(value, bool):
678
+ return "true" if value else "false"
679
+ if isinstance(value, int | float):
680
+ return json.dumps(value)
681
+ if isinstance(value, date):
682
+ return value.isoformat()
683
+ if (_BARE.fullmatch(value) and not value.endswith(":") and _classify(value) == value
684
+ and value not in _RESERVED_WORDS):
685
+ return value
686
+ return json.dumps(value)
687
+
688
+
689
+ def _render_modifiers(cond: Any) -> str:
690
+ parts: list[str] = []
691
+ q = getattr(cond, "qualifiers", None)
692
+ if q is not None:
693
+ if q.measured_by:
694
+ parts.append(f"@{q.measured_by}")
695
+ if q.effort in ("default", "max"):
696
+ parts.append(f"@{q.effort}_effort")
697
+ elif q.effort:
698
+ parts.append(f"@effort({q.effort})")
699
+ if q.harness:
700
+ parts.append(f"@harness({q.harness})")
701
+ if q.direct:
702
+ parts.append("@direct")
703
+ if q.measured_after:
704
+ parts.append(f"measured_after {q.measured_after.isoformat()}")
705
+ if cond.soft is not None:
706
+ parts.append(f"soft({json.dumps(cond.soft.penalty)})")
707
+ if getattr(cond, "unknown", None) is not None:
708
+ parts.append(f"unknown({cond.unknown})")
709
+ return "".join(f" {p}" for p in parts)
710
+
711
+
712
+ def render_condition(cond: Any) -> str:
713
+ """The canonical compact form. ``parse_condition(render_condition(c)) == c``."""
714
+ if isinstance(cond, Compare):
715
+ body = f"{cond.facet} {cond.op} {_render_value(cond.value)}"
716
+ elif isinstance(cond, Window):
717
+ low, high = cond.between
718
+ body = f"{cond.facet} in [{_render_value(low)}, {_render_value(high)}]"
719
+ elif isinstance(cond, InSet):
720
+ values = ", ".join(_render_value(v) for v in (cond.in_ or cond.not_in or []))
721
+ body = f"{cond.facet} {'in' if cond.in_ is not None else 'not in'} {{{values}}}"
722
+ elif isinstance(cond, Known):
723
+ body = f"known({cond.known})"
724
+ elif isinstance(cond, AnyOf | AllOf):
725
+ name = "any" if isinstance(cond, AnyOf) else "all"
726
+ children = cond.any if isinstance(cond, AnyOf) else cond.all
727
+ body = f"{name}({'; '.join(render_condition(ch) for ch in children)})"
728
+ elif isinstance(cond, NotOf):
729
+ body = f"not({render_condition(cond.not_)})"
730
+ else:
731
+ raise TypeError(f"not a condition: {cond!r}")
732
+ return body + _render_modifiers(cond)
733
+
734
+
735
+ # ── the objective ──────────────────────────────────────────────────────────
736
+
737
+
738
+ class Tolerance(_Strict):
739
+ """How far below the best a lexicographic step may be and still tie: ``5%`` or a number."""
740
+
741
+ relative: float | None = Field(default=None, gt=0, lt=1)
742
+ absolute: float | None = Field(default=None, gt=0)
743
+
744
+ @model_validator(mode="before")
745
+ @classmethod
746
+ def _from_shorthand(cls, value: Any) -> Any:
747
+ if isinstance(value, str):
748
+ text = value.strip()
749
+ if text.endswith("%"):
750
+ return {"relative": float(text[:-1]) / 100}
751
+ return {"absolute": float(text)}
752
+ if isinstance(value, int | float) and not isinstance(value, bool):
753
+ return {"absolute": value}
754
+ return value
755
+
756
+ @model_validator(mode="after")
757
+ def _one(self) -> Tolerance:
758
+ if (self.relative is None) == (self.absolute is None):
759
+ raise ValueError("a tolerance is exactly one of relative, absolute")
760
+ return self
761
+
762
+
763
+ class LexStep(_Strict):
764
+ max: FacetId | None = None
765
+ min: FacetId | None = None
766
+ within: Tolerance | None = None
767
+
768
+ @model_validator(mode="before")
769
+ @classmethod
770
+ def _split_within(cls, value: Any) -> Any:
771
+ if isinstance(value, Mapping):
772
+ value = dict(value)
773
+ for side in ("max", "min"):
774
+ text = value.get(side)
775
+ if isinstance(text, str) and " within " in text:
776
+ facet, tolerance = text.split(" within ", 1)
777
+ value[side] = facet.strip()
778
+ value["within"] = tolerance.strip()
779
+ return value
780
+
781
+ @model_validator(mode="after")
782
+ def _one(self) -> LexStep:
783
+ if (self.max is None) == (self.min is None):
784
+ raise ValueError("a lexicographic step is exactly one of max, min")
785
+ return self
786
+
787
+ @property
788
+ def facet(self) -> str:
789
+ return self.max or self.min or ""
790
+
791
+
792
+ def _base(signed: str) -> str:
793
+ return signed.removeprefix("-")
794
+
795
+
796
+ def _objective_term(text: str) -> tuple[str, EvidenceQualifiers | None]:
797
+ try:
798
+ reader = _Reader(text.strip())
799
+ head = reader.take()
800
+ if head is None or head.kind != "word":
801
+ raise _SyntaxError(None, "an objective term starts with a facet")
802
+ reader.field = head.text.removeprefix("-")
803
+ raw: dict[str, Any] = {}
804
+ _compact_modifiers(reader, raw)
805
+ if set(raw) - {"qualifiers"}:
806
+ raise _SyntaxError(reader.field, "objective terms take evidence qualifiers only")
807
+ qualifiers = EvidenceQualifiers.model_validate(raw["qualifiers"]) \
808
+ if raw.get("qualifiers") else None
809
+ return head.text, qualifiers
810
+ except _SyntaxError as exc:
811
+ raise ValueError(exc.reason) from None
812
+
813
+
814
+ class Objective(_Strict):
815
+ """Exactly one of ``max``, ``min``, ``lexicographic``, ``weights``, ``pareto``."""
816
+
817
+ max: FacetId | None = None
818
+ min: FacetId | None = None
819
+ lexicographic: list[LexStep] | None = None
820
+ weights: dict[SignedFacetId, float] | None = None
821
+ pareto: list[SignedFacetId] | None = None
822
+ qualifiers: dict[FacetId, EvidenceQualifiers] = Field(default_factory=dict)
823
+
824
+ @model_validator(mode="before")
825
+ @classmethod
826
+ def _qualified_terms(cls, value: Any) -> Any:
827
+ if not isinstance(value, Mapping):
828
+ return value
829
+ data = dict(value)
830
+ qualifiers = dict(data.get("qualifiers") or {})
831
+
832
+ def parse(term: Any) -> Any:
833
+ if not isinstance(term, str):
834
+ return term
835
+ facet, found = _objective_term(term)
836
+ base = _base(facet)
837
+ if found is not None:
838
+ prior = qualifiers.get(base)
839
+ dumped = found.model_dump(exclude_none=True)
840
+ if prior is not None and EvidenceQualifiers.model_validate(prior) != found:
841
+ raise ValueError(f"conflicting evidence qualifiers for {base}")
842
+ qualifiers[base] = dumped
843
+ return facet
844
+
845
+ for side in ("max", "min"):
846
+ if side in data:
847
+ data[side] = parse(data[side])
848
+ if isinstance(data.get("weights"), Mapping):
849
+ data["weights"] = {parse(term): weight for term, weight in data["weights"].items()}
850
+ if isinstance(data.get("pareto"), list):
851
+ data["pareto"] = [parse(term) for term in data["pareto"]]
852
+ if isinstance(data.get("lexicographic"), list):
853
+ steps = []
854
+ for raw in data["lexicographic"]:
855
+ if not isinstance(raw, Mapping):
856
+ steps.append(raw)
857
+ continue
858
+ step = dict(raw)
859
+ for side in ("max", "min"):
860
+ if side in step:
861
+ term = step[side]
862
+ if isinstance(term, str) and " within " in term:
863
+ term, tolerance = term.split(" within ", 1)
864
+ step["within"] = tolerance.strip()
865
+ step[side] = parse(term)
866
+ steps.append(step)
867
+ data["lexicographic"] = steps
868
+ if qualifiers:
869
+ data["qualifiers"] = qualifiers
870
+ return data
871
+
872
+ @field_validator("lexicographic")
873
+ @classmethod
874
+ def _lex(cls, steps: list[LexStep] | None) -> list[LexStep] | None:
875
+ if steps is None:
876
+ return None
877
+ if len(steps) < 2:
878
+ raise ValueError("lexicographic needs at least two steps; use max or min for one")
879
+ if steps[-1].within is not None:
880
+ raise ValueError("the last lexicographic step has nothing after it to break ties, "
881
+ "so it takes no within")
882
+ return steps
883
+
884
+ @field_validator("weights")
885
+ @classmethod
886
+ def _weights(cls, weights: dict[str, float] | None) -> dict[str, float] | None:
887
+ if weights is None:
888
+ return None
889
+ if not weights:
890
+ raise ValueError("weights is empty")
891
+ for facet, weight in weights.items():
892
+ if not weight > 0:
893
+ raise ValueError(f"weight for {facet} must be positive; put - on the facet "
894
+ "to minimise it")
895
+ _no_twice(list(weights))
896
+ return weights
897
+
898
+ @field_validator("pareto")
899
+ @classmethod
900
+ def _pareto(cls, dims: list[str] | None) -> list[str] | None:
901
+ if dims is None:
902
+ return None
903
+ if len(dims) < 2:
904
+ raise ValueError("pareto needs at least two dimensions")
905
+ _no_twice(dims)
906
+ return dims
907
+
908
+ @model_validator(mode="after")
909
+ def _exactly_one(self) -> Objective:
910
+ forms = [name for name in ("max", "min", "lexicographic", "weights", "pareto")
911
+ if getattr(self, name) is not None]
912
+ if len(forms) != 1:
913
+ got = ", ".join(forms) or "none"
914
+ raise ValueError("optimize takes exactly one of max, min, lexicographic, weights, "
915
+ f"pareto; got {got}")
916
+ return self
917
+
918
+
919
+ def _no_twice(signed: list[str]) -> None:
920
+ seen: set[str] = set()
921
+ for facet in signed:
922
+ if _base(facet) in seen:
923
+ raise ValueError(f"{_base(facet)} appears twice")
924
+ seen.add(_base(facet))
925
+
926
+
927
+ # ── the inventory profile ──────────────────────────────────────────────────
928
+
929
+
930
+ class ProfileOffering(_Strict):
931
+ model: ModelId
932
+ provider: str
933
+ region: str | None = None
934
+ tier: str | None = None
935
+
936
+
937
+ class Hardware(_Strict):
938
+ class_: str = Field(alias="class")
939
+ count: int = Field(default=1, ge=1)
940
+ memory_gb: float | None = Field(default=None, gt=0)
941
+
942
+
943
+ class LocalModel(_Strict):
944
+ model: ModelId
945
+ hardware: Hardware | None = None
946
+ runtime: str | None = None
947
+
948
+
949
+ class Budget(_Strict):
950
+ max_cost_per_task_usd: float | None = Field(default=None, gt=0)
951
+
952
+
953
+ class InventoryProfile(_Strict):
954
+ profile_version: Literal[1]
955
+ id: ProfileId | None = None
956
+ offerings: list[ProfileOffering] = Field(default_factory=list)
957
+ local: list[LocalModel] = Field(default_factory=list)
958
+ harnesses: list[HarnessId] = Field(default_factory=list)
959
+ rules: list[Condition] = Field(default_factory=list)
960
+ budget: Budget | None = None
961
+
962
+
963
+ def _profile_tag(value: Any) -> str:
964
+ return "<profile-id>" if isinstance(value, str) else "<profile-inline>"
965
+
966
+
967
+ ProfileRef = Annotated[
968
+ Annotated[ProfileId, Tag("<profile-id>")] | Annotated[InventoryProfile,
969
+ Tag("<profile-inline>")],
970
+ Discriminator(_profile_tag),
971
+ ]
972
+
973
+ # ── the spec ───────────────────────────────────────────────────────────────
974
+
975
+
976
+ class TaskTokens(_Strict):
977
+ """How many tokens one task takes. ``offering.cost_per_task`` is priced from it."""
978
+
979
+ input: int = Field(ge=0)
980
+ output: int = Field(ge=0)
981
+
982
+
983
+ #: What ``offering.cost_per_task`` is priced at when a spec gives no ``task_tokens``.
984
+ DEFAULT_TASK_TOKENS = TaskTokens(input=40000, output=4000)
985
+
986
+
987
+ class Spec(_Strict):
988
+ """A request for a decision."""
989
+
990
+ spec_version: Literal[1]
991
+ snapshot: SnapshotRef = "latest"
992
+ profile: ProfileRef | None = None
993
+ task: str | None = None
994
+ task_type: TaskType | None = None
995
+ capabilities: dict[FacetId, CapabilityLevel] | None = None
996
+ #: Tokens per task, for ``offering.cost_per_task``. Added in 1.3.
997
+ task_tokens: TaskTokens | None = None
998
+ where: list[Condition] = Field(default_factory=list)
999
+ optimize: Objective
1000
+ unknowns: Literal["default"] = "default"
1001
+ explain: Explain = "summary"
1002
+ limit: int = Field(default=20, ge=1, le=500)
1003
+ save_as: SaveAs | None = None
1004
+
1005
+
1006
+ # ── the decision ───────────────────────────────────────────────────────────
1007
+
1008
+
1009
+ class OfferingRef(_Strict):
1010
+ model: ModelId
1011
+ provider: str | None = None
1012
+ region: str | None = None
1013
+ tier: str | None = None
1014
+
1015
+
1016
+ class EvidenceItem(_Strict):
1017
+ requested_domain: FacetId | None = None
1018
+ record_id: str | None = None
1019
+ benchmark: str
1020
+ version: str | None = None
1021
+ sub_category: str | None = None
1022
+ value: float
1023
+ unit: str | None = None
1024
+ n: int | None = Field(default=None, ge=1)
1025
+ measured_by: MeasuredBy
1026
+ effort: Effort | None = None
1027
+ harness: HarnessId | None = None
1028
+ #: The evidence names a harness the registry does not know (MODEL-133's
1029
+ #: ``unregistered``). ``harness`` is then null. Added in 1.2.
1030
+ harness_unregistered: bool = False
1031
+ date: date
1032
+ date_type: DateType
1033
+ source: Url
1034
+ source_snapshot: str | None = None
1035
+ directness: Directness
1036
+ #: Directness loading used in the capability estimate. Added in 1.7.
1037
+ loading: float | None = None
1038
+ #: Share of the estimate's tagged measurement precision. Added in 1.7.
1039
+ estimate_weight: float | None = None
1040
+ #: Precision multiplier from the observation's age. Added in 1.7.
1041
+ recency_weight: float | None = Field(default=None, ge=0, le=1)
1042
+
1043
+ @model_validator(mode="after")
1044
+ def _one_harness(self) -> EvidenceItem:
1045
+ if self.harness_unregistered and self.harness is not None:
1046
+ raise ValueError("an unregistered harness has no harness ID")
1047
+ return self
1048
+
1049
+
1050
+ class DomainEvidence(_Strict):
1051
+ domain: FacetId
1052
+ items: list[EvidenceItem]
1053
+
1054
+
1055
+ class Estimate(_Strict):
1056
+ domain: FacetId
1057
+ value: float
1058
+ interval: tuple[float, float]
1059
+ harness: HarnessId | None = None
1060
+ effort: Effort | None = None
1061
+
1062
+
1063
+ class Contribution(_Strict):
1064
+ raw_value: float | None = None
1065
+ unit: str | None = None
1066
+ records: list[str] = Field(default_factory=list)
1067
+ dimension: SignedFacetId
1068
+ weight: float | None = None
1069
+ value: float | None = None
1070
+ normalisation: str | None = None
1071
+ evidence: list[EvidenceItem] = Field(default_factory=list)
1072
+ #: How a computed raw value was reached, with the numbers. Added in 1.3.
1073
+ formula: str | None = None
1074
+
1075
+
1076
+ class Result(_Strict):
1077
+ rank: int = Field(ge=1)
1078
+ offering: OfferingRef
1079
+ harness: HarnessId | None = None
1080
+ effort: Effort | None = None
1081
+ evidence: list[DomainEvidence] = Field(default_factory=list)
1082
+ estimates: list[Estimate] | None = None
1083
+ p_best: float | None = Field(default=None, ge=0, le=1)
1084
+ top3_stability: float | None = Field(default=None, ge=0, le=1)
1085
+ soft_penalty: float = Field(default=0.0, ge=0)
1086
+ contributions: list[Contribution] = Field(default_factory=list)
1087
+ warnings: list[Code] = Field(default_factory=list)
1088
+
1089
+
1090
+ class MayQualify(_Strict):
1091
+ model: ModelId
1092
+ offering: OfferingRef | None = None
1093
+ unknown: list[FacetId]
1094
+
1095
+
1096
+ class FunnelStep(_Strict):
1097
+ condition: str
1098
+ before: int = Field(ge=0)
1099
+ after: int = Field(ge=0)
1100
+ may_qualify: int = Field(default=0, ge=0)
1101
+ #: Candidate-grained counts above remain for compatibility. These counts
1102
+ #: expose the same stage at the two grains people make decisions about.
1103
+ #: Added in 1.6.
1104
+ models_before: int = Field(default=0, ge=0)
1105
+ models_after: int = Field(default=0, ge=0)
1106
+ offerings_before: int = Field(default=0, ge=0)
1107
+ offerings_after: int = Field(default=0, ge=0)
1108
+ models_may_qualify: int = Field(default=0, ge=0)
1109
+ offerings_may_qualify: int = Field(default=0, ge=0)
1110
+
1111
+
1112
+ class ModelElimination(_Strict):
1113
+ values: list[Scalar] = Field(default_factory=list)
1114
+ offering: OfferingRef | None = None
1115
+ unit: str | None = None
1116
+ records: list[str] = Field(default_factory=list)
1117
+ model: ModelId
1118
+ condition: str
1119
+ value: Scalar | None = None
1120
+ #: How a computed value was reached, with the numbers. Added in 1.3.
1121
+ formula: str | None = None
1122
+
1123
+
1124
+ class OfferingElimination(_Strict):
1125
+ """Why one offering of an eliminated model left the lineup. Added in 1.6."""
1126
+
1127
+ values: list[Scalar] = Field(default_factory=list)
1128
+ offering: OfferingRef
1129
+ unit: str | None = None
1130
+ records: list[str] = Field(default_factory=list)
1131
+ condition: str
1132
+ value: Scalar | None = None
1133
+ formula: str | None = None
1134
+
1135
+
1136
+ class ModelEliminationGroup(_Strict):
1137
+ """One eliminated model, with its model row or offering rows. Added in 1.6."""
1138
+
1139
+ model: ModelId
1140
+ model_elimination: ModelElimination | None = None
1141
+ offerings: list[OfferingElimination] = Field(default_factory=list)
1142
+
1143
+
1144
+ class Eliminated(_Strict):
1145
+ funnel: list[FunnelStep] = Field(default_factory=list)
1146
+ models: list[ModelElimination] = Field(default_factory=list)
1147
+ #: The candidate-grained ``models`` list remains for older clients.
1148
+ #: Added in 1.6.
1149
+ model_groups: list[ModelEliminationGroup] = Field(default_factory=list)
1150
+
1151
+
1152
+ class ConstraintCost(_Strict):
1153
+ units: dict[str, str | None] = Field(default_factory=dict)
1154
+ records: list[str] = Field(default_factory=list)
1155
+ condition: str
1156
+ admits: int = Field(ge=0)
1157
+ gain: dict[SignedFacetId, float] = Field(default_factory=dict)
1158
+
1159
+
1160
+ class TippingPoint(_Strict):
1161
+ description: str
1162
+ dimension: SignedFacetId | None = None
1163
+ threshold: float | None = None
1164
+ new_top: ModelId | None = None
1165
+
1166
+
1167
+ class NearMiss(_Strict):
1168
+ values: list[Scalar] = Field(default_factory=list)
1169
+ offering: OfferingRef
1170
+ condition: str
1171
+ facet: str | None = None
1172
+ value: Scalar | None = None
1173
+ distance: float | None = None
1174
+ unit: str | None = None
1175
+ records: list[str] = Field(default_factory=list)
1176
+ #: How a computed value was reached, with the numbers. Added in 1.3.
1177
+ formula: str | None = None
1178
+
1179
+
1180
+ class ShownFact(_Strict):
1181
+ facet: str
1182
+ value: Scalar | list[Scalar] | None = None
1183
+ unit: str | None = None
1184
+ record_id: str | None = None
1185
+ #: A computed fact has no record of its own: the records it was computed
1186
+ #: from, and the formula with the numbers. Added in 1.3.
1187
+ records: list[str] = Field(default_factory=list)
1188
+ formula: str | None = None
1189
+ #: The registered sources behind ``record_id`` or ``records``, as IDs into
1190
+ #: ``Decision.sources``. Added in 1.4.
1191
+ source_ids: list[str] = Field(default_factory=list)
1192
+
1193
+
1194
+ class CandidateValues(_Strict):
1195
+ offering: OfferingRef
1196
+ facts: list[ShownFact] = Field(default_factory=list)
1197
+ contributions: list[Contribution] = Field(default_factory=list)
1198
+ evidence: list[DomainEvidence] = Field(default_factory=list)
1199
+
1200
+
1201
+ class CitedSource(_Strict):
1202
+ """One registered source, listed once per decision. Added in 1.4."""
1203
+
1204
+ id: str
1205
+ url: Url
1206
+ #: The snapshot records no titles yet: null, never a guess.
1207
+ title: str | None = None
1208
+ #: The latest date a record in this decision citing it was verified.
1209
+ date: Day | None = None
1210
+
1211
+
1212
+ class NumberOrigin(_Strict):
1213
+ path: str
1214
+ basis: str
1215
+ records: list[str] = Field(default_factory=list)
1216
+ #: Always empty from 1.4: ``source_ids`` name entries of ``Decision.sources``.
1217
+ sources: list[Url] = Field(default_factory=list)
1218
+ #: The registered sources behind ``records``. Added in 1.4.
1219
+ source_ids: list[str] = Field(default_factory=list)
1220
+
1221
+
1222
+ class Relaxation(_Strict):
1223
+ """The smallest change to one numeric cap or floor that admits a model. Added in 1.5.
1224
+
1225
+ ``condition`` is the spec's condition; ``relaxed`` is the same facet and
1226
+ direction at ``value``, in ``unit``, the nearest value any excluded
1227
+ candidate has; ``admits`` counts the models that then qualify.
1228
+ """
1229
+
1230
+ condition: str
1231
+ relaxed: str
1232
+ facet: str
1233
+ value: float
1234
+ unit: str | None = None
1235
+ admits: int = Field(ge=1)
1236
+
1237
+
1238
+ class Decision(_Strict):
1239
+ """The engine's answer to one spec against one snapshot."""
1240
+
1241
+ near_misses: list[NearMiss] = Field(default_factory=list)
1242
+ top: list[CandidateValues] = Field(default_factory=list)
1243
+ chart: str | None = None
1244
+ number_origins: list[NumberOrigin] = Field(default_factory=list)
1245
+ #: Every source the number origins cite, once each. Added in 1.4.
1246
+ sources: list[CitedSource] = Field(default_factory=list)
1247
+ contract_version: Literal["1.7"] = CONTRACT_VERSION
1248
+ decision_id: DecisionId
1249
+ snapshot: SnapshotId
1250
+ spec_hash: SpecHash
1251
+ explain: Explain
1252
+ status: Status
1253
+ results: list[Result] = Field(default_factory=list)
1254
+ may_qualify: list[MayQualify] = Field(default_factory=list)
1255
+ eliminated: Eliminated = Field(default_factory=Eliminated)
1256
+ constraint_costs: list[ConstraintCost] = Field(default_factory=list)
1257
+ tipping_points: list[TippingPoint] = Field(default_factory=list)
1258
+ relax: list[str] = Field(default_factory=list)
1259
+ #: For ``no_feasible`` only: the smallest change to each numeric cap or
1260
+ #: floor that admits a model. Added in 1.5.
1261
+ relax_to: list[Relaxation] = Field(default_factory=list)
1262
+ warnings: list[Code] = Field(default_factory=list)
1263
+ #: Active models the snapshot leaves out of the lineup. Added in 1.2.
1264
+ out_of_lineup: int = Field(default=0, ge=0)
1265
+
1266
+ @model_validator(mode="after")
1267
+ def _status_agrees(self) -> Decision:
1268
+ if self.status == "no_feasible":
1269
+ if self.results:
1270
+ raise ValueError("a no_feasible decision has no results")
1271
+ if not self.relax:
1272
+ raise ValueError("a no_feasible decision names the fewest conditions to relax")
1273
+ elif self.relax or self.relax_to:
1274
+ raise ValueError(f"relax is only for no_feasible, not {self.status}")
1275
+ ranks = [r.rank for r in self.results]
1276
+ if ranks != list(range(1, len(ranks) + 1)):
1277
+ raise ValueError(f"result ranks must run 1..n in order; got {ranks}")
1278
+ return self
1279
+
1280
+
1281
+ CONTRACT_TYPES: tuple[type[BaseModel], ...] = (
1282
+ Spec, TaskTokens, Objective, LexStep, Tolerance, EvidenceQualifiers, Soft, ModelRef,
1283
+ Compare, Window, InSet, Known, AnyOf, AllOf, NotOf,
1284
+ InventoryProfile, ProfileOffering, LocalModel, Hardware, Budget,
1285
+ Decision, Result, OfferingRef, DomainEvidence, EvidenceItem, Estimate, Contribution,
1286
+ MayQualify, Eliminated, FunnelStep, ModelElimination, OfferingElimination,
1287
+ ModelEliminationGroup, ConstraintCost, TippingPoint,
1288
+ NearMiss, ShownFact, CandidateValues, NumberOrigin, CitedSource, Relaxation,
1289
+ )
1290
+
1291
+
1292
+ def closed_values() -> list[str]:
1293
+ """Every value of every closed vocabulary in the contract, for the doc agreement test."""
1294
+ values: list[str] = []
1295
+ for alias in (Op, UnknownPolicy, MeasuredByQualifier, MeasuredBy, Explain, Status, DateType,
1296
+ Directness, CapabilityLevel, TaskType):
1297
+ values.extend(str(v) for v in typing.get_args(alias))
1298
+ values.extend(QUALIFIER_KEYWORDS)
1299
+ return sorted(set(values))
1300
+
1301
+
1302
+ # ── errors ─────────────────────────────────────────────────────────────────
1303
+
1304
+
1305
+ @dataclass(frozen=True)
1306
+ class Issue:
1307
+ """One thing wrong with a spec."""
1308
+
1309
+ condition: str | None
1310
+ field: str | None
1311
+ reason: str
1312
+ path: str
1313
+
1314
+ def __str__(self) -> str:
1315
+ parts = [self.path]
1316
+ if self.condition is not None:
1317
+ parts.append(f"condition {self.condition!r}")
1318
+ if self.field is not None and self.field != self.path:
1319
+ parts.append(f"field {self.field}")
1320
+ return f"{', '.join(parts)}: {self.reason}"
1321
+
1322
+
1323
+ class SpecError(ValueError):
1324
+ """An invalid spec. ``issues`` lists every problem found."""
1325
+
1326
+ def __init__(self, issues: list[Issue]) -> None:
1327
+ self.issues = issues
1328
+ super().__init__("invalid spec:\n" + "\n".join(f" {issue}" for issue in issues))
1329
+
1330
+
1331
+ def _loc(loc: tuple[Any, ...]) -> str:
1332
+ out = ""
1333
+ for part in loc:
1334
+ if isinstance(part, int):
1335
+ out += f"[{part}]"
1336
+ elif part.startswith("<") or "[" in part or part in _UNION_TAGS:
1337
+ continue
1338
+ else:
1339
+ part = _ALIASES.get(part, part)
1340
+ out = f"{out}.{part}" if out else str(part)
1341
+ return out
1342
+
1343
+
1344
+ _ALIASES = {"in_": "in", "not_": "not", "class_": "class"}
1345
+
1346
+
1347
+ _UNION_TAGS = {"bool", "int", "float", "date", "str", "ModelRef", "tuple"}
1348
+
1349
+
1350
+ def _message(error: Mapping[str, Any]) -> str:
1351
+ if error["type"] == "extra_forbidden":
1352
+ return "not a spec field" if len(error["loc"]) == 1 else "unexpected field"
1353
+ if error["type"] == "missing":
1354
+ return "required"
1355
+ return str(error["msg"]).removeprefix("Value error, ")
1356
+
1357
+
1358
+ def _issues(exc: ValidationError) -> list[Issue]:
1359
+ issues: list[Issue] = []
1360
+ for error in exc.errors():
1361
+ where = _loc(error["loc"])
1362
+ nested = error.get("ctx", {}).get("error")
1363
+ if isinstance(nested, ConditionError):
1364
+ path = _join(where, nested.path) if nested.path else where
1365
+ issues.append(Issue(nested.condition, nested.field, nested.reason, path))
1366
+ else:
1367
+ issues.append(Issue(None, where, _message(error), where))
1368
+ return issues
1369
+
1370
+
1371
+ # ── the registry check ─────────────────────────────────────────────────────
1372
+
1373
+
1374
+ FacetLookup = Callable[[str], Any]
1375
+
1376
+
1377
+ def check_facets(spec: Spec, facets: FacetLookup) -> list[Issue]:
1378
+ """Every facet a spec names must be registered, and ordered where it is ordered."""
1379
+ issues: list[Issue] = []
1380
+
1381
+ def use(
1382
+ facet_id: str,
1383
+ ordered: bool,
1384
+ path: str,
1385
+ condition: str | None,
1386
+ qualifiers: EvidenceQualifiers | None = None,
1387
+ ) -> None:
1388
+ try:
1389
+ info = facets(facet_id)
1390
+ except KeyError as exc:
1391
+ detail = str(exc)
1392
+ reason = detail if detail.startswith("unknown facet ") else (
1393
+ f"unknown facet {facet_id!r}: not in the facet registry"
1394
+ )
1395
+ issues.append(Issue(condition, facet_id, reason, path))
1396
+ return
1397
+ value_type = info.value_type
1398
+ kind = value_type if isinstance(value_type, str) else value_type.kind
1399
+ if qualifiers is not None and getattr(info, "subject", None) != "evidence":
1400
+ issues.append(Issue(
1401
+ condition,
1402
+ facet_id,
1403
+ "evidence qualifiers are only valid on evidence facets",
1404
+ path,
1405
+ ))
1406
+ if ordered and kind in UNORDERED_VALUE_TYPES:
1407
+ issues.append(Issue(condition, facet_id,
1408
+ f"{facet_id} is a {kind} facet, which has no order; "
1409
+ "use = or in {…}", path))
1410
+
1411
+ def walk(cond: Any, path: str) -> None:
1412
+ if isinstance(cond, AnyOf | AllOf):
1413
+ name = "any" if isinstance(cond, AnyOf) else "all"
1414
+ for i, child in enumerate(cond.any if isinstance(cond, AnyOf) else cond.all):
1415
+ walk(child, f"{path}.{name}[{i}]")
1416
+ elif isinstance(cond, NotOf):
1417
+ walk(cond.not_, f"{path}.not")
1418
+ elif isinstance(cond, Known):
1419
+ use(cond.known, False, path, render_condition(cond))
1420
+ else:
1421
+ ordered = isinstance(cond, Window) or (isinstance(cond, Compare)
1422
+ and cond.op in ORDERED_OPS)
1423
+ use(
1424
+ cond.facet,
1425
+ ordered,
1426
+ path,
1427
+ render_condition(cond),
1428
+ getattr(cond, "qualifiers", None),
1429
+ )
1430
+
1431
+ for i, cond in enumerate(spec.where):
1432
+ walk(cond, f"where[{i}]")
1433
+ if isinstance(spec.profile, InventoryProfile):
1434
+ for i, cond in enumerate(spec.profile.rules):
1435
+ walk(cond, f"profile.rules[{i}]")
1436
+ objective = spec.optimize
1437
+ for side in ("max", "min"):
1438
+ if getattr(objective, side):
1439
+ facet_id = getattr(objective, side)
1440
+ use(
1441
+ facet_id,
1442
+ True,
1443
+ f"optimize.{side}",
1444
+ None,
1445
+ objective.qualifiers.get(facet_id),
1446
+ )
1447
+ for i, step in enumerate(objective.lexicographic or []):
1448
+ use(
1449
+ step.facet,
1450
+ True,
1451
+ f"optimize.lexicographic[{i}]",
1452
+ None,
1453
+ objective.qualifiers.get(step.facet),
1454
+ )
1455
+ for form in ("weights", "pareto"):
1456
+ for signed in getattr(objective, form) or []:
1457
+ facet_id = _base(signed)
1458
+ use(
1459
+ facet_id,
1460
+ True,
1461
+ f"optimize.{form}",
1462
+ None,
1463
+ objective.qualifiers.get(facet_id),
1464
+ )
1465
+ return issues
1466
+
1467
+
1468
+ # ── parsing a spec ─────────────────────────────────────────────────────────
1469
+
1470
+
1471
+ class _StrictLoader(yaml.SafeLoader):
1472
+ """Safe YAML that refuses duplicate keys instead of keeping the last one."""
1473
+
1474
+
1475
+ def _no_duplicates(loader: yaml.SafeLoader, node: yaml.MappingNode, deep: bool = False) -> Any:
1476
+ seen: set[Any] = set()
1477
+ for key_node, _ in node.value:
1478
+ key = loader.construct_object(key_node, deep=deep)
1479
+ if key in seen:
1480
+ raise SpecError([Issue(None, str(key), f"duplicate key {key!r}", str(key))])
1481
+ seen.add(key)
1482
+ return yaml.SafeLoader.construct_mapping(loader, node, deep)
1483
+
1484
+
1485
+ _StrictLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, _no_duplicates)
1486
+
1487
+
1488
+ def load_yaml(text: str) -> Any:
1489
+ try:
1490
+ return yaml.load(text, Loader=_StrictLoader) # noqa: S506 - SafeLoader subclass
1491
+ except yaml.YAMLError as exc:
1492
+ raise SpecError([Issue(None, None, f"not valid YAML: {exc}", "")]) from None
1493
+
1494
+
1495
+ def parse_spec(raw: str | Mapping[str, Any], *, facets: FacetLookup | None) -> Spec:
1496
+ """Parse and validate a spec (YAML text or a mapping).
1497
+
1498
+ ``facets`` is the registry lookup (``decision.registry.facet``). ``None``
1499
+ checks structure only. Raises ``SpecError`` naming every problem.
1500
+ """
1501
+ data = load_yaml(raw) if isinstance(raw, str) else raw
1502
+ if not isinstance(data, Mapping):
1503
+ raise SpecError([Issue(None, None, "a spec is a mapping", "")])
1504
+ try:
1505
+ spec = Spec.model_validate(data)
1506
+ except ValidationError as exc:
1507
+ raise SpecError(_issues(exc)) from None
1508
+ issues = check_facets(spec, facets) if facets is not None else []
1509
+ if spec.task is not None:
1510
+ issues.append(Issue(None, "task",
1511
+ "free-text task is not yet in slice 1; send task_type and "
1512
+ "capabilities instead", "task"))
1513
+ if issues:
1514
+ raise SpecError(issues)
1515
+ return spec
1516
+
1517
+
1518
+ # ── the canonical hash ─────────────────────────────────────────────────────
1519
+
1520
+
1521
+ def _normalise(value: Any) -> Any:
1522
+ if isinstance(value, float) and value.is_integer():
1523
+ return int(value)
1524
+ if isinstance(value, dict):
1525
+ return {k: _normalise(v) for k, v in value.items()}
1526
+ if isinstance(value, list):
1527
+ return [_normalise(v) for v in value]
1528
+ return value
1529
+
1530
+
1531
+ def canonical_json(spec: Spec) -> str:
1532
+ """The spec in its structured form, defaults filled, keys sorted, no whitespace."""
1533
+ data = _normalise(spec.model_dump(mode="json", by_alias=True, exclude_none=True))
1534
+ # Contract 1.1 adds objective qualifiers without changing the canonical
1535
+ # representation of an existing 1.0 spec.
1536
+ if not data["optimize"].get("qualifiers"):
1537
+ data["optimize"].pop("qualifiers", None)
1538
+ return json.dumps(data, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
1539
+
1540
+
1541
+ def spec_hash(spec: Spec) -> str:
1542
+ return "sha256:" + hashlib.sha256(canonical_json(spec).encode("utf-8")).hexdigest()
1543
+
1544
+
1545
+ # ── the JSON Schema ────────────────────────────────────────────────────────
1546
+
1547
+
1548
+ def json_schema() -> dict[str, Any]:
1549
+ from pydantic.json_schema import models_json_schema
1550
+
1551
+ refs, defs = models_json_schema([(Spec, "validation"), (Decision, "serialization")],
1552
+ ref_template="#/$defs/{model}")
1553
+ return {
1554
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
1555
+ "title": "ModelSpec decision contract",
1556
+ "x-contract-version": CONTRACT_VERSION,
1557
+ "description": "Generated from decision/contract.py by `python -m decision.schema`. "
1558
+ "Do not edit by hand. The prose is docs/decision-contract.md.",
1559
+ "anyOf": [refs[(Spec, "validation")], refs[(Decision, "serialization")]],
1560
+ **defs,
1561
+ }
1562
+
1563
+
1564
+ def render_json_schema() -> str:
1565
+ return json.dumps(json_schema(), indent=2, sort_keys=True, ensure_ascii=False) + "\n"
1566
+
1567
+
1568
+ def _kind(value: Any) -> str:
1569
+ if isinstance(value, bool):
1570
+ return "bool"
1571
+ if isinstance(value, int | float):
1572
+ return "number"
1573
+ if isinstance(value, date):
1574
+ return "date"
1575
+ return "str"