mayhem-cli 0.5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. mayhem/agent/__init__.py +1 -0
  2. mayhem/agent/cli.py +36 -0
  3. mayhem/agents/__init__.py +1 -0
  4. mayhem/agents/capabilities.py +106 -0
  5. mayhem/agents/executors.py +430 -0
  6. mayhem/agents/impact.py +729 -0
  7. mayhem/agents/lease_client.py +141 -0
  8. mayhem/agents/probes.py +284 -0
  9. mayhem/agents/protocol.py +134 -0
  10. mayhem/agents/server.py +281 -0
  11. mayhem/agents/sinks.py +60 -0
  12. mayhem/agents/transports.py +134 -0
  13. mayhem/agents/watchdog.py +140 -0
  14. mayhem/cli/__init__.py +11 -0
  15. mayhem/cli/app.py +154 -0
  16. mayhem/cli/campaign.py +496 -0
  17. mayhem/cli/config_cmd.py +47 -0
  18. mayhem/cli/context.py +23 -0
  19. mayhem/cli/dependency.py +429 -0
  20. mayhem/cli/exit_codes.py +24 -0
  21. mayhem/cli/experiment.py +24 -0
  22. mayhem/cli/lifecycle.py +805 -0
  23. mayhem/cli/resolver.py +72 -0
  24. mayhem/cli/services.py +459 -0
  25. mayhem/cli/style.py +101 -0
  26. mayhem/cli/toolkit.py +41 -0
  27. mayhem/cli/topology.py +127 -0
  28. mayhem/config.py +208 -0
  29. mayhem/controller/__init__.py +1 -0
  30. mayhem/controller/compensation.py +2156 -0
  31. mayhem/controller/executor.py +1719 -0
  32. mayhem/controller/janitor.py +196 -0
  33. mayhem/controller/observability_collector.py +382 -0
  34. mayhem/controller/observations.py +102 -0
  35. mayhem/controller/planner.py +715 -0
  36. mayhem/controller/recovery.py +245 -0
  37. mayhem/controller/resilience_report.py +585 -0
  38. mayhem/controller/resource_manager.py +457 -0
  39. mayhem/controller/safety.py +392 -0
  40. mayhem/domain/__init__.py +6 -0
  41. mayhem/domain/campaigns.py +118 -0
  42. mayhem/domain/cancellation.py +110 -0
  43. mayhem/domain/candidates.py +101 -0
  44. mayhem/domain/capabilities.py +86 -0
  45. mayhem/domain/catalog.py +727 -0
  46. mayhem/domain/checks.py +173 -0
  47. mayhem/domain/common.py +104 -0
  48. mayhem/domain/coverage.py +106 -0
  49. mayhem/domain/decisions.py +57 -0
  50. mayhem/domain/errors.py +87 -0
  51. mayhem/domain/events.py +61 -0
  52. mayhem/domain/execution_context.py +120 -0
  53. mayhem/domain/execution_loci.py +94 -0
  54. mayhem/domain/experiments.py +370 -0
  55. mayhem/domain/faults.py +239 -0
  56. mayhem/domain/identity.py +200 -0
  57. mayhem/domain/k8s_adapter.py +132 -0
  58. mayhem/domain/leases.py +186 -0
  59. mayhem/domain/load_strategy.py +98 -0
  60. mayhem/domain/m5_campaign.py +120 -0
  61. mayhem/domain/maniac.py +93 -0
  62. mayhem/domain/observability.py +146 -0
  63. mayhem/domain/outcomes.py +92 -0
  64. mayhem/domain/remote_agent_interface.py +70 -0
  65. mayhem/domain/resources.py +245 -0
  66. mayhem/domain/risks.py +61 -0
  67. mayhem/domain/run_outcome.py +146 -0
  68. mayhem/domain/runtime_adapter.py +256 -0
  69. mayhem/domain/success.py +329 -0
  70. mayhem/domain/topology.py +452 -0
  71. mayhem/infra/__init__.py +1 -0
  72. mayhem/infra/campaign_engine.py +205 -0
  73. mayhem/infra/candidate_gates.py +124 -0
  74. mayhem/infra/candidate_generator.py +110 -0
  75. mayhem/infra/coverage_repository.py +101 -0
  76. mayhem/infra/lease_repository.py +129 -0
  77. mayhem/infra/maniac.py +103 -0
  78. mayhem/infra/migrations.py +596 -0
  79. mayhem/infra/migrator.py +149 -0
  80. mayhem/infra/report.py +227 -0
  81. mayhem/infra/store.py +200 -0
  82. mayhem/py.typed +0 -0
  83. mayhem/spec.py +52 -0
  84. mayhem/toolkit/__init__.py +1 -0
  85. mayhem/toolkit/fingerprint.py +69 -0
  86. mayhem/toolkit/hashing.py +32 -0
  87. mayhem/toolkit/manifests/docker.yaml +11 -0
  88. mayhem/toolkit/manifests/podman.yaml +11 -0
  89. mayhem/toolkit/manifests/stress-ng.yaml +11 -0
  90. mayhem/toolkit/manifests/tc-netem.yaml +11 -0
  91. mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
  92. mayhem/toolkit/registry.py +185 -0
  93. mayhem/toolkit/tool_runner.py +129 -0
  94. mayhem/topology/__init__.py +10 -0
  95. mayhem/topology/providers/__init__.py +0 -0
  96. mayhem/topology/providers/adapter_registry.py +60 -0
  97. mayhem/topology/providers/base.py +31 -0
  98. mayhem/topology/providers/compose.py +207 -0
  99. mayhem/topology/providers/docker_adapter.py +277 -0
  100. mayhem/topology/providers/docker_runtime.py +461 -0
  101. mayhem/topology/providers/podman_adapter.py +328 -0
  102. mayhem/topology/resolve.py +196 -0
  103. mayhem/topology/service.py +158 -0
  104. mayhem_cli-0.5.1.dist-info/METADATA +555 -0
  105. mayhem_cli-0.5.1.dist-info/RECORD +107 -0
  106. mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
  107. mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,256 @@
1
+ """RuntimeAdapter contract and capability verdict matrix (ADR-M3-1, ADR-M3-2).
2
+
3
+ ``RuntimeAdapter`` is the normalized interface that docker, podman, and
4
+ future runtimes implement. ``AdapterCapabilities`` provides a static
5
+ snapshot of what the adapter supports; ``CapabilityRequirements`` captures
6
+ what a fault plan needs; and ``evaluate`` produces a verdict that the
7
+ planner uses to accept or refuse execution.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from abc import ABC, abstractmethod
13
+ from dataclasses import dataclass, field
14
+ from enum import StrEnum
15
+ from typing import TYPE_CHECKING, Any
16
+
17
+ from pydantic import BaseModel, ConfigDict, Field
18
+
19
+ if TYPE_CHECKING:
20
+ from mayhem.domain.identity import RuntimeIdentity, RuntimeMetadata
21
+ from mayhem.topology.providers.base import PartialGraph
22
+
23
+
24
+ # ---------------------------------------------------------------------------
25
+ # Enums
26
+ # ---------------------------------------------------------------------------
27
+
28
+
29
+ class RuntimeCapability(StrEnum):
30
+ """Named capability a runtime adapter may or may not support."""
31
+
32
+ EXEC = "exec"
33
+ PID = "pid"
34
+ SIGNAL = "signal"
35
+ NETNS = "netns"
36
+ RESOURCE_LIMITS = "resource_limits"
37
+ INSPECT = "inspect"
38
+ COMPOSE_FILTER = "compose_filter"
39
+
40
+
41
+ class CapabilityVerdict(StrEnum):
42
+ """Per-capability answer from the adapter's feasibility matrix."""
43
+
44
+ SUPPORTED = "supported"
45
+ ALTERNATIVE = "alternative" # degraded but safe
46
+ UNSUPPORTED = "unsupported"
47
+ UNKNOWN = "unknown"
48
+
49
+
50
+ # ---------------------------------------------------------------------------
51
+ # Capability snapshot
52
+ # ---------------------------------------------------------------------------
53
+
54
+
55
+ class AdapterCapabilities(BaseModel):
56
+ """Static capability snapshot produced by a ``RuntimeAdapter``.
57
+
58
+ Every adapter populates *supported* and *alternatives* at construction
59
+ time (derived from engine version, rootless detection, etc.) so that
60
+ ``verdict()`` is a pure lookup with no subprocess calls.
61
+ """
62
+
63
+ model_config = ConfigDict(frozen=True)
64
+
65
+ engine: str
66
+ rootless: bool = False
67
+ supported: frozenset[RuntimeCapability] = Field(
68
+ default_factory=frozenset,
69
+ )
70
+ alternatives: frozenset[RuntimeCapability] = Field(
71
+ default_factory=frozenset,
72
+ )
73
+ version: str | None = None
74
+
75
+ # -- lookup --------------------------------------------------------------
76
+
77
+ def verdict(self, cap: RuntimeCapability) -> CapabilityVerdict:
78
+ """Return the verdict for *cap* without subprocess calls."""
79
+ if cap in self.supported:
80
+ return CapabilityVerdict.SUPPORTED
81
+ if cap in self.alternatives:
82
+ return CapabilityVerdict.ALTERNATIVE
83
+ return CapabilityVerdict.UNSUPPORTED
84
+
85
+
86
+ # ---------------------------------------------------------------------------
87
+ # Requirements
88
+ # ---------------------------------------------------------------------------
89
+
90
+
91
+ @dataclass(frozen=True)
92
+ class CapabilityRequirements:
93
+ """What a fault plan requires from the runtime adapter.
94
+
95
+ The planner builds this from the plan's execution contexts and resolved
96
+ targets; the adapter's ``evaluate`` method answers each requirement.
97
+ """
98
+
99
+ platforms: frozenset[str] = field(default_factory=frozenset)
100
+ runtimes: frozenset[str] = field(default_factory=frozenset)
101
+ target_kinds: frozenset[str] = field(default_factory=frozenset)
102
+ privileges: frozenset[str] = field(default_factory=frozenset)
103
+ namespaces: frozenset[str] = field(default_factory=frozenset)
104
+ tools: frozenset[str] = field(default_factory=frozenset)
105
+ kernel_features: frozenset[str] = field(default_factory=frozenset)
106
+ permissions: frozenset[str] = field(default_factory=frozenset)
107
+
108
+
109
+ # ---------------------------------------------------------------------------
110
+ # Verdict result
111
+ # ---------------------------------------------------------------------------
112
+
113
+
114
+ class VerdictResult(BaseModel):
115
+ """Output of ``RuntimeAdapter.evaluate``.
116
+
117
+ ``blocking`` is True when at least one requirement maps to UNSUPPORTED;
118
+ the planner must refuse the plan in that case. ALTERNATIVE verdicts are
119
+ non-blocking but logged as warnings.
120
+ """
121
+
122
+ model_config = ConfigDict(frozen=True)
123
+
124
+ engine: str
125
+ requirements: CapabilityRequirements
126
+ verdicts: dict[str, str] # CapabilityRequirement summary -> CapabilityVerdict value
127
+ blocking: bool
128
+
129
+ def refuse_with_message(self) -> str | None:
130
+ """Return a human-readable refusal message if *blocking*, else None."""
131
+ if not self.blocking:
132
+ return None
133
+ unsupported = [k for k, v in self.verdicts.items() if v == CapabilityVerdict.UNSUPPORTED]
134
+ return (
135
+ f"runtime '{self.engine}' UNSUPPORTED capabilities block this plan: "
136
+ f"{', '.join(sorted(unsupported))}"
137
+ )
138
+
139
+
140
+ # ---------------------------------------------------------------------------
141
+ # Abstract adapter
142
+ # ---------------------------------------------------------------------------
143
+
144
+
145
+ class RuntimeAdapter(ABC):
146
+ """Normalized runtime interface (ADR-M3-1).
147
+
148
+ Docker and Podman are concrete implementations. Remote and Kubernetes
149
+ are stubs that return UNSUPPORTED verdicts (ADR-M3-5, ADR-M3-6).
150
+ """
151
+
152
+ # -- identity ------------------------------------------------------------
153
+
154
+ @property
155
+ @abstractmethod
156
+ def id(self) -> str:
157
+ """Stable adapter identifier, e.g. ``"docker"`` or ``"podman"``.
158
+
159
+ Exposed as a property so adapters remain structurally compatible with
160
+ the ``TopologyProvider`` protocol (``id`` is a property there).
161
+ """
162
+
163
+ @abstractmethod
164
+ def is_available(self) -> bool:
165
+ """True when the engine binary is reachable on this host."""
166
+
167
+ # -- capabilities --------------------------------------------------------
168
+
169
+ @abstractmethod
170
+ def capabilities(self) -> AdapterCapabilities:
171
+ """Static capability snapshot for this adapter instance."""
172
+
173
+ def evaluate(self, reqs: CapabilityRequirements) -> VerdictResult:
174
+ """Answer a set of requirements against the adapter's capabilities.
175
+
176
+ Default implementation maps ``reqs.namespaces`` to NETNS,
177
+ ``reqs.tools`` to tool availability, etc. Override for richer
178
+ engine-specific logic.
179
+ """
180
+ caps = self.capabilities()
181
+ verdicts: dict[str, str] = {}
182
+ blocking = False
183
+
184
+ # namespace requirements -> NETNS capability
185
+ if reqs.namespaces:
186
+ v = caps.verdict(RuntimeCapability.NETNS)
187
+ verdicts["namespaces"] = v.value
188
+ if v == CapabilityVerdict.UNSUPPORTED:
189
+ blocking = True
190
+
191
+ # tool requirements -> EXEC capability (tools need exec to run)
192
+ if reqs.tools:
193
+ v = caps.verdict(RuntimeCapability.EXEC)
194
+ verdicts["tools"] = v.value
195
+ if v == CapabilityVerdict.UNSUPPORTED:
196
+ blocking = True
197
+
198
+ # resource limit requirements
199
+ if reqs.permissions:
200
+ v = caps.verdict(RuntimeCapability.RESOURCE_LIMITS)
201
+ verdicts["permissions"] = v.value
202
+ if v == CapabilityVerdict.UNSUPPORTED:
203
+ blocking = True
204
+
205
+ return VerdictResult(
206
+ engine=self.id,
207
+ requirements=reqs,
208
+ verdicts=verdicts,
209
+ blocking=blocking,
210
+ )
211
+
212
+ # -- container operations ------------------------------------------------
213
+
214
+ @abstractmethod
215
+ def ps(self) -> list[dict[str, Any]]:
216
+ """List containers visible to this adapter (engine-native format)."""
217
+
218
+ @abstractmethod
219
+ def inspect(self, container_id: str) -> tuple[RuntimeIdentity, RuntimeMetadata | None]:
220
+ """Resolve identity + metadata for a running container."""
221
+
222
+ @abstractmethod
223
+ def exec(self, container_id: str, cmd: list[str], *, timeout_s: float = 30) -> str:
224
+ """Run *cmd* inside *container_id*, return stdout."""
225
+
226
+ @abstractmethod
227
+ def pid(self, container_id: str) -> int | None:
228
+ """Return the main PID of *container_id*, or None if unavailable."""
229
+
230
+ @abstractmethod
231
+ def signal(self, container_id: str, signo: int) -> None:
232
+ """Send signal *signo* to *container_id*'s main process."""
233
+
234
+ @abstractmethod
235
+ def netns(self, container_id: str) -> str | None:
236
+ """Return the network-namespace path for *container_id*, or None."""
237
+
238
+ # -- filtering (mutates state) -------------------------------------------
239
+
240
+ @abstractmethod
241
+ def filter_by_compose(
242
+ self,
243
+ project: str,
244
+ services: tuple[str, ...] | None = None,
245
+ ) -> None:
246
+ """Scope to containers belonging to a compose project."""
247
+
248
+ @abstractmethod
249
+ def filter_by_names(self, names: list[str]) -> None:
250
+ """Scope to containers matching explicit names."""
251
+
252
+ # -- discovery -----------------------------------------------------------
253
+
254
+ @abstractmethod
255
+ def discover(self) -> PartialGraph:
256
+ """Build a PartialGraph fragment from the engine's live state."""
@@ -0,0 +1,329 @@
1
+ """Machine-evaluable success criteria over recorded observations (ADR-M4-3 §18).
2
+
3
+ Success is an objective, machine-evaluable predicate over recorded
4
+ observations — never a human eyeball. Criteria are typed assertions
5
+ (status / latency / metric / count / boolean) evaluated by the engine over
6
+ the observation rows a drill recorded; a drill's success/failure verdict is
7
+ derived from them.
8
+
9
+ Evaluating a criterion against a *missing* observation is **false** — absence
10
+ of evidence is not success — and it never raises.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from enum import StrEnum
16
+ from typing import TYPE_CHECKING, Annotated, Literal
17
+
18
+ from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, model_validator
19
+
20
+ if TYPE_CHECKING:
21
+ from collections.abc import Mapping
22
+
23
+
24
+ class ObservationKind(StrEnum):
25
+ """The scalar type an observation row records."""
26
+
27
+ STATUS = "status" # integer status (HTTP/exit code)
28
+ LATENCY = "latency_ms" # milliseconds, float
29
+ METRIC = "metric" # arbitrary numeric sample
30
+ COUNT = "count" # integer count
31
+ BOOLEAN = "boolean" # boolean outcome
32
+
33
+
34
+ class Observation(BaseModel):
35
+ """One recorded observation row for a named ``source_id`` (step/check id)."""
36
+
37
+ model_config = ConfigDict(frozen=True)
38
+
39
+ source_id: str
40
+ kind: ObservationKind
41
+ value: float | int | bool | str
42
+ detail: str = ""
43
+
44
+ @model_validator(mode="after")
45
+ def _kind_matches_value(self) -> Observation:
46
+ if isinstance(self.value, str):
47
+ return self
48
+ if self.kind is ObservationKind.BOOLEAN:
49
+ if not isinstance(self.value, bool):
50
+ raise ValueError("boolean observation must carry a bool value")
51
+ elif isinstance(self.value, bool) or not isinstance(self.value, (int, float)):
52
+ raise ValueError(f"{self.kind.value} observation must carry a numeric value")
53
+ return self
54
+
55
+
56
+ def observations_for_step(
57
+ source_id: str,
58
+ *,
59
+ ok: bool,
60
+ measured: Mapping[str, object] | None = None,
61
+ detail: str = "",
62
+ ) -> tuple[Observation, ...]:
63
+ """Normalise a step's outcome + measured extras into observation rows.
64
+
65
+ The primary row (bare ``source_id``) records the boolean step outcome;
66
+ each measured scalar is recorded under ``<source_id>.<name>`` so a status
67
+ criterion can address ``step.status`` and a latency criterion
68
+ ``step.latency_ms`` from the same step.
69
+ """
70
+ rows: list[Observation] = [
71
+ Observation(source_id=source_id, kind=ObservationKind.BOOLEAN, value=ok, detail=detail)
72
+ ]
73
+ for name, raw in (measured or {}).items():
74
+ if isinstance(raw, bool):
75
+ rows.append(
76
+ Observation(
77
+ source_id=f"{source_id}.{name}",
78
+ kind=ObservationKind.BOOLEAN,
79
+ value=raw,
80
+ detail=detail,
81
+ )
82
+ )
83
+ elif isinstance(raw, int):
84
+ kind = ObservationKind.COUNT if not name.endswith("status") else ObservationKind.STATUS
85
+ rows.append(
86
+ Observation(source_id=f"{source_id}.{name}", kind=kind, value=raw, detail=detail)
87
+ )
88
+ elif isinstance(raw, float):
89
+ kind = (
90
+ ObservationKind.LATENCY
91
+ if name.endswith(("ms", "latency", "latency_ms"))
92
+ else ObservationKind.METRIC
93
+ )
94
+ rows.append(
95
+ Observation(source_id=f"{source_id}.{name}", kind=kind, value=raw, detail=detail)
96
+ )
97
+ return tuple(rows)
98
+
99
+
100
+ class CriterionType(StrEnum):
101
+ """Kinds of machine-evaluable assertions (ADR-M4-3)."""
102
+
103
+ STATUS = "status" # measured status equals an expected value
104
+ LATENCY = "latency" # measured latency (ms) below a bound
105
+ METRIC = "metric" # numeric metric within bounds
106
+ COUNT = "count" # recorded count at least a minimum
107
+ BOOLEAN = "boolean" # recorded outcome equals a boolean
108
+
109
+
110
+ class StatusCriterion(BaseModel):
111
+ model_config = ConfigDict(frozen=True)
112
+
113
+ type: Literal[CriterionType.STATUS] = CriterionType.STATUS
114
+ source_id: str # observation source, e.g. "check-0000-0.status"
115
+ expected: int
116
+
117
+
118
+ class LatencyCriterion(BaseModel):
119
+ model_config = ConfigDict(frozen=True)
120
+
121
+ type: Literal[CriterionType.LATENCY] = CriterionType.LATENCY
122
+ source_id: str # observation source, e.g. "check-0000-0.latency_ms"
123
+ lt_ms: float = Field(gt=0)
124
+
125
+
126
+ class MetricCriterion(BaseModel):
127
+ model_config = ConfigDict(frozen=True)
128
+
129
+ type: Literal[CriterionType.METRIC] = CriterionType.METRIC
130
+ source_id: str
131
+ gt: float | None = None # exclusive lower bound
132
+ lt: float | None = None # exclusive upper bound
133
+
134
+ @model_validator(mode="after")
135
+ def _has_bound(self) -> MetricCriterion:
136
+ if self.gt is None and self.lt is None:
137
+ raise ValueError("metric criterion requires gt and/or lt")
138
+ return self
139
+
140
+
141
+ class CountCriterion(BaseModel):
142
+ model_config = ConfigDict(frozen=True)
143
+
144
+ type: Literal[CriterionType.COUNT] = CriterionType.COUNT
145
+ source_id: str
146
+ gte: int = 1 # inclusive minimum
147
+
148
+
149
+ class BooleanCriterion(BaseModel):
150
+ model_config = ConfigDict(frozen=True)
151
+
152
+ type: Literal[CriterionType.BOOLEAN] = CriterionType.BOOLEAN
153
+ source_id: str # the bare step id — its primary boolean outcome
154
+ value: bool = True
155
+
156
+
157
+ Criterion = Annotated[
158
+ StatusCriterion | LatencyCriterion | MetricCriterion | CountCriterion | BooleanCriterion,
159
+ Field(discriminator="type"),
160
+ ]
161
+ _criterion_adapter: TypeAdapter[Criterion] = TypeAdapter(Criterion)
162
+
163
+
164
+ def parse_criterion(data: object) -> Criterion:
165
+ """Parse untyped criterion data into the closed union."""
166
+ return _criterion_adapter.validate_python(data)
167
+
168
+
169
+ class SuccessCriteria(BaseModel):
170
+ """Typed assertions a drill must satisfy to succeed (ADR-M4-3)."""
171
+
172
+ model_config = ConfigDict(frozen=True)
173
+
174
+ criteria: tuple[Criterion, ...] = ()
175
+ require_all: bool = True
176
+
177
+ @property
178
+ def empty(self) -> bool:
179
+ return not self.criteria
180
+
181
+
182
+ class CriterionResult(BaseModel):
183
+ """Evaluation of a single criterion against the recorded observations."""
184
+
185
+ model_config = ConfigDict(frozen=True)
186
+
187
+ type: CriterionType
188
+ source_id: str
189
+ satisfied: bool
190
+ detail: str = ""
191
+
192
+ def summary(self) -> str:
193
+ mark = "PASS" if self.satisfied else "FAIL"
194
+ return f"{self.type.value}:{self.source_id} {mark} ({self.detail})"
195
+
196
+
197
+ class CriteriaEvaluation(BaseModel):
198
+ """Deterministic evaluation of a drill's success criteria."""
199
+
200
+ model_config = ConfigDict(frozen=True)
201
+
202
+ results: tuple[CriterionResult, ...] = ()
203
+ all_satisfied: bool = True
204
+
205
+ @property
206
+ def empty(self) -> bool:
207
+ return not self.results
208
+
209
+ def summary_md(self) -> str:
210
+ headline = (
211
+ "**success criteria**: ALL PASS"
212
+ if self.all_satisfied
213
+ else "**success criteria**: FAILED"
214
+ )
215
+ lines = [headline]
216
+ for result in self.results:
217
+ mark = "PASS" if result.satisfied else "FAIL"
218
+ lines.append(f"- [{mark}] {result.summary()}")
219
+ return "\n".join(lines)
220
+
221
+
222
+ def _coerce(observation: Observation, kind: CriterionType) -> float | int | bool | None: # noqa: PLR0911, PLR0912
223
+ """Coerce an observation to a criterion's numeric/boolean domain.
224
+
225
+ Returns None when the recorded kind is not coercible to the criterion's
226
+ kind (e.g. a boolean step outcome asked for a latency bound).
227
+ """
228
+ if kind in (CriterionType.STATUS, CriterionType.COUNT):
229
+ if observation.kind is ObservationKind.BOOLEAN:
230
+ return None
231
+ raw = observation.value
232
+ if isinstance(raw, str):
233
+ try:
234
+ return int(raw)
235
+ except ValueError:
236
+ return None
237
+ return int(raw) if isinstance(raw, (int, float)) and not isinstance(raw, bool) else None
238
+ if kind is CriterionType.LATENCY or kind is CriterionType.METRIC:
239
+ if observation.kind is ObservationKind.BOOLEAN:
240
+ return None
241
+ raw = observation.value
242
+ if isinstance(raw, str):
243
+ try:
244
+ return float(raw)
245
+ except ValueError:
246
+ return None
247
+ return float(raw) if isinstance(raw, (int, float)) and not isinstance(raw, bool) else None
248
+ assert kind is CriterionType.BOOLEAN # closed enum: last member
249
+ raw = observation.value
250
+ if isinstance(raw, bool):
251
+ return raw
252
+ if isinstance(raw, str):
253
+ normalized = raw.strip().lower()
254
+ if normalized in ("1", "true", "yes", "y"):
255
+ return True
256
+ if normalized in ("0", "false", "no", "n"):
257
+ return False
258
+ if isinstance(raw, (int, float)) and not isinstance(raw, bool):
259
+ return raw in (1, 1.0)
260
+ return None
261
+
262
+
263
+ def _evaluate_one(criterion: Criterion, observations: Mapping[str, Observation]) -> CriterionResult:
264
+ source_id = criterion.source_id
265
+ observation = observations.get(source_id)
266
+ if observation is None:
267
+ return CriterionResult(
268
+ type=criterion.type,
269
+ source_id=source_id,
270
+ satisfied=False,
271
+ detail="no observation recorded",
272
+ )
273
+ value = _coerce(observation, criterion.type)
274
+ if value is None:
275
+ return CriterionResult(
276
+ type=criterion.type,
277
+ source_id=source_id,
278
+ satisfied=False,
279
+ detail=f"recorded kind {observation.kind.value} not comparable",
280
+ )
281
+ if isinstance(criterion, StatusCriterion):
282
+ matched = value == criterion.expected
283
+ detail = f"status={value} expected={criterion.expected}"
284
+ elif isinstance(criterion, LatencyCriterion):
285
+ matched = float(value) < criterion.lt_ms
286
+ detail = f"latency={value}ms bound=<{criterion.lt_ms}ms"
287
+ elif isinstance(criterion, MetricCriterion):
288
+ lower_ok = criterion.gt is None or float(value) > criterion.gt
289
+ upper_ok = criterion.lt is None or float(value) < criterion.lt
290
+ matched = lower_ok and upper_ok
291
+ detail = (
292
+ f"metric={value}"
293
+ + (f" bound=>{criterion.gt}" if criterion.gt is not None else "")
294
+ + (f" bound=<{criterion.lt}" if criterion.lt is not None else "")
295
+ )
296
+ elif isinstance(criterion, CountCriterion):
297
+ matched = int(value) >= criterion.gte
298
+ detail = f"count={value} required>={criterion.gte}"
299
+ elif isinstance(criterion, BooleanCriterion):
300
+ matched = bool(value) is criterion.value
301
+ detail = f"outcome={value} required={criterion.value}"
302
+ else: # pragma: no cover - closed union
303
+ raise AssertionError(f"unhandled criterion {type(criterion).__name__}")
304
+ return CriterionResult(
305
+ type=criterion.type,
306
+ source_id=source_id,
307
+ satisfied=matched,
308
+ detail=detail,
309
+ )
310
+
311
+
312
+ def evaluate_criteria(
313
+ criteria: SuccessCriteria | None,
314
+ observations: Mapping[str, Observation],
315
+ ) -> CriteriaEvaluation:
316
+ """Evaluate success criteria over recorded observations (deterministic).
317
+
318
+ Missing or incomparable observations evaluate to false; an absent/empty
319
+ criteria block is satisfied (the caller decides whether any verdict is
320
+ derived at all).
321
+ """
322
+ if criteria is None or criteria.empty:
323
+ return CriteriaEvaluation()
324
+ results = tuple(_evaluate_one(criterion, observations) for criterion in criteria.criteria)
325
+ if criteria.require_all:
326
+ all_satisfied = all(r.satisfied for r in results)
327
+ else:
328
+ all_satisfied = any(r.satisfied for r in results)
329
+ return CriteriaEvaluation(results=results, all_satisfied=all_satisfied)