mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
"""RuntimeAdapter contract and capability verdict matrix (ADR-M3-1, ADR-M3-2).
|
|
2
|
+
|
|
3
|
+
``RuntimeAdapter`` is the normalized interface that docker, podman, and
|
|
4
|
+
future runtimes implement. ``AdapterCapabilities`` provides a static
|
|
5
|
+
snapshot of what the adapter supports; ``CapabilityRequirements`` captures
|
|
6
|
+
what a fault plan needs; and ``evaluate`` produces a verdict that the
|
|
7
|
+
planner uses to accept or refuse execution.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from abc import ABC, abstractmethod
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from enum import StrEnum
|
|
15
|
+
from typing import TYPE_CHECKING, Any
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
18
|
+
|
|
19
|
+
if TYPE_CHECKING:
|
|
20
|
+
from mayhem.domain.identity import RuntimeIdentity, RuntimeMetadata
|
|
21
|
+
from mayhem.topology.providers.base import PartialGraph
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# ---------------------------------------------------------------------------
|
|
25
|
+
# Enums
|
|
26
|
+
# ---------------------------------------------------------------------------
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class RuntimeCapability(StrEnum):
|
|
30
|
+
"""Named capability a runtime adapter may or may not support."""
|
|
31
|
+
|
|
32
|
+
EXEC = "exec"
|
|
33
|
+
PID = "pid"
|
|
34
|
+
SIGNAL = "signal"
|
|
35
|
+
NETNS = "netns"
|
|
36
|
+
RESOURCE_LIMITS = "resource_limits"
|
|
37
|
+
INSPECT = "inspect"
|
|
38
|
+
COMPOSE_FILTER = "compose_filter"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class CapabilityVerdict(StrEnum):
|
|
42
|
+
"""Per-capability answer from the adapter's feasibility matrix."""
|
|
43
|
+
|
|
44
|
+
SUPPORTED = "supported"
|
|
45
|
+
ALTERNATIVE = "alternative" # degraded but safe
|
|
46
|
+
UNSUPPORTED = "unsupported"
|
|
47
|
+
UNKNOWN = "unknown"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# ---------------------------------------------------------------------------
|
|
51
|
+
# Capability snapshot
|
|
52
|
+
# ---------------------------------------------------------------------------
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class AdapterCapabilities(BaseModel):
|
|
56
|
+
"""Static capability snapshot produced by a ``RuntimeAdapter``.
|
|
57
|
+
|
|
58
|
+
Every adapter populates *supported* and *alternatives* at construction
|
|
59
|
+
time (derived from engine version, rootless detection, etc.) so that
|
|
60
|
+
``verdict()`` is a pure lookup with no subprocess calls.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
model_config = ConfigDict(frozen=True)
|
|
64
|
+
|
|
65
|
+
engine: str
|
|
66
|
+
rootless: bool = False
|
|
67
|
+
supported: frozenset[RuntimeCapability] = Field(
|
|
68
|
+
default_factory=frozenset,
|
|
69
|
+
)
|
|
70
|
+
alternatives: frozenset[RuntimeCapability] = Field(
|
|
71
|
+
default_factory=frozenset,
|
|
72
|
+
)
|
|
73
|
+
version: str | None = None
|
|
74
|
+
|
|
75
|
+
# -- lookup --------------------------------------------------------------
|
|
76
|
+
|
|
77
|
+
def verdict(self, cap: RuntimeCapability) -> CapabilityVerdict:
|
|
78
|
+
"""Return the verdict for *cap* without subprocess calls."""
|
|
79
|
+
if cap in self.supported:
|
|
80
|
+
return CapabilityVerdict.SUPPORTED
|
|
81
|
+
if cap in self.alternatives:
|
|
82
|
+
return CapabilityVerdict.ALTERNATIVE
|
|
83
|
+
return CapabilityVerdict.UNSUPPORTED
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# ---------------------------------------------------------------------------
|
|
87
|
+
# Requirements
|
|
88
|
+
# ---------------------------------------------------------------------------
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass(frozen=True)
|
|
92
|
+
class CapabilityRequirements:
|
|
93
|
+
"""What a fault plan requires from the runtime adapter.
|
|
94
|
+
|
|
95
|
+
The planner builds this from the plan's execution contexts and resolved
|
|
96
|
+
targets; the adapter's ``evaluate`` method answers each requirement.
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
platforms: frozenset[str] = field(default_factory=frozenset)
|
|
100
|
+
runtimes: frozenset[str] = field(default_factory=frozenset)
|
|
101
|
+
target_kinds: frozenset[str] = field(default_factory=frozenset)
|
|
102
|
+
privileges: frozenset[str] = field(default_factory=frozenset)
|
|
103
|
+
namespaces: frozenset[str] = field(default_factory=frozenset)
|
|
104
|
+
tools: frozenset[str] = field(default_factory=frozenset)
|
|
105
|
+
kernel_features: frozenset[str] = field(default_factory=frozenset)
|
|
106
|
+
permissions: frozenset[str] = field(default_factory=frozenset)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# ---------------------------------------------------------------------------
|
|
110
|
+
# Verdict result
|
|
111
|
+
# ---------------------------------------------------------------------------
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class VerdictResult(BaseModel):
|
|
115
|
+
"""Output of ``RuntimeAdapter.evaluate``.
|
|
116
|
+
|
|
117
|
+
``blocking`` is True when at least one requirement maps to UNSUPPORTED;
|
|
118
|
+
the planner must refuse the plan in that case. ALTERNATIVE verdicts are
|
|
119
|
+
non-blocking but logged as warnings.
|
|
120
|
+
"""
|
|
121
|
+
|
|
122
|
+
model_config = ConfigDict(frozen=True)
|
|
123
|
+
|
|
124
|
+
engine: str
|
|
125
|
+
requirements: CapabilityRequirements
|
|
126
|
+
verdicts: dict[str, str] # CapabilityRequirement summary -> CapabilityVerdict value
|
|
127
|
+
blocking: bool
|
|
128
|
+
|
|
129
|
+
def refuse_with_message(self) -> str | None:
|
|
130
|
+
"""Return a human-readable refusal message if *blocking*, else None."""
|
|
131
|
+
if not self.blocking:
|
|
132
|
+
return None
|
|
133
|
+
unsupported = [k for k, v in self.verdicts.items() if v == CapabilityVerdict.UNSUPPORTED]
|
|
134
|
+
return (
|
|
135
|
+
f"runtime '{self.engine}' UNSUPPORTED capabilities block this plan: "
|
|
136
|
+
f"{', '.join(sorted(unsupported))}"
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
# ---------------------------------------------------------------------------
|
|
141
|
+
# Abstract adapter
|
|
142
|
+
# ---------------------------------------------------------------------------
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
class RuntimeAdapter(ABC):
|
|
146
|
+
"""Normalized runtime interface (ADR-M3-1).
|
|
147
|
+
|
|
148
|
+
Docker and Podman are concrete implementations. Remote and Kubernetes
|
|
149
|
+
are stubs that return UNSUPPORTED verdicts (ADR-M3-5, ADR-M3-6).
|
|
150
|
+
"""
|
|
151
|
+
|
|
152
|
+
# -- identity ------------------------------------------------------------
|
|
153
|
+
|
|
154
|
+
@property
|
|
155
|
+
@abstractmethod
|
|
156
|
+
def id(self) -> str:
|
|
157
|
+
"""Stable adapter identifier, e.g. ``"docker"`` or ``"podman"``.
|
|
158
|
+
|
|
159
|
+
Exposed as a property so adapters remain structurally compatible with
|
|
160
|
+
the ``TopologyProvider`` protocol (``id`` is a property there).
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
@abstractmethod
|
|
164
|
+
def is_available(self) -> bool:
|
|
165
|
+
"""True when the engine binary is reachable on this host."""
|
|
166
|
+
|
|
167
|
+
# -- capabilities --------------------------------------------------------
|
|
168
|
+
|
|
169
|
+
@abstractmethod
|
|
170
|
+
def capabilities(self) -> AdapterCapabilities:
|
|
171
|
+
"""Static capability snapshot for this adapter instance."""
|
|
172
|
+
|
|
173
|
+
def evaluate(self, reqs: CapabilityRequirements) -> VerdictResult:
|
|
174
|
+
"""Answer a set of requirements against the adapter's capabilities.
|
|
175
|
+
|
|
176
|
+
Default implementation maps ``reqs.namespaces`` to NETNS,
|
|
177
|
+
``reqs.tools`` to tool availability, etc. Override for richer
|
|
178
|
+
engine-specific logic.
|
|
179
|
+
"""
|
|
180
|
+
caps = self.capabilities()
|
|
181
|
+
verdicts: dict[str, str] = {}
|
|
182
|
+
blocking = False
|
|
183
|
+
|
|
184
|
+
# namespace requirements -> NETNS capability
|
|
185
|
+
if reqs.namespaces:
|
|
186
|
+
v = caps.verdict(RuntimeCapability.NETNS)
|
|
187
|
+
verdicts["namespaces"] = v.value
|
|
188
|
+
if v == CapabilityVerdict.UNSUPPORTED:
|
|
189
|
+
blocking = True
|
|
190
|
+
|
|
191
|
+
# tool requirements -> EXEC capability (tools need exec to run)
|
|
192
|
+
if reqs.tools:
|
|
193
|
+
v = caps.verdict(RuntimeCapability.EXEC)
|
|
194
|
+
verdicts["tools"] = v.value
|
|
195
|
+
if v == CapabilityVerdict.UNSUPPORTED:
|
|
196
|
+
blocking = True
|
|
197
|
+
|
|
198
|
+
# resource limit requirements
|
|
199
|
+
if reqs.permissions:
|
|
200
|
+
v = caps.verdict(RuntimeCapability.RESOURCE_LIMITS)
|
|
201
|
+
verdicts["permissions"] = v.value
|
|
202
|
+
if v == CapabilityVerdict.UNSUPPORTED:
|
|
203
|
+
blocking = True
|
|
204
|
+
|
|
205
|
+
return VerdictResult(
|
|
206
|
+
engine=self.id,
|
|
207
|
+
requirements=reqs,
|
|
208
|
+
verdicts=verdicts,
|
|
209
|
+
blocking=blocking,
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
# -- container operations ------------------------------------------------
|
|
213
|
+
|
|
214
|
+
@abstractmethod
|
|
215
|
+
def ps(self) -> list[dict[str, Any]]:
|
|
216
|
+
"""List containers visible to this adapter (engine-native format)."""
|
|
217
|
+
|
|
218
|
+
@abstractmethod
|
|
219
|
+
def inspect(self, container_id: str) -> tuple[RuntimeIdentity, RuntimeMetadata | None]:
|
|
220
|
+
"""Resolve identity + metadata for a running container."""
|
|
221
|
+
|
|
222
|
+
@abstractmethod
|
|
223
|
+
def exec(self, container_id: str, cmd: list[str], *, timeout_s: float = 30) -> str:
|
|
224
|
+
"""Run *cmd* inside *container_id*, return stdout."""
|
|
225
|
+
|
|
226
|
+
@abstractmethod
|
|
227
|
+
def pid(self, container_id: str) -> int | None:
|
|
228
|
+
"""Return the main PID of *container_id*, or None if unavailable."""
|
|
229
|
+
|
|
230
|
+
@abstractmethod
|
|
231
|
+
def signal(self, container_id: str, signo: int) -> None:
|
|
232
|
+
"""Send signal *signo* to *container_id*'s main process."""
|
|
233
|
+
|
|
234
|
+
@abstractmethod
|
|
235
|
+
def netns(self, container_id: str) -> str | None:
|
|
236
|
+
"""Return the network-namespace path for *container_id*, or None."""
|
|
237
|
+
|
|
238
|
+
# -- filtering (mutates state) -------------------------------------------
|
|
239
|
+
|
|
240
|
+
@abstractmethod
|
|
241
|
+
def filter_by_compose(
|
|
242
|
+
self,
|
|
243
|
+
project: str,
|
|
244
|
+
services: tuple[str, ...] | None = None,
|
|
245
|
+
) -> None:
|
|
246
|
+
"""Scope to containers belonging to a compose project."""
|
|
247
|
+
|
|
248
|
+
@abstractmethod
|
|
249
|
+
def filter_by_names(self, names: list[str]) -> None:
|
|
250
|
+
"""Scope to containers matching explicit names."""
|
|
251
|
+
|
|
252
|
+
# -- discovery -----------------------------------------------------------
|
|
253
|
+
|
|
254
|
+
@abstractmethod
|
|
255
|
+
def discover(self) -> PartialGraph:
|
|
256
|
+
"""Build a PartialGraph fragment from the engine's live state."""
|
mayhem/domain/success.py
ADDED
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""Machine-evaluable success criteria over recorded observations (ADR-M4-3 §18).
|
|
2
|
+
|
|
3
|
+
Success is an objective, machine-evaluable predicate over recorded
|
|
4
|
+
observations — never a human eyeball. Criteria are typed assertions
|
|
5
|
+
(status / latency / metric / count / boolean) evaluated by the engine over
|
|
6
|
+
the observation rows a drill recorded; a drill's success/failure verdict is
|
|
7
|
+
derived from them.
|
|
8
|
+
|
|
9
|
+
Evaluating a criterion against a *missing* observation is **false** — absence
|
|
10
|
+
of evidence is not success — and it never raises.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from enum import StrEnum
|
|
16
|
+
from typing import TYPE_CHECKING, Annotated, Literal
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, model_validator
|
|
19
|
+
|
|
20
|
+
if TYPE_CHECKING:
|
|
21
|
+
from collections.abc import Mapping
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ObservationKind(StrEnum):
|
|
25
|
+
"""The scalar type an observation row records."""
|
|
26
|
+
|
|
27
|
+
STATUS = "status" # integer status (HTTP/exit code)
|
|
28
|
+
LATENCY = "latency_ms" # milliseconds, float
|
|
29
|
+
METRIC = "metric" # arbitrary numeric sample
|
|
30
|
+
COUNT = "count" # integer count
|
|
31
|
+
BOOLEAN = "boolean" # boolean outcome
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class Observation(BaseModel):
|
|
35
|
+
"""One recorded observation row for a named ``source_id`` (step/check id)."""
|
|
36
|
+
|
|
37
|
+
model_config = ConfigDict(frozen=True)
|
|
38
|
+
|
|
39
|
+
source_id: str
|
|
40
|
+
kind: ObservationKind
|
|
41
|
+
value: float | int | bool | str
|
|
42
|
+
detail: str = ""
|
|
43
|
+
|
|
44
|
+
@model_validator(mode="after")
|
|
45
|
+
def _kind_matches_value(self) -> Observation:
|
|
46
|
+
if isinstance(self.value, str):
|
|
47
|
+
return self
|
|
48
|
+
if self.kind is ObservationKind.BOOLEAN:
|
|
49
|
+
if not isinstance(self.value, bool):
|
|
50
|
+
raise ValueError("boolean observation must carry a bool value")
|
|
51
|
+
elif isinstance(self.value, bool) or not isinstance(self.value, (int, float)):
|
|
52
|
+
raise ValueError(f"{self.kind.value} observation must carry a numeric value")
|
|
53
|
+
return self
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def observations_for_step(
|
|
57
|
+
source_id: str,
|
|
58
|
+
*,
|
|
59
|
+
ok: bool,
|
|
60
|
+
measured: Mapping[str, object] | None = None,
|
|
61
|
+
detail: str = "",
|
|
62
|
+
) -> tuple[Observation, ...]:
|
|
63
|
+
"""Normalise a step's outcome + measured extras into observation rows.
|
|
64
|
+
|
|
65
|
+
The primary row (bare ``source_id``) records the boolean step outcome;
|
|
66
|
+
each measured scalar is recorded under ``<source_id>.<name>`` so a status
|
|
67
|
+
criterion can address ``step.status`` and a latency criterion
|
|
68
|
+
``step.latency_ms`` from the same step.
|
|
69
|
+
"""
|
|
70
|
+
rows: list[Observation] = [
|
|
71
|
+
Observation(source_id=source_id, kind=ObservationKind.BOOLEAN, value=ok, detail=detail)
|
|
72
|
+
]
|
|
73
|
+
for name, raw in (measured or {}).items():
|
|
74
|
+
if isinstance(raw, bool):
|
|
75
|
+
rows.append(
|
|
76
|
+
Observation(
|
|
77
|
+
source_id=f"{source_id}.{name}",
|
|
78
|
+
kind=ObservationKind.BOOLEAN,
|
|
79
|
+
value=raw,
|
|
80
|
+
detail=detail,
|
|
81
|
+
)
|
|
82
|
+
)
|
|
83
|
+
elif isinstance(raw, int):
|
|
84
|
+
kind = ObservationKind.COUNT if not name.endswith("status") else ObservationKind.STATUS
|
|
85
|
+
rows.append(
|
|
86
|
+
Observation(source_id=f"{source_id}.{name}", kind=kind, value=raw, detail=detail)
|
|
87
|
+
)
|
|
88
|
+
elif isinstance(raw, float):
|
|
89
|
+
kind = (
|
|
90
|
+
ObservationKind.LATENCY
|
|
91
|
+
if name.endswith(("ms", "latency", "latency_ms"))
|
|
92
|
+
else ObservationKind.METRIC
|
|
93
|
+
)
|
|
94
|
+
rows.append(
|
|
95
|
+
Observation(source_id=f"{source_id}.{name}", kind=kind, value=raw, detail=detail)
|
|
96
|
+
)
|
|
97
|
+
return tuple(rows)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class CriterionType(StrEnum):
|
|
101
|
+
"""Kinds of machine-evaluable assertions (ADR-M4-3)."""
|
|
102
|
+
|
|
103
|
+
STATUS = "status" # measured status equals an expected value
|
|
104
|
+
LATENCY = "latency" # measured latency (ms) below a bound
|
|
105
|
+
METRIC = "metric" # numeric metric within bounds
|
|
106
|
+
COUNT = "count" # recorded count at least a minimum
|
|
107
|
+
BOOLEAN = "boolean" # recorded outcome equals a boolean
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class StatusCriterion(BaseModel):
|
|
111
|
+
model_config = ConfigDict(frozen=True)
|
|
112
|
+
|
|
113
|
+
type: Literal[CriterionType.STATUS] = CriterionType.STATUS
|
|
114
|
+
source_id: str # observation source, e.g. "check-0000-0.status"
|
|
115
|
+
expected: int
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class LatencyCriterion(BaseModel):
|
|
119
|
+
model_config = ConfigDict(frozen=True)
|
|
120
|
+
|
|
121
|
+
type: Literal[CriterionType.LATENCY] = CriterionType.LATENCY
|
|
122
|
+
source_id: str # observation source, e.g. "check-0000-0.latency_ms"
|
|
123
|
+
lt_ms: float = Field(gt=0)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class MetricCriterion(BaseModel):
|
|
127
|
+
model_config = ConfigDict(frozen=True)
|
|
128
|
+
|
|
129
|
+
type: Literal[CriterionType.METRIC] = CriterionType.METRIC
|
|
130
|
+
source_id: str
|
|
131
|
+
gt: float | None = None # exclusive lower bound
|
|
132
|
+
lt: float | None = None # exclusive upper bound
|
|
133
|
+
|
|
134
|
+
@model_validator(mode="after")
|
|
135
|
+
def _has_bound(self) -> MetricCriterion:
|
|
136
|
+
if self.gt is None and self.lt is None:
|
|
137
|
+
raise ValueError("metric criterion requires gt and/or lt")
|
|
138
|
+
return self
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class CountCriterion(BaseModel):
|
|
142
|
+
model_config = ConfigDict(frozen=True)
|
|
143
|
+
|
|
144
|
+
type: Literal[CriterionType.COUNT] = CriterionType.COUNT
|
|
145
|
+
source_id: str
|
|
146
|
+
gte: int = 1 # inclusive minimum
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class BooleanCriterion(BaseModel):
|
|
150
|
+
model_config = ConfigDict(frozen=True)
|
|
151
|
+
|
|
152
|
+
type: Literal[CriterionType.BOOLEAN] = CriterionType.BOOLEAN
|
|
153
|
+
source_id: str # the bare step id — its primary boolean outcome
|
|
154
|
+
value: bool = True
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
Criterion = Annotated[
|
|
158
|
+
StatusCriterion | LatencyCriterion | MetricCriterion | CountCriterion | BooleanCriterion,
|
|
159
|
+
Field(discriminator="type"),
|
|
160
|
+
]
|
|
161
|
+
_criterion_adapter: TypeAdapter[Criterion] = TypeAdapter(Criterion)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def parse_criterion(data: object) -> Criterion:
|
|
165
|
+
"""Parse untyped criterion data into the closed union."""
|
|
166
|
+
return _criterion_adapter.validate_python(data)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
class SuccessCriteria(BaseModel):
|
|
170
|
+
"""Typed assertions a drill must satisfy to succeed (ADR-M4-3)."""
|
|
171
|
+
|
|
172
|
+
model_config = ConfigDict(frozen=True)
|
|
173
|
+
|
|
174
|
+
criteria: tuple[Criterion, ...] = ()
|
|
175
|
+
require_all: bool = True
|
|
176
|
+
|
|
177
|
+
@property
|
|
178
|
+
def empty(self) -> bool:
|
|
179
|
+
return not self.criteria
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
class CriterionResult(BaseModel):
|
|
183
|
+
"""Evaluation of a single criterion against the recorded observations."""
|
|
184
|
+
|
|
185
|
+
model_config = ConfigDict(frozen=True)
|
|
186
|
+
|
|
187
|
+
type: CriterionType
|
|
188
|
+
source_id: str
|
|
189
|
+
satisfied: bool
|
|
190
|
+
detail: str = ""
|
|
191
|
+
|
|
192
|
+
def summary(self) -> str:
|
|
193
|
+
mark = "PASS" if self.satisfied else "FAIL"
|
|
194
|
+
return f"{self.type.value}:{self.source_id} {mark} ({self.detail})"
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class CriteriaEvaluation(BaseModel):
|
|
198
|
+
"""Deterministic evaluation of a drill's success criteria."""
|
|
199
|
+
|
|
200
|
+
model_config = ConfigDict(frozen=True)
|
|
201
|
+
|
|
202
|
+
results: tuple[CriterionResult, ...] = ()
|
|
203
|
+
all_satisfied: bool = True
|
|
204
|
+
|
|
205
|
+
@property
|
|
206
|
+
def empty(self) -> bool:
|
|
207
|
+
return not self.results
|
|
208
|
+
|
|
209
|
+
def summary_md(self) -> str:
|
|
210
|
+
headline = (
|
|
211
|
+
"**success criteria**: ALL PASS"
|
|
212
|
+
if self.all_satisfied
|
|
213
|
+
else "**success criteria**: FAILED"
|
|
214
|
+
)
|
|
215
|
+
lines = [headline]
|
|
216
|
+
for result in self.results:
|
|
217
|
+
mark = "PASS" if result.satisfied else "FAIL"
|
|
218
|
+
lines.append(f"- [{mark}] {result.summary()}")
|
|
219
|
+
return "\n".join(lines)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _coerce(observation: Observation, kind: CriterionType) -> float | int | bool | None: # noqa: PLR0911, PLR0912
|
|
223
|
+
"""Coerce an observation to a criterion's numeric/boolean domain.
|
|
224
|
+
|
|
225
|
+
Returns None when the recorded kind is not coercible to the criterion's
|
|
226
|
+
kind (e.g. a boolean step outcome asked for a latency bound).
|
|
227
|
+
"""
|
|
228
|
+
if kind in (CriterionType.STATUS, CriterionType.COUNT):
|
|
229
|
+
if observation.kind is ObservationKind.BOOLEAN:
|
|
230
|
+
return None
|
|
231
|
+
raw = observation.value
|
|
232
|
+
if isinstance(raw, str):
|
|
233
|
+
try:
|
|
234
|
+
return int(raw)
|
|
235
|
+
except ValueError:
|
|
236
|
+
return None
|
|
237
|
+
return int(raw) if isinstance(raw, (int, float)) and not isinstance(raw, bool) else None
|
|
238
|
+
if kind is CriterionType.LATENCY or kind is CriterionType.METRIC:
|
|
239
|
+
if observation.kind is ObservationKind.BOOLEAN:
|
|
240
|
+
return None
|
|
241
|
+
raw = observation.value
|
|
242
|
+
if isinstance(raw, str):
|
|
243
|
+
try:
|
|
244
|
+
return float(raw)
|
|
245
|
+
except ValueError:
|
|
246
|
+
return None
|
|
247
|
+
return float(raw) if isinstance(raw, (int, float)) and not isinstance(raw, bool) else None
|
|
248
|
+
assert kind is CriterionType.BOOLEAN # closed enum: last member
|
|
249
|
+
raw = observation.value
|
|
250
|
+
if isinstance(raw, bool):
|
|
251
|
+
return raw
|
|
252
|
+
if isinstance(raw, str):
|
|
253
|
+
normalized = raw.strip().lower()
|
|
254
|
+
if normalized in ("1", "true", "yes", "y"):
|
|
255
|
+
return True
|
|
256
|
+
if normalized in ("0", "false", "no", "n"):
|
|
257
|
+
return False
|
|
258
|
+
if isinstance(raw, (int, float)) and not isinstance(raw, bool):
|
|
259
|
+
return raw in (1, 1.0)
|
|
260
|
+
return None
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _evaluate_one(criterion: Criterion, observations: Mapping[str, Observation]) -> CriterionResult:
|
|
264
|
+
source_id = criterion.source_id
|
|
265
|
+
observation = observations.get(source_id)
|
|
266
|
+
if observation is None:
|
|
267
|
+
return CriterionResult(
|
|
268
|
+
type=criterion.type,
|
|
269
|
+
source_id=source_id,
|
|
270
|
+
satisfied=False,
|
|
271
|
+
detail="no observation recorded",
|
|
272
|
+
)
|
|
273
|
+
value = _coerce(observation, criterion.type)
|
|
274
|
+
if value is None:
|
|
275
|
+
return CriterionResult(
|
|
276
|
+
type=criterion.type,
|
|
277
|
+
source_id=source_id,
|
|
278
|
+
satisfied=False,
|
|
279
|
+
detail=f"recorded kind {observation.kind.value} not comparable",
|
|
280
|
+
)
|
|
281
|
+
if isinstance(criterion, StatusCriterion):
|
|
282
|
+
matched = value == criterion.expected
|
|
283
|
+
detail = f"status={value} expected={criterion.expected}"
|
|
284
|
+
elif isinstance(criterion, LatencyCriterion):
|
|
285
|
+
matched = float(value) < criterion.lt_ms
|
|
286
|
+
detail = f"latency={value}ms bound=<{criterion.lt_ms}ms"
|
|
287
|
+
elif isinstance(criterion, MetricCriterion):
|
|
288
|
+
lower_ok = criterion.gt is None or float(value) > criterion.gt
|
|
289
|
+
upper_ok = criterion.lt is None or float(value) < criterion.lt
|
|
290
|
+
matched = lower_ok and upper_ok
|
|
291
|
+
detail = (
|
|
292
|
+
f"metric={value}"
|
|
293
|
+
+ (f" bound=>{criterion.gt}" if criterion.gt is not None else "")
|
|
294
|
+
+ (f" bound=<{criterion.lt}" if criterion.lt is not None else "")
|
|
295
|
+
)
|
|
296
|
+
elif isinstance(criterion, CountCriterion):
|
|
297
|
+
matched = int(value) >= criterion.gte
|
|
298
|
+
detail = f"count={value} required>={criterion.gte}"
|
|
299
|
+
elif isinstance(criterion, BooleanCriterion):
|
|
300
|
+
matched = bool(value) is criterion.value
|
|
301
|
+
detail = f"outcome={value} required={criterion.value}"
|
|
302
|
+
else: # pragma: no cover - closed union
|
|
303
|
+
raise AssertionError(f"unhandled criterion {type(criterion).__name__}")
|
|
304
|
+
return CriterionResult(
|
|
305
|
+
type=criterion.type,
|
|
306
|
+
source_id=source_id,
|
|
307
|
+
satisfied=matched,
|
|
308
|
+
detail=detail,
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def evaluate_criteria(
|
|
313
|
+
criteria: SuccessCriteria | None,
|
|
314
|
+
observations: Mapping[str, Observation],
|
|
315
|
+
) -> CriteriaEvaluation:
|
|
316
|
+
"""Evaluate success criteria over recorded observations (deterministic).
|
|
317
|
+
|
|
318
|
+
Missing or incomparable observations evaluate to false; an absent/empty
|
|
319
|
+
criteria block is satisfied (the caller decides whether any verdict is
|
|
320
|
+
derived at all).
|
|
321
|
+
"""
|
|
322
|
+
if criteria is None or criteria.empty:
|
|
323
|
+
return CriteriaEvaluation()
|
|
324
|
+
results = tuple(_evaluate_one(criterion, observations) for criterion in criteria.criteria)
|
|
325
|
+
if criteria.require_all:
|
|
326
|
+
all_satisfied = all(r.satisfied for r in results)
|
|
327
|
+
else:
|
|
328
|
+
all_satisfied = any(r.satisfied for r in results)
|
|
329
|
+
return CriteriaEvaluation(results=results, all_satisfied=all_satisfied)
|