mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,392 @@
|
|
|
1
|
+
"""Mechanical safety gates between planner and executor (ADR-0012, ADR-0014,
|
|
2
|
+
architecture/safety.md).
|
|
3
|
+
|
|
4
|
+
Gate stack enforced here:
|
|
5
|
+
G1 config policy — allowlists/denylists, risk ladder, critical opt-in
|
|
6
|
+
G2 plan validation — budgets fit the topology graph, fingerprint match
|
|
7
|
+
G3 pre-exec assertion — resolved targets re-checked against *live* topology
|
|
8
|
+
G4 execution context — declared context must be feasible for target node kinds
|
|
9
|
+
|
|
10
|
+
Precedence: denylist beats allowlist beats selector beats default.
|
|
11
|
+
Every refusal is typed and carries a machine-readable reason.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import hashlib
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from typing import TYPE_CHECKING
|
|
19
|
+
|
|
20
|
+
from mayhem.domain.errors import InvariantViolationError, TargetResolutionError
|
|
21
|
+
from mayhem.domain.execution_context import ExecutionContext
|
|
22
|
+
from mayhem.domain.risks import RiskLevel
|
|
23
|
+
from mayhem.domain.runtime_adapter import (
|
|
24
|
+
CapabilityRequirements,
|
|
25
|
+
CapabilityVerdict,
|
|
26
|
+
RuntimeAdapter,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from collections.abc import Iterable
|
|
31
|
+
|
|
32
|
+
from mayhem.config import PolicyCfg
|
|
33
|
+
from mayhem.domain.experiments import BlastRadiusBudget, ExecutionPlan, PlannedFault
|
|
34
|
+
from mayhem.domain.topology import NodeKind, TargetSelector, TopologyGraph
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class SafetyRefusedError(InvariantViolationError):
|
|
38
|
+
"""A mechanical safety gate refused a fault or plan (``safety.refused``)."""
|
|
39
|
+
|
|
40
|
+
def __init__(self, reason_code: str, message: str) -> None:
|
|
41
|
+
super().__init__(reason_code, message)
|
|
42
|
+
self.reason_code = reason_code
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def environment_fingerprint(
|
|
46
|
+
*,
|
|
47
|
+
host_names: Iterable[str],
|
|
48
|
+
compose_digest: str,
|
|
49
|
+
profile: str | None = None,
|
|
50
|
+
) -> str:
|
|
51
|
+
"""SHA256(sorted host set + compose digest + profile name)."""
|
|
52
|
+
payload = "|".join(
|
|
53
|
+
(
|
|
54
|
+
",".join(sorted(host_names)),
|
|
55
|
+
compose_digest,
|
|
56
|
+
profile or "default",
|
|
57
|
+
)
|
|
58
|
+
)
|
|
59
|
+
return hashlib.sha256(payload.encode()).hexdigest()
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass(frozen=True)
|
|
63
|
+
class SafetyContext:
|
|
64
|
+
"""Everything G1/G2 need; built once per CLI invocation."""
|
|
65
|
+
|
|
66
|
+
policy: PolicyCfg
|
|
67
|
+
budget: BlastRadiusBudget
|
|
68
|
+
fingerprint: str
|
|
69
|
+
allow_critical_cli: bool = False
|
|
70
|
+
warnings: list[str] = field(default_factory=list)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
DEFAULT_DENY = frozenset({"node.reboot"})
|
|
74
|
+
"""Host-reboot class faults are forbidden unless explicitly allowlisted
|
|
75
|
+
([ADR-0012]: an explicit allowlist overrides default deny; the bare default
|
|
76
|
+
policy never grants them)."""
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def check_fault_admission(fault_id: str, definition_risk: RiskLevel, ctx: SafetyContext) -> None:
|
|
80
|
+
"""G1: denylist → allowlist → risk ceiling → critical double-opt-in."""
|
|
81
|
+
if fault_id in ctx.policy.deny_faults:
|
|
82
|
+
raise SafetyRefusedError("safety.refused", f"{fault_id}: denied by policy denylist")
|
|
83
|
+
if ctx.policy.allow_faults is not None:
|
|
84
|
+
if fault_id not in ctx.policy.allow_faults:
|
|
85
|
+
raise SafetyRefusedError(
|
|
86
|
+
"safety.refused", f"{fault_id}: not present in policy allowlist"
|
|
87
|
+
)
|
|
88
|
+
elif fault_id in DEFAULT_DENY:
|
|
89
|
+
raise SafetyRefusedError(
|
|
90
|
+
"safety.refused",
|
|
91
|
+
f"{fault_id}: denied by default; add it to policy.allow_faults to opt in",
|
|
92
|
+
)
|
|
93
|
+
ceiling = ctx.policy.risk_ceiling
|
|
94
|
+
if ceiling is not None and definition_risk.at_least(ceiling.next_higher()):
|
|
95
|
+
raise SafetyRefusedError(
|
|
96
|
+
"safety.refused",
|
|
97
|
+
f"{fault_id}: risk {definition_risk.value} exceeds policy ceiling {ceiling.value}",
|
|
98
|
+
)
|
|
99
|
+
if definition_risk is RiskLevel.CRITICAL and not (
|
|
100
|
+
ctx.policy.allow_critical and ctx.allow_critical_cli
|
|
101
|
+
):
|
|
102
|
+
raise SafetyRefusedError(
|
|
103
|
+
"safety.refused",
|
|
104
|
+
f"{fault_id}: critical risk requires config policy.allow_critical AND --allow-critical",
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _affected_node_ids(graph: TopologyGraph, targets: Iterable[str]) -> frozenset[str]:
|
|
109
|
+
affected: set[str] = set()
|
|
110
|
+
for node_id in targets:
|
|
111
|
+
affected.add(node_id)
|
|
112
|
+
affected |= graph.dependents_closure(node_id)
|
|
113
|
+
return frozenset(affected)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def check_blast_radius(
|
|
117
|
+
graph: TopologyGraph,
|
|
118
|
+
target_node_ids: Iterable[str],
|
|
119
|
+
duration_s: float,
|
|
120
|
+
fault_ids_so_far: tuple[str, ...],
|
|
121
|
+
new_fault_id: str,
|
|
122
|
+
*,
|
|
123
|
+
ctx: SafetyContext,
|
|
124
|
+
) -> dict[str, float]:
|
|
125
|
+
"""G2 (budget half): topology-derived blast radius must fit; returns measured stats."""
|
|
126
|
+
budget = ctx.budget
|
|
127
|
+
affected = _affected_node_ids(graph, target_node_ids)
|
|
128
|
+
services_total = len(graph.of_kind(_service_kind()))
|
|
129
|
+
services_hit = sum(1 for n in graph.of_kind(_service_kind()) if n.id in affected)
|
|
130
|
+
hosts_hit = sum(1 for n in graph.of_kind(_host_kind()) if n.id in affected)
|
|
131
|
+
pct = (services_hit / services_total * 100.0) if services_total else 0.0
|
|
132
|
+
stats = {
|
|
133
|
+
"services_pct": round(pct, 1),
|
|
134
|
+
"hosts": float(hosts_hit),
|
|
135
|
+
"concurrent_faults": float(len(fault_ids_so_far) + 1),
|
|
136
|
+
"duration_per_fault": duration_s,
|
|
137
|
+
}
|
|
138
|
+
if pct > budget.max_services_pct:
|
|
139
|
+
raise SafetyRefusedError(
|
|
140
|
+
"safety.refused",
|
|
141
|
+
f"blast radius: {new_fault_id} would affect {pct:.0f}% of services"
|
|
142
|
+
f" > budget {budget.max_services_pct}%",
|
|
143
|
+
)
|
|
144
|
+
if hosts_hit > budget.max_hosts:
|
|
145
|
+
raise SafetyRefusedError(
|
|
146
|
+
"safety.refused",
|
|
147
|
+
f"blast radius: {new_fault_id} touches {hosts_hit} hosts > budget {budget.max_hosts}",
|
|
148
|
+
)
|
|
149
|
+
if len(fault_ids_so_far) + 1 > budget.max_concurrent_faults:
|
|
150
|
+
raise SafetyRefusedError("safety.refused", "blast radius: max_concurrent_faults exceeded")
|
|
151
|
+
if duration_s > budget.max_duration_per_fault_s:
|
|
152
|
+
raise SafetyRefusedError(
|
|
153
|
+
"safety.refused",
|
|
154
|
+
f"{new_fault_id} duration {duration_s:.0f}s exceeds per-fault cap"
|
|
155
|
+
f" {budget.max_duration_per_fault_s:.0f}s",
|
|
156
|
+
)
|
|
157
|
+
pair = frozenset((*fault_ids_so_far, new_fault_id)) if fault_ids_so_far else None
|
|
158
|
+
if pair and pair in budget.forbidden_fault_pairs:
|
|
159
|
+
raise SafetyRefusedError(
|
|
160
|
+
"safety.refused",
|
|
161
|
+
f"blast radius: forbidden fault pair {sorted(pair)}",
|
|
162
|
+
)
|
|
163
|
+
return stats
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _check_execution_context(fault: PlannedFault, graph: TopologyGraph) -> None:
|
|
167
|
+
"""G4: validate that the declared execution context is feasible for all targets.
|
|
168
|
+
|
|
169
|
+
Prevents accidental host-level execution when the experiment intended
|
|
170
|
+
container-level execution, or vice versa.
|
|
171
|
+
"""
|
|
172
|
+
if fault.execution_context is None:
|
|
173
|
+
return
|
|
174
|
+
node_kinds: set[NodeKind] = set()
|
|
175
|
+
for target in fault.targets:
|
|
176
|
+
for node_id in target.node_ids:
|
|
177
|
+
node = graph.by_id(node_id)
|
|
178
|
+
if node is not None:
|
|
179
|
+
node_kinds.add(node.kind)
|
|
180
|
+
if not node_kinds:
|
|
181
|
+
return # no nodes resolved — G3 will catch this
|
|
182
|
+
try:
|
|
183
|
+
fault.execution_context.assert_compatible(frozenset(node_kinds))
|
|
184
|
+
except InvariantViolationError as exc:
|
|
185
|
+
raise SafetyRefusedError(
|
|
186
|
+
"execution_context.refused",
|
|
187
|
+
f"{fault.fault_id}: {exc}",
|
|
188
|
+
) from exc
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
_K8S_NODE_KIND_VALUES: frozenset[str] = frozenset({"pod", "k8s_node"})
|
|
192
|
+
_K8S_REFUSE_MSG = (
|
|
193
|
+
"kubernetes execution not yet supported; see the RuntimeAdapter contract at ADR-M7-1"
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
_REMOTE_NODE_KIND_VALUES: frozenset[str] = frozenset({"external_dependency"})
|
|
197
|
+
_REMOTE_REFUSE_MSG = (
|
|
198
|
+
"remote execution not yet supported; ADR-M3-5 ships only the "
|
|
199
|
+
"RemoteAgentInterface contract — no transport is wired in this milestone"
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _check_k8s_targets(plan: ExecutionPlan, graph: TopologyGraph) -> None:
|
|
204
|
+
"""Refuse any plan that targets K8s node kinds without a live driver (ADR-M7).
|
|
205
|
+
|
|
206
|
+
K8s execution is out-of-scope for this milestone — the adapter contract
|
|
207
|
+
exists so future drivers can implement against a stable seam, but no
|
|
208
|
+
live-cluster fault injection is wired yet. A plan targeting K8s nodes
|
|
209
|
+
must fail loud and early with an actionable message.
|
|
210
|
+
"""
|
|
211
|
+
for step in plan.steps:
|
|
212
|
+
fault = step.fault
|
|
213
|
+
if fault is None:
|
|
214
|
+
continue
|
|
215
|
+
for target in fault.targets:
|
|
216
|
+
for node_id in target.node_ids:
|
|
217
|
+
node = graph.by_id(node_id)
|
|
218
|
+
if node is not None and node.kind in _K8S_NODE_KIND_VALUES:
|
|
219
|
+
raise SafetyRefusedError(
|
|
220
|
+
"k8s.unsupported",
|
|
221
|
+
f"{fault.fault_id}: {_K8S_REFUSE_MSG}",
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _check_remote_targets(plan: ExecutionPlan, graph: TopologyGraph) -> None:
|
|
226
|
+
"""Hard planning gate for remote targets (ADR-M3-5) — defect register #4.
|
|
227
|
+
|
|
228
|
+
Remote execution is interface-only (``RemoteAgentInterface``): the adapter
|
|
229
|
+
rejects requirements at ``evaluate`` time, but that refusal was never
|
|
230
|
+
auto-wired into ``validate_plan``, so a remote spec would plan and only
|
|
231
|
+
fail mid-execution. This gate mirrors the K8s one: a plan targeting an
|
|
232
|
+
``external_dependency`` node fails loud and early at plan time.
|
|
233
|
+
"""
|
|
234
|
+
for step in plan.steps:
|
|
235
|
+
fault = step.fault
|
|
236
|
+
if fault is None:
|
|
237
|
+
continue
|
|
238
|
+
for target in fault.targets:
|
|
239
|
+
for node_id in target.node_ids:
|
|
240
|
+
node = graph.by_id(node_id)
|
|
241
|
+
if node is not None and node.kind in _REMOTE_NODE_KIND_VALUES:
|
|
242
|
+
raise SafetyRefusedError(
|
|
243
|
+
"remote.unsupported",
|
|
244
|
+
f"{fault.fault_id}: {_REMOTE_REFUSE_MSG}",
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def validate_plan(
|
|
249
|
+
plan: ExecutionPlan,
|
|
250
|
+
graph: TopologyGraph,
|
|
251
|
+
ctx: SafetyContext,
|
|
252
|
+
adapter: RuntimeAdapter | None = None,
|
|
253
|
+
) -> None:
|
|
254
|
+
"""G1+G2+G4 over every fault step of an already-compiled plan.
|
|
255
|
+
|
|
256
|
+
When *adapter* is provided, capability requirements are derived from the
|
|
257
|
+
plan's execution contexts and evaluated against the adapter. UNSUPPORTED
|
|
258
|
+
verdicts block the plan; ALTERNATIVE verdicts are tolerated but emit a
|
|
259
|
+
warning (ADR-M3-2).
|
|
260
|
+
"""
|
|
261
|
+
if plan.environment_fingerprint != ctx.fingerprint:
|
|
262
|
+
raise SafetyRefusedError(
|
|
263
|
+
"environment.mismatch",
|
|
264
|
+
"plan fingerprint does not match current environment identity;"
|
|
265
|
+
" re-plan against live topology",
|
|
266
|
+
)
|
|
267
|
+
_check_k8s_targets(plan, graph)
|
|
268
|
+
_check_remote_targets(plan, graph)
|
|
269
|
+
if adapter is not None:
|
|
270
|
+
_validate_capability_requirements(plan, adapter, ctx)
|
|
271
|
+
seen_faults: list[str] = []
|
|
272
|
+
for step in plan.steps:
|
|
273
|
+
fault = step.fault
|
|
274
|
+
if fault is None:
|
|
275
|
+
continue
|
|
276
|
+
definition_risk = _risk_of(fault.fault_id)
|
|
277
|
+
check_fault_admission(fault.fault_id, definition_risk, ctx)
|
|
278
|
+
target_ids = frozenset().union(*(t.node_ids for t in fault.targets))
|
|
279
|
+
check_blast_radius(
|
|
280
|
+
graph,
|
|
281
|
+
target_ids,
|
|
282
|
+
float(fault.duration),
|
|
283
|
+
tuple(seen_faults),
|
|
284
|
+
fault.fault_id,
|
|
285
|
+
ctx=ctx,
|
|
286
|
+
)
|
|
287
|
+
seen_faults.append(fault.fault_id)
|
|
288
|
+
# G4: validate execution context compatibility
|
|
289
|
+
if fault.execution_context is not None:
|
|
290
|
+
_check_execution_context(fault, graph)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _validate_capability_requirements(
|
|
294
|
+
plan: ExecutionPlan,
|
|
295
|
+
adapter: RuntimeAdapter,
|
|
296
|
+
ctx: SafetyContext,
|
|
297
|
+
) -> None:
|
|
298
|
+
"""ADR-M3-2: evaluate plan capability requirements against an adapter.
|
|
299
|
+
|
|
300
|
+
Unsupported verdicts block the plan; alternatives warn.
|
|
301
|
+
"""
|
|
302
|
+
namespaces: set[str] = set()
|
|
303
|
+
tools: set[str] = set()
|
|
304
|
+
permissions: set[str] = set()
|
|
305
|
+
for step in plan.steps:
|
|
306
|
+
fault = step.fault
|
|
307
|
+
if fault is None:
|
|
308
|
+
continue
|
|
309
|
+
if fault.execution_loci is not None:
|
|
310
|
+
target = fault.execution_loci.get("target")
|
|
311
|
+
if isinstance(target, str) and target.startswith("network_namespace"):
|
|
312
|
+
namespaces.add(target)
|
|
313
|
+
if fault.execution_context is not None and (
|
|
314
|
+
fault.execution_context.context
|
|
315
|
+
in (ExecutionContext.NETWORK_NAMESPACE, ExecutionContext.PROCESS)
|
|
316
|
+
):
|
|
317
|
+
namespaces.add(fault.execution_context.context.value)
|
|
318
|
+
for target in fault.targets:
|
|
319
|
+
for node_id in target.node_ids:
|
|
320
|
+
if node_id.startswith("net"):
|
|
321
|
+
namespaces.add(node_id)
|
|
322
|
+
if node_id.startswith(("p-", "proc")):
|
|
323
|
+
permissions.add("limit")
|
|
324
|
+
|
|
325
|
+
reqs = CapabilityRequirements(
|
|
326
|
+
namespaces=frozenset(namespaces),
|
|
327
|
+
tools=frozenset(tools),
|
|
328
|
+
permissions=frozenset(permissions),
|
|
329
|
+
)
|
|
330
|
+
fallback = CapabilityRequirements(
|
|
331
|
+
namespaces=frozenset({"network"}),
|
|
332
|
+
tools=frozenset({"tool"}),
|
|
333
|
+
permissions=frozenset({"limit"}),
|
|
334
|
+
)
|
|
335
|
+
if not (namespaces or tools or permissions):
|
|
336
|
+
result = adapter.evaluate(fallback)
|
|
337
|
+
else:
|
|
338
|
+
result = adapter.evaluate(reqs)
|
|
339
|
+
if result.blocking:
|
|
340
|
+
message = result.refuse_with_message() or "unsupported capability requirements"
|
|
341
|
+
raise SafetyRefusedError("capability.unsupported", f"{adapter.id}: {message}")
|
|
342
|
+
for key, verdict in result.verdicts.items():
|
|
343
|
+
if verdict == CapabilityVerdict.ALTERNATIVE:
|
|
344
|
+
ctx.warnings.append(f"{adapter.id}: capability '{key}' satisfied via ALTERNATIVE path")
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def pre_exec_assertion(
|
|
348
|
+
target_selector_pairs: Iterable[tuple[TargetSelector, frozenset[str] | tuple[str, ...]]],
|
|
349
|
+
live_graph: TopologyGraph,
|
|
350
|
+
) -> None:
|
|
351
|
+
"""G3: seconds before injection, re-check selectors against live topology."""
|
|
352
|
+
for selector, expected_ids in target_selector_pairs:
|
|
353
|
+
try:
|
|
354
|
+
live = live_graph.resolve(selector)
|
|
355
|
+
except TargetResolutionError as exc:
|
|
356
|
+
raise SafetyRefusedError(
|
|
357
|
+
"target.drift", f"G3 drift refusal for {selector}: {exc}"
|
|
358
|
+
) from None
|
|
359
|
+
live_ids = frozenset(n.id for n in live)
|
|
360
|
+
missing = set(expected_ids) - live_ids
|
|
361
|
+
if missing:
|
|
362
|
+
raise SafetyRefusedError(
|
|
363
|
+
"target.drift",
|
|
364
|
+
f"G3 drift refusal: nodes vanished since planning: {sorted(missing)}",
|
|
365
|
+
)
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
# -- helpers ---------------------------------------------------------------------------
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _risk_of(fault_id: str) -> RiskLevel:
|
|
372
|
+
from mayhem.domain.catalog import (
|
|
373
|
+
definition_for,
|
|
374
|
+
) # local: keep module import graph flat
|
|
375
|
+
from mayhem.domain.errors import SchemaValidationError
|
|
376
|
+
|
|
377
|
+
try:
|
|
378
|
+
return definition_for(fault_id).risk
|
|
379
|
+
except SchemaValidationError:
|
|
380
|
+
return RiskLevel.LOW
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _service_kind() -> NodeKind:
|
|
384
|
+
from mayhem.domain.topology import NodeKind
|
|
385
|
+
|
|
386
|
+
return NodeKind.SERVICE
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _host_kind() -> NodeKind:
|
|
390
|
+
from mayhem.domain.topology import NodeKind
|
|
391
|
+
|
|
392
|
+
return NodeKind.HOST
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Campaign model — orchestrating multiple experiments (ADR-0022).
|
|
2
|
+
|
|
3
|
+
A ``Campaign`` groups multiple experiments under a single scheduling and
|
|
4
|
+
execution umbrella, with start/stop times, concurrency limits, and a
|
|
5
|
+
policy for what happens when an experiment fails.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from enum import StrEnum
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
|
13
|
+
|
|
14
|
+
from mayhem.domain.common import Duration
|
|
15
|
+
from mayhem.domain.errors import InvariantViolationError
|
|
16
|
+
from mayhem.domain.risks import RiskLevel
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class CampaignStatus(StrEnum):
|
|
20
|
+
DRAFT = "draft"
|
|
21
|
+
SCHEDULED = "scheduled"
|
|
22
|
+
RUNNING = "running"
|
|
23
|
+
PAUSED = "paused"
|
|
24
|
+
COMPLETED = "completed"
|
|
25
|
+
ABORTED = "aborted"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class ExperimentOnFailure(StrEnum):
|
|
29
|
+
"""What happens when an experiment in a campaign fails."""
|
|
30
|
+
|
|
31
|
+
ABORT_CAMPAIGN = "abort_campaign"
|
|
32
|
+
SKIP_AND_CONTINUE = "skip_and_continue"
|
|
33
|
+
RETRY_THEN_ABORT = "retry_then_abort"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class CampaignExperiment(BaseModel):
|
|
37
|
+
"""A single experiment entry within a campaign.
|
|
38
|
+
|
|
39
|
+
References an experiment spec by name or ID. The actual spec is resolved
|
|
40
|
+
at scheduling time.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
model_config = ConfigDict(frozen=True)
|
|
44
|
+
|
|
45
|
+
experiment_ref: str # name or ID of the experiment
|
|
46
|
+
priority: int = Field(default=0, ge=0) # higher = runs first
|
|
47
|
+
delay_seconds: Duration = 0.0 # delay after previous experiment completes
|
|
48
|
+
weight: float = 1.0 # for weighted random selection
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class CampaignWindow(BaseModel):
|
|
52
|
+
"""Time window for campaign execution."""
|
|
53
|
+
|
|
54
|
+
model_config = ConfigDict(frozen=True)
|
|
55
|
+
|
|
56
|
+
start_epoch_s: float | None = None
|
|
57
|
+
end_epoch_s: float | None = None
|
|
58
|
+
max_duration_s: Duration = 3600.0 # hard stop
|
|
59
|
+
cooldown_between_experiments_s: Duration = 5.0
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class CampaignPolicy(BaseModel):
|
|
63
|
+
"""Execution policies for a campaign."""
|
|
64
|
+
|
|
65
|
+
model_config = ConfigDict(frozen=True)
|
|
66
|
+
|
|
67
|
+
on_experiment_failure: ExperimentOnFailure = ExperimentOnFailure.ABORT_CAMPAIGN
|
|
68
|
+
max_concurrent_experiments: int = Field(default=1, ge=1)
|
|
69
|
+
max_risk_level: RiskLevel | None = None # risk ceiling for all experiments
|
|
70
|
+
total_budget_usd: float | None = None # cost ceiling (future use)
|
|
71
|
+
require_approval_above: RiskLevel | None = RiskLevel.HIGH
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class Campaign(BaseModel):
|
|
75
|
+
"""A campaign is a named, schedulable collection of experiments."""
|
|
76
|
+
|
|
77
|
+
model_config = ConfigDict(frozen=True)
|
|
78
|
+
|
|
79
|
+
id: str
|
|
80
|
+
name: str
|
|
81
|
+
description: str = ""
|
|
82
|
+
status: CampaignStatus = CampaignStatus.DRAFT
|
|
83
|
+
experiments: tuple[CampaignExperiment, ...] = ()
|
|
84
|
+
window: CampaignWindow = Field(default_factory=CampaignWindow)
|
|
85
|
+
policy: CampaignPolicy = Field(default_factory=CampaignPolicy)
|
|
86
|
+
labels: dict[str, str] = Field(default_factory=dict)
|
|
87
|
+
|
|
88
|
+
@field_validator("experiments")
|
|
89
|
+
@classmethod
|
|
90
|
+
def _non_empty_experiments(
|
|
91
|
+
cls, value: tuple[CampaignExperiment, ...]
|
|
92
|
+
) -> tuple[CampaignExperiment, ...]:
|
|
93
|
+
if not value:
|
|
94
|
+
raise InvariantViolationError(
|
|
95
|
+
"campaign_requires_experiments",
|
|
96
|
+
"a campaign must have at least one experiment",
|
|
97
|
+
)
|
|
98
|
+
return value
|
|
99
|
+
|
|
100
|
+
def sorted_experiments(self) -> tuple[CampaignExperiment, ...]:
|
|
101
|
+
"""Experiments ordered by priority (descending)."""
|
|
102
|
+
return tuple(sorted(self.experiments, key=lambda e: e.priority, reverse=True))
|
|
103
|
+
|
|
104
|
+
def total_weight(self) -> float:
|
|
105
|
+
"""Sum of all experiment weights."""
|
|
106
|
+
return sum(e.weight for e in self.experiments)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class CampaignSchedule(BaseModel):
|
|
110
|
+
"""A scheduled campaign with execution metadata."""
|
|
111
|
+
|
|
112
|
+
model_config = ConfigDict(frozen=True)
|
|
113
|
+
|
|
114
|
+
campaign: Campaign
|
|
115
|
+
scheduled_epoch_s: float | None = None
|
|
116
|
+
last_run_epoch_s: float | None = None
|
|
117
|
+
run_count: int = 0
|
|
118
|
+
next_experiment_idx: int = 0
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Cancellation escalation ladder (ADR-M2 Phase 2.5).
|
|
2
|
+
|
|
3
|
+
Users can request abort/cancel of a running run and escalate if the current
|
|
4
|
+
mechanism fails. The ladder is monotonic:
|
|
5
|
+
|
|
6
|
+
grace -> term -> kill
|
|
7
|
+
|
|
8
|
+
* ``grace`` — cooperative, safe-abort at the next checkpoint (between steps /
|
|
9
|
+
fault boundaries). In-flight faults are allowed to finish their undo so the
|
|
10
|
+
run never leaves a mutated system behind.
|
|
11
|
+
* ``term`` — SIGTERM to live payload toolkit processes; the engine re-asserts
|
|
12
|
+
after a short grace wait.
|
|
13
|
+
* ``kill`` — SIGKILL to payload toolkit processes; immediate abort.
|
|
14
|
+
|
|
15
|
+
A :class:`CancellationToken` is the shared, thread-safe object the signal
|
|
16
|
+
handler escalates and the agent loop reads at every safe point. Levels never
|
|
17
|
+
decrease: requesting ``term`` after ``kill`` is a no-op.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from enum import IntEnum
|
|
23
|
+
from threading import Lock
|
|
24
|
+
|
|
25
|
+
_LADDER: tuple[CancellationLevel, ...] = (
|
|
26
|
+
"grace",
|
|
27
|
+
"term",
|
|
28
|
+
"kill",
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class CancellationLevel(IntEnum):
|
|
33
|
+
"""Monotonic cancellation intensity; IntEnum so ``>=`` comparisons order it."""
|
|
34
|
+
|
|
35
|
+
NONE = 0
|
|
36
|
+
GRACE = 1
|
|
37
|
+
TERM = 2
|
|
38
|
+
KILL = 3
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def next(self) -> CancellationLevel | None:
|
|
42
|
+
idx = int(self)
|
|
43
|
+
nxt = _LADDER[idx] if idx < len(_LADDER) else None
|
|
44
|
+
return CancellationLevel[nxt.upper()] if nxt is not None else None
|
|
45
|
+
|
|
46
|
+
def __str__(self) -> str: # paint nicely in logs/events
|
|
47
|
+
return self.name.lower()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class CancellationToken:
|
|
51
|
+
"""Thread-safe cancellation signal readable by the agent loop.
|
|
52
|
+
|
|
53
|
+
The level only ever rises. Readers observe ``level`` and the convenience
|
|
54
|
+
flags; the signal layer escalates via :meth:`request` / :meth:`escalate`.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
__slots__ = ("_level", "_lock", "_mutations")
|
|
58
|
+
|
|
59
|
+
def __init__(self, level: CancellationLevel = CancellationLevel.NONE) -> None:
|
|
60
|
+
self._level: CancellationLevel = CancellationLevel(level)
|
|
61
|
+
self._lock = Lock()
|
|
62
|
+
self._mutations = 0
|
|
63
|
+
|
|
64
|
+
# -- escalation ----------------------------------------------------------
|
|
65
|
+
|
|
66
|
+
def request(self, level: CancellationLevel) -> bool:
|
|
67
|
+
"""Raise the level to *level* (never lower it). Idempotent.
|
|
68
|
+
|
|
69
|
+
Returns ``True`` if the level actually changed.
|
|
70
|
+
"""
|
|
71
|
+
level = CancellationLevel(level)
|
|
72
|
+
with self._lock:
|
|
73
|
+
if level > self._level:
|
|
74
|
+
self._level = level
|
|
75
|
+
self._mutations += 1
|
|
76
|
+
return True
|
|
77
|
+
return False
|
|
78
|
+
|
|
79
|
+
def escalate(self) -> CancellationLevel:
|
|
80
|
+
"""Advance one rung up the ladder; ``kill`` is the ceiling.
|
|
81
|
+
|
|
82
|
+
Returns the new (effective) level.
|
|
83
|
+
"""
|
|
84
|
+
with self._lock:
|
|
85
|
+
nxt = self._level.next
|
|
86
|
+
if nxt is not None:
|
|
87
|
+
self._level = nxt
|
|
88
|
+
self._mutations += 1
|
|
89
|
+
return self._level
|
|
90
|
+
|
|
91
|
+
# -- observation ---------------------------------------------------------
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def level(self) -> CancellationLevel:
|
|
95
|
+
with self._lock:
|
|
96
|
+
return self._level
|
|
97
|
+
|
|
98
|
+
@property
|
|
99
|
+
def cancelled(self) -> bool:
|
|
100
|
+
"""True at any non-NONE level — agents must stop cooperative work."""
|
|
101
|
+
return self.level != CancellationLevel.NONE
|
|
102
|
+
|
|
103
|
+
@property
|
|
104
|
+
def is_kill(self) -> bool:
|
|
105
|
+
return self.level == CancellationLevel.KILL
|
|
106
|
+
|
|
107
|
+
def snapshot(self) -> tuple[CancellationLevel, int]:
|
|
108
|
+
"""Atomically read ``(level, mutation_count)`` for change detection."""
|
|
109
|
+
with self._lock:
|
|
110
|
+
return self._level, self._mutations
|