mayhem-cli 0.5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. mayhem/agent/__init__.py +1 -0
  2. mayhem/agent/cli.py +36 -0
  3. mayhem/agents/__init__.py +1 -0
  4. mayhem/agents/capabilities.py +106 -0
  5. mayhem/agents/executors.py +430 -0
  6. mayhem/agents/impact.py +729 -0
  7. mayhem/agents/lease_client.py +141 -0
  8. mayhem/agents/probes.py +284 -0
  9. mayhem/agents/protocol.py +134 -0
  10. mayhem/agents/server.py +281 -0
  11. mayhem/agents/sinks.py +60 -0
  12. mayhem/agents/transports.py +134 -0
  13. mayhem/agents/watchdog.py +140 -0
  14. mayhem/cli/__init__.py +11 -0
  15. mayhem/cli/app.py +154 -0
  16. mayhem/cli/campaign.py +496 -0
  17. mayhem/cli/config_cmd.py +47 -0
  18. mayhem/cli/context.py +23 -0
  19. mayhem/cli/dependency.py +429 -0
  20. mayhem/cli/exit_codes.py +24 -0
  21. mayhem/cli/experiment.py +24 -0
  22. mayhem/cli/lifecycle.py +805 -0
  23. mayhem/cli/resolver.py +72 -0
  24. mayhem/cli/services.py +459 -0
  25. mayhem/cli/style.py +101 -0
  26. mayhem/cli/toolkit.py +41 -0
  27. mayhem/cli/topology.py +127 -0
  28. mayhem/config.py +208 -0
  29. mayhem/controller/__init__.py +1 -0
  30. mayhem/controller/compensation.py +2156 -0
  31. mayhem/controller/executor.py +1719 -0
  32. mayhem/controller/janitor.py +196 -0
  33. mayhem/controller/observability_collector.py +382 -0
  34. mayhem/controller/observations.py +102 -0
  35. mayhem/controller/planner.py +715 -0
  36. mayhem/controller/recovery.py +245 -0
  37. mayhem/controller/resilience_report.py +585 -0
  38. mayhem/controller/resource_manager.py +457 -0
  39. mayhem/controller/safety.py +392 -0
  40. mayhem/domain/__init__.py +6 -0
  41. mayhem/domain/campaigns.py +118 -0
  42. mayhem/domain/cancellation.py +110 -0
  43. mayhem/domain/candidates.py +101 -0
  44. mayhem/domain/capabilities.py +86 -0
  45. mayhem/domain/catalog.py +727 -0
  46. mayhem/domain/checks.py +173 -0
  47. mayhem/domain/common.py +104 -0
  48. mayhem/domain/coverage.py +106 -0
  49. mayhem/domain/decisions.py +57 -0
  50. mayhem/domain/errors.py +87 -0
  51. mayhem/domain/events.py +61 -0
  52. mayhem/domain/execution_context.py +120 -0
  53. mayhem/domain/execution_loci.py +94 -0
  54. mayhem/domain/experiments.py +370 -0
  55. mayhem/domain/faults.py +239 -0
  56. mayhem/domain/identity.py +200 -0
  57. mayhem/domain/k8s_adapter.py +132 -0
  58. mayhem/domain/leases.py +186 -0
  59. mayhem/domain/load_strategy.py +98 -0
  60. mayhem/domain/m5_campaign.py +120 -0
  61. mayhem/domain/maniac.py +93 -0
  62. mayhem/domain/observability.py +146 -0
  63. mayhem/domain/outcomes.py +92 -0
  64. mayhem/domain/remote_agent_interface.py +70 -0
  65. mayhem/domain/resources.py +245 -0
  66. mayhem/domain/risks.py +61 -0
  67. mayhem/domain/run_outcome.py +146 -0
  68. mayhem/domain/runtime_adapter.py +256 -0
  69. mayhem/domain/success.py +329 -0
  70. mayhem/domain/topology.py +452 -0
  71. mayhem/infra/__init__.py +1 -0
  72. mayhem/infra/campaign_engine.py +205 -0
  73. mayhem/infra/candidate_gates.py +124 -0
  74. mayhem/infra/candidate_generator.py +110 -0
  75. mayhem/infra/coverage_repository.py +101 -0
  76. mayhem/infra/lease_repository.py +129 -0
  77. mayhem/infra/maniac.py +103 -0
  78. mayhem/infra/migrations.py +596 -0
  79. mayhem/infra/migrator.py +149 -0
  80. mayhem/infra/report.py +227 -0
  81. mayhem/infra/store.py +200 -0
  82. mayhem/py.typed +0 -0
  83. mayhem/spec.py +52 -0
  84. mayhem/toolkit/__init__.py +1 -0
  85. mayhem/toolkit/fingerprint.py +69 -0
  86. mayhem/toolkit/hashing.py +32 -0
  87. mayhem/toolkit/manifests/docker.yaml +11 -0
  88. mayhem/toolkit/manifests/podman.yaml +11 -0
  89. mayhem/toolkit/manifests/stress-ng.yaml +11 -0
  90. mayhem/toolkit/manifests/tc-netem.yaml +11 -0
  91. mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
  92. mayhem/toolkit/registry.py +185 -0
  93. mayhem/toolkit/tool_runner.py +129 -0
  94. mayhem/topology/__init__.py +10 -0
  95. mayhem/topology/providers/__init__.py +0 -0
  96. mayhem/topology/providers/adapter_registry.py +60 -0
  97. mayhem/topology/providers/base.py +31 -0
  98. mayhem/topology/providers/compose.py +207 -0
  99. mayhem/topology/providers/docker_adapter.py +277 -0
  100. mayhem/topology/providers/docker_runtime.py +461 -0
  101. mayhem/topology/providers/podman_adapter.py +328 -0
  102. mayhem/topology/resolve.py +196 -0
  103. mayhem/topology/service.py +158 -0
  104. mayhem_cli-0.5.1.dist-info/METADATA +555 -0
  105. mayhem_cli-0.5.1.dist-info/RECORD +107 -0
  106. mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
  107. mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,392 @@
1
+ """Mechanical safety gates between planner and executor (ADR-0012, ADR-0014,
2
+ architecture/safety.md).
3
+
4
+ Gate stack enforced here:
5
+ G1 config policy — allowlists/denylists, risk ladder, critical opt-in
6
+ G2 plan validation — budgets fit the topology graph, fingerprint match
7
+ G3 pre-exec assertion — resolved targets re-checked against *live* topology
8
+ G4 execution context — declared context must be feasible for target node kinds
9
+
10
+ Precedence: denylist beats allowlist beats selector beats default.
11
+ Every refusal is typed and carries a machine-readable reason.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import hashlib
17
+ from dataclasses import dataclass, field
18
+ from typing import TYPE_CHECKING
19
+
20
+ from mayhem.domain.errors import InvariantViolationError, TargetResolutionError
21
+ from mayhem.domain.execution_context import ExecutionContext
22
+ from mayhem.domain.risks import RiskLevel
23
+ from mayhem.domain.runtime_adapter import (
24
+ CapabilityRequirements,
25
+ CapabilityVerdict,
26
+ RuntimeAdapter,
27
+ )
28
+
29
+ if TYPE_CHECKING:
30
+ from collections.abc import Iterable
31
+
32
+ from mayhem.config import PolicyCfg
33
+ from mayhem.domain.experiments import BlastRadiusBudget, ExecutionPlan, PlannedFault
34
+ from mayhem.domain.topology import NodeKind, TargetSelector, TopologyGraph
35
+
36
+
37
+ class SafetyRefusedError(InvariantViolationError):
38
+ """A mechanical safety gate refused a fault or plan (``safety.refused``)."""
39
+
40
+ def __init__(self, reason_code: str, message: str) -> None:
41
+ super().__init__(reason_code, message)
42
+ self.reason_code = reason_code
43
+
44
+
45
+ def environment_fingerprint(
46
+ *,
47
+ host_names: Iterable[str],
48
+ compose_digest: str,
49
+ profile: str | None = None,
50
+ ) -> str:
51
+ """SHA256(sorted host set + compose digest + profile name)."""
52
+ payload = "|".join(
53
+ (
54
+ ",".join(sorted(host_names)),
55
+ compose_digest,
56
+ profile or "default",
57
+ )
58
+ )
59
+ return hashlib.sha256(payload.encode()).hexdigest()
60
+
61
+
62
+ @dataclass(frozen=True)
63
+ class SafetyContext:
64
+ """Everything G1/G2 need; built once per CLI invocation."""
65
+
66
+ policy: PolicyCfg
67
+ budget: BlastRadiusBudget
68
+ fingerprint: str
69
+ allow_critical_cli: bool = False
70
+ warnings: list[str] = field(default_factory=list)
71
+
72
+
73
+ DEFAULT_DENY = frozenset({"node.reboot"})
74
+ """Host-reboot class faults are forbidden unless explicitly allowlisted
75
+ ([ADR-0012]: an explicit allowlist overrides default deny; the bare default
76
+ policy never grants them)."""
77
+
78
+
79
+ def check_fault_admission(fault_id: str, definition_risk: RiskLevel, ctx: SafetyContext) -> None:
80
+ """G1: denylist → allowlist → risk ceiling → critical double-opt-in."""
81
+ if fault_id in ctx.policy.deny_faults:
82
+ raise SafetyRefusedError("safety.refused", f"{fault_id}: denied by policy denylist")
83
+ if ctx.policy.allow_faults is not None:
84
+ if fault_id not in ctx.policy.allow_faults:
85
+ raise SafetyRefusedError(
86
+ "safety.refused", f"{fault_id}: not present in policy allowlist"
87
+ )
88
+ elif fault_id in DEFAULT_DENY:
89
+ raise SafetyRefusedError(
90
+ "safety.refused",
91
+ f"{fault_id}: denied by default; add it to policy.allow_faults to opt in",
92
+ )
93
+ ceiling = ctx.policy.risk_ceiling
94
+ if ceiling is not None and definition_risk.at_least(ceiling.next_higher()):
95
+ raise SafetyRefusedError(
96
+ "safety.refused",
97
+ f"{fault_id}: risk {definition_risk.value} exceeds policy ceiling {ceiling.value}",
98
+ )
99
+ if definition_risk is RiskLevel.CRITICAL and not (
100
+ ctx.policy.allow_critical and ctx.allow_critical_cli
101
+ ):
102
+ raise SafetyRefusedError(
103
+ "safety.refused",
104
+ f"{fault_id}: critical risk requires config policy.allow_critical AND --allow-critical",
105
+ )
106
+
107
+
108
+ def _affected_node_ids(graph: TopologyGraph, targets: Iterable[str]) -> frozenset[str]:
109
+ affected: set[str] = set()
110
+ for node_id in targets:
111
+ affected.add(node_id)
112
+ affected |= graph.dependents_closure(node_id)
113
+ return frozenset(affected)
114
+
115
+
116
+ def check_blast_radius(
117
+ graph: TopologyGraph,
118
+ target_node_ids: Iterable[str],
119
+ duration_s: float,
120
+ fault_ids_so_far: tuple[str, ...],
121
+ new_fault_id: str,
122
+ *,
123
+ ctx: SafetyContext,
124
+ ) -> dict[str, float]:
125
+ """G2 (budget half): topology-derived blast radius must fit; returns measured stats."""
126
+ budget = ctx.budget
127
+ affected = _affected_node_ids(graph, target_node_ids)
128
+ services_total = len(graph.of_kind(_service_kind()))
129
+ services_hit = sum(1 for n in graph.of_kind(_service_kind()) if n.id in affected)
130
+ hosts_hit = sum(1 for n in graph.of_kind(_host_kind()) if n.id in affected)
131
+ pct = (services_hit / services_total * 100.0) if services_total else 0.0
132
+ stats = {
133
+ "services_pct": round(pct, 1),
134
+ "hosts": float(hosts_hit),
135
+ "concurrent_faults": float(len(fault_ids_so_far) + 1),
136
+ "duration_per_fault": duration_s,
137
+ }
138
+ if pct > budget.max_services_pct:
139
+ raise SafetyRefusedError(
140
+ "safety.refused",
141
+ f"blast radius: {new_fault_id} would affect {pct:.0f}% of services"
142
+ f" > budget {budget.max_services_pct}%",
143
+ )
144
+ if hosts_hit > budget.max_hosts:
145
+ raise SafetyRefusedError(
146
+ "safety.refused",
147
+ f"blast radius: {new_fault_id} touches {hosts_hit} hosts > budget {budget.max_hosts}",
148
+ )
149
+ if len(fault_ids_so_far) + 1 > budget.max_concurrent_faults:
150
+ raise SafetyRefusedError("safety.refused", "blast radius: max_concurrent_faults exceeded")
151
+ if duration_s > budget.max_duration_per_fault_s:
152
+ raise SafetyRefusedError(
153
+ "safety.refused",
154
+ f"{new_fault_id} duration {duration_s:.0f}s exceeds per-fault cap"
155
+ f" {budget.max_duration_per_fault_s:.0f}s",
156
+ )
157
+ pair = frozenset((*fault_ids_so_far, new_fault_id)) if fault_ids_so_far else None
158
+ if pair and pair in budget.forbidden_fault_pairs:
159
+ raise SafetyRefusedError(
160
+ "safety.refused",
161
+ f"blast radius: forbidden fault pair {sorted(pair)}",
162
+ )
163
+ return stats
164
+
165
+
166
+ def _check_execution_context(fault: PlannedFault, graph: TopologyGraph) -> None:
167
+ """G4: validate that the declared execution context is feasible for all targets.
168
+
169
+ Prevents accidental host-level execution when the experiment intended
170
+ container-level execution, or vice versa.
171
+ """
172
+ if fault.execution_context is None:
173
+ return
174
+ node_kinds: set[NodeKind] = set()
175
+ for target in fault.targets:
176
+ for node_id in target.node_ids:
177
+ node = graph.by_id(node_id)
178
+ if node is not None:
179
+ node_kinds.add(node.kind)
180
+ if not node_kinds:
181
+ return # no nodes resolved — G3 will catch this
182
+ try:
183
+ fault.execution_context.assert_compatible(frozenset(node_kinds))
184
+ except InvariantViolationError as exc:
185
+ raise SafetyRefusedError(
186
+ "execution_context.refused",
187
+ f"{fault.fault_id}: {exc}",
188
+ ) from exc
189
+
190
+
191
+ _K8S_NODE_KIND_VALUES: frozenset[str] = frozenset({"pod", "k8s_node"})
192
+ _K8S_REFUSE_MSG = (
193
+ "kubernetes execution not yet supported; see the RuntimeAdapter contract at ADR-M7-1"
194
+ )
195
+
196
+ _REMOTE_NODE_KIND_VALUES: frozenset[str] = frozenset({"external_dependency"})
197
+ _REMOTE_REFUSE_MSG = (
198
+ "remote execution not yet supported; ADR-M3-5 ships only the "
199
+ "RemoteAgentInterface contract — no transport is wired in this milestone"
200
+ )
201
+
202
+
203
+ def _check_k8s_targets(plan: ExecutionPlan, graph: TopologyGraph) -> None:
204
+ """Refuse any plan that targets K8s node kinds without a live driver (ADR-M7).
205
+
206
+ K8s execution is out-of-scope for this milestone — the adapter contract
207
+ exists so future drivers can implement against a stable seam, but no
208
+ live-cluster fault injection is wired yet. A plan targeting K8s nodes
209
+ must fail loud and early with an actionable message.
210
+ """
211
+ for step in plan.steps:
212
+ fault = step.fault
213
+ if fault is None:
214
+ continue
215
+ for target in fault.targets:
216
+ for node_id in target.node_ids:
217
+ node = graph.by_id(node_id)
218
+ if node is not None and node.kind in _K8S_NODE_KIND_VALUES:
219
+ raise SafetyRefusedError(
220
+ "k8s.unsupported",
221
+ f"{fault.fault_id}: {_K8S_REFUSE_MSG}",
222
+ )
223
+
224
+
225
+ def _check_remote_targets(plan: ExecutionPlan, graph: TopologyGraph) -> None:
226
+ """Hard planning gate for remote targets (ADR-M3-5) — defect register #4.
227
+
228
+ Remote execution is interface-only (``RemoteAgentInterface``): the adapter
229
+ rejects requirements at ``evaluate`` time, but that refusal was never
230
+ auto-wired into ``validate_plan``, so a remote spec would plan and only
231
+ fail mid-execution. This gate mirrors the K8s one: a plan targeting an
232
+ ``external_dependency`` node fails loud and early at plan time.
233
+ """
234
+ for step in plan.steps:
235
+ fault = step.fault
236
+ if fault is None:
237
+ continue
238
+ for target in fault.targets:
239
+ for node_id in target.node_ids:
240
+ node = graph.by_id(node_id)
241
+ if node is not None and node.kind in _REMOTE_NODE_KIND_VALUES:
242
+ raise SafetyRefusedError(
243
+ "remote.unsupported",
244
+ f"{fault.fault_id}: {_REMOTE_REFUSE_MSG}",
245
+ )
246
+
247
+
248
+ def validate_plan(
249
+ plan: ExecutionPlan,
250
+ graph: TopologyGraph,
251
+ ctx: SafetyContext,
252
+ adapter: RuntimeAdapter | None = None,
253
+ ) -> None:
254
+ """G1+G2+G4 over every fault step of an already-compiled plan.
255
+
256
+ When *adapter* is provided, capability requirements are derived from the
257
+ plan's execution contexts and evaluated against the adapter. UNSUPPORTED
258
+ verdicts block the plan; ALTERNATIVE verdicts are tolerated but emit a
259
+ warning (ADR-M3-2).
260
+ """
261
+ if plan.environment_fingerprint != ctx.fingerprint:
262
+ raise SafetyRefusedError(
263
+ "environment.mismatch",
264
+ "plan fingerprint does not match current environment identity;"
265
+ " re-plan against live topology",
266
+ )
267
+ _check_k8s_targets(plan, graph)
268
+ _check_remote_targets(plan, graph)
269
+ if adapter is not None:
270
+ _validate_capability_requirements(plan, adapter, ctx)
271
+ seen_faults: list[str] = []
272
+ for step in plan.steps:
273
+ fault = step.fault
274
+ if fault is None:
275
+ continue
276
+ definition_risk = _risk_of(fault.fault_id)
277
+ check_fault_admission(fault.fault_id, definition_risk, ctx)
278
+ target_ids = frozenset().union(*(t.node_ids for t in fault.targets))
279
+ check_blast_radius(
280
+ graph,
281
+ target_ids,
282
+ float(fault.duration),
283
+ tuple(seen_faults),
284
+ fault.fault_id,
285
+ ctx=ctx,
286
+ )
287
+ seen_faults.append(fault.fault_id)
288
+ # G4: validate execution context compatibility
289
+ if fault.execution_context is not None:
290
+ _check_execution_context(fault, graph)
291
+
292
+
293
+ def _validate_capability_requirements(
294
+ plan: ExecutionPlan,
295
+ adapter: RuntimeAdapter,
296
+ ctx: SafetyContext,
297
+ ) -> None:
298
+ """ADR-M3-2: evaluate plan capability requirements against an adapter.
299
+
300
+ Unsupported verdicts block the plan; alternatives warn.
301
+ """
302
+ namespaces: set[str] = set()
303
+ tools: set[str] = set()
304
+ permissions: set[str] = set()
305
+ for step in plan.steps:
306
+ fault = step.fault
307
+ if fault is None:
308
+ continue
309
+ if fault.execution_loci is not None:
310
+ target = fault.execution_loci.get("target")
311
+ if isinstance(target, str) and target.startswith("network_namespace"):
312
+ namespaces.add(target)
313
+ if fault.execution_context is not None and (
314
+ fault.execution_context.context
315
+ in (ExecutionContext.NETWORK_NAMESPACE, ExecutionContext.PROCESS)
316
+ ):
317
+ namespaces.add(fault.execution_context.context.value)
318
+ for target in fault.targets:
319
+ for node_id in target.node_ids:
320
+ if node_id.startswith("net"):
321
+ namespaces.add(node_id)
322
+ if node_id.startswith(("p-", "proc")):
323
+ permissions.add("limit")
324
+
325
+ reqs = CapabilityRequirements(
326
+ namespaces=frozenset(namespaces),
327
+ tools=frozenset(tools),
328
+ permissions=frozenset(permissions),
329
+ )
330
+ fallback = CapabilityRequirements(
331
+ namespaces=frozenset({"network"}),
332
+ tools=frozenset({"tool"}),
333
+ permissions=frozenset({"limit"}),
334
+ )
335
+ if not (namespaces or tools or permissions):
336
+ result = adapter.evaluate(fallback)
337
+ else:
338
+ result = adapter.evaluate(reqs)
339
+ if result.blocking:
340
+ message = result.refuse_with_message() or "unsupported capability requirements"
341
+ raise SafetyRefusedError("capability.unsupported", f"{adapter.id}: {message}")
342
+ for key, verdict in result.verdicts.items():
343
+ if verdict == CapabilityVerdict.ALTERNATIVE:
344
+ ctx.warnings.append(f"{adapter.id}: capability '{key}' satisfied via ALTERNATIVE path")
345
+
346
+
347
+ def pre_exec_assertion(
348
+ target_selector_pairs: Iterable[tuple[TargetSelector, frozenset[str] | tuple[str, ...]]],
349
+ live_graph: TopologyGraph,
350
+ ) -> None:
351
+ """G3: seconds before injection, re-check selectors against live topology."""
352
+ for selector, expected_ids in target_selector_pairs:
353
+ try:
354
+ live = live_graph.resolve(selector)
355
+ except TargetResolutionError as exc:
356
+ raise SafetyRefusedError(
357
+ "target.drift", f"G3 drift refusal for {selector}: {exc}"
358
+ ) from None
359
+ live_ids = frozenset(n.id for n in live)
360
+ missing = set(expected_ids) - live_ids
361
+ if missing:
362
+ raise SafetyRefusedError(
363
+ "target.drift",
364
+ f"G3 drift refusal: nodes vanished since planning: {sorted(missing)}",
365
+ )
366
+
367
+
368
+ # -- helpers ---------------------------------------------------------------------------
369
+
370
+
371
+ def _risk_of(fault_id: str) -> RiskLevel:
372
+ from mayhem.domain.catalog import (
373
+ definition_for,
374
+ ) # local: keep module import graph flat
375
+ from mayhem.domain.errors import SchemaValidationError
376
+
377
+ try:
378
+ return definition_for(fault_id).risk
379
+ except SchemaValidationError:
380
+ return RiskLevel.LOW
381
+
382
+
383
+ def _service_kind() -> NodeKind:
384
+ from mayhem.domain.topology import NodeKind
385
+
386
+ return NodeKind.SERVICE
387
+
388
+
389
+ def _host_kind() -> NodeKind:
390
+ from mayhem.domain.topology import NodeKind
391
+
392
+ return NodeKind.HOST
@@ -0,0 +1,6 @@
1
+ """Tgondi domain models.
2
+
3
+ Pure, importable-by-everything data layer (ADR-0002): no IO, no upward imports.
4
+ All models serialize deterministically via ``model_dump(mode="json")``; enums
5
+ serialize as their string values and timestamps are ISO-8601 UTC.
6
+ """
@@ -0,0 +1,118 @@
1
+ """Campaign model — orchestrating multiple experiments (ADR-0022).
2
+
3
+ A ``Campaign`` groups multiple experiments under a single scheduling and
4
+ execution umbrella, with start/stop times, concurrency limits, and a
5
+ policy for what happens when an experiment fails.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from enum import StrEnum
11
+
12
+ from pydantic import BaseModel, ConfigDict, Field, field_validator
13
+
14
+ from mayhem.domain.common import Duration
15
+ from mayhem.domain.errors import InvariantViolationError
16
+ from mayhem.domain.risks import RiskLevel
17
+
18
+
19
+ class CampaignStatus(StrEnum):
20
+ DRAFT = "draft"
21
+ SCHEDULED = "scheduled"
22
+ RUNNING = "running"
23
+ PAUSED = "paused"
24
+ COMPLETED = "completed"
25
+ ABORTED = "aborted"
26
+
27
+
28
+ class ExperimentOnFailure(StrEnum):
29
+ """What happens when an experiment in a campaign fails."""
30
+
31
+ ABORT_CAMPAIGN = "abort_campaign"
32
+ SKIP_AND_CONTINUE = "skip_and_continue"
33
+ RETRY_THEN_ABORT = "retry_then_abort"
34
+
35
+
36
+ class CampaignExperiment(BaseModel):
37
+ """A single experiment entry within a campaign.
38
+
39
+ References an experiment spec by name or ID. The actual spec is resolved
40
+ at scheduling time.
41
+ """
42
+
43
+ model_config = ConfigDict(frozen=True)
44
+
45
+ experiment_ref: str # name or ID of the experiment
46
+ priority: int = Field(default=0, ge=0) # higher = runs first
47
+ delay_seconds: Duration = 0.0 # delay after previous experiment completes
48
+ weight: float = 1.0 # for weighted random selection
49
+
50
+
51
+ class CampaignWindow(BaseModel):
52
+ """Time window for campaign execution."""
53
+
54
+ model_config = ConfigDict(frozen=True)
55
+
56
+ start_epoch_s: float | None = None
57
+ end_epoch_s: float | None = None
58
+ max_duration_s: Duration = 3600.0 # hard stop
59
+ cooldown_between_experiments_s: Duration = 5.0
60
+
61
+
62
+ class CampaignPolicy(BaseModel):
63
+ """Execution policies for a campaign."""
64
+
65
+ model_config = ConfigDict(frozen=True)
66
+
67
+ on_experiment_failure: ExperimentOnFailure = ExperimentOnFailure.ABORT_CAMPAIGN
68
+ max_concurrent_experiments: int = Field(default=1, ge=1)
69
+ max_risk_level: RiskLevel | None = None # risk ceiling for all experiments
70
+ total_budget_usd: float | None = None # cost ceiling (future use)
71
+ require_approval_above: RiskLevel | None = RiskLevel.HIGH
72
+
73
+
74
+ class Campaign(BaseModel):
75
+ """A campaign is a named, schedulable collection of experiments."""
76
+
77
+ model_config = ConfigDict(frozen=True)
78
+
79
+ id: str
80
+ name: str
81
+ description: str = ""
82
+ status: CampaignStatus = CampaignStatus.DRAFT
83
+ experiments: tuple[CampaignExperiment, ...] = ()
84
+ window: CampaignWindow = Field(default_factory=CampaignWindow)
85
+ policy: CampaignPolicy = Field(default_factory=CampaignPolicy)
86
+ labels: dict[str, str] = Field(default_factory=dict)
87
+
88
+ @field_validator("experiments")
89
+ @classmethod
90
+ def _non_empty_experiments(
91
+ cls, value: tuple[CampaignExperiment, ...]
92
+ ) -> tuple[CampaignExperiment, ...]:
93
+ if not value:
94
+ raise InvariantViolationError(
95
+ "campaign_requires_experiments",
96
+ "a campaign must have at least one experiment",
97
+ )
98
+ return value
99
+
100
+ def sorted_experiments(self) -> tuple[CampaignExperiment, ...]:
101
+ """Experiments ordered by priority (descending)."""
102
+ return tuple(sorted(self.experiments, key=lambda e: e.priority, reverse=True))
103
+
104
+ def total_weight(self) -> float:
105
+ """Sum of all experiment weights."""
106
+ return sum(e.weight for e in self.experiments)
107
+
108
+
109
+ class CampaignSchedule(BaseModel):
110
+ """A scheduled campaign with execution metadata."""
111
+
112
+ model_config = ConfigDict(frozen=True)
113
+
114
+ campaign: Campaign
115
+ scheduled_epoch_s: float | None = None
116
+ last_run_epoch_s: float | None = None
117
+ run_count: int = 0
118
+ next_experiment_idx: int = 0
@@ -0,0 +1,110 @@
1
+ """Cancellation escalation ladder (ADR-M2 Phase 2.5).
2
+
3
+ Users can request abort/cancel of a running run and escalate if the current
4
+ mechanism fails. The ladder is monotonic:
5
+
6
+ grace -> term -> kill
7
+
8
+ * ``grace`` — cooperative, safe-abort at the next checkpoint (between steps /
9
+ fault boundaries). In-flight faults are allowed to finish their undo so the
10
+ run never leaves a mutated system behind.
11
+ * ``term`` — SIGTERM to live payload toolkit processes; the engine re-asserts
12
+ after a short grace wait.
13
+ * ``kill`` — SIGKILL to payload toolkit processes; immediate abort.
14
+
15
+ A :class:`CancellationToken` is the shared, thread-safe object the signal
16
+ handler escalates and the agent loop reads at every safe point. Levels never
17
+ decrease: requesting ``term`` after ``kill`` is a no-op.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from enum import IntEnum
23
+ from threading import Lock
24
+
25
+ _LADDER: tuple[CancellationLevel, ...] = (
26
+ "grace",
27
+ "term",
28
+ "kill",
29
+ )
30
+
31
+
32
+ class CancellationLevel(IntEnum):
33
+ """Monotonic cancellation intensity; IntEnum so ``>=`` comparisons order it."""
34
+
35
+ NONE = 0
36
+ GRACE = 1
37
+ TERM = 2
38
+ KILL = 3
39
+
40
+ @property
41
+ def next(self) -> CancellationLevel | None:
42
+ idx = int(self)
43
+ nxt = _LADDER[idx] if idx < len(_LADDER) else None
44
+ return CancellationLevel[nxt.upper()] if nxt is not None else None
45
+
46
+ def __str__(self) -> str: # paint nicely in logs/events
47
+ return self.name.lower()
48
+
49
+
50
+ class CancellationToken:
51
+ """Thread-safe cancellation signal readable by the agent loop.
52
+
53
+ The level only ever rises. Readers observe ``level`` and the convenience
54
+ flags; the signal layer escalates via :meth:`request` / :meth:`escalate`.
55
+ """
56
+
57
+ __slots__ = ("_level", "_lock", "_mutations")
58
+
59
+ def __init__(self, level: CancellationLevel = CancellationLevel.NONE) -> None:
60
+ self._level: CancellationLevel = CancellationLevel(level)
61
+ self._lock = Lock()
62
+ self._mutations = 0
63
+
64
+ # -- escalation ----------------------------------------------------------
65
+
66
+ def request(self, level: CancellationLevel) -> bool:
67
+ """Raise the level to *level* (never lower it). Idempotent.
68
+
69
+ Returns ``True`` if the level actually changed.
70
+ """
71
+ level = CancellationLevel(level)
72
+ with self._lock:
73
+ if level > self._level:
74
+ self._level = level
75
+ self._mutations += 1
76
+ return True
77
+ return False
78
+
79
+ def escalate(self) -> CancellationLevel:
80
+ """Advance one rung up the ladder; ``kill`` is the ceiling.
81
+
82
+ Returns the new (effective) level.
83
+ """
84
+ with self._lock:
85
+ nxt = self._level.next
86
+ if nxt is not None:
87
+ self._level = nxt
88
+ self._mutations += 1
89
+ return self._level
90
+
91
+ # -- observation ---------------------------------------------------------
92
+
93
+ @property
94
+ def level(self) -> CancellationLevel:
95
+ with self._lock:
96
+ return self._level
97
+
98
+ @property
99
+ def cancelled(self) -> bool:
100
+ """True at any non-NONE level — agents must stop cooperative work."""
101
+ return self.level != CancellationLevel.NONE
102
+
103
+ @property
104
+ def is_kill(self) -> bool:
105
+ return self.level == CancellationLevel.KILL
106
+
107
+ def snapshot(self) -> tuple[CancellationLevel, int]:
108
+ """Atomically read ``(level, mutation_count)`` for change detection."""
109
+ with self._lock:
110
+ return self._level, self._mutations