mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Recovery state machine and audit trail (ADR-0016).
|
|
2
|
+
|
|
3
|
+
Every resource recovery is a state machine:
|
|
4
|
+
IDLE → RECOVERING → VERIFIED | DIRTY
|
|
5
|
+
↑ retry ↑
|
|
6
|
+
DIRTY → RECOVERING → VERIFIED | DIRTY (max retries)
|
|
7
|
+
|
|
8
|
+
The state machine is the *only* way to transition recovery states —
|
|
9
|
+
all transitions are validated, timestamped, and persisted to the
|
|
10
|
+
recovery_audit_log table. This gives us:
|
|
11
|
+
1. Idempotent recovery — safe to retry from crash
|
|
12
|
+
2. Exhaustive audit trail — every transition recorded
|
|
13
|
+
3. Retry exhaustion detection — stops after max retries
|
|
14
|
+
4. Ownership-aware — only owner can transition
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from enum import StrEnum
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
22
|
+
|
|
23
|
+
from mayhem.domain.common import utc_now
|
|
24
|
+
from mayhem.domain.errors import InvariantViolationError
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class RecoveryStatus(StrEnum):
|
|
28
|
+
"""States in the recovery lifecycle."""
|
|
29
|
+
|
|
30
|
+
IDLE = "idle" # resource is active, not recovering
|
|
31
|
+
RECOVERING = "recovering" # cleanup in progress
|
|
32
|
+
VERIFIED = "verified" # cleanup succeeded and verified
|
|
33
|
+
DIRTY = "dirty" # cleanup failed or verify failed
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# Valid transitions: source → set of targets
|
|
37
|
+
_VALID_TRANSITIONS: dict[RecoveryStatus, frozenset[RecoveryStatus]] = {
|
|
38
|
+
RecoveryStatus.IDLE: frozenset({RecoveryStatus.RECOVERING}),
|
|
39
|
+
RecoveryStatus.RECOVERING: frozenset({RecoveryStatus.VERIFIED, RecoveryStatus.DIRTY}),
|
|
40
|
+
RecoveryStatus.DIRTY: frozenset({RecoveryStatus.RECOVERING}),
|
|
41
|
+
RecoveryStatus.VERIFIED: frozenset(), # terminal state — no transitions out
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class RecoveryTransition(BaseModel):
|
|
46
|
+
"""One atomic state transition in the recovery lifecycle."""
|
|
47
|
+
|
|
48
|
+
model_config = ConfigDict(frozen=True)
|
|
49
|
+
|
|
50
|
+
resource_id: str
|
|
51
|
+
from_status: RecoveryStatus
|
|
52
|
+
to_status: RecoveryStatus
|
|
53
|
+
reason: str
|
|
54
|
+
attempt: int = 1
|
|
55
|
+
timestamp: str = Field(default_factory=lambda: utc_now().isoformat())
|
|
56
|
+
runtime_identity: str | None = None # canonical identity key (ADR-M1-1/1-3)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class RecoveryAuditLog:
|
|
60
|
+
"""Append-only audit trail of all recovery transitions.
|
|
61
|
+
|
|
62
|
+
Backed by SQLite ``recovery_audit_log`` table. Reads are in-memory
|
|
63
|
+
for speed; writes go through the store for durability.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
def __init__(self, store: Store | None = None) -> None: # noqa: F821
|
|
67
|
+
self._store = store
|
|
68
|
+
self._transitions: list[RecoveryTransition] = []
|
|
69
|
+
if store is not None:
|
|
70
|
+
self._ensure_table()
|
|
71
|
+
self._load()
|
|
72
|
+
|
|
73
|
+
def _ensure_table(self) -> None:
|
|
74
|
+
with self._store.write() as conn: # type: ignore[union-type]
|
|
75
|
+
conn.execute("""
|
|
76
|
+
CREATE TABLE IF NOT EXISTS recovery_audit_log (
|
|
77
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
78
|
+
resource_id TEXT NOT NULL,
|
|
79
|
+
from_status TEXT NOT NULL,
|
|
80
|
+
to_status TEXT NOT NULL,
|
|
81
|
+
reason TEXT NOT NULL,
|
|
82
|
+
attempt INTEGER NOT NULL DEFAULT 1,
|
|
83
|
+
timestamp TEXT NOT NULL,
|
|
84
|
+
runtime_identity TEXT
|
|
85
|
+
)
|
|
86
|
+
""")
|
|
87
|
+
conn.execute(
|
|
88
|
+
"CREATE INDEX IF NOT EXISTS idx_ral_resource ON recovery_audit_log(resource_id)"
|
|
89
|
+
)
|
|
90
|
+
# Back-fill the identity column on tables created before it existed.
|
|
91
|
+
cols = {
|
|
92
|
+
r["name"] for r in conn.execute("PRAGMA table_info(recovery_audit_log)").fetchall()
|
|
93
|
+
}
|
|
94
|
+
if "runtime_identity" not in cols:
|
|
95
|
+
conn.execute("ALTER TABLE recovery_audit_log ADD COLUMN runtime_identity TEXT")
|
|
96
|
+
|
|
97
|
+
def _load(self) -> None:
|
|
98
|
+
with self._store.write() as conn: # type: ignore[union-type]
|
|
99
|
+
rows = conn.execute("SELECT * FROM recovery_audit_log ORDER BY id").fetchall()
|
|
100
|
+
for row in rows:
|
|
101
|
+
self._transitions.append(
|
|
102
|
+
RecoveryTransition(
|
|
103
|
+
resource_id=row["resource_id"],
|
|
104
|
+
from_status=RecoveryStatus(row["from_status"]),
|
|
105
|
+
to_status=RecoveryStatus(row["to_status"]),
|
|
106
|
+
reason=row["reason"],
|
|
107
|
+
attempt=row["attempt"],
|
|
108
|
+
timestamp=row["timestamp"],
|
|
109
|
+
runtime_identity=row.get("runtime_identity"),
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
def record(self, transition: RecoveryTransition) -> None:
|
|
114
|
+
"""Append a transition to the audit log."""
|
|
115
|
+
if self._store is not None:
|
|
116
|
+
with self._store.write() as conn:
|
|
117
|
+
conn.execute(
|
|
118
|
+
"INSERT INTO recovery_audit_log "
|
|
119
|
+
"(resource_id, from_status, to_status, reason, attempt, timestamp,"
|
|
120
|
+
" runtime_identity) "
|
|
121
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?)",
|
|
122
|
+
(
|
|
123
|
+
transition.resource_id,
|
|
124
|
+
transition.from_status.value,
|
|
125
|
+
transition.to_status.value,
|
|
126
|
+
transition.reason,
|
|
127
|
+
transition.attempt,
|
|
128
|
+
transition.timestamp,
|
|
129
|
+
transition.runtime_identity,
|
|
130
|
+
),
|
|
131
|
+
)
|
|
132
|
+
self._transitions.append(transition)
|
|
133
|
+
|
|
134
|
+
def for_resource(self, resource_id: str) -> list[RecoveryTransition]:
|
|
135
|
+
"""All transitions for a given resource, in order."""
|
|
136
|
+
return [t for t in self._transitions if t.resource_id == resource_id]
|
|
137
|
+
|
|
138
|
+
def current_status(self, resource_id: str) -> RecoveryStatus:
|
|
139
|
+
"""Derive current status from the latest transition."""
|
|
140
|
+
transitions = self.for_resource(resource_id)
|
|
141
|
+
if not transitions:
|
|
142
|
+
return RecoveryStatus.IDLE
|
|
143
|
+
return transitions[-1].to_status
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class RecoveryStateMachine:
|
|
147
|
+
"""Enforces valid recovery transitions and records them.
|
|
148
|
+
|
|
149
|
+
The state machine does not own resources — it is a pure transition
|
|
150
|
+
validator + audit logger that sits between the ResourceManager
|
|
151
|
+
and the recovery audit log.
|
|
152
|
+
"""
|
|
153
|
+
|
|
154
|
+
MAX_RETRIES = 3
|
|
155
|
+
|
|
156
|
+
def __init__(self, audit_log: RecoveryAuditLog) -> None:
|
|
157
|
+
self._audit = audit_log
|
|
158
|
+
|
|
159
|
+
def start_recovery(
|
|
160
|
+
self, resource_id: str, reason: str = "cleanup initiated"
|
|
161
|
+
) -> RecoveryTransition:
|
|
162
|
+
"""Begin recovery: IDLE → RECOVERING or DIRTY → RECOVERING."""
|
|
163
|
+
current = self._audit.current_status(resource_id)
|
|
164
|
+
target = RecoveryStatus.RECOVERING
|
|
165
|
+
if target not in _VALID_TRANSITIONS.get(current, frozenset()):
|
|
166
|
+
raise InvariantViolationError(
|
|
167
|
+
"recovery_invalid_transition",
|
|
168
|
+
f"resource '{resource_id}': cannot transition from "
|
|
169
|
+
f"'{current.value}' to '{target.value}'",
|
|
170
|
+
)
|
|
171
|
+
attempt = self._retry_count(resource_id) + 1
|
|
172
|
+
transition = RecoveryTransition(
|
|
173
|
+
resource_id=resource_id,
|
|
174
|
+
from_status=current,
|
|
175
|
+
to_status=target,
|
|
176
|
+
reason=reason,
|
|
177
|
+
attempt=attempt,
|
|
178
|
+
)
|
|
179
|
+
self._audit.record(transition)
|
|
180
|
+
return transition
|
|
181
|
+
|
|
182
|
+
def mark_verified(
|
|
183
|
+
self, resource_id: str, reason: str = "probe satisfied"
|
|
184
|
+
) -> RecoveryTransition:
|
|
185
|
+
"""Cleanup succeeded: RECOVERING → VERIFIED."""
|
|
186
|
+
return self._transition(
|
|
187
|
+
resource_id,
|
|
188
|
+
RecoveryStatus.RECOVERING,
|
|
189
|
+
RecoveryStatus.VERIFIED,
|
|
190
|
+
reason,
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
def mark_dirty(
|
|
194
|
+
self, resource_id: str, reason: str = "cleanup or verify failed"
|
|
195
|
+
) -> RecoveryTransition:
|
|
196
|
+
"""Cleanup failed: RECOVERING → DIRTY."""
|
|
197
|
+
return self._transition(
|
|
198
|
+
resource_id,
|
|
199
|
+
RecoveryStatus.RECOVERING,
|
|
200
|
+
RecoveryStatus.DIRTY,
|
|
201
|
+
reason,
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
def can_retry(self, resource_id: str) -> bool:
|
|
205
|
+
"""True if the resource has retries remaining."""
|
|
206
|
+
return self._retry_count(resource_id) < self.MAX_RETRIES
|
|
207
|
+
|
|
208
|
+
def _transition(
|
|
209
|
+
self,
|
|
210
|
+
resource_id: str,
|
|
211
|
+
from_status: RecoveryStatus,
|
|
212
|
+
to_status: RecoveryStatus,
|
|
213
|
+
reason: str,
|
|
214
|
+
) -> RecoveryTransition:
|
|
215
|
+
current = self._audit.current_status(resource_id)
|
|
216
|
+
if current != from_status:
|
|
217
|
+
raise InvariantViolationError(
|
|
218
|
+
"recovery_invalid_transition",
|
|
219
|
+
f"resource '{resource_id}': expected from '{from_status.value}', "
|
|
220
|
+
f"found '{current.value}'",
|
|
221
|
+
)
|
|
222
|
+
if to_status not in _VALID_TRANSITIONS.get(from_status, frozenset()):
|
|
223
|
+
raise InvariantViolationError(
|
|
224
|
+
"recovery_invalid_transition",
|
|
225
|
+
f"resource '{resource_id}': cannot transition from "
|
|
226
|
+
f"'{from_status.value}' to '{to_status.value}'",
|
|
227
|
+
)
|
|
228
|
+
attempt = self._retry_count(resource_id) + 1
|
|
229
|
+
transition = RecoveryTransition(
|
|
230
|
+
resource_id=resource_id,
|
|
231
|
+
from_status=from_status,
|
|
232
|
+
to_status=to_status,
|
|
233
|
+
reason=reason,
|
|
234
|
+
attempt=attempt,
|
|
235
|
+
)
|
|
236
|
+
self._audit.record(transition)
|
|
237
|
+
return transition
|
|
238
|
+
|
|
239
|
+
def _retry_count(self, resource_id: str) -> int:
|
|
240
|
+
"""How many RECOVERING attempts have been made."""
|
|
241
|
+
return sum(
|
|
242
|
+
1
|
|
243
|
+
for t in self._audit.for_resource(resource_id)
|
|
244
|
+
if t.to_status == RecoveryStatus.RECOVERING
|
|
245
|
+
)
|