mayhem-cli 0.5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. mayhem/agent/__init__.py +1 -0
  2. mayhem/agent/cli.py +36 -0
  3. mayhem/agents/__init__.py +1 -0
  4. mayhem/agents/capabilities.py +106 -0
  5. mayhem/agents/executors.py +430 -0
  6. mayhem/agents/impact.py +729 -0
  7. mayhem/agents/lease_client.py +141 -0
  8. mayhem/agents/probes.py +284 -0
  9. mayhem/agents/protocol.py +134 -0
  10. mayhem/agents/server.py +281 -0
  11. mayhem/agents/sinks.py +60 -0
  12. mayhem/agents/transports.py +134 -0
  13. mayhem/agents/watchdog.py +140 -0
  14. mayhem/cli/__init__.py +11 -0
  15. mayhem/cli/app.py +154 -0
  16. mayhem/cli/campaign.py +496 -0
  17. mayhem/cli/config_cmd.py +47 -0
  18. mayhem/cli/context.py +23 -0
  19. mayhem/cli/dependency.py +429 -0
  20. mayhem/cli/exit_codes.py +24 -0
  21. mayhem/cli/experiment.py +24 -0
  22. mayhem/cli/lifecycle.py +805 -0
  23. mayhem/cli/resolver.py +72 -0
  24. mayhem/cli/services.py +459 -0
  25. mayhem/cli/style.py +101 -0
  26. mayhem/cli/toolkit.py +41 -0
  27. mayhem/cli/topology.py +127 -0
  28. mayhem/config.py +208 -0
  29. mayhem/controller/__init__.py +1 -0
  30. mayhem/controller/compensation.py +2156 -0
  31. mayhem/controller/executor.py +1719 -0
  32. mayhem/controller/janitor.py +196 -0
  33. mayhem/controller/observability_collector.py +382 -0
  34. mayhem/controller/observations.py +102 -0
  35. mayhem/controller/planner.py +715 -0
  36. mayhem/controller/recovery.py +245 -0
  37. mayhem/controller/resilience_report.py +585 -0
  38. mayhem/controller/resource_manager.py +457 -0
  39. mayhem/controller/safety.py +392 -0
  40. mayhem/domain/__init__.py +6 -0
  41. mayhem/domain/campaigns.py +118 -0
  42. mayhem/domain/cancellation.py +110 -0
  43. mayhem/domain/candidates.py +101 -0
  44. mayhem/domain/capabilities.py +86 -0
  45. mayhem/domain/catalog.py +727 -0
  46. mayhem/domain/checks.py +173 -0
  47. mayhem/domain/common.py +104 -0
  48. mayhem/domain/coverage.py +106 -0
  49. mayhem/domain/decisions.py +57 -0
  50. mayhem/domain/errors.py +87 -0
  51. mayhem/domain/events.py +61 -0
  52. mayhem/domain/execution_context.py +120 -0
  53. mayhem/domain/execution_loci.py +94 -0
  54. mayhem/domain/experiments.py +370 -0
  55. mayhem/domain/faults.py +239 -0
  56. mayhem/domain/identity.py +200 -0
  57. mayhem/domain/k8s_adapter.py +132 -0
  58. mayhem/domain/leases.py +186 -0
  59. mayhem/domain/load_strategy.py +98 -0
  60. mayhem/domain/m5_campaign.py +120 -0
  61. mayhem/domain/maniac.py +93 -0
  62. mayhem/domain/observability.py +146 -0
  63. mayhem/domain/outcomes.py +92 -0
  64. mayhem/domain/remote_agent_interface.py +70 -0
  65. mayhem/domain/resources.py +245 -0
  66. mayhem/domain/risks.py +61 -0
  67. mayhem/domain/run_outcome.py +146 -0
  68. mayhem/domain/runtime_adapter.py +256 -0
  69. mayhem/domain/success.py +329 -0
  70. mayhem/domain/topology.py +452 -0
  71. mayhem/infra/__init__.py +1 -0
  72. mayhem/infra/campaign_engine.py +205 -0
  73. mayhem/infra/candidate_gates.py +124 -0
  74. mayhem/infra/candidate_generator.py +110 -0
  75. mayhem/infra/coverage_repository.py +101 -0
  76. mayhem/infra/lease_repository.py +129 -0
  77. mayhem/infra/maniac.py +103 -0
  78. mayhem/infra/migrations.py +596 -0
  79. mayhem/infra/migrator.py +149 -0
  80. mayhem/infra/report.py +227 -0
  81. mayhem/infra/store.py +200 -0
  82. mayhem/py.typed +0 -0
  83. mayhem/spec.py +52 -0
  84. mayhem/toolkit/__init__.py +1 -0
  85. mayhem/toolkit/fingerprint.py +69 -0
  86. mayhem/toolkit/hashing.py +32 -0
  87. mayhem/toolkit/manifests/docker.yaml +11 -0
  88. mayhem/toolkit/manifests/podman.yaml +11 -0
  89. mayhem/toolkit/manifests/stress-ng.yaml +11 -0
  90. mayhem/toolkit/manifests/tc-netem.yaml +11 -0
  91. mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
  92. mayhem/toolkit/registry.py +185 -0
  93. mayhem/toolkit/tool_runner.py +129 -0
  94. mayhem/topology/__init__.py +10 -0
  95. mayhem/topology/providers/__init__.py +0 -0
  96. mayhem/topology/providers/adapter_registry.py +60 -0
  97. mayhem/topology/providers/base.py +31 -0
  98. mayhem/topology/providers/compose.py +207 -0
  99. mayhem/topology/providers/docker_adapter.py +277 -0
  100. mayhem/topology/providers/docker_runtime.py +461 -0
  101. mayhem/topology/providers/podman_adapter.py +328 -0
  102. mayhem/topology/resolve.py +196 -0
  103. mayhem/topology/service.py +158 -0
  104. mayhem_cli-0.5.1.dist-info/METADATA +555 -0
  105. mayhem_cli-0.5.1.dist-info/RECORD +107 -0
  106. mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
  107. mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,245 @@
1
+ """Recovery state machine and audit trail (ADR-0016).
2
+
3
+ Every resource recovery is a state machine:
4
+ IDLE → RECOVERING → VERIFIED | DIRTY
5
+ ↑ retry ↑
6
+ DIRTY → RECOVERING → VERIFIED | DIRTY (max retries)
7
+
8
+ The state machine is the *only* way to transition recovery states —
9
+ all transitions are validated, timestamped, and persisted to the
10
+ recovery_audit_log table. This gives us:
11
+ 1. Idempotent recovery — safe to retry from crash
12
+ 2. Exhaustive audit trail — every transition recorded
13
+ 3. Retry exhaustion detection — stops after max retries
14
+ 4. Ownership-aware — only owner can transition
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from enum import StrEnum
20
+
21
+ from pydantic import BaseModel, ConfigDict, Field
22
+
23
+ from mayhem.domain.common import utc_now
24
+ from mayhem.domain.errors import InvariantViolationError
25
+
26
+
27
+ class RecoveryStatus(StrEnum):
28
+ """States in the recovery lifecycle."""
29
+
30
+ IDLE = "idle" # resource is active, not recovering
31
+ RECOVERING = "recovering" # cleanup in progress
32
+ VERIFIED = "verified" # cleanup succeeded and verified
33
+ DIRTY = "dirty" # cleanup failed or verify failed
34
+
35
+
36
+ # Valid transitions: source → set of targets
37
+ _VALID_TRANSITIONS: dict[RecoveryStatus, frozenset[RecoveryStatus]] = {
38
+ RecoveryStatus.IDLE: frozenset({RecoveryStatus.RECOVERING}),
39
+ RecoveryStatus.RECOVERING: frozenset({RecoveryStatus.VERIFIED, RecoveryStatus.DIRTY}),
40
+ RecoveryStatus.DIRTY: frozenset({RecoveryStatus.RECOVERING}),
41
+ RecoveryStatus.VERIFIED: frozenset(), # terminal state — no transitions out
42
+ }
43
+
44
+
45
+ class RecoveryTransition(BaseModel):
46
+ """One atomic state transition in the recovery lifecycle."""
47
+
48
+ model_config = ConfigDict(frozen=True)
49
+
50
+ resource_id: str
51
+ from_status: RecoveryStatus
52
+ to_status: RecoveryStatus
53
+ reason: str
54
+ attempt: int = 1
55
+ timestamp: str = Field(default_factory=lambda: utc_now().isoformat())
56
+ runtime_identity: str | None = None # canonical identity key (ADR-M1-1/1-3)
57
+
58
+
59
+ class RecoveryAuditLog:
60
+ """Append-only audit trail of all recovery transitions.
61
+
62
+ Backed by SQLite ``recovery_audit_log`` table. Reads are in-memory
63
+ for speed; writes go through the store for durability.
64
+ """
65
+
66
+ def __init__(self, store: Store | None = None) -> None: # noqa: F821
67
+ self._store = store
68
+ self._transitions: list[RecoveryTransition] = []
69
+ if store is not None:
70
+ self._ensure_table()
71
+ self._load()
72
+
73
+ def _ensure_table(self) -> None:
74
+ with self._store.write() as conn: # type: ignore[union-type]
75
+ conn.execute("""
76
+ CREATE TABLE IF NOT EXISTS recovery_audit_log (
77
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
78
+ resource_id TEXT NOT NULL,
79
+ from_status TEXT NOT NULL,
80
+ to_status TEXT NOT NULL,
81
+ reason TEXT NOT NULL,
82
+ attempt INTEGER NOT NULL DEFAULT 1,
83
+ timestamp TEXT NOT NULL,
84
+ runtime_identity TEXT
85
+ )
86
+ """)
87
+ conn.execute(
88
+ "CREATE INDEX IF NOT EXISTS idx_ral_resource ON recovery_audit_log(resource_id)"
89
+ )
90
+ # Back-fill the identity column on tables created before it existed.
91
+ cols = {
92
+ r["name"] for r in conn.execute("PRAGMA table_info(recovery_audit_log)").fetchall()
93
+ }
94
+ if "runtime_identity" not in cols:
95
+ conn.execute("ALTER TABLE recovery_audit_log ADD COLUMN runtime_identity TEXT")
96
+
97
+ def _load(self) -> None:
98
+ with self._store.write() as conn: # type: ignore[union-type]
99
+ rows = conn.execute("SELECT * FROM recovery_audit_log ORDER BY id").fetchall()
100
+ for row in rows:
101
+ self._transitions.append(
102
+ RecoveryTransition(
103
+ resource_id=row["resource_id"],
104
+ from_status=RecoveryStatus(row["from_status"]),
105
+ to_status=RecoveryStatus(row["to_status"]),
106
+ reason=row["reason"],
107
+ attempt=row["attempt"],
108
+ timestamp=row["timestamp"],
109
+ runtime_identity=row.get("runtime_identity"),
110
+ )
111
+ )
112
+
113
+ def record(self, transition: RecoveryTransition) -> None:
114
+ """Append a transition to the audit log."""
115
+ if self._store is not None:
116
+ with self._store.write() as conn:
117
+ conn.execute(
118
+ "INSERT INTO recovery_audit_log "
119
+ "(resource_id, from_status, to_status, reason, attempt, timestamp,"
120
+ " runtime_identity) "
121
+ "VALUES (?, ?, ?, ?, ?, ?, ?)",
122
+ (
123
+ transition.resource_id,
124
+ transition.from_status.value,
125
+ transition.to_status.value,
126
+ transition.reason,
127
+ transition.attempt,
128
+ transition.timestamp,
129
+ transition.runtime_identity,
130
+ ),
131
+ )
132
+ self._transitions.append(transition)
133
+
134
+ def for_resource(self, resource_id: str) -> list[RecoveryTransition]:
135
+ """All transitions for a given resource, in order."""
136
+ return [t for t in self._transitions if t.resource_id == resource_id]
137
+
138
+ def current_status(self, resource_id: str) -> RecoveryStatus:
139
+ """Derive current status from the latest transition."""
140
+ transitions = self.for_resource(resource_id)
141
+ if not transitions:
142
+ return RecoveryStatus.IDLE
143
+ return transitions[-1].to_status
144
+
145
+
146
+ class RecoveryStateMachine:
147
+ """Enforces valid recovery transitions and records them.
148
+
149
+ The state machine does not own resources — it is a pure transition
150
+ validator + audit logger that sits between the ResourceManager
151
+ and the recovery audit log.
152
+ """
153
+
154
+ MAX_RETRIES = 3
155
+
156
+ def __init__(self, audit_log: RecoveryAuditLog) -> None:
157
+ self._audit = audit_log
158
+
159
+ def start_recovery(
160
+ self, resource_id: str, reason: str = "cleanup initiated"
161
+ ) -> RecoveryTransition:
162
+ """Begin recovery: IDLE → RECOVERING or DIRTY → RECOVERING."""
163
+ current = self._audit.current_status(resource_id)
164
+ target = RecoveryStatus.RECOVERING
165
+ if target not in _VALID_TRANSITIONS.get(current, frozenset()):
166
+ raise InvariantViolationError(
167
+ "recovery_invalid_transition",
168
+ f"resource '{resource_id}': cannot transition from "
169
+ f"'{current.value}' to '{target.value}'",
170
+ )
171
+ attempt = self._retry_count(resource_id) + 1
172
+ transition = RecoveryTransition(
173
+ resource_id=resource_id,
174
+ from_status=current,
175
+ to_status=target,
176
+ reason=reason,
177
+ attempt=attempt,
178
+ )
179
+ self._audit.record(transition)
180
+ return transition
181
+
182
+ def mark_verified(
183
+ self, resource_id: str, reason: str = "probe satisfied"
184
+ ) -> RecoveryTransition:
185
+ """Cleanup succeeded: RECOVERING → VERIFIED."""
186
+ return self._transition(
187
+ resource_id,
188
+ RecoveryStatus.RECOVERING,
189
+ RecoveryStatus.VERIFIED,
190
+ reason,
191
+ )
192
+
193
+ def mark_dirty(
194
+ self, resource_id: str, reason: str = "cleanup or verify failed"
195
+ ) -> RecoveryTransition:
196
+ """Cleanup failed: RECOVERING → DIRTY."""
197
+ return self._transition(
198
+ resource_id,
199
+ RecoveryStatus.RECOVERING,
200
+ RecoveryStatus.DIRTY,
201
+ reason,
202
+ )
203
+
204
+ def can_retry(self, resource_id: str) -> bool:
205
+ """True if the resource has retries remaining."""
206
+ return self._retry_count(resource_id) < self.MAX_RETRIES
207
+
208
+ def _transition(
209
+ self,
210
+ resource_id: str,
211
+ from_status: RecoveryStatus,
212
+ to_status: RecoveryStatus,
213
+ reason: str,
214
+ ) -> RecoveryTransition:
215
+ current = self._audit.current_status(resource_id)
216
+ if current != from_status:
217
+ raise InvariantViolationError(
218
+ "recovery_invalid_transition",
219
+ f"resource '{resource_id}': expected from '{from_status.value}', "
220
+ f"found '{current.value}'",
221
+ )
222
+ if to_status not in _VALID_TRANSITIONS.get(from_status, frozenset()):
223
+ raise InvariantViolationError(
224
+ "recovery_invalid_transition",
225
+ f"resource '{resource_id}': cannot transition from "
226
+ f"'{from_status.value}' to '{to_status.value}'",
227
+ )
228
+ attempt = self._retry_count(resource_id) + 1
229
+ transition = RecoveryTransition(
230
+ resource_id=resource_id,
231
+ from_status=from_status,
232
+ to_status=to_status,
233
+ reason=reason,
234
+ attempt=attempt,
235
+ )
236
+ self._audit.record(transition)
237
+ return transition
238
+
239
+ def _retry_count(self, resource_id: str) -> int:
240
+ """How many RECOVERING attempts have been made."""
241
+ return sum(
242
+ 1
243
+ for t in self._audit.for_resource(resource_id)
244
+ if t.to_status == RecoveryStatus.RECOVERING
245
+ )