mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,457 @@
|
|
|
1
|
+
"""Resource manager — persistence + ownership-aware recovery (ADR-0015).
|
|
2
|
+
|
|
3
|
+
Bridges the in-memory ``ResourceOwnershipGraph`` with the SQLite store so
|
|
4
|
+
resource tracking survives controller restarts. Every mutation is write-ahead
|
|
5
|
+
persisted before the resource graph is updated.
|
|
6
|
+
|
|
7
|
+
The manager is instantiated once per controller lifecycle and shared by the
|
|
8
|
+
``RunEngine`` and ``Janitor``.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import uuid
|
|
15
|
+
from typing import TYPE_CHECKING
|
|
16
|
+
|
|
17
|
+
from mayhem.domain.common import utc_now
|
|
18
|
+
from mayhem.domain.errors import InvariantViolationError
|
|
19
|
+
from mayhem.domain.leases import UndoOp, VerifyProbe
|
|
20
|
+
from mayhem.domain.resources import (
|
|
21
|
+
MutationJournalEntry,
|
|
22
|
+
RecoveryResult,
|
|
23
|
+
ResourceOwnershipGraph,
|
|
24
|
+
ResourceState,
|
|
25
|
+
ResourceType,
|
|
26
|
+
TrackedResource,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
import sqlite3
|
|
31
|
+
|
|
32
|
+
from mayhem.infra.store import Store
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class ResourceManager:
|
|
36
|
+
"""Tracks every resource created by fault injection.
|
|
37
|
+
|
|
38
|
+
Lifecycle:
|
|
39
|
+
1. ``register()`` — write-ahead persist + add to graph
|
|
40
|
+
2. ``activate()`` — mark as ACTIVE after successful injection
|
|
41
|
+
3. ``recover_owned()`` — ownership-aware cleanup for a run
|
|
42
|
+
4. ``recover_orphans()`` — janitor picks up stale resources
|
|
43
|
+
5. ``verify_recovery()`` — post-cleanup probe confirms removal
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
def __init__(self, store: Store) -> None:
|
|
47
|
+
self._store = store
|
|
48
|
+
self._graph = ResourceOwnershipGraph()
|
|
49
|
+
self._ensure_table()
|
|
50
|
+
self._load_graph()
|
|
51
|
+
|
|
52
|
+
# -- schema ---------------------------------------------------------------
|
|
53
|
+
|
|
54
|
+
def _ensure_table(self) -> None:
|
|
55
|
+
with self._store.write() as conn:
|
|
56
|
+
conn.execute("""
|
|
57
|
+
CREATE TABLE IF NOT EXISTS tracked_resources (
|
|
58
|
+
id TEXT PRIMARY KEY,
|
|
59
|
+
resource_type TEXT NOT NULL,
|
|
60
|
+
owner_run_id TEXT NOT NULL,
|
|
61
|
+
owner_step_id TEXT NOT NULL,
|
|
62
|
+
owner_fault_id TEXT NOT NULL,
|
|
63
|
+
state TEXT NOT NULL DEFAULT 'pending',
|
|
64
|
+
target_identity TEXT NOT NULL,
|
|
65
|
+
cleanup_op_json TEXT NOT NULL,
|
|
66
|
+
verify_probe_json TEXT NOT NULL,
|
|
67
|
+
created_at TEXT NOT NULL,
|
|
68
|
+
recovered_at TEXT,
|
|
69
|
+
metadata_json TEXT NOT NULL DEFAULT '{}',
|
|
70
|
+
fingerprint TEXT NOT NULL DEFAULT ''
|
|
71
|
+
)
|
|
72
|
+
""")
|
|
73
|
+
# Defensive column add for DBs created before ADR-M3-7: a pre-existing
|
|
74
|
+
# tracked_resources table won't be re-CREATEd, so ensure the column
|
|
75
|
+
# exists without dropping any data.
|
|
76
|
+
cols = {row[1] for row in conn.execute("PRAGMA table_info(tracked_resources)")}
|
|
77
|
+
if "fingerprint" not in cols:
|
|
78
|
+
conn.execute(
|
|
79
|
+
"ALTER TABLE tracked_resources ADD COLUMN fingerprint TEXT NOT NULL DEFAULT ''"
|
|
80
|
+
)
|
|
81
|
+
conn.execute("CREATE INDEX IF NOT EXISTS idx_tr_run ON tracked_resources(owner_run_id)")
|
|
82
|
+
conn.execute(
|
|
83
|
+
"CREATE INDEX IF NOT EXISTS idx_tr_target ON tracked_resources(target_identity)"
|
|
84
|
+
)
|
|
85
|
+
conn.execute("CREATE INDEX IF NOT EXISTS idx_tr_state ON tracked_resources(state)")
|
|
86
|
+
# Mutation-boundary journal (ADR-M2 Phase 2.6/2.7)
|
|
87
|
+
conn.execute("""
|
|
88
|
+
CREATE TABLE IF NOT EXISTS mutation_journal (
|
|
89
|
+
id TEXT PRIMARY KEY,
|
|
90
|
+
lease_id TEXT NOT NULL,
|
|
91
|
+
resource_id TEXT NOT NULL,
|
|
92
|
+
run_id TEXT NOT NULL,
|
|
93
|
+
step_id TEXT NOT NULL,
|
|
94
|
+
fault_id TEXT NOT NULL,
|
|
95
|
+
defining_op_json TEXT NOT NULL,
|
|
96
|
+
target_identity TEXT NOT NULL,
|
|
97
|
+
journaled_at TEXT NOT NULL
|
|
98
|
+
)
|
|
99
|
+
""")
|
|
100
|
+
conn.execute(
|
|
101
|
+
"CREATE INDEX IF NOT EXISTS idx_mj_target ON mutation_journal(target_identity)"
|
|
102
|
+
)
|
|
103
|
+
conn.execute("CREATE INDEX IF NOT EXISTS idx_mj_lease ON mutation_journal(lease_id)")
|
|
104
|
+
|
|
105
|
+
def _load_graph(self) -> None:
|
|
106
|
+
"""Reload the in-memory graph from persistent storage."""
|
|
107
|
+
with self._store.write() as conn:
|
|
108
|
+
rows = conn.execute(
|
|
109
|
+
"SELECT * FROM tracked_resources WHERE state IN ('pending', 'active', 'recovering')"
|
|
110
|
+
).fetchall()
|
|
111
|
+
for row in rows:
|
|
112
|
+
resource = self._row_to_resource(row)
|
|
113
|
+
self._graph.add(resource)
|
|
114
|
+
|
|
115
|
+
# -- public API -----------------------------------------------------------
|
|
116
|
+
|
|
117
|
+
def register(
|
|
118
|
+
self,
|
|
119
|
+
resource_type: ResourceType,
|
|
120
|
+
run_id: str,
|
|
121
|
+
step_id: str,
|
|
122
|
+
fault_id: str,
|
|
123
|
+
target_identity: str,
|
|
124
|
+
cleanup_op: UndoOp,
|
|
125
|
+
verify_probe: VerifyProbe,
|
|
126
|
+
metadata: dict[str, object] | None = None,
|
|
127
|
+
) -> TrackedResource:
|
|
128
|
+
"""Declare and persist a new resource before injection.
|
|
129
|
+
|
|
130
|
+
Raises InvariantViolationError if there is a serializable conflict
|
|
131
|
+
with an existing resource owned by a *different* run.
|
|
132
|
+
"""
|
|
133
|
+
resource = TrackedResource(
|
|
134
|
+
id=str(uuid.uuid4()),
|
|
135
|
+
resource_type=resource_type,
|
|
136
|
+
owner_run_id=run_id,
|
|
137
|
+
owner_step_id=step_id,
|
|
138
|
+
owner_fault_id=fault_id,
|
|
139
|
+
target_identity=target_identity,
|
|
140
|
+
cleanup_op=cleanup_op,
|
|
141
|
+
verify_probe=verify_probe,
|
|
142
|
+
metadata=metadata or {},
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
# Conflict detection — raise before persisting
|
|
146
|
+
conflicts = self._graph.detect_conflicts(resource)
|
|
147
|
+
serializable = [c for c in conflicts if c.kind.value in ("serialize", "reject")]
|
|
148
|
+
if serializable:
|
|
149
|
+
raise InvariantViolationError(
|
|
150
|
+
"resource_conflict",
|
|
151
|
+
f"resource '{resource_type.value}' on target "
|
|
152
|
+
f"'{target_identity}' conflicts with run "
|
|
153
|
+
f"'{serializable[0].existing_resource.owner_run_id}': "
|
|
154
|
+
f"{serializable[0].reason}",
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
self._persist(resource)
|
|
158
|
+
self._graph.add(resource)
|
|
159
|
+
return resource
|
|
160
|
+
|
|
161
|
+
def activate(self, resource_id: str) -> TrackedResource:
|
|
162
|
+
"""Mark a resource as ACTIVE after successful injection."""
|
|
163
|
+
return self._transition(resource_id, ResourceState.ACTIVE)
|
|
164
|
+
|
|
165
|
+
# -- mutation journal (ADR-M2 Phase 2.6/2.7) --------------------------------
|
|
166
|
+
|
|
167
|
+
def journal_mutation(
|
|
168
|
+
self,
|
|
169
|
+
lease_id: str,
|
|
170
|
+
resource_id: str,
|
|
171
|
+
run_id: str,
|
|
172
|
+
step_id: str,
|
|
173
|
+
fault_id: str,
|
|
174
|
+
defining_op: UndoOp,
|
|
175
|
+
target_identity: str,
|
|
176
|
+
) -> MutationJournalEntry:
|
|
177
|
+
"""Record a mutation at the boundary when the last undo-fallible op is
|
|
178
|
+
applied (Phase 2.6). Carries the resource owner, the defining op, and
|
|
179
|
+
the lease reference.
|
|
180
|
+
|
|
181
|
+
Must be called AFTER the inject succeeds — never before.
|
|
182
|
+
|
|
183
|
+
The resource transitions from PENDING to ACTIVE at this point (not
|
|
184
|
+
during ``register()``). A resource that was registered but never
|
|
185
|
+
journaled is orphaned during cleanup.
|
|
186
|
+
"""
|
|
187
|
+
entry = MutationJournalEntry(
|
|
188
|
+
id=f"mj-{uuid.uuid4().hex[:12]}",
|
|
189
|
+
lease_id=lease_id,
|
|
190
|
+
resource_id=resource_id,
|
|
191
|
+
run_id=run_id,
|
|
192
|
+
step_id=step_id,
|
|
193
|
+
fault_id=fault_id,
|
|
194
|
+
defining_op=defining_op,
|
|
195
|
+
target_identity=target_identity,
|
|
196
|
+
)
|
|
197
|
+
with self._store.write() as conn:
|
|
198
|
+
conn.execute(
|
|
199
|
+
"INSERT INTO mutation_journal "
|
|
200
|
+
"(id, lease_id, resource_id, run_id, step_id, fault_id, "
|
|
201
|
+
" defining_op_json, target_identity, journaled_at) "
|
|
202
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
203
|
+
(
|
|
204
|
+
entry.id,
|
|
205
|
+
entry.lease_id,
|
|
206
|
+
entry.resource_id,
|
|
207
|
+
entry.run_id,
|
|
208
|
+
entry.step_id,
|
|
209
|
+
entry.fault_id,
|
|
210
|
+
entry.defining_op.model_dump_json(),
|
|
211
|
+
entry.target_identity,
|
|
212
|
+
entry.journaled_at.isoformat(),
|
|
213
|
+
),
|
|
214
|
+
)
|
|
215
|
+
self.activate(resource_id)
|
|
216
|
+
return entry
|
|
217
|
+
|
|
218
|
+
def check_no_inflight_writer(
|
|
219
|
+
self,
|
|
220
|
+
target_identity: str,
|
|
221
|
+
exclude_run_id: str | None = None,
|
|
222
|
+
) -> str | None:
|
|
223
|
+
"""Phase 2.7 — one-inflight-writer check.
|
|
224
|
+
|
|
225
|
+
Returns ``None`` when no conflict is found. When another lease holds
|
|
226
|
+
an active mutation on *target_identity*, returns the human-readable
|
|
227
|
+
conflict reason (caller reports RESOURCE_CONFLICT and aborts).
|
|
228
|
+
"""
|
|
229
|
+
with self._store.write() as conn:
|
|
230
|
+
rows = conn.execute(
|
|
231
|
+
"SELECT mj.lease_id, mj.run_id, mj.fault_id, tr.state "
|
|
232
|
+
"FROM mutation_journal mj "
|
|
233
|
+
"JOIN tracked_resources tr ON tr.id = mj.resource_id "
|
|
234
|
+
"WHERE mj.target_identity = ? "
|
|
235
|
+
"AND tr.state IN ('pending', 'active')",
|
|
236
|
+
(target_identity,),
|
|
237
|
+
).fetchall()
|
|
238
|
+
for row in rows:
|
|
239
|
+
if exclude_run_id and row["run_id"] == exclude_run_id:
|
|
240
|
+
continue
|
|
241
|
+
lease_id = row["lease_id"]
|
|
242
|
+
resource_state = row["state"]
|
|
243
|
+
return (
|
|
244
|
+
f"RESOURCE_CONFLICT: target '{target_identity}' is held by "
|
|
245
|
+
f"lease '{lease_id}' (run {row['run_id']}, fault "
|
|
246
|
+
f"{row['fault_id']}, resource state {resource_state})"
|
|
247
|
+
)
|
|
248
|
+
return None
|
|
249
|
+
|
|
250
|
+
def in_flight_mutations_for_target(self, target_identity: str) -> list[MutationJournalEntry]:
|
|
251
|
+
"""Return all journaled mutations on a target whose resource is still
|
|
252
|
+
held (PENDING or ACTIVE state). Used for diagnostics and conflict
|
|
253
|
+
reporting.
|
|
254
|
+
"""
|
|
255
|
+
with self._store.write() as conn:
|
|
256
|
+
rows = conn.execute(
|
|
257
|
+
"SELECT mj.* FROM mutation_journal mj "
|
|
258
|
+
"JOIN tracked_resources tr ON tr.id = mj.resource_id "
|
|
259
|
+
"WHERE mj.target_identity = ? "
|
|
260
|
+
"AND tr.state IN ('pending', 'active')",
|
|
261
|
+
(target_identity,),
|
|
262
|
+
).fetchall()
|
|
263
|
+
entries: list[MutationJournalEntry] = []
|
|
264
|
+
for row in rows:
|
|
265
|
+
entry = MutationJournalEntry(
|
|
266
|
+
id=row["id"],
|
|
267
|
+
lease_id=row["lease_id"],
|
|
268
|
+
resource_id=row["resource_id"],
|
|
269
|
+
run_id=row["run_id"],
|
|
270
|
+
step_id=row["step_id"],
|
|
271
|
+
fault_id=row["fault_id"],
|
|
272
|
+
defining_op=UndoOp.model_validate_json(row["defining_op_json"]),
|
|
273
|
+
target_identity=row["target_identity"],
|
|
274
|
+
journaled_at=utc_now(), # approx; the actual value is in the DB
|
|
275
|
+
)
|
|
276
|
+
entries.append(entry)
|
|
277
|
+
return entries
|
|
278
|
+
|
|
279
|
+
def start_recovery(self, resource_id: str) -> TrackedResource:
|
|
280
|
+
"""Mark a resource as RECOVERING before cleanup starts."""
|
|
281
|
+
return self._transition(resource_id, ResourceState.RECOVERING)
|
|
282
|
+
|
|
283
|
+
def mark_recovered(self, resource_id: str, verified: bool) -> TrackedResource:
|
|
284
|
+
"""Mark a resource as RECOVERED or DIRTY based on verification."""
|
|
285
|
+
target = ResourceState.RECOVERED if verified else ResourceState.DIRTY
|
|
286
|
+
resource = self._transition(resource_id, target)
|
|
287
|
+
# Persist recovery timestamp
|
|
288
|
+
now = utc_now().isoformat()
|
|
289
|
+
with self._store.write() as conn:
|
|
290
|
+
conn.execute(
|
|
291
|
+
"UPDATE tracked_resources SET recovered_at = ? WHERE id = ?",
|
|
292
|
+
(now, resource_id),
|
|
293
|
+
)
|
|
294
|
+
return resource
|
|
295
|
+
|
|
296
|
+
def recover_owned(self, run_id: str) -> list[RecoveryResult]:
|
|
297
|
+
"""Recover all resources owned by a run, in dependency order.
|
|
298
|
+
|
|
299
|
+
Ownership-aware: skips resources that are also owned by another
|
|
300
|
+
active run (should not happen, but defends against bugs).
|
|
301
|
+
"""
|
|
302
|
+
resources = self._graph.active_for_run(run_id)
|
|
303
|
+
results: list[RecoveryResult] = []
|
|
304
|
+
for resource in sorted(resources, key=_recovery_order):
|
|
305
|
+
result = self._recover_single(resource)
|
|
306
|
+
results.append(result)
|
|
307
|
+
return results
|
|
308
|
+
|
|
309
|
+
def recover_orphans(self) -> list[RecoveryResult]:
|
|
310
|
+
"""Recover resources with no active owner (post-crash cleanup)."""
|
|
311
|
+
orphans = self._graph.orphans()
|
|
312
|
+
results: list[RecoveryResult] = []
|
|
313
|
+
for resource in orphans:
|
|
314
|
+
# Mark as orphaned first
|
|
315
|
+
self._transition(resource.id, ResourceState.ORPHANED)
|
|
316
|
+
result = self._recover_single(resource, is_orphan=True)
|
|
317
|
+
results.append(result)
|
|
318
|
+
return results
|
|
319
|
+
|
|
320
|
+
def verify_recovery(self, resource_id: str) -> bool:
|
|
321
|
+
"""Run the resource's verify probe to confirm cleanup."""
|
|
322
|
+
resource = self._graph.get(resource_id)
|
|
323
|
+
if resource is None:
|
|
324
|
+
return True # already cleaned
|
|
325
|
+
if resource.state != ResourceState.RECOVERED:
|
|
326
|
+
return False
|
|
327
|
+
# The actual probe execution happens through the agent runtime.
|
|
328
|
+
# Here we just check the recorded state — the caller (RunEngine)
|
|
329
|
+
# invokes the verify probe and calls mark_recovered.
|
|
330
|
+
return True
|
|
331
|
+
|
|
332
|
+
def list_resources(
|
|
333
|
+
self, run_id: str | None = None, state: ResourceState | None = None
|
|
334
|
+
) -> list[TrackedResource]:
|
|
335
|
+
"""List tracked resources with optional filters."""
|
|
336
|
+
if run_id:
|
|
337
|
+
# When filtering by non-active state, we need all resources for the run
|
|
338
|
+
if state and state not in (ResourceState.ACTIVE, ResourceState.PENDING):
|
|
339
|
+
resources = [r for r in self._graph._resources.values() if r.owner_run_id == run_id]
|
|
340
|
+
else:
|
|
341
|
+
resources = self._graph.active_for_run(run_id)
|
|
342
|
+
else:
|
|
343
|
+
resources = list(self._graph._resources.values())
|
|
344
|
+
if state:
|
|
345
|
+
resources = [r for r in resources if r.state == state]
|
|
346
|
+
return resources
|
|
347
|
+
|
|
348
|
+
@property
|
|
349
|
+
def graph(self) -> ResourceOwnershipGraph:
|
|
350
|
+
return self._graph
|
|
351
|
+
|
|
352
|
+
# -- internals ------------------------------------------------------------
|
|
353
|
+
|
|
354
|
+
def _transition(self, resource_id: str, target: ResourceState) -> TrackedResource:
|
|
355
|
+
resource = self._graph.get(resource_id)
|
|
356
|
+
if resource is None:
|
|
357
|
+
raise InvariantViolationError(
|
|
358
|
+
"resource_unknown", f"resource '{resource_id}' not tracked"
|
|
359
|
+
)
|
|
360
|
+
# Build new resource with updated state
|
|
361
|
+
updated = resource.model_copy(update={"state": target}) # type: ignore[call-arg]
|
|
362
|
+
self._graph.remove(resource_id)
|
|
363
|
+
self._graph.add(updated)
|
|
364
|
+
with self._store.write() as conn:
|
|
365
|
+
conn.execute(
|
|
366
|
+
"UPDATE tracked_resources SET state = ? WHERE id = ?",
|
|
367
|
+
(target.value, resource_id),
|
|
368
|
+
)
|
|
369
|
+
return updated
|
|
370
|
+
|
|
371
|
+
def _persist(self, resource: TrackedResource) -> None:
|
|
372
|
+
with self._store.write() as conn:
|
|
373
|
+
conn.execute(
|
|
374
|
+
"INSERT INTO tracked_resources "
|
|
375
|
+
"(id, resource_type, owner_run_id, owner_step_id, owner_fault_id, "
|
|
376
|
+
" state, target_identity, cleanup_op_json, verify_probe_json, "
|
|
377
|
+
" created_at, metadata_json, fingerprint) "
|
|
378
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
379
|
+
(
|
|
380
|
+
resource.id,
|
|
381
|
+
resource.resource_type.value,
|
|
382
|
+
resource.owner_run_id,
|
|
383
|
+
resource.owner_step_id,
|
|
384
|
+
resource.owner_fault_id,
|
|
385
|
+
resource.state.value,
|
|
386
|
+
resource.target_identity,
|
|
387
|
+
resource.cleanup_op.model_dump_json(),
|
|
388
|
+
resource.verify_probe.model_dump_json(),
|
|
389
|
+
resource.created_at.isoformat(),
|
|
390
|
+
json.dumps(resource.metadata),
|
|
391
|
+
resource.fingerprint,
|
|
392
|
+
),
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
def _recover_single(
|
|
396
|
+
self, resource: TrackedResource, *, is_orphan: bool = False
|
|
397
|
+
) -> RecoveryResult:
|
|
398
|
+
"""Execute cleanup for one resource.
|
|
399
|
+
|
|
400
|
+
In a full implementation this sends the cleanup_op to the agent.
|
|
401
|
+
For durability, the resource is transitioned to RECOVERING before
|
|
402
|
+
cleanup and DIRTY/RECOVERED after.
|
|
403
|
+
"""
|
|
404
|
+
try:
|
|
405
|
+
self.start_recovery(resource.id)
|
|
406
|
+
# The actual cleanup execution happens in RunEngine via the agent
|
|
407
|
+
# runtime. The resource manager just tracks state here.
|
|
408
|
+
# For orphan recovery the Janitor dispatches to agents.
|
|
409
|
+
return RecoveryResult(
|
|
410
|
+
resource_id=resource.id,
|
|
411
|
+
success=True,
|
|
412
|
+
verified=False, # verification happens after agent executes cleanup
|
|
413
|
+
)
|
|
414
|
+
except Exception as exc:
|
|
415
|
+
return RecoveryResult(
|
|
416
|
+
resource_id=resource.id,
|
|
417
|
+
success=False,
|
|
418
|
+
error=str(exc),
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
@staticmethod
|
|
422
|
+
def _row_to_resource(row: sqlite3.Row) -> TrackedResource:
|
|
423
|
+
return TrackedResource(
|
|
424
|
+
id=row["id"],
|
|
425
|
+
resource_type=ResourceType(row["resource_type"]),
|
|
426
|
+
owner_run_id=row["owner_run_id"],
|
|
427
|
+
owner_step_id=row["owner_step_id"],
|
|
428
|
+
owner_fault_id=row["owner_fault_id"],
|
|
429
|
+
state=ResourceState(row["state"]),
|
|
430
|
+
target_identity=row["target_identity"],
|
|
431
|
+
cleanup_op=UndoOp.model_validate_json(row["cleanup_op_json"]),
|
|
432
|
+
verify_probe=VerifyProbe.model_validate_json(row["verify_probe_json"]),
|
|
433
|
+
created_at=utc_now(), # row["created_at"] is ISO string; deserialize if needed
|
|
434
|
+
recovered_at=None,
|
|
435
|
+
metadata=json.loads(row["metadata_json"]),
|
|
436
|
+
)
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def _recovery_order(resource: TrackedResource) -> int:
|
|
440
|
+
"""Dependency order for recovery — network rules before process signals."""
|
|
441
|
+
_ORDER = {
|
|
442
|
+
ResourceType.TC_RULE: 0,
|
|
443
|
+
ResourceType.IPTABLES_RULE: 1,
|
|
444
|
+
ResourceType.NFTABLES_RULE: 2,
|
|
445
|
+
ResourceType.TOXIPROXY_TOXIC: 3,
|
|
446
|
+
ResourceType.LOAD_GENERATOR: 4,
|
|
447
|
+
ResourceType.CONTAINER_STATE: 5,
|
|
448
|
+
ResourceType.CGROUP_LIMIT: 6,
|
|
449
|
+
ResourceType.RESOURCE_LIMIT: 7,
|
|
450
|
+
ResourceType.PROCESS_SIGNAL: 8,
|
|
451
|
+
ResourceType.PROCESS_SPAWN: 9,
|
|
452
|
+
ResourceType.NETWORK_NAMESPACE: 10,
|
|
453
|
+
ResourceType.FILESYSTEM_MOUNT: 11,
|
|
454
|
+
ResourceType.TEMPORARY_FILE: 12,
|
|
455
|
+
ResourceType.GENERIC: 99,
|
|
456
|
+
}
|
|
457
|
+
return _ORDER.get(resource.resource_type, 99)
|