mayhem-cli 0.5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. mayhem/agent/__init__.py +1 -0
  2. mayhem/agent/cli.py +36 -0
  3. mayhem/agents/__init__.py +1 -0
  4. mayhem/agents/capabilities.py +106 -0
  5. mayhem/agents/executors.py +430 -0
  6. mayhem/agents/impact.py +729 -0
  7. mayhem/agents/lease_client.py +141 -0
  8. mayhem/agents/probes.py +284 -0
  9. mayhem/agents/protocol.py +134 -0
  10. mayhem/agents/server.py +281 -0
  11. mayhem/agents/sinks.py +60 -0
  12. mayhem/agents/transports.py +134 -0
  13. mayhem/agents/watchdog.py +140 -0
  14. mayhem/cli/__init__.py +11 -0
  15. mayhem/cli/app.py +154 -0
  16. mayhem/cli/campaign.py +496 -0
  17. mayhem/cli/config_cmd.py +47 -0
  18. mayhem/cli/context.py +23 -0
  19. mayhem/cli/dependency.py +429 -0
  20. mayhem/cli/exit_codes.py +24 -0
  21. mayhem/cli/experiment.py +24 -0
  22. mayhem/cli/lifecycle.py +805 -0
  23. mayhem/cli/resolver.py +72 -0
  24. mayhem/cli/services.py +459 -0
  25. mayhem/cli/style.py +101 -0
  26. mayhem/cli/toolkit.py +41 -0
  27. mayhem/cli/topology.py +127 -0
  28. mayhem/config.py +208 -0
  29. mayhem/controller/__init__.py +1 -0
  30. mayhem/controller/compensation.py +2156 -0
  31. mayhem/controller/executor.py +1719 -0
  32. mayhem/controller/janitor.py +196 -0
  33. mayhem/controller/observability_collector.py +382 -0
  34. mayhem/controller/observations.py +102 -0
  35. mayhem/controller/planner.py +715 -0
  36. mayhem/controller/recovery.py +245 -0
  37. mayhem/controller/resilience_report.py +585 -0
  38. mayhem/controller/resource_manager.py +457 -0
  39. mayhem/controller/safety.py +392 -0
  40. mayhem/domain/__init__.py +6 -0
  41. mayhem/domain/campaigns.py +118 -0
  42. mayhem/domain/cancellation.py +110 -0
  43. mayhem/domain/candidates.py +101 -0
  44. mayhem/domain/capabilities.py +86 -0
  45. mayhem/domain/catalog.py +727 -0
  46. mayhem/domain/checks.py +173 -0
  47. mayhem/domain/common.py +104 -0
  48. mayhem/domain/coverage.py +106 -0
  49. mayhem/domain/decisions.py +57 -0
  50. mayhem/domain/errors.py +87 -0
  51. mayhem/domain/events.py +61 -0
  52. mayhem/domain/execution_context.py +120 -0
  53. mayhem/domain/execution_loci.py +94 -0
  54. mayhem/domain/experiments.py +370 -0
  55. mayhem/domain/faults.py +239 -0
  56. mayhem/domain/identity.py +200 -0
  57. mayhem/domain/k8s_adapter.py +132 -0
  58. mayhem/domain/leases.py +186 -0
  59. mayhem/domain/load_strategy.py +98 -0
  60. mayhem/domain/m5_campaign.py +120 -0
  61. mayhem/domain/maniac.py +93 -0
  62. mayhem/domain/observability.py +146 -0
  63. mayhem/domain/outcomes.py +92 -0
  64. mayhem/domain/remote_agent_interface.py +70 -0
  65. mayhem/domain/resources.py +245 -0
  66. mayhem/domain/risks.py +61 -0
  67. mayhem/domain/run_outcome.py +146 -0
  68. mayhem/domain/runtime_adapter.py +256 -0
  69. mayhem/domain/success.py +329 -0
  70. mayhem/domain/topology.py +452 -0
  71. mayhem/infra/__init__.py +1 -0
  72. mayhem/infra/campaign_engine.py +205 -0
  73. mayhem/infra/candidate_gates.py +124 -0
  74. mayhem/infra/candidate_generator.py +110 -0
  75. mayhem/infra/coverage_repository.py +101 -0
  76. mayhem/infra/lease_repository.py +129 -0
  77. mayhem/infra/maniac.py +103 -0
  78. mayhem/infra/migrations.py +596 -0
  79. mayhem/infra/migrator.py +149 -0
  80. mayhem/infra/report.py +227 -0
  81. mayhem/infra/store.py +200 -0
  82. mayhem/py.typed +0 -0
  83. mayhem/spec.py +52 -0
  84. mayhem/toolkit/__init__.py +1 -0
  85. mayhem/toolkit/fingerprint.py +69 -0
  86. mayhem/toolkit/hashing.py +32 -0
  87. mayhem/toolkit/manifests/docker.yaml +11 -0
  88. mayhem/toolkit/manifests/podman.yaml +11 -0
  89. mayhem/toolkit/manifests/stress-ng.yaml +11 -0
  90. mayhem/toolkit/manifests/tc-netem.yaml +11 -0
  91. mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
  92. mayhem/toolkit/registry.py +185 -0
  93. mayhem/toolkit/tool_runner.py +129 -0
  94. mayhem/topology/__init__.py +10 -0
  95. mayhem/topology/providers/__init__.py +0 -0
  96. mayhem/topology/providers/adapter_registry.py +60 -0
  97. mayhem/topology/providers/base.py +31 -0
  98. mayhem/topology/providers/compose.py +207 -0
  99. mayhem/topology/providers/docker_adapter.py +277 -0
  100. mayhem/topology/providers/docker_runtime.py +461 -0
  101. mayhem/topology/providers/podman_adapter.py +328 -0
  102. mayhem/topology/resolve.py +196 -0
  103. mayhem/topology/service.py +158 -0
  104. mayhem_cli-0.5.1.dist-info/METADATA +555 -0
  105. mayhem_cli-0.5.1.dist-info/RECORD +107 -0
  106. mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
  107. mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,124 @@
1
+ """Candidate gate pipeline (ADR-M5-2, M5 Phase 5.3).
2
+
3
+ Routes an ``ExperimentCandidate`` through the three gates:
4
+
5
+ Safety → Feasibility → Resource-conflict
6
+
7
+ A candidate that fails a gate is rejected with that gate and a reason; it is
8
+ never executed. Each gate is an injectable checker so the pipeline is fully
9
+ deterministic and unit-testable without a live runtime, while the concrete
10
+ checkers mirror the M2 (resource-conflict), M3 (capability verdicts), and
11
+ safety machinery already present in the codebase.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from typing import Protocol
17
+
18
+ from mayhem.domain.candidates import (
19
+ CandidateDecision,
20
+ CandidateGate,
21
+ ExperimentCandidate,
22
+ accept,
23
+ reject,
24
+ )
25
+
26
+
27
+ class GateChecker(Protocol):
28
+ """A single gate: return ``None`` to pass, or a reason string to reject."""
29
+
30
+ def check(self, candidate: ExperimentCandidate) -> str | None:
31
+ """Return ``None`` if the candidate passes, else the rejection reason."""
32
+ ...
33
+
34
+
35
+ class SafetyGate:
36
+ """Gate 1 — safety (ADR-M5 / controller safety checks)."""
37
+
38
+ def __init__(self, *, forbidden_faults: tuple[str, ...] = ()) -> None:
39
+ # forbidden faults are e.g. host reboot / node kill not allowlisted.
40
+ self._forbidden_faults = set(forbidden_faults)
41
+
42
+ def check(self, candidate: ExperimentCandidate) -> str | None:
43
+ for fault in candidate.fault_kinds:
44
+ if fault in self._forbidden_faults:
45
+ return f"fault {fault!r} is forbidden by safety policy"
46
+ if not candidate.target:
47
+ return "candidate has no target"
48
+ return None
49
+
50
+
51
+ class FeasibilityGate:
52
+ """Gate 2 — feasibility (M3 runtime-capability verdicts).
53
+
54
+ ``S``/``A``/``U`` mark each fault as Supported / Alternative / Unsupported.
55
+ Any ``U`` (unsupported) fault is not feasible.
56
+ """
57
+
58
+ def __init__(self, supported: tuple[str, ...], unsupported: tuple[str, ...] = ()) -> None:
59
+ self._supported = set(supported)
60
+ self._unsupported = set(unsupported)
61
+
62
+ def check(self, candidate: ExperimentCandidate) -> str | None:
63
+ for fault in candidate.fault_kinds:
64
+ if fault in self._unsupported:
65
+ return f"fault {fault!r} is UNSUPPORTED by the runtime (not feasible)"
66
+ if fault not in self._supported:
67
+ return f"fault {fault!r} is not in the supported capability set"
68
+ return None
69
+
70
+
71
+ class ResourceConflictGate:
72
+ """Gate 3 — resource conflict (M2).
73
+
74
+ ``busy_targets`` are targets that already hold an in-flight mutation, so a
75
+ new experiment against them would collide.
76
+ """
77
+
78
+ def __init__(self, busy_targets: tuple[str, ...] = ()) -> None:
79
+ self._busy = set(busy_targets)
80
+
81
+ def check(self, candidate: ExperimentCandidate) -> str | None:
82
+ if candidate.target in self._busy:
83
+ return f"target {candidate.target!r} has an in-flight resource conflict"
84
+ return None
85
+
86
+
87
+ class CandidateGatePipeline:
88
+ """Composes the three gates; returns the first rejection, else acceptance."""
89
+
90
+ def __init__(
91
+ self,
92
+ *,
93
+ safety: GateChecker | None = None,
94
+ feasibility: GateChecker | None = None,
95
+ resource_conflict: GateChecker | None = None,
96
+ ) -> None:
97
+ self._safety = safety if safety is not None else SafetyGate()
98
+ self._feasibility = (
99
+ feasibility if feasibility is not None else FeasibilityGate(supported=())
100
+ )
101
+ self._resource_conflict = (
102
+ resource_conflict if resource_conflict is not None else ResourceConflictGate()
103
+ )
104
+
105
+ def gate(self, candidate: ExperimentCandidate) -> CandidateDecision:
106
+ """Route a candidate through the three gates in fixed order."""
107
+ # 1. Safety
108
+ reason = self._safety.check(candidate)
109
+ if reason is not None:
110
+ return reject(candidate, CandidateGate.SAFETY, reason)
111
+ # 2. Feasibility
112
+ reason = self._feasibility.check(candidate)
113
+ if reason is not None:
114
+ return reject(candidate, CandidateGate.FEASIBILITY, reason)
115
+ # 3. Resource-conflict
116
+ reason = self._resource_conflict.check(candidate)
117
+ if reason is not None:
118
+ return reject(candidate, CandidateGate.RESOURCE_CONFLICT, reason)
119
+ return accept(candidate)
120
+
121
+ def gate_many(
122
+ self, candidates: tuple[ExperimentCandidate, ...]
123
+ ) -> tuple[CandidateDecision, ...]:
124
+ return tuple(self.gate(c) for c in candidates)
@@ -0,0 +1,110 @@
1
+ """Seeded, bounded candidate generator (ADR-M5-2, M5 Phase 5.3).
2
+
3
+ Produces ``ExperimentCandidate`` proposals over the
4
+ (target, fault_kind, execution_context, parameter_band) landscape. Generation
5
+ is deterministic for a fixed seed, and honors a per-band risk ceiling so no
6
+ candidate above the campaign's risk tolerance is ever proposed.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import random
12
+ from dataclasses import dataclass
13
+ from typing import Any
14
+
15
+ from mayhem.domain.candidates import ExperimentCandidate
16
+ from mayhem.domain.risks import RiskLevel
17
+
18
+ # Risk of a fault_kind, keyed by fault id prefix (mirrors safety projection).
19
+ _DEFAULT_FAULT_RISK: dict[str, RiskLevel] = {
20
+ "net": RiskLevel.MEDIUM,
21
+ "cpu": RiskLevel.MEDIUM,
22
+ "mem": RiskLevel.MEDIUM,
23
+ "fs": RiskLevel.HIGH,
24
+ "disk": RiskLevel.HIGH,
25
+ "storage": RiskLevel.HIGH,
26
+ "process": RiskLevel.LOW,
27
+ "container": RiskLevel.LOW,
28
+ "node": RiskLevel.CRITICAL,
29
+ "http_api": RiskLevel.LOW,
30
+ "database": RiskLevel.HIGH,
31
+ "load": RiskLevel.MEDIUM,
32
+ "fuzz": RiskLevel.HIGH,
33
+ "dns": RiskLevel.LOW,
34
+ "tls": RiskLevel.LOW,
35
+ "clock": RiskLevel.MEDIUM,
36
+ "fd": RiskLevel.MEDIUM,
37
+ }
38
+
39
+
40
+ def _risk_of_fault(fault_kind: str) -> RiskLevel:
41
+ prefix = fault_kind.split(".", 1)[0]
42
+ return _DEFAULT_FAULT_RISK.get(prefix, RiskLevel.MEDIUM)
43
+
44
+
45
+ @dataclass
46
+ class CandidateLandscape:
47
+ """The space the generator draws from, plus the allowed risk ceiling."""
48
+
49
+ targets: tuple[str, ...] = ()
50
+ fault_kinds: tuple[str, ...] = ()
51
+ execution_contexts: tuple[str, ...] = ("container",)
52
+ parameter_bands: tuple[str, ...] = ("default",)
53
+ risk_ceiling: RiskLevel = RiskLevel.HIGH
54
+
55
+
56
+ @dataclass
57
+ class SeededCandidateGenerator:
58
+ """Deterministic generator; the same seed yields the same candidates."""
59
+
60
+ landscape: CandidateLandscape
61
+ seed: int = 0
62
+ rng: random.Random | None = None
63
+
64
+ def __post_init__(self) -> None:
65
+ if self.rng is None:
66
+ self.rng = random.Random(self.seed)
67
+
68
+ def _in_ceiling(self, risk: RiskLevel) -> bool:
69
+ return _risk_rank(risk) <= _risk_rank(self.landscape.risk_ceiling)
70
+
71
+ def generate(self, *, limit: int | None = None) -> tuple[ExperimentCandidate, ...]:
72
+ """Yield candidates until the landscape is exhausted or ``limit`` reached.
73
+
74
+ Candidates above the risk ceiling are skipped, never emitted.
75
+ """
76
+ out: list[ExperimentCandidate] = []
77
+ for target in self.landscape.targets:
78
+ for fault_kind in self.landscape.fault_kinds:
79
+ risk = _risk_of_fault(fault_kind)
80
+ if not self._in_ceiling(risk):
81
+ continue
82
+ for ctx in self.landscape.execution_contexts:
83
+ for band in self.landscape.parameter_bands:
84
+ params: dict[str, Any] = {"band": band}
85
+ candidate = ExperimentCandidate(
86
+ target=target,
87
+ fault_kinds=(fault_kind,),
88
+ params=params,
89
+ execution_context=ctx,
90
+ expected_effect=f"{fault_kind} on {target} in {band}",
91
+ risk_band=risk,
92
+ seed_hint=self.seed,
93
+ )
94
+ out.append(candidate)
95
+ if limit is not None and len(out) >= limit:
96
+ return tuple(out)
97
+ # Deterministic shuffle so selection order isn't trivially the
98
+ # landscape enumeration order — still fully reproducible from seed.
99
+ rng = self.rng if self.rng is not None else random.Random(self.seed)
100
+ rng.shuffle(out)
101
+ return tuple(out)
102
+
103
+
104
+ def _risk_rank(level: RiskLevel) -> int:
105
+ return {
106
+ RiskLevel.LOW: 0,
107
+ RiskLevel.MEDIUM: 1,
108
+ RiskLevel.HIGH: 2,
109
+ RiskLevel.CRITICAL: 3,
110
+ }[level]
@@ -0,0 +1,101 @@
1
+ """SQLite-backed coverage accounting (ADR-M5-3, M5 Phase 5.2).
2
+
3
+ A recorded ``Outcome`` marks the run's coverage cell as *seen*. Marking the
4
+ same cell again (a re-run) is idempotent — ``INSERT OR IGNORE`` on the primary
5
+ key means re-running a cell never double-counts. Maniac can then enumerate the
6
+ ``UNKNOWN`` cells of a landscape (cells not yet covered) to prioritize coverage
7
+ gain.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ from typing import TYPE_CHECKING
14
+
15
+ from mayhem.domain.coverage import (
16
+ CoverageCell,
17
+ CoverageRecord,
18
+ CoverageSummary,
19
+ )
20
+
21
+ if TYPE_CHECKING:
22
+ from collections.abc import Iterable
23
+
24
+ from mayhem.infra.store import Store
25
+
26
+
27
+ class SQLiteCoverageRepository:
28
+ """Persists seen coverage cells over the ``m5_coverage`` table."""
29
+
30
+ def __init__(self, store: Store) -> None:
31
+ self._store = store
32
+
33
+ def mark_seen(
34
+ self,
35
+ cell: CoverageCell,
36
+ run_id: str,
37
+ *,
38
+ extra: dict[str, object] | None = None,
39
+ ) -> None:
40
+ """Record that ``cell`` was observed by ``run_id``.
41
+
42
+ Idempotent per cell: re-running the same cell does not double-count;
43
+ ``CREATE TABLE ... PRIMARY KEY`` + ``INSERT OR IGNORE`` guarantees it.
44
+ """
45
+ with self._store.write() as conn:
46
+ conn.execute(
47
+ """
48
+ INSERT OR IGNORE INTO m5_coverage (
49
+ cell_key, target, fault_kind, execution_context,
50
+ parameter_band, run_id, covered, extra_json
51
+ ) VALUES (?, ?, ?, ?, ?, ?, 1, ?)
52
+ """,
53
+ (
54
+ cell.key,
55
+ cell.target,
56
+ cell.fault_kind,
57
+ cell.execution_context,
58
+ cell.parameter_band,
59
+ run_id,
60
+ json.dumps(extra or {}),
61
+ ),
62
+ )
63
+
64
+ def is_covered(self, cell: CoverageCell) -> bool:
65
+ rows = self._store.query(
66
+ "SELECT 1 FROM m5_coverage WHERE cell_key = ? AND covered = 1",
67
+ (cell.key,),
68
+ )
69
+ return bool(rows)
70
+
71
+ def covered_keys(self) -> frozenset[str]:
72
+ rows = self._store.query("SELECT cell_key FROM m5_coverage WHERE covered = 1")
73
+ return frozenset(r["cell_key"] for r in rows)
74
+
75
+ def covered_records(self) -> tuple[CoverageRecord, ...]:
76
+ rows = self._store.query(
77
+ "SELECT target, fault_kind, execution_context, parameter_band, run_id, extra_json "
78
+ "FROM m5_coverage WHERE covered = 1"
79
+ )
80
+ records: list[CoverageRecord] = []
81
+ for r in rows:
82
+ cell = CoverageCell(
83
+ target=r["target"],
84
+ fault_kind=r["fault_kind"],
85
+ execution_context=r["execution_context"],
86
+ parameter_band=r["parameter_band"],
87
+ )
88
+ records.append(
89
+ CoverageRecord(cell=cell, run_id=r["run_id"], extra=json.loads(r["extra_json"]))
90
+ )
91
+ return tuple(records)
92
+
93
+ def summary(self, landscape: Iterable[CoverageCell]) -> CoverageSummary:
94
+ """Aggregate covered vs UNKNOWN cells over ``landscape``."""
95
+ landscape_tuple = tuple(landscape)
96
+ covered = self.covered_keys()
97
+ return CoverageSummary(covered_keys=covered, total_cells=len(landscape_tuple))
98
+
99
+ def unknown_cells(self, landscape: Iterable[CoverageCell]) -> tuple[CoverageCell, ...]:
100
+ """Cells in ``landscape`` not yet covered, in landscape order."""
101
+ return self.summary(landscape).unknown_cells(tuple(landscape))
@@ -0,0 +1,129 @@
1
+ """SQLite-backed LeaseSink — the durable half of the lease protocol (ADR-0007).
2
+
3
+ The controller is the single writer; agents receive this sink through the
4
+ SDK boundary and never open the database themselves. Rows are self-contained:
5
+ run/fault/targets context is denormalized onto the lease row (migration 002)
6
+ so janitor and watchdog sweeps never need joins.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ from datetime import UTC, datetime, timedelta
13
+ from typing import TYPE_CHECKING
14
+
15
+ from mayhem.domain.leases import FaultLease, LeaseState
16
+
17
+ if TYPE_CHECKING:
18
+ from collections.abc import Mapping
19
+
20
+ from mayhem.infra.store import Store
21
+
22
+ _TERMINAL_STATES = ("released", "expired")
23
+
24
+
25
+ class SQLiteLeaseSink:
26
+ """Implements ``agents.sinks.LeaseSink`` over the fault_leases table."""
27
+
28
+ def __init__(self, store: Store) -> None:
29
+ self._store = store
30
+
31
+ def save(self, lease: FaultLease) -> None:
32
+ created = lease.created_at
33
+ expires_at = created + timedelta(seconds=float(lease.ttl_seconds))
34
+ with self._store.write() as conn:
35
+ conn.execute(
36
+ """
37
+ INSERT OR REPLACE INTO fault_leases (
38
+ id, state, owner_agent, undo_json, verify_json,
39
+ ttl_seconds, expires_at, injected_at, released_at,
40
+ release_mechanism, escalation_notes,
41
+ run_id, fault_id, targets_json, created_epoch_s,
42
+ runtime_identity
43
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
44
+ """,
45
+ (
46
+ lease.id,
47
+ lease.state.value,
48
+ lease.owner_agent,
49
+ json.dumps([op.model_dump(mode="json") for op in lease.undo_ops]),
50
+ json.dumps([probe.model_dump(mode="json") for probe in lease.verify_probes]),
51
+ float(lease.ttl_seconds),
52
+ expires_at.isoformat(),
53
+ _iso_or_none(lease.injected_at),
54
+ _iso_or_none(lease.released_at),
55
+ lease.release_mechanism,
56
+ lease.escalation_notes,
57
+ lease.run_id,
58
+ lease.fault_id,
59
+ json.dumps(sorted(lease.targets)),
60
+ created.timestamp(),
61
+ lease.runtime_identity,
62
+ ),
63
+ )
64
+
65
+ def load(self, lease_id: str) -> FaultLease | None:
66
+ rows = self._store.query("SELECT * FROM fault_leases WHERE id = ?", (lease_id,))
67
+ return _row_to_lease(dict(rows[0])) if rows else None
68
+
69
+ def active_leases(self) -> tuple[FaultLease, ...]:
70
+ placeholders = ", ".join("?" for _ in _TERMINAL_STATES)
71
+ rows = self._store.query(
72
+ f"SELECT * FROM fault_leases WHERE state NOT IN ({placeholders})"
73
+ " ORDER BY created_epoch_s",
74
+ _TERMINAL_STATES,
75
+ )
76
+ return tuple(_row_to_lease(dict(row)) for row in rows)
77
+
78
+ def expired_pending(self, now_epoch_s: float) -> tuple[FaultLease, ...]:
79
+ rows = self._store.query(
80
+ "SELECT * FROM fault_leases WHERE created_epoch_s + ttl_seconds < ? "
81
+ "AND state IN ('pending', 'active')",
82
+ (now_epoch_s,),
83
+ )
84
+ return tuple(_row_to_lease(dict(row)) for row in rows)
85
+
86
+ def next_sequence(self) -> int:
87
+ # Max trailing integer of every `l-<n>` id ever persisted. Seed the next
88
+ # run's counter here so ids never collide in a shared DB.
89
+ rows = self._store.query(
90
+ "SELECT MAX(CAST(substr(id, 3) AS INTEGER)) AS seq FROM fault_leases"
91
+ " WHERE id LIKE 'l-%'"
92
+ )
93
+ if not rows:
94
+ return 0
95
+ seq = rows[0]["seq"]
96
+ return int(seq) if seq is not None else 0
97
+
98
+
99
+ def _row_to_lease(row: Mapping[str, object]) -> FaultLease:
100
+ return FaultLease.model_validate(
101
+ {
102
+ "id": str(row["id"]),
103
+ "state": LeaseState(str(row["state"])),
104
+ "owner_agent": str(row["owner_agent"]),
105
+ "undo_ops": json.loads(str(row["undo_json"])),
106
+ "verify_probes": json.loads(str(row["verify_json"])),
107
+ "ttl_seconds": float(str(row["ttl_seconds"])),
108
+ "injected_at": _dt_or_none(row.get("injected_at")),
109
+ "released_at": _dt_or_none(row.get("released_at")),
110
+ "release_mechanism": row["release_mechanism"],
111
+ "escalation_notes": row["escalation_notes"],
112
+ "run_id": str(row["run_id"] or ""),
113
+ "fault_id": str(row["fault_id"] or ""),
114
+ "targets": frozenset(json.loads(str(row["targets_json"]))),
115
+ "created_at": datetime.fromtimestamp(float(str(row["created_epoch_s"])), tz=UTC),
116
+ "runtime_identity": row.get("runtime_identity"),
117
+ }
118
+ )
119
+
120
+
121
+ def _iso_or_none(moment: datetime | None) -> str | None:
122
+ return moment.isoformat() if moment is not None else None
123
+
124
+
125
+ def _dt_or_none(raw: object) -> datetime | None:
126
+ if raw is None:
127
+ return None
128
+ parsed = datetime.fromisoformat(str(raw))
129
+ return parsed if parsed.tzinfo else parsed.replace(tzinfo=UTC)
mayhem/infra/maniac.py ADDED
@@ -0,0 +1,103 @@
1
+ """Maniac selection/optimization engine (ADR-M5-3, M5 Phase 5.5).
2
+
3
+ Maniac chooses *which experiment to run next* from the candidate set, grounded
4
+ in recorded outcome coverage and the campaign's bounds. It is:
5
+
6
+ - **stateless** across campaign runs (no in-memory carry-over; every selection
7
+ is re-derived from its explicit inputs),
8
+ - **deterministic and reproducible**: identical inputs (candidates, covered
9
+ cells, gates, bounds, seed) always yield the identical selection,
10
+ - **knowledge-transferable**: recorded coverage from prior campaigns is passed
11
+ in explicitly, so learnings carry across campaign boundaries.
12
+
13
+ Selection is a greedy coverage-maximizing walk: gate every candidate, prefer
14
+ cells not yet covered, and stop once the budget or coverage target is met.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import random
20
+ from dataclasses import dataclass
21
+ from typing import TYPE_CHECKING
22
+
23
+ from mayhem.domain.coverage import CoverageCell
24
+ from mayhem.infra.candidate_gates import CandidateGatePipeline
25
+
26
+ if TYPE_CHECKING:
27
+ from mayhem.domain.candidates import CandidateDecision, ExperimentCandidate
28
+
29
+
30
+ def coverage_cell_for_candidate(candidate: ExperimentCandidate) -> CoverageCell:
31
+ """The coverage cell a candidate explores, derived deterministically."""
32
+ band = str(candidate.params.get("band") or "default") if candidate.params else "default"
33
+ return CoverageCell(
34
+ candidate.target,
35
+ candidate.primary_fault,
36
+ candidate.execution_context,
37
+ band,
38
+ )
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class SelectionInputs:
43
+ """Every input Maniac is allowed to look at (no hidden state)."""
44
+
45
+ candidates: tuple[ExperimentCandidate, ...]
46
+ covered_keys: frozenset[str] = frozenset()
47
+ gates: CandidateGatePipeline | None = None
48
+ max_runs: int = 100
49
+ coverage_target: int = 1
50
+ seed: int = 0
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class SelectionResult:
55
+ selected: tuple[ExperimentCandidate, ...]
56
+ rejected: tuple[CandidateDecision, ...] = ()
57
+ predicted_new_coverage: int = 0
58
+
59
+ @property
60
+ def selected_count(self) -> int:
61
+ return len(self.selected)
62
+
63
+
64
+ def select_next(inputs: SelectionInputs) -> SelectionResult:
65
+ """Pure, deterministic greedy selection (stateless, reproducible)."""
66
+ gates = inputs.gates if inputs.gates is not None else CandidateGatePipeline()
67
+ covered = set(inputs.covered_keys)
68
+
69
+ accepted: list[ExperimentCandidate] = []
70
+ rejected: list[CandidateDecision] = []
71
+ for cand in inputs.candidates:
72
+ decision = gates.gate(cand)
73
+ if decision.accepted:
74
+ accepted.append(cand)
75
+ else:
76
+ rejected.append(decision)
77
+
78
+ # Deterministic ordering: seeded random tie-break over the candidates,
79
+ # then rank so that already-covered cells come last (coverage first).
80
+ rng = random.Random(inputs.seed)
81
+ ordered = sorted(accepted, key=lambda c: rng.random())
82
+ uncovered_first = sorted(
83
+ ordered,
84
+ key=lambda c: coverage_cell_for_candidate(c).key in covered,
85
+ )
86
+
87
+ selected: list[ExperimentCandidate] = []
88
+ covered_now = set(covered)
89
+ for cand in uncovered_first:
90
+ if len(selected) >= inputs.max_runs:
91
+ break
92
+ if len(covered_now) >= inputs.coverage_target:
93
+ break
94
+ cell = coverage_cell_for_candidate(cand)
95
+ covered_now.add(cell.key)
96
+ selected.append(cand)
97
+
98
+ projected = len(covered_now) - len(covered)
99
+ return SelectionResult(
100
+ selected=tuple(selected),
101
+ rejected=tuple(rejected),
102
+ predicted_new_coverage=projected,
103
+ )