mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Candidate gate pipeline (ADR-M5-2, M5 Phase 5.3).
|
|
2
|
+
|
|
3
|
+
Routes an ``ExperimentCandidate`` through the three gates:
|
|
4
|
+
|
|
5
|
+
Safety → Feasibility → Resource-conflict
|
|
6
|
+
|
|
7
|
+
A candidate that fails a gate is rejected with that gate and a reason; it is
|
|
8
|
+
never executed. Each gate is an injectable checker so the pipeline is fully
|
|
9
|
+
deterministic and unit-testable without a live runtime, while the concrete
|
|
10
|
+
checkers mirror the M2 (resource-conflict), M3 (capability verdicts), and
|
|
11
|
+
safety machinery already present in the codebase.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from typing import Protocol
|
|
17
|
+
|
|
18
|
+
from mayhem.domain.candidates import (
|
|
19
|
+
CandidateDecision,
|
|
20
|
+
CandidateGate,
|
|
21
|
+
ExperimentCandidate,
|
|
22
|
+
accept,
|
|
23
|
+
reject,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class GateChecker(Protocol):
|
|
28
|
+
"""A single gate: return ``None`` to pass, or a reason string to reject."""
|
|
29
|
+
|
|
30
|
+
def check(self, candidate: ExperimentCandidate) -> str | None:
|
|
31
|
+
"""Return ``None`` if the candidate passes, else the rejection reason."""
|
|
32
|
+
...
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class SafetyGate:
|
|
36
|
+
"""Gate 1 — safety (ADR-M5 / controller safety checks)."""
|
|
37
|
+
|
|
38
|
+
def __init__(self, *, forbidden_faults: tuple[str, ...] = ()) -> None:
|
|
39
|
+
# forbidden faults are e.g. host reboot / node kill not allowlisted.
|
|
40
|
+
self._forbidden_faults = set(forbidden_faults)
|
|
41
|
+
|
|
42
|
+
def check(self, candidate: ExperimentCandidate) -> str | None:
|
|
43
|
+
for fault in candidate.fault_kinds:
|
|
44
|
+
if fault in self._forbidden_faults:
|
|
45
|
+
return f"fault {fault!r} is forbidden by safety policy"
|
|
46
|
+
if not candidate.target:
|
|
47
|
+
return "candidate has no target"
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class FeasibilityGate:
|
|
52
|
+
"""Gate 2 — feasibility (M3 runtime-capability verdicts).
|
|
53
|
+
|
|
54
|
+
``S``/``A``/``U`` mark each fault as Supported / Alternative / Unsupported.
|
|
55
|
+
Any ``U`` (unsupported) fault is not feasible.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def __init__(self, supported: tuple[str, ...], unsupported: tuple[str, ...] = ()) -> None:
|
|
59
|
+
self._supported = set(supported)
|
|
60
|
+
self._unsupported = set(unsupported)
|
|
61
|
+
|
|
62
|
+
def check(self, candidate: ExperimentCandidate) -> str | None:
|
|
63
|
+
for fault in candidate.fault_kinds:
|
|
64
|
+
if fault in self._unsupported:
|
|
65
|
+
return f"fault {fault!r} is UNSUPPORTED by the runtime (not feasible)"
|
|
66
|
+
if fault not in self._supported:
|
|
67
|
+
return f"fault {fault!r} is not in the supported capability set"
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class ResourceConflictGate:
|
|
72
|
+
"""Gate 3 — resource conflict (M2).
|
|
73
|
+
|
|
74
|
+
``busy_targets`` are targets that already hold an in-flight mutation, so a
|
|
75
|
+
new experiment against them would collide.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
def __init__(self, busy_targets: tuple[str, ...] = ()) -> None:
|
|
79
|
+
self._busy = set(busy_targets)
|
|
80
|
+
|
|
81
|
+
def check(self, candidate: ExperimentCandidate) -> str | None:
|
|
82
|
+
if candidate.target in self._busy:
|
|
83
|
+
return f"target {candidate.target!r} has an in-flight resource conflict"
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class CandidateGatePipeline:
|
|
88
|
+
"""Composes the three gates; returns the first rejection, else acceptance."""
|
|
89
|
+
|
|
90
|
+
def __init__(
|
|
91
|
+
self,
|
|
92
|
+
*,
|
|
93
|
+
safety: GateChecker | None = None,
|
|
94
|
+
feasibility: GateChecker | None = None,
|
|
95
|
+
resource_conflict: GateChecker | None = None,
|
|
96
|
+
) -> None:
|
|
97
|
+
self._safety = safety if safety is not None else SafetyGate()
|
|
98
|
+
self._feasibility = (
|
|
99
|
+
feasibility if feasibility is not None else FeasibilityGate(supported=())
|
|
100
|
+
)
|
|
101
|
+
self._resource_conflict = (
|
|
102
|
+
resource_conflict if resource_conflict is not None else ResourceConflictGate()
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
def gate(self, candidate: ExperimentCandidate) -> CandidateDecision:
|
|
106
|
+
"""Route a candidate through the three gates in fixed order."""
|
|
107
|
+
# 1. Safety
|
|
108
|
+
reason = self._safety.check(candidate)
|
|
109
|
+
if reason is not None:
|
|
110
|
+
return reject(candidate, CandidateGate.SAFETY, reason)
|
|
111
|
+
# 2. Feasibility
|
|
112
|
+
reason = self._feasibility.check(candidate)
|
|
113
|
+
if reason is not None:
|
|
114
|
+
return reject(candidate, CandidateGate.FEASIBILITY, reason)
|
|
115
|
+
# 3. Resource-conflict
|
|
116
|
+
reason = self._resource_conflict.check(candidate)
|
|
117
|
+
if reason is not None:
|
|
118
|
+
return reject(candidate, CandidateGate.RESOURCE_CONFLICT, reason)
|
|
119
|
+
return accept(candidate)
|
|
120
|
+
|
|
121
|
+
def gate_many(
|
|
122
|
+
self, candidates: tuple[ExperimentCandidate, ...]
|
|
123
|
+
) -> tuple[CandidateDecision, ...]:
|
|
124
|
+
return tuple(self.gate(c) for c in candidates)
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Seeded, bounded candidate generator (ADR-M5-2, M5 Phase 5.3).
|
|
2
|
+
|
|
3
|
+
Produces ``ExperimentCandidate`` proposals over the
|
|
4
|
+
(target, fault_kind, execution_context, parameter_band) landscape. Generation
|
|
5
|
+
is deterministic for a fixed seed, and honors a per-band risk ceiling so no
|
|
6
|
+
candidate above the campaign's risk tolerance is ever proposed.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import random
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from mayhem.domain.candidates import ExperimentCandidate
|
|
16
|
+
from mayhem.domain.risks import RiskLevel
|
|
17
|
+
|
|
18
|
+
# Risk of a fault_kind, keyed by fault id prefix (mirrors safety projection).
|
|
19
|
+
_DEFAULT_FAULT_RISK: dict[str, RiskLevel] = {
|
|
20
|
+
"net": RiskLevel.MEDIUM,
|
|
21
|
+
"cpu": RiskLevel.MEDIUM,
|
|
22
|
+
"mem": RiskLevel.MEDIUM,
|
|
23
|
+
"fs": RiskLevel.HIGH,
|
|
24
|
+
"disk": RiskLevel.HIGH,
|
|
25
|
+
"storage": RiskLevel.HIGH,
|
|
26
|
+
"process": RiskLevel.LOW,
|
|
27
|
+
"container": RiskLevel.LOW,
|
|
28
|
+
"node": RiskLevel.CRITICAL,
|
|
29
|
+
"http_api": RiskLevel.LOW,
|
|
30
|
+
"database": RiskLevel.HIGH,
|
|
31
|
+
"load": RiskLevel.MEDIUM,
|
|
32
|
+
"fuzz": RiskLevel.HIGH,
|
|
33
|
+
"dns": RiskLevel.LOW,
|
|
34
|
+
"tls": RiskLevel.LOW,
|
|
35
|
+
"clock": RiskLevel.MEDIUM,
|
|
36
|
+
"fd": RiskLevel.MEDIUM,
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _risk_of_fault(fault_kind: str) -> RiskLevel:
|
|
41
|
+
prefix = fault_kind.split(".", 1)[0]
|
|
42
|
+
return _DEFAULT_FAULT_RISK.get(prefix, RiskLevel.MEDIUM)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class CandidateLandscape:
|
|
47
|
+
"""The space the generator draws from, plus the allowed risk ceiling."""
|
|
48
|
+
|
|
49
|
+
targets: tuple[str, ...] = ()
|
|
50
|
+
fault_kinds: tuple[str, ...] = ()
|
|
51
|
+
execution_contexts: tuple[str, ...] = ("container",)
|
|
52
|
+
parameter_bands: tuple[str, ...] = ("default",)
|
|
53
|
+
risk_ceiling: RiskLevel = RiskLevel.HIGH
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass
|
|
57
|
+
class SeededCandidateGenerator:
|
|
58
|
+
"""Deterministic generator; the same seed yields the same candidates."""
|
|
59
|
+
|
|
60
|
+
landscape: CandidateLandscape
|
|
61
|
+
seed: int = 0
|
|
62
|
+
rng: random.Random | None = None
|
|
63
|
+
|
|
64
|
+
def __post_init__(self) -> None:
|
|
65
|
+
if self.rng is None:
|
|
66
|
+
self.rng = random.Random(self.seed)
|
|
67
|
+
|
|
68
|
+
def _in_ceiling(self, risk: RiskLevel) -> bool:
|
|
69
|
+
return _risk_rank(risk) <= _risk_rank(self.landscape.risk_ceiling)
|
|
70
|
+
|
|
71
|
+
def generate(self, *, limit: int | None = None) -> tuple[ExperimentCandidate, ...]:
|
|
72
|
+
"""Yield candidates until the landscape is exhausted or ``limit`` reached.
|
|
73
|
+
|
|
74
|
+
Candidates above the risk ceiling are skipped, never emitted.
|
|
75
|
+
"""
|
|
76
|
+
out: list[ExperimentCandidate] = []
|
|
77
|
+
for target in self.landscape.targets:
|
|
78
|
+
for fault_kind in self.landscape.fault_kinds:
|
|
79
|
+
risk = _risk_of_fault(fault_kind)
|
|
80
|
+
if not self._in_ceiling(risk):
|
|
81
|
+
continue
|
|
82
|
+
for ctx in self.landscape.execution_contexts:
|
|
83
|
+
for band in self.landscape.parameter_bands:
|
|
84
|
+
params: dict[str, Any] = {"band": band}
|
|
85
|
+
candidate = ExperimentCandidate(
|
|
86
|
+
target=target,
|
|
87
|
+
fault_kinds=(fault_kind,),
|
|
88
|
+
params=params,
|
|
89
|
+
execution_context=ctx,
|
|
90
|
+
expected_effect=f"{fault_kind} on {target} in {band}",
|
|
91
|
+
risk_band=risk,
|
|
92
|
+
seed_hint=self.seed,
|
|
93
|
+
)
|
|
94
|
+
out.append(candidate)
|
|
95
|
+
if limit is not None and len(out) >= limit:
|
|
96
|
+
return tuple(out)
|
|
97
|
+
# Deterministic shuffle so selection order isn't trivially the
|
|
98
|
+
# landscape enumeration order — still fully reproducible from seed.
|
|
99
|
+
rng = self.rng if self.rng is not None else random.Random(self.seed)
|
|
100
|
+
rng.shuffle(out)
|
|
101
|
+
return tuple(out)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _risk_rank(level: RiskLevel) -> int:
|
|
105
|
+
return {
|
|
106
|
+
RiskLevel.LOW: 0,
|
|
107
|
+
RiskLevel.MEDIUM: 1,
|
|
108
|
+
RiskLevel.HIGH: 2,
|
|
109
|
+
RiskLevel.CRITICAL: 3,
|
|
110
|
+
}[level]
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""SQLite-backed coverage accounting (ADR-M5-3, M5 Phase 5.2).
|
|
2
|
+
|
|
3
|
+
A recorded ``Outcome`` marks the run's coverage cell as *seen*. Marking the
|
|
4
|
+
same cell again (a re-run) is idempotent — ``INSERT OR IGNORE`` on the primary
|
|
5
|
+
key means re-running a cell never double-counts. Maniac can then enumerate the
|
|
6
|
+
``UNKNOWN`` cells of a landscape (cells not yet covered) to prioritize coverage
|
|
7
|
+
gain.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
from typing import TYPE_CHECKING
|
|
14
|
+
|
|
15
|
+
from mayhem.domain.coverage import (
|
|
16
|
+
CoverageCell,
|
|
17
|
+
CoverageRecord,
|
|
18
|
+
CoverageSummary,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
from collections.abc import Iterable
|
|
23
|
+
|
|
24
|
+
from mayhem.infra.store import Store
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SQLiteCoverageRepository:
|
|
28
|
+
"""Persists seen coverage cells over the ``m5_coverage`` table."""
|
|
29
|
+
|
|
30
|
+
def __init__(self, store: Store) -> None:
|
|
31
|
+
self._store = store
|
|
32
|
+
|
|
33
|
+
def mark_seen(
|
|
34
|
+
self,
|
|
35
|
+
cell: CoverageCell,
|
|
36
|
+
run_id: str,
|
|
37
|
+
*,
|
|
38
|
+
extra: dict[str, object] | None = None,
|
|
39
|
+
) -> None:
|
|
40
|
+
"""Record that ``cell`` was observed by ``run_id``.
|
|
41
|
+
|
|
42
|
+
Idempotent per cell: re-running the same cell does not double-count;
|
|
43
|
+
``CREATE TABLE ... PRIMARY KEY`` + ``INSERT OR IGNORE`` guarantees it.
|
|
44
|
+
"""
|
|
45
|
+
with self._store.write() as conn:
|
|
46
|
+
conn.execute(
|
|
47
|
+
"""
|
|
48
|
+
INSERT OR IGNORE INTO m5_coverage (
|
|
49
|
+
cell_key, target, fault_kind, execution_context,
|
|
50
|
+
parameter_band, run_id, covered, extra_json
|
|
51
|
+
) VALUES (?, ?, ?, ?, ?, ?, 1, ?)
|
|
52
|
+
""",
|
|
53
|
+
(
|
|
54
|
+
cell.key,
|
|
55
|
+
cell.target,
|
|
56
|
+
cell.fault_kind,
|
|
57
|
+
cell.execution_context,
|
|
58
|
+
cell.parameter_band,
|
|
59
|
+
run_id,
|
|
60
|
+
json.dumps(extra or {}),
|
|
61
|
+
),
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
def is_covered(self, cell: CoverageCell) -> bool:
|
|
65
|
+
rows = self._store.query(
|
|
66
|
+
"SELECT 1 FROM m5_coverage WHERE cell_key = ? AND covered = 1",
|
|
67
|
+
(cell.key,),
|
|
68
|
+
)
|
|
69
|
+
return bool(rows)
|
|
70
|
+
|
|
71
|
+
def covered_keys(self) -> frozenset[str]:
|
|
72
|
+
rows = self._store.query("SELECT cell_key FROM m5_coverage WHERE covered = 1")
|
|
73
|
+
return frozenset(r["cell_key"] for r in rows)
|
|
74
|
+
|
|
75
|
+
def covered_records(self) -> tuple[CoverageRecord, ...]:
|
|
76
|
+
rows = self._store.query(
|
|
77
|
+
"SELECT target, fault_kind, execution_context, parameter_band, run_id, extra_json "
|
|
78
|
+
"FROM m5_coverage WHERE covered = 1"
|
|
79
|
+
)
|
|
80
|
+
records: list[CoverageRecord] = []
|
|
81
|
+
for r in rows:
|
|
82
|
+
cell = CoverageCell(
|
|
83
|
+
target=r["target"],
|
|
84
|
+
fault_kind=r["fault_kind"],
|
|
85
|
+
execution_context=r["execution_context"],
|
|
86
|
+
parameter_band=r["parameter_band"],
|
|
87
|
+
)
|
|
88
|
+
records.append(
|
|
89
|
+
CoverageRecord(cell=cell, run_id=r["run_id"], extra=json.loads(r["extra_json"]))
|
|
90
|
+
)
|
|
91
|
+
return tuple(records)
|
|
92
|
+
|
|
93
|
+
def summary(self, landscape: Iterable[CoverageCell]) -> CoverageSummary:
|
|
94
|
+
"""Aggregate covered vs UNKNOWN cells over ``landscape``."""
|
|
95
|
+
landscape_tuple = tuple(landscape)
|
|
96
|
+
covered = self.covered_keys()
|
|
97
|
+
return CoverageSummary(covered_keys=covered, total_cells=len(landscape_tuple))
|
|
98
|
+
|
|
99
|
+
def unknown_cells(self, landscape: Iterable[CoverageCell]) -> tuple[CoverageCell, ...]:
|
|
100
|
+
"""Cells in ``landscape`` not yet covered, in landscape order."""
|
|
101
|
+
return self.summary(landscape).unknown_cells(tuple(landscape))
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""SQLite-backed LeaseSink — the durable half of the lease protocol (ADR-0007).
|
|
2
|
+
|
|
3
|
+
The controller is the single writer; agents receive this sink through the
|
|
4
|
+
SDK boundary and never open the database themselves. Rows are self-contained:
|
|
5
|
+
run/fault/targets context is denormalized onto the lease row (migration 002)
|
|
6
|
+
so janitor and watchdog sweeps never need joins.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
from datetime import UTC, datetime, timedelta
|
|
13
|
+
from typing import TYPE_CHECKING
|
|
14
|
+
|
|
15
|
+
from mayhem.domain.leases import FaultLease, LeaseState
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
from collections.abc import Mapping
|
|
19
|
+
|
|
20
|
+
from mayhem.infra.store import Store
|
|
21
|
+
|
|
22
|
+
_TERMINAL_STATES = ("released", "expired")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class SQLiteLeaseSink:
|
|
26
|
+
"""Implements ``agents.sinks.LeaseSink`` over the fault_leases table."""
|
|
27
|
+
|
|
28
|
+
def __init__(self, store: Store) -> None:
|
|
29
|
+
self._store = store
|
|
30
|
+
|
|
31
|
+
def save(self, lease: FaultLease) -> None:
|
|
32
|
+
created = lease.created_at
|
|
33
|
+
expires_at = created + timedelta(seconds=float(lease.ttl_seconds))
|
|
34
|
+
with self._store.write() as conn:
|
|
35
|
+
conn.execute(
|
|
36
|
+
"""
|
|
37
|
+
INSERT OR REPLACE INTO fault_leases (
|
|
38
|
+
id, state, owner_agent, undo_json, verify_json,
|
|
39
|
+
ttl_seconds, expires_at, injected_at, released_at,
|
|
40
|
+
release_mechanism, escalation_notes,
|
|
41
|
+
run_id, fault_id, targets_json, created_epoch_s,
|
|
42
|
+
runtime_identity
|
|
43
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
44
|
+
""",
|
|
45
|
+
(
|
|
46
|
+
lease.id,
|
|
47
|
+
lease.state.value,
|
|
48
|
+
lease.owner_agent,
|
|
49
|
+
json.dumps([op.model_dump(mode="json") for op in lease.undo_ops]),
|
|
50
|
+
json.dumps([probe.model_dump(mode="json") for probe in lease.verify_probes]),
|
|
51
|
+
float(lease.ttl_seconds),
|
|
52
|
+
expires_at.isoformat(),
|
|
53
|
+
_iso_or_none(lease.injected_at),
|
|
54
|
+
_iso_or_none(lease.released_at),
|
|
55
|
+
lease.release_mechanism,
|
|
56
|
+
lease.escalation_notes,
|
|
57
|
+
lease.run_id,
|
|
58
|
+
lease.fault_id,
|
|
59
|
+
json.dumps(sorted(lease.targets)),
|
|
60
|
+
created.timestamp(),
|
|
61
|
+
lease.runtime_identity,
|
|
62
|
+
),
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
def load(self, lease_id: str) -> FaultLease | None:
|
|
66
|
+
rows = self._store.query("SELECT * FROM fault_leases WHERE id = ?", (lease_id,))
|
|
67
|
+
return _row_to_lease(dict(rows[0])) if rows else None
|
|
68
|
+
|
|
69
|
+
def active_leases(self) -> tuple[FaultLease, ...]:
|
|
70
|
+
placeholders = ", ".join("?" for _ in _TERMINAL_STATES)
|
|
71
|
+
rows = self._store.query(
|
|
72
|
+
f"SELECT * FROM fault_leases WHERE state NOT IN ({placeholders})"
|
|
73
|
+
" ORDER BY created_epoch_s",
|
|
74
|
+
_TERMINAL_STATES,
|
|
75
|
+
)
|
|
76
|
+
return tuple(_row_to_lease(dict(row)) for row in rows)
|
|
77
|
+
|
|
78
|
+
def expired_pending(self, now_epoch_s: float) -> tuple[FaultLease, ...]:
|
|
79
|
+
rows = self._store.query(
|
|
80
|
+
"SELECT * FROM fault_leases WHERE created_epoch_s + ttl_seconds < ? "
|
|
81
|
+
"AND state IN ('pending', 'active')",
|
|
82
|
+
(now_epoch_s,),
|
|
83
|
+
)
|
|
84
|
+
return tuple(_row_to_lease(dict(row)) for row in rows)
|
|
85
|
+
|
|
86
|
+
def next_sequence(self) -> int:
|
|
87
|
+
# Max trailing integer of every `l-<n>` id ever persisted. Seed the next
|
|
88
|
+
# run's counter here so ids never collide in a shared DB.
|
|
89
|
+
rows = self._store.query(
|
|
90
|
+
"SELECT MAX(CAST(substr(id, 3) AS INTEGER)) AS seq FROM fault_leases"
|
|
91
|
+
" WHERE id LIKE 'l-%'"
|
|
92
|
+
)
|
|
93
|
+
if not rows:
|
|
94
|
+
return 0
|
|
95
|
+
seq = rows[0]["seq"]
|
|
96
|
+
return int(seq) if seq is not None else 0
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _row_to_lease(row: Mapping[str, object]) -> FaultLease:
|
|
100
|
+
return FaultLease.model_validate(
|
|
101
|
+
{
|
|
102
|
+
"id": str(row["id"]),
|
|
103
|
+
"state": LeaseState(str(row["state"])),
|
|
104
|
+
"owner_agent": str(row["owner_agent"]),
|
|
105
|
+
"undo_ops": json.loads(str(row["undo_json"])),
|
|
106
|
+
"verify_probes": json.loads(str(row["verify_json"])),
|
|
107
|
+
"ttl_seconds": float(str(row["ttl_seconds"])),
|
|
108
|
+
"injected_at": _dt_or_none(row.get("injected_at")),
|
|
109
|
+
"released_at": _dt_or_none(row.get("released_at")),
|
|
110
|
+
"release_mechanism": row["release_mechanism"],
|
|
111
|
+
"escalation_notes": row["escalation_notes"],
|
|
112
|
+
"run_id": str(row["run_id"] or ""),
|
|
113
|
+
"fault_id": str(row["fault_id"] or ""),
|
|
114
|
+
"targets": frozenset(json.loads(str(row["targets_json"]))),
|
|
115
|
+
"created_at": datetime.fromtimestamp(float(str(row["created_epoch_s"])), tz=UTC),
|
|
116
|
+
"runtime_identity": row.get("runtime_identity"),
|
|
117
|
+
}
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _iso_or_none(moment: datetime | None) -> str | None:
|
|
122
|
+
return moment.isoformat() if moment is not None else None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _dt_or_none(raw: object) -> datetime | None:
|
|
126
|
+
if raw is None:
|
|
127
|
+
return None
|
|
128
|
+
parsed = datetime.fromisoformat(str(raw))
|
|
129
|
+
return parsed if parsed.tzinfo else parsed.replace(tzinfo=UTC)
|
mayhem/infra/maniac.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Maniac selection/optimization engine (ADR-M5-3, M5 Phase 5.5).
|
|
2
|
+
|
|
3
|
+
Maniac chooses *which experiment to run next* from the candidate set, grounded
|
|
4
|
+
in recorded outcome coverage and the campaign's bounds. It is:
|
|
5
|
+
|
|
6
|
+
- **stateless** across campaign runs (no in-memory carry-over; every selection
|
|
7
|
+
is re-derived from its explicit inputs),
|
|
8
|
+
- **deterministic and reproducible**: identical inputs (candidates, covered
|
|
9
|
+
cells, gates, bounds, seed) always yield the identical selection,
|
|
10
|
+
- **knowledge-transferable**: recorded coverage from prior campaigns is passed
|
|
11
|
+
in explicitly, so learnings carry across campaign boundaries.
|
|
12
|
+
|
|
13
|
+
Selection is a greedy coverage-maximizing walk: gate every candidate, prefer
|
|
14
|
+
cells not yet covered, and stop once the budget or coverage target is met.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import random
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from typing import TYPE_CHECKING
|
|
22
|
+
|
|
23
|
+
from mayhem.domain.coverage import CoverageCell
|
|
24
|
+
from mayhem.infra.candidate_gates import CandidateGatePipeline
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from mayhem.domain.candidates import CandidateDecision, ExperimentCandidate
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def coverage_cell_for_candidate(candidate: ExperimentCandidate) -> CoverageCell:
|
|
31
|
+
"""The coverage cell a candidate explores, derived deterministically."""
|
|
32
|
+
band = str(candidate.params.get("band") or "default") if candidate.params else "default"
|
|
33
|
+
return CoverageCell(
|
|
34
|
+
candidate.target,
|
|
35
|
+
candidate.primary_fault,
|
|
36
|
+
candidate.execution_context,
|
|
37
|
+
band,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class SelectionInputs:
|
|
43
|
+
"""Every input Maniac is allowed to look at (no hidden state)."""
|
|
44
|
+
|
|
45
|
+
candidates: tuple[ExperimentCandidate, ...]
|
|
46
|
+
covered_keys: frozenset[str] = frozenset()
|
|
47
|
+
gates: CandidateGatePipeline | None = None
|
|
48
|
+
max_runs: int = 100
|
|
49
|
+
coverage_target: int = 1
|
|
50
|
+
seed: int = 0
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class SelectionResult:
|
|
55
|
+
selected: tuple[ExperimentCandidate, ...]
|
|
56
|
+
rejected: tuple[CandidateDecision, ...] = ()
|
|
57
|
+
predicted_new_coverage: int = 0
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def selected_count(self) -> int:
|
|
61
|
+
return len(self.selected)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def select_next(inputs: SelectionInputs) -> SelectionResult:
|
|
65
|
+
"""Pure, deterministic greedy selection (stateless, reproducible)."""
|
|
66
|
+
gates = inputs.gates if inputs.gates is not None else CandidateGatePipeline()
|
|
67
|
+
covered = set(inputs.covered_keys)
|
|
68
|
+
|
|
69
|
+
accepted: list[ExperimentCandidate] = []
|
|
70
|
+
rejected: list[CandidateDecision] = []
|
|
71
|
+
for cand in inputs.candidates:
|
|
72
|
+
decision = gates.gate(cand)
|
|
73
|
+
if decision.accepted:
|
|
74
|
+
accepted.append(cand)
|
|
75
|
+
else:
|
|
76
|
+
rejected.append(decision)
|
|
77
|
+
|
|
78
|
+
# Deterministic ordering: seeded random tie-break over the candidates,
|
|
79
|
+
# then rank so that already-covered cells come last (coverage first).
|
|
80
|
+
rng = random.Random(inputs.seed)
|
|
81
|
+
ordered = sorted(accepted, key=lambda c: rng.random())
|
|
82
|
+
uncovered_first = sorted(
|
|
83
|
+
ordered,
|
|
84
|
+
key=lambda c: coverage_cell_for_candidate(c).key in covered,
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
selected: list[ExperimentCandidate] = []
|
|
88
|
+
covered_now = set(covered)
|
|
89
|
+
for cand in uncovered_first:
|
|
90
|
+
if len(selected) >= inputs.max_runs:
|
|
91
|
+
break
|
|
92
|
+
if len(covered_now) >= inputs.coverage_target:
|
|
93
|
+
break
|
|
94
|
+
cell = coverage_cell_for_candidate(cand)
|
|
95
|
+
covered_now.add(cell.key)
|
|
96
|
+
selected.append(cand)
|
|
97
|
+
|
|
98
|
+
projected = len(covered_now) - len(covered)
|
|
99
|
+
return SelectionResult(
|
|
100
|
+
selected=tuple(selected),
|
|
101
|
+
rejected=tuple(rejected),
|
|
102
|
+
predicted_new_coverage=projected,
|
|
103
|
+
)
|