mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
mayhem/domain/leases.py
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Fault leases: the recovery guarantee's core object (ADR-0005).
|
|
2
|
+
|
|
3
|
+
A lease is created *before* any mutation and persisted with write-ahead undo
|
|
4
|
+
data. The state machine below is the single source of truth for legal
|
|
5
|
+
transitions; the janitor, watchdog, and release paths all go through it.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
from enum import StrEnum
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
|
|
14
|
+
|
|
15
|
+
from mayhem.domain.common import Duration, utc_now
|
|
16
|
+
from mayhem.domain.errors import InvalidTransitionError, InvariantViolationError
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class LeaseState(StrEnum):
|
|
20
|
+
PENDING = "pending" # undo written, injection not yet started
|
|
21
|
+
ACTIVE = "active" # fault applied
|
|
22
|
+
RELEASING = "releasing" # compensation in progress
|
|
23
|
+
RELEASED = "released" # compensated AND verified (safe terminal)
|
|
24
|
+
EXPIRED = "expired" # watchdog TTL fired, agent self-compensated (safe terminal)
|
|
25
|
+
ORPHANED = "orphaned" # owner unreachable; janitor will reclaim
|
|
26
|
+
DIRTY = "dirty" # compensation failed; LOUD escalation (ack required)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
_TRANSITIONS: dict[LeaseState, frozenset[LeaseState]] = {
|
|
30
|
+
LeaseState.PENDING: frozenset({LeaseState.ACTIVE, LeaseState.EXPIRED}),
|
|
31
|
+
LeaseState.ACTIVE: frozenset({LeaseState.RELEASING, LeaseState.EXPIRED, LeaseState.ORPHANED}),
|
|
32
|
+
LeaseState.ORPHANED: frozenset({LeaseState.RELEASING}),
|
|
33
|
+
LeaseState.RELEASING: frozenset({LeaseState.RELEASED, LeaseState.DIRTY}),
|
|
34
|
+
# Safe terminals: released, expired. Dirty may only be *expired* past its
|
|
35
|
+
# TTL by the janitor — the fault is then abandoned (ADR-0007 surrender).
|
|
36
|
+
LeaseState.RELEASED: frozenset(),
|
|
37
|
+
LeaseState.EXPIRED: frozenset(),
|
|
38
|
+
LeaseState.DIRTY: frozenset({LeaseState.EXPIRED}),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
_SAFE_TERMINALS: frozenset[LeaseState] = frozenset({LeaseState.RELEASED, LeaseState.EXPIRED})
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class UndoOp(BaseModel):
|
|
45
|
+
"""One compensation primitive; MUST be idempotent-first by convention."""
|
|
46
|
+
|
|
47
|
+
model_config = ConfigDict(frozen=True)
|
|
48
|
+
|
|
49
|
+
op: str # executor operation name, e.g. "tc.del_qdisc"
|
|
50
|
+
args: dict[str, str] = Field(default_factory=dict)
|
|
51
|
+
idempotent: bool = True
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class VerifyProbe(BaseModel):
|
|
55
|
+
"""Post-recovery assertion proving the fault is really gone."""
|
|
56
|
+
|
|
57
|
+
model_config = ConfigDict(frozen=True)
|
|
58
|
+
|
|
59
|
+
probe: str # e.g. "exec", "tc.qdisc_absent", "iptables.chain_absent"
|
|
60
|
+
args: dict[str, object] = Field(default_factory=dict)
|
|
61
|
+
expect_present: bool = False
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class FaultLease(BaseModel):
|
|
65
|
+
"""Immutable lease value object; transitions produce new instances."""
|
|
66
|
+
|
|
67
|
+
model_config = ConfigDict(frozen=True)
|
|
68
|
+
|
|
69
|
+
id: str # l-<hex>
|
|
70
|
+
run_id: str
|
|
71
|
+
fault_id: str
|
|
72
|
+
owner_agent: str
|
|
73
|
+
targets: frozenset[str]
|
|
74
|
+
undo_ops: tuple[UndoOp, ...] = ()
|
|
75
|
+
verify_probes: tuple[VerifyProbe, ...] = ()
|
|
76
|
+
ttl_seconds: Duration = 120.0
|
|
77
|
+
state: LeaseState = LeaseState.PENDING
|
|
78
|
+
created_at: datetime = Field(default_factory=utc_now)
|
|
79
|
+
injected_at: datetime | None = None
|
|
80
|
+
released_at: datetime | None = None
|
|
81
|
+
release_mechanism: str | None = None # normal|watchdog|janitor|manual
|
|
82
|
+
escalation_notes: str | None = None
|
|
83
|
+
runtime_identity: str | None = None # canonical identity key (ADR-M1-1/1-3)
|
|
84
|
+
|
|
85
|
+
@field_validator("id")
|
|
86
|
+
@classmethod
|
|
87
|
+
def _id_prefix(cls, value: str) -> str:
|
|
88
|
+
if not value.startswith("l-"):
|
|
89
|
+
msg = f"lease ids start with 'l-', got {value!r}"
|
|
90
|
+
raise InvariantViolationError("lease_id_prefix", msg)
|
|
91
|
+
return value
|
|
92
|
+
|
|
93
|
+
@model_validator(mode="after")
|
|
94
|
+
def _check_invariants(self) -> FaultLease:
|
|
95
|
+
if self.state is LeaseState.ACTIVE and not self.undo_ops:
|
|
96
|
+
raise InvariantViolationError(
|
|
97
|
+
"undo_required_before_active",
|
|
98
|
+
f"lease {self.id} cannot be ACTIVE without write-ahead undo ops",
|
|
99
|
+
)
|
|
100
|
+
if self.state in (LeaseState.ACTIVE, LeaseState.RELEASING) and not self.verify_probes:
|
|
101
|
+
raise InvariantViolationError(
|
|
102
|
+
"verify_required_before_release",
|
|
103
|
+
f"lease {self.id} in {self.state.value} needs verify probes",
|
|
104
|
+
)
|
|
105
|
+
if self.state is LeaseState.DIRTY and not self.escalation_notes:
|
|
106
|
+
raise InvariantViolationError(
|
|
107
|
+
"dirty_requires_escalation_notes",
|
|
108
|
+
f"lease {self.id} marked dirty without escalation notes",
|
|
109
|
+
)
|
|
110
|
+
if self.injected_at is not None and self.created_at > self.injected_at:
|
|
111
|
+
raise InvariantViolationError(
|
|
112
|
+
"lease_time_ordering", f"lease {self.id} injected before creation"
|
|
113
|
+
)
|
|
114
|
+
if self.released_at is not None and self.state not in _SAFE_TERMINALS:
|
|
115
|
+
raise InvariantViolationError(
|
|
116
|
+
"lease_time_ordering",
|
|
117
|
+
f"lease {self.id} has released_at while in state {self.state.value}",
|
|
118
|
+
)
|
|
119
|
+
return self
|
|
120
|
+
|
|
121
|
+
# -- state machine ------------------------------------------------------------
|
|
122
|
+
def can_transition(self, target: LeaseState) -> bool:
|
|
123
|
+
return target in _TRANSITIONS[self.state]
|
|
124
|
+
|
|
125
|
+
def transition(
|
|
126
|
+
self,
|
|
127
|
+
target: LeaseState,
|
|
128
|
+
*,
|
|
129
|
+
mechanism: str | None = None,
|
|
130
|
+
now: datetime | None = None,
|
|
131
|
+
escalation_notes: str | None = None,
|
|
132
|
+
) -> FaultLease:
|
|
133
|
+
"""Return a new lease advanced to ``target``.
|
|
134
|
+
|
|
135
|
+
Raises:
|
|
136
|
+
InvalidTransitionError: If the transition is illegal.
|
|
137
|
+
InvariantViolationError: If target-specific preconditions are unmet.
|
|
138
|
+
"""
|
|
139
|
+
if not self.can_transition(target):
|
|
140
|
+
raise InvalidTransitionError("fault_lease", self.id, self.state.value, target.value)
|
|
141
|
+
updates: dict[str, object] = {"state": target}
|
|
142
|
+
moment = now if now is not None else utc_now()
|
|
143
|
+
if target is LeaseState.ACTIVE:
|
|
144
|
+
updates["injected_at"] = moment
|
|
145
|
+
if target in (LeaseState.RELEASED, LeaseState.EXPIRED):
|
|
146
|
+
updates["released_at"] = moment
|
|
147
|
+
if mechanism is not None:
|
|
148
|
+
updates["release_mechanism"] = mechanism
|
|
149
|
+
if escalation_notes is not None:
|
|
150
|
+
updates["escalation_notes"] = escalation_notes
|
|
151
|
+
# model_copy skips validators; transitions MUST re-check every invariant.
|
|
152
|
+
return self.__class__.model_validate({**self.model_dump(), **updates})
|
|
153
|
+
|
|
154
|
+
# -- classification -------------------------------------------------------------
|
|
155
|
+
@property
|
|
156
|
+
def is_terminal(self) -> bool:
|
|
157
|
+
return not _TRANSITIONS[self.state]
|
|
158
|
+
|
|
159
|
+
@property
|
|
160
|
+
def is_safe_terminal(self) -> bool:
|
|
161
|
+
return self.state in _SAFE_TERMINALS
|
|
162
|
+
|
|
163
|
+
@property
|
|
164
|
+
def needs_janitor_attention(self) -> bool:
|
|
165
|
+
return self.state in (
|
|
166
|
+
LeaseState.PENDING,
|
|
167
|
+
LeaseState.ACTIVE,
|
|
168
|
+
LeaseState.ORPHANED,
|
|
169
|
+
LeaseState.RELEASING,
|
|
170
|
+
LeaseState.DIRTY,
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def assert_all_recovered(leases: list[FaultLease]) -> None:
|
|
175
|
+
"""Run-completion invariant: no non-safe-terminal leases may remain.
|
|
176
|
+
|
|
177
|
+
Raises:
|
|
178
|
+
InvariantViolationError: Naming every offending lease.
|
|
179
|
+
"""
|
|
180
|
+
offenders = [lease for lease in leases if not lease.is_safe_terminal]
|
|
181
|
+
if offenders:
|
|
182
|
+
detail = ", ".join(f"{o.id}:{o.state.value}" for o in offenders)
|
|
183
|
+
raise InvariantViolationError(
|
|
184
|
+
"all_leases_recovered_before_run_completion",
|
|
185
|
+
f"non-recovered leases remain: {detail}",
|
|
186
|
+
)
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Load generation strategy models (ADR-0021).
|
|
2
|
+
|
|
3
|
+
Defines load patterns that the controller can orchestrate during chaos
|
|
4
|
+
experiments — ramping, steady, burst, and spike profiles with configurable
|
|
5
|
+
concurrency, duration, and ramp phases.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from enum import StrEnum
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
13
|
+
|
|
14
|
+
from mayhem.domain.common import Duration
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class LoadPattern(StrEnum):
|
|
18
|
+
"""Named load patterns for experiment orchestration."""
|
|
19
|
+
|
|
20
|
+
CONSTANT = "constant" # fixed rate for the entire duration
|
|
21
|
+
RAMP_UP = "ramp_up" # linearly increase from min to max
|
|
22
|
+
RAMP_DOWN = "ramp_down" # linearly decrease from max to min
|
|
23
|
+
BURST = "burst" # short high-intensity burst
|
|
24
|
+
SPIKE = "spike" # sudden jump then return
|
|
25
|
+
STEP = "step" # step function: low → high → low
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class LoadPhase(BaseModel):
|
|
29
|
+
"""One phase in a multi-phase load profile."""
|
|
30
|
+
|
|
31
|
+
model_config = ConfigDict(frozen=True)
|
|
32
|
+
|
|
33
|
+
pattern: LoadPattern
|
|
34
|
+
vus: int = Field(ge=1, default=1) # virtual users / concurrency
|
|
35
|
+
rps: int | None = Field(default=None, ge=1) # requests per second (alternative to VUs)
|
|
36
|
+
duration: Duration = 10.0
|
|
37
|
+
ramp_seconds: Duration = 0.0 # ramp time for RAMP_UP/RAMP_DOWN
|
|
38
|
+
target_vus: int | None = Field(default=None, ge=1) # endpoint for ramp patterns
|
|
39
|
+
target_rps: int | None = Field(default=None, ge=1)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class LoadStrategy(BaseModel):
|
|
43
|
+
"""A complete load generation strategy: one or more phases.
|
|
44
|
+
|
|
45
|
+
Phases execute sequentially. The total experiment duration is the sum
|
|
46
|
+
of all phase durations. Between phases, there is no pause (zero-load gap
|
|
47
|
+
is explicit via a phase with vus=1, duration=gap).
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
model_config = ConfigDict(frozen=True)
|
|
51
|
+
|
|
52
|
+
name: str = ""
|
|
53
|
+
phases: tuple[LoadPhase, ...] = (LoadPhase(pattern=LoadPattern.CONSTANT),)
|
|
54
|
+
endpoint: str = "" # target URL or service
|
|
55
|
+
method: str = "GET"
|
|
56
|
+
headers: dict[str, str] = Field(default_factory=dict)
|
|
57
|
+
timeout: Duration = 10.0
|
|
58
|
+
|
|
59
|
+
def total_duration(self) -> float:
|
|
60
|
+
"""Sum of all phase durations in seconds."""
|
|
61
|
+
return sum(float(p.duration) for p in self.phases)
|
|
62
|
+
|
|
63
|
+
def max_concurrency(self) -> int:
|
|
64
|
+
"""Peak VUs across all phases."""
|
|
65
|
+
return max(p.vus for p in self.phases)
|
|
66
|
+
|
|
67
|
+
@classmethod
|
|
68
|
+
def constant_profile(cls, vus: int, duration: Duration) -> LoadStrategy:
|
|
69
|
+
"""Shorthand: single constant-rate phase."""
|
|
70
|
+
return cls(phases=(LoadPhase(pattern=LoadPattern.CONSTANT, vus=vus, duration=duration),))
|
|
71
|
+
|
|
72
|
+
@classmethod
|
|
73
|
+
def ramp_profile(cls, start_vus: int, end_vus: int, duration: Duration) -> LoadStrategy:
|
|
74
|
+
"""Shorthand: single ramp-up phase."""
|
|
75
|
+
return cls(
|
|
76
|
+
phases=(
|
|
77
|
+
LoadPhase(
|
|
78
|
+
pattern=LoadPattern.RAMP_UP,
|
|
79
|
+
vus=start_vus,
|
|
80
|
+
target_vus=end_vus,
|
|
81
|
+
duration=duration,
|
|
82
|
+
ramp_seconds=duration,
|
|
83
|
+
),
|
|
84
|
+
)
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class FuzzStrategy(BaseModel):
|
|
89
|
+
"""Configuration for protocol fuzzing during an experiment."""
|
|
90
|
+
|
|
91
|
+
model_config = ConfigDict(frozen=True)
|
|
92
|
+
|
|
93
|
+
target_field: str = "" # which field to mutate (empty = all)
|
|
94
|
+
mutations_per_request: int = Field(default=1, ge=1)
|
|
95
|
+
max_requests: int = Field(default=100, ge=1)
|
|
96
|
+
timeout: Duration = 30.0
|
|
97
|
+
seed: int | None = None # deterministic fuzzing
|
|
98
|
+
dictionary: tuple[str, ...] = () # custom mutation dictionary
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""M5 campaign semantics (ADR-M5-4, M5 Phase 5.4).
|
|
2
|
+
|
|
3
|
+
A campaign is a *goal + bounds*: a target set, a coverage target or stop
|
|
4
|
+
condition, an execution budget (blast-radius bound), a ``mode``
|
|
5
|
+
(``supervised`` | ``autonomous``), an optional deadline, and a per-campaign
|
|
6
|
+
risk ceiling. A campaign produces many Run/Outcome pairs and completes when the
|
|
7
|
+
stop condition is reached, the budget is exhausted, or the deadline passes.
|
|
8
|
+
|
|
9
|
+
This is a distinct model from the ADR-0022 ``Campaign`` used by the scheduler;
|
|
10
|
+
ADR-M5-4 campaigns are the intelligence-loop container.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from enum import StrEnum
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from mayhem.domain.risks import RiskLevel
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class CampaignMode(StrEnum):
|
|
23
|
+
SUPERVISED = "supervised" # human approves each drill (default)
|
|
24
|
+
AUTONOMOUS = "autonomous" # auto-run within bounds, explicit opt-in
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class CampaignState(StrEnum):
|
|
28
|
+
IDLE = "idle"
|
|
29
|
+
RUNNING = "running"
|
|
30
|
+
COMPLETED = "completed"
|
|
31
|
+
STOPPED = "stopped"
|
|
32
|
+
ABORTED = "aborted"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class StopReason(StrEnum):
|
|
36
|
+
COVERAGE_REACHED = "coverage_reached"
|
|
37
|
+
BUDGET_EXHAUSTED = "budget_exhausted"
|
|
38
|
+
DEADLINE_PASSED = "deadline_passed"
|
|
39
|
+
STOP_CONDITION = "stop_condition"
|
|
40
|
+
RESOURCE_CONFLICT = "resource_conflict" # hard bound-abort; never executed
|
|
41
|
+
NOT_STARTED = "not_started"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class ApproveDecision(StrEnum):
|
|
45
|
+
PENDING = "pending"
|
|
46
|
+
APPROVED = "approved"
|
|
47
|
+
DENIED = "denied"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class CampaignLogEntry:
|
|
52
|
+
"""One recorded decision in a campaign's approve/deny trail."""
|
|
53
|
+
|
|
54
|
+
candidate_id: str
|
|
55
|
+
decision: ApproveDecision
|
|
56
|
+
rationale: str = ""
|
|
57
|
+
run_id: str | None = None
|
|
58
|
+
outcome_id: str | None = None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass(frozen=True)
|
|
62
|
+
class M5Campaign:
|
|
63
|
+
"""Goal + bounds for one intelligence-loop campaign (ADR-M5-4)."""
|
|
64
|
+
|
|
65
|
+
id: str
|
|
66
|
+
name: str
|
|
67
|
+
targets: tuple[str, ...] = ()
|
|
68
|
+
coverage_target: int = 1 # distinct cells to cover
|
|
69
|
+
max_runs: int = 100 # blast-radius budget = max executed runs
|
|
70
|
+
mode: CampaignMode = CampaignMode.SUPERVISED
|
|
71
|
+
risk_ceiling: RiskLevel = RiskLevel.HIGH
|
|
72
|
+
deadline_epoch_s: float | None = None
|
|
73
|
+
stop_condition: str = "" # free-form durable note; caller enforces it
|
|
74
|
+
|
|
75
|
+
def with_defaults(self, **kwargs: Any) -> M5Campaign:
|
|
76
|
+
"""Return a copy with overridden bounds (immutable-equivalent helper)."""
|
|
77
|
+
return M5Campaign(**{**self.__dict__, **kwargs})
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class CampaignProgress:
|
|
82
|
+
"""Live counters of one campaign iteration."""
|
|
83
|
+
|
|
84
|
+
runs_executed: int = 0
|
|
85
|
+
cells_covered: int = 0
|
|
86
|
+
candidates_rejected: int = 0
|
|
87
|
+
candidates_pending_approval: int = 0
|
|
88
|
+
candidates_denied: int = 0
|
|
89
|
+
candidates_aborted: int = 0
|
|
90
|
+
log: list[CampaignLogEntry] = field(default_factory=list)
|
|
91
|
+
|
|
92
|
+
def snapshot(self) -> dict[str, int]:
|
|
93
|
+
return {
|
|
94
|
+
"runs_executed": self.runs_executed,
|
|
95
|
+
"cells_covered": self.cells_covered,
|
|
96
|
+
"candidates_rejected": self.candidates_rejected,
|
|
97
|
+
"candidates_pending_approval": self.candidates_pending_approval,
|
|
98
|
+
"candidates_denied": self.candidates_denied,
|
|
99
|
+
"candidates_aborted": self.candidates_aborted,
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
def record(
|
|
103
|
+
self,
|
|
104
|
+
candidate_id: str,
|
|
105
|
+
decision: ApproveDecision,
|
|
106
|
+
*,
|
|
107
|
+
rationale: str = "",
|
|
108
|
+
run_id: str | None = None,
|
|
109
|
+
outcome_id: str | None = None,
|
|
110
|
+
) -> None:
|
|
111
|
+
"""Append one decision to the approve/deny trail."""
|
|
112
|
+
self.log.append(
|
|
113
|
+
CampaignLogEntry(
|
|
114
|
+
candidate_id=candidate_id,
|
|
115
|
+
decision=decision,
|
|
116
|
+
rationale=rationale,
|
|
117
|
+
run_id=run_id,
|
|
118
|
+
outcome_id=outcome_id,
|
|
119
|
+
)
|
|
120
|
+
)
|
mayhem/domain/maniac.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Maniac mode — random fault injection (ADR-M5-1).
|
|
2
|
+
|
|
3
|
+
``mayhem maniac`` compiles a drill spec exactly like ``mayhem run``, then
|
|
4
|
+
replaces the authored execution with a random draw: ``run_level`` rounds, each
|
|
5
|
+
picking a random container (from the spec's own ``containers`` map) and a
|
|
6
|
+
random fault. The spec keeps providing the container pool, the fault catalog
|
|
7
|
+
entries, the success criteria and the observability sources, so the machine
|
|
8
|
+
verdict and evidence are produced exactly as in a deterministic run.
|
|
9
|
+
|
|
10
|
+
The ``level`` dial (1-5) scales how far each round strays from the spec's
|
|
11
|
+
authored intent (see the table below). Drawing is reproducible when ``seed``
|
|
12
|
+
is set.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import random
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from typing import TYPE_CHECKING
|
|
20
|
+
|
|
21
|
+
from mayhem.domain.catalog import definition_for
|
|
22
|
+
from mayhem.domain.common import parse_duration
|
|
23
|
+
from mayhem.domain.errors import SchemaValidationError
|
|
24
|
+
|
|
25
|
+
if TYPE_CHECKING:
|
|
26
|
+
from mayhem.domain.experiments import DrillFault, DrillSpec
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ManiacError(Exception):
|
|
30
|
+
"""Raised when maniac mode cannot be compiled."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class ManiacDraw:
|
|
35
|
+
"""One maniac round: a fault to inject on a container."""
|
|
36
|
+
|
|
37
|
+
container: str # target container name (matches the topology)
|
|
38
|
+
fault: DrillFault # duration may be jittered (levels 4-5)
|
|
39
|
+
round: int # 1-based injection round
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _jitter_duration(fault: DrillFault, pct: float, rng: random.Random) -> DrillFault:
|
|
43
|
+
"""Jitter ``fault.duration`` by ``±pct``, clamped to the catalog cap / 1 s."""
|
|
44
|
+
raw = (
|
|
45
|
+
parse_duration(fault.duration) if isinstance(fault.duration, str) else float(fault.duration)
|
|
46
|
+
)
|
|
47
|
+
jittered = raw * (1.0 + rng.uniform(-pct, pct))
|
|
48
|
+
try:
|
|
49
|
+
cap = definition_for(fault.fault).max_duration_s
|
|
50
|
+
jittered = min(jittered, cap)
|
|
51
|
+
except (SchemaValidationError, LookupError):
|
|
52
|
+
pass # the planner surfaces unknown-fault errors with authority
|
|
53
|
+
return fault.model_copy(update={"duration": round(max(jittered, 1.0), 3)})
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def draw_maniac_rounds(
|
|
57
|
+
spec: DrillSpec,
|
|
58
|
+
*,
|
|
59
|
+
level: int,
|
|
60
|
+
run_level: int,
|
|
61
|
+
seed: int | None = None,
|
|
62
|
+
) -> tuple[ManiacDraw, ...]:
|
|
63
|
+
"""Draw ``run_level`` random (container, fault) rounds from ``spec``.
|
|
64
|
+
|
|
65
|
+
Only containers that actually author faults are in the pool; cross-locus
|
|
66
|
+
draws (``level >= 3``) may apply any spec-wide fault to any pool container.
|
|
67
|
+
Durations are jittered at ``level >= 4`` (ADR-M5-1 §semantics).
|
|
68
|
+
"""
|
|
69
|
+
pool = {name: container for name, container in spec.containers.items() if container.faults}
|
|
70
|
+
if not pool:
|
|
71
|
+
raise ManiacError(
|
|
72
|
+
"maniac mode needs at least one container with faults — "
|
|
73
|
+
"none of the spec's containers define any"
|
|
74
|
+
)
|
|
75
|
+
rng = random.Random(seed)
|
|
76
|
+
names = sorted(pool)
|
|
77
|
+
all_faults = tuple(f for c in pool.values() for f in c.faults)
|
|
78
|
+
jitter_pct = 0.20 if level >= 5 else (0.10 if level == 4 else 0.0)
|
|
79
|
+
|
|
80
|
+
draws: list[ManiacDraw] = []
|
|
81
|
+
for round_no in range(1, run_level + 1):
|
|
82
|
+
target = rng.choice(names)
|
|
83
|
+
if level >= 3:
|
|
84
|
+
# Cross-locus: any spec-wide fault may land on any pool container.
|
|
85
|
+
fault = rng.choice(all_faults)
|
|
86
|
+
elif level == 2:
|
|
87
|
+
fault = rng.choice(pool[target].faults)
|
|
88
|
+
else:
|
|
89
|
+
fault = pool[target].faults[0]
|
|
90
|
+
if jitter_pct > 0:
|
|
91
|
+
fault = _jitter_duration(fault, jitter_pct, rng)
|
|
92
|
+
draws.append(ManiacDraw(container=target, fault=fault, round=round_no))
|
|
93
|
+
return tuple(draws)
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Declarative observability/metrics source DSL (ADR-M4-4 §22).
|
|
2
|
+
|
|
3
|
+
An ``observability`` section on a drill spec collects *evidence* into the
|
|
4
|
+
outcome record — the values an evaluator later needs — without driving the
|
|
5
|
+
run's own verdict. Every source is:
|
|
6
|
+
|
|
7
|
+
- **named** (``source_id``) so observations can be cross-referenced,
|
|
8
|
+
- **bounded** (per-source ``timeout`` and an overall ``total_timeout``), and
|
|
9
|
+
- **best-effort** (a failing source is recorded as a skip note, never fatal).
|
|
10
|
+
|
|
11
|
+
Source kinds: container logs tail, container inspect, an external probe
|
|
12
|
+
(HTTP/exec/TCP/process/metric/file — the same closed union as checks) and a
|
|
13
|
+
Prometheus text-format metrics scrape.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from enum import StrEnum
|
|
19
|
+
from typing import Annotated, Literal
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
22
|
+
|
|
23
|
+
from mayhem.domain.checks import Probe
|
|
24
|
+
from mayhem.domain.common import Duration, parse_duration
|
|
25
|
+
from mayhem.domain.errors import SchemaValidationError
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def duration_seconds(value: object) -> float:
|
|
29
|
+
"""Coerce a ``Duration`` to seconds, tolerating un-validated string defaults.
|
|
30
|
+
|
|
31
|
+
Pydantic v2 returns a class-level string default (e.g. ``"10s"``) *without*
|
|
32
|
+
running the duration validators, so naive ``float()`` calls would raise.
|
|
33
|
+
"""
|
|
34
|
+
if isinstance(value, str):
|
|
35
|
+
return parse_duration(value)
|
|
36
|
+
if isinstance(value, (int, float)):
|
|
37
|
+
return float(value)
|
|
38
|
+
raise TypeError(f"cannot interpret {value!r} as a duration")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ObservabilitySourceKind(StrEnum):
|
|
42
|
+
"""Kinds of observability evidences a drill can collect (ADR-M4-4)."""
|
|
43
|
+
|
|
44
|
+
LOGS = "logs"
|
|
45
|
+
INSPECTION = "inspection"
|
|
46
|
+
PROBE = "probe"
|
|
47
|
+
METRICS = "metrics"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class LogsSource(BaseModel):
|
|
51
|
+
"""Container log tail (Docker/Podman ``logs``) captured as evidence."""
|
|
52
|
+
|
|
53
|
+
model_config = ConfigDict(frozen=True)
|
|
54
|
+
|
|
55
|
+
kind: Literal[ObservabilitySourceKind.LOGS] = ObservabilitySourceKind.LOGS
|
|
56
|
+
source_id: str
|
|
57
|
+
container: str # container name from the drill's containers:
|
|
58
|
+
tail: int = Field(default=100, ge=1, le=10_000) # bounded tail, never unbounded
|
|
59
|
+
since: str = "" # optional engine `--since` filter; empty = latest tail
|
|
60
|
+
timeout: Duration = 10.0
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class InspectionSource(BaseModel):
|
|
64
|
+
"""Container runtime inspect JSON captured as key/value evidence."""
|
|
65
|
+
|
|
66
|
+
model_config = ConfigDict(frozen=True)
|
|
67
|
+
|
|
68
|
+
kind: Literal[ObservabilitySourceKind.INSPECTION] = ObservabilitySourceKind.INSPECTION
|
|
69
|
+
source_id: str
|
|
70
|
+
container: str
|
|
71
|
+
timeout: Duration = 10.0
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class ProbeSource(BaseModel):
|
|
75
|
+
"""An external probe observed on a cadence inside the run window."""
|
|
76
|
+
|
|
77
|
+
model_config = ConfigDict(frozen=True)
|
|
78
|
+
|
|
79
|
+
kind: Literal[ObservabilitySourceKind.PROBE] = ObservabilitySourceKind.PROBE
|
|
80
|
+
source_id: str
|
|
81
|
+
probe: Probe
|
|
82
|
+
cadence: Duration = 0.0 # >0 polls repeatedly; 0 = collect once
|
|
83
|
+
timeout: Duration = 10.0
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class MetricsSource(BaseModel):
|
|
87
|
+
"""Prometheus text-format scrape of one metric, sampled on a cadence."""
|
|
88
|
+
|
|
89
|
+
model_config = ConfigDict(frozen=True)
|
|
90
|
+
|
|
91
|
+
kind: Literal[ObservabilitySourceKind.METRICS] = ObservabilitySourceKind.METRICS
|
|
92
|
+
source_id: str
|
|
93
|
+
endpoint: str # e.g. http://127.0.0.1:9100/metrics
|
|
94
|
+
metric: str # Prometheus metric name to resolve to a sample value
|
|
95
|
+
cadence: Duration = 0.0
|
|
96
|
+
timeout: Duration = 10.0
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
ObservabilitySource = Annotated[
|
|
100
|
+
LogsSource | InspectionSource | ProbeSource | MetricsSource,
|
|
101
|
+
Field(discriminator="kind"),
|
|
102
|
+
]
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class ObservabilityConfig(BaseModel):
|
|
106
|
+
"""One bounding ``observability`` section on a drill spec."""
|
|
107
|
+
|
|
108
|
+
model_config = ConfigDict(frozen=True)
|
|
109
|
+
|
|
110
|
+
sources: tuple[ObservabilitySource, ...] = ()
|
|
111
|
+
cadence: Duration = 5.0 # default polling cadence where cadence > 0
|
|
112
|
+
total_timeout: Duration = 30.0 # hard bound across the whole pass
|
|
113
|
+
|
|
114
|
+
@model_validator(mode="after")
|
|
115
|
+
def _bounds_hold(self) -> ObservabilityConfig:
|
|
116
|
+
total = duration_seconds(self.total_timeout)
|
|
117
|
+
default_cadence = duration_seconds(self.cadence)
|
|
118
|
+
if total <= 0:
|
|
119
|
+
raise SchemaValidationError("total_timeout", "must be > 0")
|
|
120
|
+
if default_cadence < 0:
|
|
121
|
+
raise SchemaValidationError("cadence", "must be >= 0")
|
|
122
|
+
visited: set[str] = set()
|
|
123
|
+
for source in self.sources:
|
|
124
|
+
if source.source_id in visited:
|
|
125
|
+
raise SchemaValidationError(
|
|
126
|
+
"sources",
|
|
127
|
+
f"duplicate source_id {source.source_id!r}",
|
|
128
|
+
)
|
|
129
|
+
visited.add(source.source_id)
|
|
130
|
+
if duration_seconds(source.timeout) > total:
|
|
131
|
+
raise SchemaValidationError(
|
|
132
|
+
f"sources[{source.source_id}]",
|
|
133
|
+
f"timeout {duration_seconds(source.timeout)}s exceeds total_timeout {total}s",
|
|
134
|
+
)
|
|
135
|
+
cadence = getattr(source, "cadence", "0s")
|
|
136
|
+
if duration_seconds(cadence) > 0 and duration_seconds(cadence) > total:
|
|
137
|
+
raise SchemaValidationError(
|
|
138
|
+
f"sources[{source.source_id}]",
|
|
139
|
+
f"cadence {duration_seconds(cadence)}s exceeds total_timeout {total}s; "
|
|
140
|
+
"no sampling window remains",
|
|
141
|
+
)
|
|
142
|
+
return self
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def empty(self) -> bool:
|
|
146
|
+
return not self.sources
|