mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Shared outcome/state taxonomy (ADR-M1-3 Phase 1.4).
|
|
2
|
+
|
|
3
|
+
The outcome vocabulary is a single source of truth that the executor,
|
|
4
|
+
planner, CLI exit mapping, recovery state machine, and a future drift
|
|
5
|
+
*publisher event* all name the same way.
|
|
6
|
+
|
|
7
|
+
The three constants are deliberately siblings with sharply distinct
|
|
8
|
+
semantics:
|
|
9
|
+
|
|
10
|
+
* :data:`TARGET_DRIFT` — the plan targeted an identity that no longer exists.
|
|
11
|
+
The planned ``RuntimeIdentity`` does not match the live one. A drifted
|
|
12
|
+
target is *mismatched*, never failed.
|
|
13
|
+
* :data:`FAILED_TO_APPLY` — a capability/permission/mutation failure on a
|
|
14
|
+
*present* target.
|
|
15
|
+
* :data:`RESOURCE_CONFLICT` — ownership/lease contention on a *present*
|
|
16
|
+
target.
|
|
17
|
+
|
|
18
|
+
``TARGET_DRIFT`` is declared here (M1) but *detected* in M2. This module
|
|
19
|
+
ships no detection logic; it only fixes the shared vocabulary.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from enum import StrEnum
|
|
26
|
+
from typing import TYPE_CHECKING, Protocol
|
|
27
|
+
|
|
28
|
+
if TYPE_CHECKING:
|
|
29
|
+
from mayhem.domain.identity import RuntimeIdentity
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class StepOutcome(StrEnum):
|
|
33
|
+
"""Lead outcome for a single step at execution time (ADR-M1-3).
|
|
34
|
+
|
|
35
|
+
Values are exactly the ``step_runs.status`` vocabulary admitted by the
|
|
36
|
+
persisted schema (migration M0006): ``completed``, ``failed``,
|
|
37
|
+
``skipped``, ``cancelled``, ``bypassed``, and ``target_drift``.
|
|
38
|
+
:attr:`TARGET_DRIFT` is a first-class persisted step state — it is *not* a
|
|
39
|
+
failure of injection, not a dirty state, and not ownership contention; it
|
|
40
|
+
means the object the plan intended to touch is no longer the object
|
|
41
|
+
present.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
COMPLETED = "completed"
|
|
45
|
+
FAILED = "failed"
|
|
46
|
+
TARGET_DRIFT = "target_drift"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class TargetOutcome(StrEnum):
|
|
50
|
+
"""Why a particular fault never reached its target (failure taxonomy).
|
|
51
|
+
|
|
52
|
+
These classify the *reason* a step did not apply to its target. They
|
|
53
|
+
refine the raw ``failed`` step outcome, and are the vocabulary used to
|
|
54
|
+
distinguish drift from capability failures from contention:
|
|
55
|
+
|
|
56
|
+
* :attr:`FAILED_TO_APPLY` — capability/permission/mutation failure on a
|
|
57
|
+
*present* target.
|
|
58
|
+
* :attr:`RESOURCE_CONFLICT` — ownership/lease contention on a *present*
|
|
59
|
+
target.
|
|
60
|
+
* :attr:`TARGET_DRIFT` — the planned identity no longer matches the live
|
|
61
|
+
identity; the target was *mismatched, not failed*.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
FAILED_TO_APPLY = "failed_to_apply"
|
|
65
|
+
RESOURCE_CONFLICT = "resource_conflict"
|
|
66
|
+
TARGET_DRIFT = "target_drift"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass(frozen=True)
|
|
70
|
+
class DriftEvent:
|
|
71
|
+
"""Payload of a future drift publisher event (declared M1, fired M2).
|
|
72
|
+
|
|
73
|
+
No detection logic ships here; this dataclass is the contract consumed by
|
|
74
|
+
a publisher callback so recovery, reporting, and the CLI all agree on the
|
|
75
|
+
shape of a drift notification.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
run_id: str
|
|
79
|
+
step_id: str
|
|
80
|
+
planned: RuntimeIdentity
|
|
81
|
+
live: RuntimeIdentity
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class DriftPublisher(Protocol):
|
|
85
|
+
"""Callback signature for the drift event publisher (ADR-M1-3).
|
|
86
|
+
|
|
87
|
+
A conforming publisher accepts a :class:`DriftEvent` and returns
|
|
88
|
+
``None``. The controller may use this to emit an observation, persist a
|
|
89
|
+
recovery hint, or notify an external surface — enforcement is wired in M2.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def __call__(self, event: DriftEvent) -> None: ...
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Remote-agent interface contract (ADR-M3-5).
|
|
2
|
+
|
|
3
|
+
Defines the seam a remote execution transport must satisfy. **No transport is
|
|
4
|
+
implemented in this milestone** — a remote target in a spec fails planning with
|
|
5
|
+
``UNSUPPORTED`` until a future milestone implements a transport (Q11).
|
|
6
|
+
|
|
7
|
+
The methods below describe the *capability handshake* and *tool run* lifecycle.
|
|
8
|
+
Because no SSH/agent transport exists yet, ``capabilities()`` returns all
|
|
9
|
+
verdicts as ``UNSUPPORTED`` so the planner can refuse remote plans with a clear
|
|
10
|
+
message rather than fail unpredictably mid-run.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Protocol, runtime_checkable
|
|
16
|
+
|
|
17
|
+
from mayhem.domain.runtime_adapter import (
|
|
18
|
+
CapabilityRequirements,
|
|
19
|
+
CapabilityVerdict,
|
|
20
|
+
VerdictResult,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@runtime_checkable
|
|
25
|
+
class RemoteAgentInterface(Protocol):
|
|
26
|
+
"""Contract for a remote execution agent (ADR-M3-5).
|
|
27
|
+
|
|
28
|
+
Implementations provide a transport (SSH, WireGuard, managed agent, ...).
|
|
29
|
+
This milestone ships the interface only.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
def connect(self) -> None: ...
|
|
33
|
+
|
|
34
|
+
def capability_handshake(self, reqs: CapabilityRequirements) -> VerdictResult: ...
|
|
35
|
+
|
|
36
|
+
def target_resolution(self, target_id: str) -> str: ...
|
|
37
|
+
|
|
38
|
+
def tool_run(self, command: list[str], *, timeout_s: float) -> str: ...
|
|
39
|
+
|
|
40
|
+
def cancel(self, run_id: str) -> None: ...
|
|
41
|
+
|
|
42
|
+
def teardown(self) -> None: ...
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class RemoteAgentAdapter:
|
|
46
|
+
"""Placeholder adapter so the planner can statically refuse remote plans.
|
|
47
|
+
|
|
48
|
+
``evaluate`` returns UNSUPPORTED for every requirement, which is what the
|
|
49
|
+
planner uses to reject remote execution with a clear message.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(self, remote_id: str = "remote") -> None:
|
|
53
|
+
self._remote_id = remote_id
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def id(self) -> str:
|
|
57
|
+
return self._remote_id
|
|
58
|
+
|
|
59
|
+
def evaluate(self, reqs: CapabilityRequirements) -> VerdictResult:
|
|
60
|
+
blocking = True
|
|
61
|
+
verdicts = {
|
|
62
|
+
"remote_execution": CapabilityVerdict.UNSUPPORTED.value,
|
|
63
|
+
"transport": CapabilityVerdict.UNSUPPORTED.value,
|
|
64
|
+
}
|
|
65
|
+
return VerdictResult(
|
|
66
|
+
engine=self._remote_id,
|
|
67
|
+
requirements=reqs,
|
|
68
|
+
verdicts=verdicts,
|
|
69
|
+
blocking=blocking,
|
|
70
|
+
)
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Resource ownership and conflict management (ADR-0015).
|
|
2
|
+
|
|
3
|
+
Every fault that mutates system state must declare exactly which resources it
|
|
4
|
+
creates or modifies. ``TrackedResource`` is the canonical record; the
|
|
5
|
+
``ResourceOwnershipGraph`` provides in-memory conflict detection; and
|
|
6
|
+
``ResourceManager`` persists + recovers through the SQLite store.
|
|
7
|
+
|
|
8
|
+
Design principles:
|
|
9
|
+
* ownership is scoped to (run_id, step_id, fault_id)
|
|
10
|
+
* recovery is ownership-aware — experiment A never touches experiment B's resources
|
|
11
|
+
* every resource must have a verify probe — recovery is not complete until verified
|
|
12
|
+
* orphan detection runs through the Janitor which inspects the ownership graph
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from datetime import datetime
|
|
18
|
+
from enum import StrEnum
|
|
19
|
+
|
|
20
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
21
|
+
|
|
22
|
+
from mayhem.domain.common import utc_now
|
|
23
|
+
from mayhem.domain.leases import UndoOp, VerifyProbe
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ResourceType(StrEnum):
|
|
27
|
+
"""Kinds of system resources a fault may create or modifies."""
|
|
28
|
+
|
|
29
|
+
TC_RULE = "tc_rule"
|
|
30
|
+
IPTABLES_RULE = "iptables_rule"
|
|
31
|
+
NFTABLES_RULE = "nftables_rule"
|
|
32
|
+
PROCESS_SIGNAL = "process_signal"
|
|
33
|
+
CGROUP_LIMIT = "cgroup_limit"
|
|
34
|
+
TOXIPROXY_TOXIC = "toxiproxy_toxic"
|
|
35
|
+
CONTAINER_STATE = "container_state"
|
|
36
|
+
TEMPORARY_FILE = "temporary_file"
|
|
37
|
+
LOAD_GENERATOR = "load_generator"
|
|
38
|
+
NETWORK_NAMESPACE = "network_namespace"
|
|
39
|
+
NETWORK_FAULT = "network_fault"
|
|
40
|
+
FILESYSTEM_MOUNT = "filesystem_mount"
|
|
41
|
+
PROCESS_SPAWN = "process_spawn"
|
|
42
|
+
RESOURCE_LIMIT = "resource_limit"
|
|
43
|
+
GENERIC = "generic"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class ResourceState(StrEnum):
|
|
47
|
+
"""Lifecycle of a tracked resource."""
|
|
48
|
+
|
|
49
|
+
PENDING = "pending" # declared, not yet created
|
|
50
|
+
ACTIVE = "active" # created and tracked
|
|
51
|
+
RECOVERING = "recovering" # cleanup in progress
|
|
52
|
+
RECOVERED = "recovered" # verified cleaned
|
|
53
|
+
ORPHANED = "orphaned" # owner experiment lost
|
|
54
|
+
DIRTY = "dirty" # cleanup failed — needs operator
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class ConflictKind(StrEnum):
|
|
58
|
+
"""Classification of resource conflicts between experiments."""
|
|
59
|
+
|
|
60
|
+
COEXIST = "coexist" # different resource types on same target — OK
|
|
61
|
+
SERIALIZE = "serialize" # same resource type on same target — must not overlap
|
|
62
|
+
REJECT = "reject" # irreconcilable — one experiment must not run
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class TrackedResource(BaseModel):
|
|
66
|
+
"""A single resource created or modified by a fault injection.
|
|
67
|
+
|
|
68
|
+
Immutable after creation except for state transitions managed by
|
|
69
|
+
``ResourceManager``.
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
model_config = ConfigDict(frozen=True)
|
|
73
|
+
|
|
74
|
+
id: str # uuid4
|
|
75
|
+
resource_type: ResourceType
|
|
76
|
+
owner_run_id: str
|
|
77
|
+
owner_step_id: str
|
|
78
|
+
owner_fault_id: str
|
|
79
|
+
state: ResourceState = ResourceState.PENDING
|
|
80
|
+
target_identity: str # immutable reference to the target node
|
|
81
|
+
cleanup_op: UndoOp
|
|
82
|
+
verify_probe: VerifyProbe
|
|
83
|
+
created_at: datetime = Field(default_factory=utc_now)
|
|
84
|
+
recovered_at: datetime | None = None
|
|
85
|
+
metadata: dict[str, object] = Field(default_factory=dict)
|
|
86
|
+
fingerprint: str = "" # ADR-M3-7 fault-path fingerprint
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class MutationJournalEntry(BaseModel):
|
|
90
|
+
"""ADR-M2 Phase 2.6 — mutation-boundary journal entry.
|
|
91
|
+
|
|
92
|
+
Recorded at the exact boundary when the last undo-fallible op is applied
|
|
93
|
+
(injection succeeds), not as a post-hoc probe result. Carries the resource
|
|
94
|
+
owner, the mutation's defining op (e.g. ``kill -9``), and the lease
|
|
95
|
+
reference so downstream can tell *which* lease holds the mutation and
|
|
96
|
+
*what* actually mutated the system.
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
model_config = ConfigDict(frozen=True)
|
|
100
|
+
|
|
101
|
+
id: str # uuid4
|
|
102
|
+
lease_id: str
|
|
103
|
+
resource_id: str
|
|
104
|
+
run_id: str
|
|
105
|
+
step_id: str
|
|
106
|
+
fault_id: str
|
|
107
|
+
defining_op: UndoOp # the op that actually mutated the system
|
|
108
|
+
target_identity: str
|
|
109
|
+
journaled_at: datetime = Field(default_factory=utc_now)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class ResourceConflict(BaseModel):
|
|
113
|
+
"""A detected conflict between a new resource and an existing one."""
|
|
114
|
+
|
|
115
|
+
model_config = ConfigDict(frozen=True)
|
|
116
|
+
|
|
117
|
+
kind: ConflictKind
|
|
118
|
+
new_resource: TrackedResource
|
|
119
|
+
existing_resource: TrackedResource
|
|
120
|
+
reason: str
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class RecoveryResult(BaseModel):
|
|
124
|
+
"""Outcome of a single resource recovery attempt."""
|
|
125
|
+
|
|
126
|
+
model_config = ConfigDict(frozen=True)
|
|
127
|
+
|
|
128
|
+
resource_id: str
|
|
129
|
+
success: bool
|
|
130
|
+
verified: bool = False
|
|
131
|
+
error: str | None = None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class ResourceOwnershipGraph:
|
|
135
|
+
"""In-memory graph of active resource ownership.
|
|
136
|
+
|
|
137
|
+
Not a persistence layer — the ``ResourceManager`` stores rows in SQLite
|
|
138
|
+
and loads them into this graph at startup. The graph provides fast
|
|
139
|
+
conflict detection and ownership queries without touching the database
|
|
140
|
+
on every check.
|
|
141
|
+
"""
|
|
142
|
+
|
|
143
|
+
def __init__(self) -> None:
|
|
144
|
+
self._resources: dict[str, TrackedResource] = {}
|
|
145
|
+
# target_identity -> set of resource ids for fast conflict lookup
|
|
146
|
+
self._by_target: dict[str, set[str]] = {}
|
|
147
|
+
|
|
148
|
+
def add(self, resource: TrackedResource) -> None:
|
|
149
|
+
"""Register a resource in the graph."""
|
|
150
|
+
self._resources[resource.id] = resource
|
|
151
|
+
by_target = self._by_target.setdefault(resource.target_identity, set())
|
|
152
|
+
by_target.add(resource.id)
|
|
153
|
+
|
|
154
|
+
def remove(self, resource_id: str) -> TrackedResource | None:
|
|
155
|
+
"""Remove a resource from the graph. Returns the resource if present."""
|
|
156
|
+
resource = self._resources.pop(resource_id, None)
|
|
157
|
+
if resource is None:
|
|
158
|
+
return None
|
|
159
|
+
by_target = self._by_target.get(resource.target_identity)
|
|
160
|
+
if by_target:
|
|
161
|
+
by_target.discard(resource_id)
|
|
162
|
+
if not by_target:
|
|
163
|
+
del self._by_target[resource.target_identity]
|
|
164
|
+
return resource
|
|
165
|
+
|
|
166
|
+
def get(self, resource_id: str) -> TrackedResource | None:
|
|
167
|
+
return self._resources.get(resource_id)
|
|
168
|
+
|
|
169
|
+
def active_for_target(self, target_identity: str) -> list[TrackedResource]:
|
|
170
|
+
"""All active/pending resources on a given target."""
|
|
171
|
+
ids = self._by_target.get(target_identity, set())
|
|
172
|
+
return [
|
|
173
|
+
r
|
|
174
|
+
for r in (self._resources.get(rid) for rid in ids)
|
|
175
|
+
if r is not None and r.state in (ResourceState.ACTIVE, ResourceState.PENDING)
|
|
176
|
+
]
|
|
177
|
+
|
|
178
|
+
def active_for_run(self, run_id: str) -> list[TrackedResource]:
|
|
179
|
+
"""All active/pending resources owned by a specific run."""
|
|
180
|
+
return [
|
|
181
|
+
r
|
|
182
|
+
for r in self._resources.values()
|
|
183
|
+
if r.owner_run_id == run_id and r.state in (ResourceState.ACTIVE, ResourceState.PENDING)
|
|
184
|
+
]
|
|
185
|
+
|
|
186
|
+
def orphans(self) -> list[TrackedResource]:
|
|
187
|
+
"""Resources that have no active owner run (stale after controller crash)."""
|
|
188
|
+
return [
|
|
189
|
+
r
|
|
190
|
+
for r in self._resources.values()
|
|
191
|
+
if r.state in (ResourceState.ACTIVE, ResourceState.PENDING, ResourceState.RECOVERING)
|
|
192
|
+
]
|
|
193
|
+
|
|
194
|
+
@property
|
|
195
|
+
def size(self) -> int:
|
|
196
|
+
return len(self._resources)
|
|
197
|
+
|
|
198
|
+
def detect_conflicts(self, new_resource: TrackedResource) -> list[ResourceConflict]:
|
|
199
|
+
"""Check whether a proposed resource conflicts with existing ones.
|
|
200
|
+
|
|
201
|
+
Conflict rules:
|
|
202
|
+
* Same resource_type + same target → SERIALIZE
|
|
203
|
+
* Different resource_type + same target → COEXIST
|
|
204
|
+
* Irreconcilable (e.g. tc qdisc + tc class on same device) → REJECT
|
|
205
|
+
"""
|
|
206
|
+
conflicts: list[ResourceConflict] = []
|
|
207
|
+
existing = self.active_for_target(new_resource.target_identity)
|
|
208
|
+
|
|
209
|
+
for ex in existing:
|
|
210
|
+
if ex.owner_run_id == new_resource.owner_run_id:
|
|
211
|
+
continue # same run — no conflict
|
|
212
|
+
if ex.resource_type == new_resource.resource_type:
|
|
213
|
+
conflicts.append(
|
|
214
|
+
ResourceConflict(
|
|
215
|
+
kind=ConflictKind.SERIALIZE,
|
|
216
|
+
new_resource=new_resource,
|
|
217
|
+
existing_resource=ex,
|
|
218
|
+
reason=(
|
|
219
|
+
f"same resource type '{new_resource.resource_type.value}' "
|
|
220
|
+
f"on target '{new_resource.target_identity}' "
|
|
221
|
+
f"owned by run '{ex.owner_run_id}'"
|
|
222
|
+
),
|
|
223
|
+
)
|
|
224
|
+
)
|
|
225
|
+
else:
|
|
226
|
+
conflicts.append(
|
|
227
|
+
ResourceConflict(
|
|
228
|
+
kind=ConflictKind.COEXIST,
|
|
229
|
+
new_resource=new_resource,
|
|
230
|
+
existing_resource=ex,
|
|
231
|
+
reason=(
|
|
232
|
+
f"different resource type '{new_resource.resource_type.value}' "
|
|
233
|
+
f"coexists with '{ex.resource_type.value}' "
|
|
234
|
+
f"on target '{new_resource.target_identity}'"
|
|
235
|
+
),
|
|
236
|
+
)
|
|
237
|
+
)
|
|
238
|
+
return conflicts
|
|
239
|
+
|
|
240
|
+
def has_serializable_conflicts(self, new_resource: TrackedResource) -> bool:
|
|
241
|
+
"""True if a SERIALIZE or REJECT conflict exists."""
|
|
242
|
+
return any(
|
|
243
|
+
c.kind in (ConflictKind.SERIALIZE, ConflictKind.REJECT)
|
|
244
|
+
for c in self.detect_conflicts(new_resource)
|
|
245
|
+
)
|
mayhem/domain/risks.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Risk ladder and environment classification (ADR-0012).
|
|
2
|
+
|
|
3
|
+
The ladder is *ordered* so gates can compare; enforcement lives in the safety
|
|
4
|
+
engine, but the ordering itself is a domain fact.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from enum import StrEnum
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class RiskLevel(StrEnum):
|
|
13
|
+
"""Ordered severity ladder: low < medium < high < critical."""
|
|
14
|
+
|
|
15
|
+
LOW = "low"
|
|
16
|
+
MEDIUM = "medium"
|
|
17
|
+
HIGH = "high"
|
|
18
|
+
CRITICAL = "critical"
|
|
19
|
+
|
|
20
|
+
@property
|
|
21
|
+
def rank(self) -> int:
|
|
22
|
+
return _RANKS[self]
|
|
23
|
+
|
|
24
|
+
def at_least(self, other: RiskLevel) -> bool:
|
|
25
|
+
"""True when this level is greater than or equal to ``other``."""
|
|
26
|
+
return self.rank >= other.rank
|
|
27
|
+
|
|
28
|
+
def exceeds(self, ceiling: RiskLevel) -> bool:
|
|
29
|
+
"""True when this level is strictly above the policy ceiling."""
|
|
30
|
+
return self.rank > ceiling.rank
|
|
31
|
+
|
|
32
|
+
def next_higher(self) -> RiskLevel:
|
|
33
|
+
"""One step up the ladder; critical returns itself."""
|
|
34
|
+
return _NEXT_HIGHER[self]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_RANKS: dict[RiskLevel, int] = {
|
|
38
|
+
RiskLevel.LOW: 0,
|
|
39
|
+
RiskLevel.MEDIUM: 1,
|
|
40
|
+
RiskLevel.HIGH: 2,
|
|
41
|
+
RiskLevel.CRITICAL: 3,
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
_NEXT_HIGHER: dict[RiskLevel, RiskLevel] = {
|
|
45
|
+
RiskLevel.LOW: RiskLevel.MEDIUM,
|
|
46
|
+
RiskLevel.MEDIUM: RiskLevel.HIGH,
|
|
47
|
+
RiskLevel.HIGH: RiskLevel.CRITICAL,
|
|
48
|
+
RiskLevel.CRITICAL: RiskLevel.CRITICAL,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class EnvironmentClass(StrEnum):
|
|
53
|
+
"""Deployment class driving safety defaults (ADR-0012 §3)."""
|
|
54
|
+
|
|
55
|
+
DEV = "dev"
|
|
56
|
+
STAGING = "staging"
|
|
57
|
+
PRODUCTION = "production"
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def is_production(self) -> bool:
|
|
61
|
+
return self is EnvironmentClass.PRODUCTION
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Run and Outcome domain models (ADR-M5-1).
|
|
2
|
+
|
|
3
|
+
A ``RunRecord`` is *what was executed* — the experiment spec, faults, groups,
|
|
4
|
+
cancellation, journal refs, verdict, duration, and tool evidence.
|
|
5
|
+
|
|
6
|
+
An ``Outcome`` is *what happened* — observed post-run system state, checks
|
|
7
|
+
passed/failed, metric deltas, residual effect, and stability/recovery signal.
|
|
8
|
+
|
|
9
|
+
They persist separately and link by a ``run_id → outcome`` reference.
|
|
10
|
+
A ``RunRecord`` is never conflated with its ``Outcome``.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from datetime import datetime
|
|
17
|
+
from enum import StrEnum
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class RunVerdict(StrEnum):
|
|
22
|
+
"""Top-level verdict for a completed run."""
|
|
23
|
+
|
|
24
|
+
PASS = "pass"
|
|
25
|
+
FAIL = "fail"
|
|
26
|
+
ERROR = "error"
|
|
27
|
+
ABORTED = "aborted"
|
|
28
|
+
BYPASSED = "bypassed"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class RunStatus(StrEnum):
|
|
32
|
+
"""Lifecycle status of a run."""
|
|
33
|
+
|
|
34
|
+
PENDING = "pending"
|
|
35
|
+
RUNNING = "running"
|
|
36
|
+
COMPLETED = "completed"
|
|
37
|
+
FAILED = "failed"
|
|
38
|
+
ABORTED = "aborted"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class RunRecord:
|
|
43
|
+
"""What was executed in a single drill run.
|
|
44
|
+
|
|
45
|
+
Persisted separately from its ``Outcome`` and linked by ``run_id``.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
run_id: str
|
|
49
|
+
experiment_name: str
|
|
50
|
+
spec_json: str
|
|
51
|
+
plan_json: str
|
|
52
|
+
seed: int | None = None
|
|
53
|
+
status: RunStatus = RunStatus.PENDING
|
|
54
|
+
environment_fingerprint: str = ""
|
|
55
|
+
config_snapshot_id: str = ""
|
|
56
|
+
started_at: str = ""
|
|
57
|
+
ended_at: str = ""
|
|
58
|
+
description: str = ""
|
|
59
|
+
verdict: RunVerdict = RunVerdict.PASS
|
|
60
|
+
tags: tuple[str, ...] = ()
|
|
61
|
+
extra: dict[str, Any] = field(default_factory=dict)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def wall_seconds(self) -> float:
|
|
65
|
+
"""Duration in seconds (requires both timestamps set)."""
|
|
66
|
+
if not self.started_at or not self.ended_at:
|
|
67
|
+
return 0.0
|
|
68
|
+
try:
|
|
69
|
+
start = datetime.fromisoformat(self.started_at)
|
|
70
|
+
end = datetime.fromisoformat(self.ended_at)
|
|
71
|
+
return max(0.0, (end - start).total_seconds())
|
|
72
|
+
except ValueError:
|
|
73
|
+
return 0.0
|
|
74
|
+
|
|
75
|
+
def summary_md(self) -> str:
|
|
76
|
+
"""Markdown summary of the run."""
|
|
77
|
+
lines = [
|
|
78
|
+
f"# Run {self.run_id}",
|
|
79
|
+
"",
|
|
80
|
+
f"**experiment**: {self.experiment_name}",
|
|
81
|
+
f"**status**: {self.status.value}",
|
|
82
|
+
f"**verdict**: {self.verdict.value}",
|
|
83
|
+
]
|
|
84
|
+
if self.started_at:
|
|
85
|
+
lines.append(f"**started**: {self.started_at}")
|
|
86
|
+
if self.ended_at:
|
|
87
|
+
lines.append(f"**ended**: {self.ended_at}")
|
|
88
|
+
if self.seed is not None:
|
|
89
|
+
lines.append(f"**seed**: {self.seed}")
|
|
90
|
+
if self.tags:
|
|
91
|
+
lines.append(f"**tags**: {', '.join(self.tags)}")
|
|
92
|
+
if self.description:
|
|
93
|
+
lines.append("")
|
|
94
|
+
lines.append(self.description)
|
|
95
|
+
return "\n".join(lines)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@dataclass(frozen=True)
|
|
99
|
+
class Outcome:
|
|
100
|
+
"""What happened after a run — observed post-run system state.
|
|
101
|
+
|
|
102
|
+
Persisted separately from the ``RunRecord`` and linked by ``run_id``.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
run_id: str
|
|
106
|
+
body_json: str = "{}"
|
|
107
|
+
body_hash: str = ""
|
|
108
|
+
checks_passed: int = 0
|
|
109
|
+
checks_failed: int = 0
|
|
110
|
+
metric_deltas: dict[str, float] = field(default_factory=dict)
|
|
111
|
+
residual_effect: str = ""
|
|
112
|
+
stability_signal: str = ""
|
|
113
|
+
extra: dict[str, Any] = field(default_factory=dict)
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def all_checks_passed(self) -> bool:
|
|
117
|
+
"""True when there were checks and every one passed."""
|
|
118
|
+
return self.checks_passed > 0 and self.checks_failed == 0
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def total_checks(self) -> int:
|
|
122
|
+
return self.checks_passed + self.checks_failed
|
|
123
|
+
|
|
124
|
+
def summary_md(self) -> str:
|
|
125
|
+
"""Markdown summary of the outcome."""
|
|
126
|
+
total = self.total_checks
|
|
127
|
+
if total == 0:
|
|
128
|
+
status_str = "no checks"
|
|
129
|
+
elif self.all_checks_passed:
|
|
130
|
+
status_str = f"all {total} checks passed"
|
|
131
|
+
else:
|
|
132
|
+
status_str = f"{self.checks_failed}/{total} checks failed"
|
|
133
|
+
lines = [
|
|
134
|
+
f"# Outcome for Run {self.run_id}",
|
|
135
|
+
"",
|
|
136
|
+
f"**checks**: {status_str}",
|
|
137
|
+
]
|
|
138
|
+
if self.metric_deltas:
|
|
139
|
+
lines.append("**metric deltas**:")
|
|
140
|
+
for name, delta in self.metric_deltas.items():
|
|
141
|
+
lines.append(f" - {name}: {delta:+.4f}")
|
|
142
|
+
if self.residual_effect:
|
|
143
|
+
lines.append(f"**residual**: {self.residual_effect}")
|
|
144
|
+
if self.stability_signal:
|
|
145
|
+
lines.append(f"**stability**: {self.stability_signal}")
|
|
146
|
+
return "\n".join(lines)
|