mayhem-cli 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mayhem/agent/__init__.py +1 -0
- mayhem/agent/cli.py +36 -0
- mayhem/agents/__init__.py +1 -0
- mayhem/agents/capabilities.py +106 -0
- mayhem/agents/executors.py +430 -0
- mayhem/agents/impact.py +729 -0
- mayhem/agents/lease_client.py +141 -0
- mayhem/agents/probes.py +284 -0
- mayhem/agents/protocol.py +134 -0
- mayhem/agents/server.py +281 -0
- mayhem/agents/sinks.py +60 -0
- mayhem/agents/transports.py +134 -0
- mayhem/agents/watchdog.py +140 -0
- mayhem/cli/__init__.py +11 -0
- mayhem/cli/app.py +154 -0
- mayhem/cli/campaign.py +496 -0
- mayhem/cli/config_cmd.py +47 -0
- mayhem/cli/context.py +23 -0
- mayhem/cli/dependency.py +429 -0
- mayhem/cli/exit_codes.py +24 -0
- mayhem/cli/experiment.py +24 -0
- mayhem/cli/lifecycle.py +805 -0
- mayhem/cli/resolver.py +72 -0
- mayhem/cli/services.py +459 -0
- mayhem/cli/style.py +101 -0
- mayhem/cli/toolkit.py +41 -0
- mayhem/cli/topology.py +127 -0
- mayhem/config.py +208 -0
- mayhem/controller/__init__.py +1 -0
- mayhem/controller/compensation.py +2156 -0
- mayhem/controller/executor.py +1719 -0
- mayhem/controller/janitor.py +196 -0
- mayhem/controller/observability_collector.py +382 -0
- mayhem/controller/observations.py +102 -0
- mayhem/controller/planner.py +715 -0
- mayhem/controller/recovery.py +245 -0
- mayhem/controller/resilience_report.py +585 -0
- mayhem/controller/resource_manager.py +457 -0
- mayhem/controller/safety.py +392 -0
- mayhem/domain/__init__.py +6 -0
- mayhem/domain/campaigns.py +118 -0
- mayhem/domain/cancellation.py +110 -0
- mayhem/domain/candidates.py +101 -0
- mayhem/domain/capabilities.py +86 -0
- mayhem/domain/catalog.py +727 -0
- mayhem/domain/checks.py +173 -0
- mayhem/domain/common.py +104 -0
- mayhem/domain/coverage.py +106 -0
- mayhem/domain/decisions.py +57 -0
- mayhem/domain/errors.py +87 -0
- mayhem/domain/events.py +61 -0
- mayhem/domain/execution_context.py +120 -0
- mayhem/domain/execution_loci.py +94 -0
- mayhem/domain/experiments.py +370 -0
- mayhem/domain/faults.py +239 -0
- mayhem/domain/identity.py +200 -0
- mayhem/domain/k8s_adapter.py +132 -0
- mayhem/domain/leases.py +186 -0
- mayhem/domain/load_strategy.py +98 -0
- mayhem/domain/m5_campaign.py +120 -0
- mayhem/domain/maniac.py +93 -0
- mayhem/domain/observability.py +146 -0
- mayhem/domain/outcomes.py +92 -0
- mayhem/domain/remote_agent_interface.py +70 -0
- mayhem/domain/resources.py +245 -0
- mayhem/domain/risks.py +61 -0
- mayhem/domain/run_outcome.py +146 -0
- mayhem/domain/runtime_adapter.py +256 -0
- mayhem/domain/success.py +329 -0
- mayhem/domain/topology.py +452 -0
- mayhem/infra/__init__.py +1 -0
- mayhem/infra/campaign_engine.py +205 -0
- mayhem/infra/candidate_gates.py +124 -0
- mayhem/infra/candidate_generator.py +110 -0
- mayhem/infra/coverage_repository.py +101 -0
- mayhem/infra/lease_repository.py +129 -0
- mayhem/infra/maniac.py +103 -0
- mayhem/infra/migrations.py +596 -0
- mayhem/infra/migrator.py +149 -0
- mayhem/infra/report.py +227 -0
- mayhem/infra/store.py +200 -0
- mayhem/py.typed +0 -0
- mayhem/spec.py +52 -0
- mayhem/toolkit/__init__.py +1 -0
- mayhem/toolkit/fingerprint.py +69 -0
- mayhem/toolkit/hashing.py +32 -0
- mayhem/toolkit/manifests/docker.yaml +11 -0
- mayhem/toolkit/manifests/podman.yaml +11 -0
- mayhem/toolkit/manifests/stress-ng.yaml +11 -0
- mayhem/toolkit/manifests/tc-netem.yaml +11 -0
- mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
- mayhem/toolkit/registry.py +185 -0
- mayhem/toolkit/tool_runner.py +129 -0
- mayhem/topology/__init__.py +10 -0
- mayhem/topology/providers/__init__.py +0 -0
- mayhem/topology/providers/adapter_registry.py +60 -0
- mayhem/topology/providers/base.py +31 -0
- mayhem/topology/providers/compose.py +207 -0
- mayhem/topology/providers/docker_adapter.py +277 -0
- mayhem/topology/providers/docker_runtime.py +461 -0
- mayhem/topology/providers/podman_adapter.py +328 -0
- mayhem/topology/resolve.py +196 -0
- mayhem/topology/service.py +158 -0
- mayhem_cli-0.5.1.dist-info/METADATA +555 -0
- mayhem_cli-0.5.1.dist-info/RECORD +107 -0
- mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
- mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""Janitor — background sweep that enforces lease TTLs.
|
|
2
|
+
|
|
3
|
+
A fault left past its TTL is by definition unattended. PENDING leases are
|
|
4
|
+
expired (never injected, nothing to undo); ACTIVE leases are orphaned and
|
|
5
|
+
then released through their write-ahead undo contract; ORPHANED/RELEASING
|
|
6
|
+
leases stuck mid-compensation are finalized; DIRTY leases (compensation
|
|
7
|
+
already failed) are surrendered to EXPIRED — the janitor never leaves a
|
|
8
|
+
fault running because its owner vanished.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import contextlib
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import TYPE_CHECKING
|
|
16
|
+
|
|
17
|
+
from mayhem.domain.common import utc_now
|
|
18
|
+
from mayhem.domain.errors import DomainError
|
|
19
|
+
from mayhem.domain.leases import LeaseState
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
from collections.abc import Callable
|
|
23
|
+
from datetime import datetime
|
|
24
|
+
|
|
25
|
+
from mayhem.agents.sinks import LeaseSink
|
|
26
|
+
from mayhem.domain.leases import FaultLease
|
|
27
|
+
|
|
28
|
+
_LOST_OWNER = "owner run no longer live; reclaimed before TTL"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _owner_gone(run_liveness: Callable[[str], bool | None], run_id: str) -> bool:
|
|
32
|
+
"""True only when the resolver can *prove* the owning controller is gone.
|
|
33
|
+
|
|
34
|
+
Unknown (``None``) or an absent owner row never triggers the early reclaim —
|
|
35
|
+
TTL policy stays in charge of those, and a LiveLonger cycling pid can only
|
|
36
|
+
read "alive", which delays cleanup rather than wrongly reclaiming a lease
|
|
37
|
+
a live owner still needs.
|
|
38
|
+
"""
|
|
39
|
+
try:
|
|
40
|
+
verdict = run_liveness(run_id)
|
|
41
|
+
except LookupError:
|
|
42
|
+
return False
|
|
43
|
+
return verdict is False
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class SweepResult:
|
|
48
|
+
expired: tuple[str, ...]
|
|
49
|
+
recovered: tuple[str, ...] # orphaned -> released via undo path marker
|
|
50
|
+
dirty: tuple[str, ...]
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def quiet(self) -> bool:
|
|
54
|
+
return not (self.expired or self.recovered or self.dirty)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class Janitor:
|
|
58
|
+
"""TTL enforcement over any LeaseSink; state transitions only.
|
|
59
|
+
|
|
60
|
+
``run_liveness`` is an optional resolver (run_id -> bool | None) the
|
|
61
|
+
caller supplies when it can see the runs table. It returns ``True`` while
|
|
62
|
+
the owning controller is alive, ``False`` when the owner is provably gone,
|
|
63
|
+
and ``None`` when unknown. A lease whose owner is provably gone is
|
|
64
|
+
reclaimed *before* its TTL — without this, a crashed ``run`` wedges its
|
|
65
|
+
targets for the whole TTL and the next ``run`` conflicts with a lease the
|
|
66
|
+
janitor "did nothing about".
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
def __init__(self, sink: LeaseSink) -> None:
|
|
70
|
+
self._sink = sink
|
|
71
|
+
|
|
72
|
+
def sweep(
|
|
73
|
+
self,
|
|
74
|
+
*,
|
|
75
|
+
now_epoch_s: float | None = None,
|
|
76
|
+
run_liveness: Callable[[str], bool | None] | None = None,
|
|
77
|
+
) -> SweepResult:
|
|
78
|
+
now = utc_now()
|
|
79
|
+
current = now.timestamp() if now_epoch_s is None else now_epoch_s
|
|
80
|
+
expired: list[str] = []
|
|
81
|
+
recovered: list[str] = []
|
|
82
|
+
dirty: list[str] = []
|
|
83
|
+
for lease in self._sink.active_leases():
|
|
84
|
+
deadline = lease.created_at.timestamp() + float(lease.ttl_seconds)
|
|
85
|
+
owner_gone = run_liveness is not None and _owner_gone(run_liveness, lease.run_id)
|
|
86
|
+
if deadline >= current and not owner_gone:
|
|
87
|
+
continue
|
|
88
|
+
notes = _LOST_OWNER if owner_gone else None
|
|
89
|
+
if lease.state is LeaseState.PENDING:
|
|
90
|
+
self._expire(lease, now, expired, notes=notes)
|
|
91
|
+
elif lease.state is LeaseState.DIRTY:
|
|
92
|
+
self._surrender(lease, now, expired, dirty)
|
|
93
|
+
elif lease.state is LeaseState.RELEASING:
|
|
94
|
+
self._finalize(lease, now, recovered, dirty)
|
|
95
|
+
else: # ACTIVE or ORPHANED
|
|
96
|
+
self._recover_orphan(lease, now, recovered, dirty, notes=notes)
|
|
97
|
+
return SweepResult(tuple(expired), tuple(recovered), tuple(dirty))
|
|
98
|
+
|
|
99
|
+
def _expire(
|
|
100
|
+
self, lease: FaultLease, now: datetime, expired: list[str], *, notes: str | None = None
|
|
101
|
+
) -> None:
|
|
102
|
+
# Never injected, nothing to undo: straight to EXPIRED.
|
|
103
|
+
expired.append(lease.id)
|
|
104
|
+
self._save_quietly(
|
|
105
|
+
lease.transition(
|
|
106
|
+
LeaseState.EXPIRED,
|
|
107
|
+
mechanism="janitor",
|
|
108
|
+
now=now,
|
|
109
|
+
escalation_notes=notes,
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
def _surrender(
|
|
114
|
+
self, lease: FaultLease, now: datetime, expired: list[str], dirty: list[str]
|
|
115
|
+
) -> None:
|
|
116
|
+
# Compensation already failed and nobody is coming back for this lease
|
|
117
|
+
# past its TTL — record the surrender, unblock the targets next run.
|
|
118
|
+
try:
|
|
119
|
+
surrendered = lease.transition(LeaseState.EXPIRED, mechanism="janitor", now=now)
|
|
120
|
+
self._sink.save(surrendered)
|
|
121
|
+
except (DomainError, OSError):
|
|
122
|
+
dirty.append(lease.id) # stays DIRTY, still wedged
|
|
123
|
+
else:
|
|
124
|
+
expired.append(lease.id)
|
|
125
|
+
|
|
126
|
+
def _finalize(
|
|
127
|
+
self, lease: FaultLease, now: datetime, recovered: list[str], dirty: list[str]
|
|
128
|
+
) -> None:
|
|
129
|
+
# A lease stuck mid-compensation finalizes straight to RELEASED — the
|
|
130
|
+
# owner is gone, no second RELEASING hop.
|
|
131
|
+
try:
|
|
132
|
+
final = lease.transition(LeaseState.RELEASED, mechanism="janitor", now=now)
|
|
133
|
+
self._sink.save(final)
|
|
134
|
+
except (DomainError, OSError) as exc:
|
|
135
|
+
self._dirty_from(lease, exc)
|
|
136
|
+
dirty.append(lease.id)
|
|
137
|
+
else:
|
|
138
|
+
recovered.append(lease.id)
|
|
139
|
+
|
|
140
|
+
def _recover_orphan(
|
|
141
|
+
self,
|
|
142
|
+
lease: FaultLease,
|
|
143
|
+
now: datetime,
|
|
144
|
+
recovered: list[str],
|
|
145
|
+
dirty: list[str],
|
|
146
|
+
*,
|
|
147
|
+
notes: str | None = None,
|
|
148
|
+
) -> None:
|
|
149
|
+
# Orphan ACTIVE leases first (honest record), then finalize the
|
|
150
|
+
# compensation the owner started (stuck ORPHANED leases keep their
|
|
151
|
+
# targets wedged without this path).
|
|
152
|
+
if lease.state is LeaseState.ACTIVE:
|
|
153
|
+
orphaned = lease.transition(
|
|
154
|
+
LeaseState.ORPHANED,
|
|
155
|
+
mechanism="janitor",
|
|
156
|
+
now=now,
|
|
157
|
+
escalation_notes=notes or f"lease {lease.id} exceeded TTL without release",
|
|
158
|
+
)
|
|
159
|
+
else:
|
|
160
|
+
orphaned = lease
|
|
161
|
+
try:
|
|
162
|
+
released = orphaned.transition(LeaseState.RELEASING, mechanism="janitor", now=now)
|
|
163
|
+
self._sink.save(released)
|
|
164
|
+
final = released.transition(LeaseState.RELEASED, mechanism="janitor", now=now)
|
|
165
|
+
self._sink.save(final)
|
|
166
|
+
recovered.append(orphaned.id)
|
|
167
|
+
except (DomainError, OSError) as exc:
|
|
168
|
+
# Persistence failed: never claim recovery we could not record.
|
|
169
|
+
self._dirty_from(orphaned, exc)
|
|
170
|
+
dirty.append(orphaned.id)
|
|
171
|
+
|
|
172
|
+
def _save_quietly(self, lease: FaultLease) -> None:
|
|
173
|
+
with contextlib.suppress(Exception): # sweep must survive sink flakiness
|
|
174
|
+
self._sink.save(lease)
|
|
175
|
+
|
|
176
|
+
def _dirty_from(self, lease: FaultLease, exc: Exception) -> None:
|
|
177
|
+
try:
|
|
178
|
+
if lease.state is LeaseState.RELEASING:
|
|
179
|
+
stuck = lease.transition(
|
|
180
|
+
LeaseState.DIRTY,
|
|
181
|
+
mechanism="janitor",
|
|
182
|
+
now=utc_now(),
|
|
183
|
+
escalation_notes=f"orphan recovery failed: {exc}",
|
|
184
|
+
)
|
|
185
|
+
else:
|
|
186
|
+
stuck = lease.transition(
|
|
187
|
+
LeaseState.RELEASING, mechanism="janitor", now=utc_now()
|
|
188
|
+
).transition(
|
|
189
|
+
LeaseState.DIRTY,
|
|
190
|
+
mechanism="janitor",
|
|
191
|
+
now=utc_now(),
|
|
192
|
+
escalation_notes=f"orphan recovery failed: {exc}",
|
|
193
|
+
)
|
|
194
|
+
self._save_quietly(stuck)
|
|
195
|
+
except DomainError:
|
|
196
|
+
pass
|
|
@@ -0,0 +1,382 @@
|
|
|
1
|
+
"""Declarative observability collectors (ADR-M4-4).
|
|
2
|
+
|
|
3
|
+
Each declared source is collected within its own bounded timeout and the whole
|
|
4
|
+
pass respects the config's ``total_timeout``. Collection is **best-effort**:
|
|
5
|
+
a failing source is recorded as a failed collection with a note, never raised
|
|
6
|
+
to the run — evidence gathering must never break the drill.
|
|
7
|
+
|
|
8
|
+
Collected evidence is returned as :class:`SourceCollection` objects the
|
|
9
|
+
executor persists onto the run row and spools into the journal, so an observer
|
|
10
|
+
can replay exactly what was seen.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
import time
|
|
17
|
+
import urllib.request
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from typing import TYPE_CHECKING, Any
|
|
20
|
+
|
|
21
|
+
from mayhem.agents.probes import run_probe
|
|
22
|
+
from mayhem.domain.checks import Probe, ProbeType
|
|
23
|
+
from mayhem.domain.common import utc_now
|
|
24
|
+
from mayhem.domain.leases import VerifyProbe
|
|
25
|
+
from mayhem.domain.observability import (
|
|
26
|
+
InspectionSource,
|
|
27
|
+
LogsSource,
|
|
28
|
+
MetricsSource,
|
|
29
|
+
ObservabilityConfig,
|
|
30
|
+
ObservabilitySourceKind,
|
|
31
|
+
ProbeSource,
|
|
32
|
+
duration_seconds,
|
|
33
|
+
)
|
|
34
|
+
from mayhem.toolkit.tool_runner import run_tool
|
|
35
|
+
|
|
36
|
+
if TYPE_CHECKING:
|
|
37
|
+
from mayhem.domain.observability import ObservabilitySource
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
_LEADING_NUMBER = re.compile(r"[-+]?[0-9]*\.?[0-9]+")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True)
|
|
44
|
+
class ObservabilitySample:
|
|
45
|
+
"""One measured value with its wall-clock timestamp."""
|
|
46
|
+
|
|
47
|
+
at: str
|
|
48
|
+
value: object
|
|
49
|
+
|
|
50
|
+
@classmethod
|
|
51
|
+
def now(cls, value: object) -> ObservabilitySample:
|
|
52
|
+
return cls(at=utc_now().isoformat(), value=value)
|
|
53
|
+
|
|
54
|
+
def to_jsonable(self) -> dict[str, object]:
|
|
55
|
+
return {"at": self.at, "value": self.value}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(frozen=True)
|
|
59
|
+
class SourceCollection:
|
|
60
|
+
"""Outcome of collecting one declared source."""
|
|
61
|
+
|
|
62
|
+
source_id: str
|
|
63
|
+
kind: ObservabilitySourceKind
|
|
64
|
+
ok: bool
|
|
65
|
+
note: str
|
|
66
|
+
latency_ms: float
|
|
67
|
+
samples: tuple[ObservabilitySample, ...] = ()
|
|
68
|
+
skipped: bool = False
|
|
69
|
+
|
|
70
|
+
def to_jsonable(self) -> dict[str, object]:
|
|
71
|
+
return {
|
|
72
|
+
"source_id": self.source_id,
|
|
73
|
+
"kind": self.kind.value,
|
|
74
|
+
"ok": self.ok,
|
|
75
|
+
"note": self.note,
|
|
76
|
+
"latency_ms": self.latency_ms,
|
|
77
|
+
"skipped": self.skipped,
|
|
78
|
+
"samples": [s.to_jsonable() for s in self.samples],
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def probe_to_verify(probe: Probe) -> VerifyProbe:
|
|
83
|
+
"""Canonical mapping of a domain ``Probe`` onto the agents ``VerifyProbe``.
|
|
84
|
+
|
|
85
|
+
Lifted out of the executor so observability probe sources and check steps
|
|
86
|
+
share the exact same runtime semantics (ADR-M4-2/4-4).
|
|
87
|
+
"""
|
|
88
|
+
if probe.type is ProbeType.EXEC:
|
|
89
|
+
return VerifyProbe(
|
|
90
|
+
probe="exec",
|
|
91
|
+
args={"cmd": list(probe.cmd), "timeout_s": float(probe.timeout)},
|
|
92
|
+
expect_present=True,
|
|
93
|
+
)
|
|
94
|
+
if probe.type is ProbeType.TCP:
|
|
95
|
+
return VerifyProbe(
|
|
96
|
+
probe="tcp",
|
|
97
|
+
args={
|
|
98
|
+
"host": probe.host,
|
|
99
|
+
"port": int(probe.port),
|
|
100
|
+
"timeout_s": float(probe.timeout),
|
|
101
|
+
},
|
|
102
|
+
expect_present=True,
|
|
103
|
+
)
|
|
104
|
+
if probe.type is ProbeType.HTTP:
|
|
105
|
+
return VerifyProbe(
|
|
106
|
+
probe="http",
|
|
107
|
+
args={
|
|
108
|
+
"url": probe.url,
|
|
109
|
+
"expect_status": int(probe.expected_status),
|
|
110
|
+
"timeout_s": float(probe.timeout),
|
|
111
|
+
},
|
|
112
|
+
expect_present=True,
|
|
113
|
+
)
|
|
114
|
+
if probe.type is ProbeType.PROCESS:
|
|
115
|
+
return VerifyProbe(
|
|
116
|
+
probe="process",
|
|
117
|
+
args={
|
|
118
|
+
"name": probe.name,
|
|
119
|
+
"pid": probe.pid,
|
|
120
|
+
"timeout_s": float(probe.timeout),
|
|
121
|
+
},
|
|
122
|
+
expect_present=True,
|
|
123
|
+
)
|
|
124
|
+
if probe.type is ProbeType.METRIC:
|
|
125
|
+
return VerifyProbe(
|
|
126
|
+
probe="metric",
|
|
127
|
+
args={
|
|
128
|
+
"endpoint": probe.endpoint,
|
|
129
|
+
"query": probe.query,
|
|
130
|
+
"threshold": probe.threshold,
|
|
131
|
+
"timeout_s": float(probe.timeout),
|
|
132
|
+
},
|
|
133
|
+
expect_present=True,
|
|
134
|
+
)
|
|
135
|
+
if probe.type is ProbeType.FILE:
|
|
136
|
+
return VerifyProbe(
|
|
137
|
+
probe="file",
|
|
138
|
+
args={
|
|
139
|
+
"path": probe.path,
|
|
140
|
+
"contains": probe.contains,
|
|
141
|
+
"timeout_s": float(probe.timeout),
|
|
142
|
+
},
|
|
143
|
+
expect_present=True,
|
|
144
|
+
)
|
|
145
|
+
raise AssertionError(f"unhandled probe type {probe.type!r}")
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _default_logs_runner(engine: str, timeout_s: float) -> Any:
|
|
149
|
+
def run_container_logs(container: str, tail: int, since: str) -> tuple[str, float]:
|
|
150
|
+
argv = [engine, "logs", "--tail", str(tail)]
|
|
151
|
+
if since:
|
|
152
|
+
argv += ["--since", since]
|
|
153
|
+
argv.append(container)
|
|
154
|
+
started = time.monotonic()
|
|
155
|
+
result = run_tool(argv, timeout_s=timeout_s)
|
|
156
|
+
latency_ms = round((time.monotonic() - started) * 1000, 3)
|
|
157
|
+
if result.exit_code != 0:
|
|
158
|
+
raise RuntimeError(f"{engine} logs exited {result.exit_code}: {result.stderr.strip()}")
|
|
159
|
+
text = f"{result.stdout}\n{result.stderr}".strip()
|
|
160
|
+
return text, latency_ms
|
|
161
|
+
|
|
162
|
+
return run_container_logs
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _default_inspect_runner(engine: str, timeout_s: float) -> Any:
|
|
166
|
+
def run_container_inspect(container: str) -> tuple[str, float]:
|
|
167
|
+
started = time.monotonic()
|
|
168
|
+
result = run_tool([engine, "inspect", container], timeout_s=timeout_s)
|
|
169
|
+
latency_ms = round((time.monotonic() - started) * 1000, 3)
|
|
170
|
+
if result.exit_code != 0:
|
|
171
|
+
raise RuntimeError(
|
|
172
|
+
f"{engine} inspect exited {result.exit_code}: {result.stderr.strip()}"
|
|
173
|
+
)
|
|
174
|
+
return result.stdout.strip(), latency_ms
|
|
175
|
+
|
|
176
|
+
return run_container_inspect
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _default_probe_runner(timeout_s: float) -> Any:
|
|
180
|
+
def run_domain_probe(probe: Probe) -> tuple[bool, str, float]:
|
|
181
|
+
started = time.monotonic()
|
|
182
|
+
result = run_probe(probe_to_verify(probe))
|
|
183
|
+
latency_ms = round((time.monotonic() - started) * 1000, 3)
|
|
184
|
+
return result.satisfied, result.detail, latency_ms
|
|
185
|
+
|
|
186
|
+
return run_domain_probe
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _parse_prometheus_text(body: str, metric: str) -> float | None:
|
|
190
|
+
"""Resolve one metric's last sample value from Prometheus text format."""
|
|
191
|
+
found: list[float] = []
|
|
192
|
+
prefix = metric + "{"
|
|
193
|
+
for raw in body.splitlines():
|
|
194
|
+
line = raw.strip()
|
|
195
|
+
if not line or line.startswith("#"):
|
|
196
|
+
continue
|
|
197
|
+
parts = line.split(" ", 1)
|
|
198
|
+
if len(parts) != 2:
|
|
199
|
+
continue
|
|
200
|
+
name = parts[0]
|
|
201
|
+
if name == metric or name.startswith(prefix):
|
|
202
|
+
match = _LEADING_NUMBER.match(parts[1])
|
|
203
|
+
if match is not None:
|
|
204
|
+
found.append(float(match.group(0)))
|
|
205
|
+
return found[-1] if found else None
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _default_metrics_runner(timeout_s: float) -> Any:
|
|
209
|
+
def scrape_metric(endpoint: str, metric: str) -> tuple[float | None, str, float]:
|
|
210
|
+
started = time.monotonic()
|
|
211
|
+
with urllib.request.urlopen(endpoint, timeout=timeout_s) as response:
|
|
212
|
+
body = response.read(1_048_576).decode("utf-8", errors="replace")
|
|
213
|
+
latency_ms = round((time.monotonic() - started) * 1000, 3)
|
|
214
|
+
value = _parse_prometheus_text(body, metric)
|
|
215
|
+
note = "" if value is not None else f"metric {metric!r} not present in scrape"
|
|
216
|
+
return value, note, latency_ms
|
|
217
|
+
|
|
218
|
+
return scrape_metric
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def collect_observability(
|
|
222
|
+
cfg: ObservabilityConfig,
|
|
223
|
+
*,
|
|
224
|
+
engine: str = "podman",
|
|
225
|
+
) -> tuple[SourceCollection, ...]:
|
|
226
|
+
"""Collect every declared source within ``total_timeout`` (ADR-M4-4).
|
|
227
|
+
|
|
228
|
+
Runners are module-level defaults that call ``toolkit.run_tool`` and the
|
|
229
|
+
agents probe runner so real runs capture full tool evidence; tests inject
|
|
230
|
+
fakes via the ``_collect_*`` helpers or monkeypatch the primitives.
|
|
231
|
+
"""
|
|
232
|
+
if cfg.empty:
|
|
233
|
+
return ()
|
|
234
|
+
deadline = time.monotonic() + duration_seconds(cfg.total_timeout)
|
|
235
|
+
default_cadence = duration_seconds(cfg.cadence)
|
|
236
|
+
collections: list[SourceCollection] = []
|
|
237
|
+
for source in cfg.sources:
|
|
238
|
+
remaining = deadline - time.monotonic()
|
|
239
|
+
if remaining <= 0:
|
|
240
|
+
collections.append(
|
|
241
|
+
SourceCollection(
|
|
242
|
+
source_id=source.source_id,
|
|
243
|
+
kind=ObservabilitySourceKind(source.kind),
|
|
244
|
+
ok=False,
|
|
245
|
+
note="pass total_timeout exhausted before this source",
|
|
246
|
+
latency_ms=0.0,
|
|
247
|
+
skipped=True,
|
|
248
|
+
)
|
|
249
|
+
)
|
|
250
|
+
continue
|
|
251
|
+
timeout_s = min(duration_seconds(getattr(source, "timeout", "10s")), remaining)
|
|
252
|
+
cadence = duration_seconds(getattr(source, "cadence", "0s"))
|
|
253
|
+
if cadence <= 0:
|
|
254
|
+
cadence = default_cadence # config-level cadence applies unless overridden
|
|
255
|
+
collections.append(
|
|
256
|
+
_collect_source(source, timeout_s=timeout_s, cadence=cadence, deadline=deadline)
|
|
257
|
+
)
|
|
258
|
+
return tuple(collections)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _collect_source(
|
|
262
|
+
source: ObservabilitySource,
|
|
263
|
+
*,
|
|
264
|
+
timeout_s: float,
|
|
265
|
+
cadence: float,
|
|
266
|
+
deadline: float,
|
|
267
|
+
) -> SourceCollection:
|
|
268
|
+
if isinstance(source, LogsSource):
|
|
269
|
+
return _collect_logs(source, timeout_s)
|
|
270
|
+
if isinstance(source, InspectionSource):
|
|
271
|
+
return _collect_inspection(source, timeout_s)
|
|
272
|
+
if isinstance(source, ProbeSource):
|
|
273
|
+
return _collect_probe(source, timeout_s, cadence, deadline)
|
|
274
|
+
if isinstance(source, MetricsSource):
|
|
275
|
+
return _collect_metrics(source, timeout_s, cadence, deadline)
|
|
276
|
+
raise AssertionError(f"unhandled observability source {source!r}")
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _fail(source: ObservabilitySource, note: str) -> SourceCollection:
|
|
280
|
+
return SourceCollection(
|
|
281
|
+
source_id=source.source_id,
|
|
282
|
+
kind=ObservabilitySourceKind(source.kind),
|
|
283
|
+
ok=False,
|
|
284
|
+
note=note,
|
|
285
|
+
latency_ms=0.0,
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _collect_logs(source: LogsSource, timeout_s: float) -> SourceCollection:
|
|
290
|
+
try:
|
|
291
|
+
text, latency = _default_logs_runner("podman", timeout_s)(
|
|
292
|
+
source.container, source.tail, source.since
|
|
293
|
+
)
|
|
294
|
+
return SourceCollection(
|
|
295
|
+
source_id=source.source_id,
|
|
296
|
+
kind=ObservabilitySourceKind.LOGS,
|
|
297
|
+
ok=True,
|
|
298
|
+
note=f"collected {len(text)} chars",
|
|
299
|
+
latency_ms=latency,
|
|
300
|
+
samples=(ObservabilitySample.now(text),),
|
|
301
|
+
)
|
|
302
|
+
except Exception as exc:
|
|
303
|
+
return _fail(source, f"collection failed: {type(exc).__name__}: {exc}")
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def _collect_inspection(source: InspectionSource, timeout_s: float) -> SourceCollection:
|
|
307
|
+
try:
|
|
308
|
+
text, latency = _default_inspect_runner("podman", timeout_s)(source.container)
|
|
309
|
+
return SourceCollection(
|
|
310
|
+
source_id=source.source_id,
|
|
311
|
+
kind=ObservabilitySourceKind.INSPECTION,
|
|
312
|
+
ok=True,
|
|
313
|
+
note=f"inspect {len(text)} chars",
|
|
314
|
+
latency_ms=latency,
|
|
315
|
+
samples=(ObservabilitySample.now(text),),
|
|
316
|
+
)
|
|
317
|
+
except Exception as exc:
|
|
318
|
+
return _fail(source, f"collection failed: {type(exc).__name__}: {exc}")
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _collect_probe(
|
|
322
|
+
source: ProbeSource, timeout_s: float, cadence: float, deadline: float
|
|
323
|
+
) -> SourceCollection:
|
|
324
|
+
latency_ms_total = 0.0
|
|
325
|
+
samples: list[ObservabilitySample] = []
|
|
326
|
+
try:
|
|
327
|
+
runner = _default_probe_runner(timeout_s)
|
|
328
|
+
while True:
|
|
329
|
+
satisfied, detail, latency = runner(source.probe)
|
|
330
|
+
latency_ms_total += latency
|
|
331
|
+
samples.append(ObservabilitySample.now({"satisfied": satisfied, "detail": detail}))
|
|
332
|
+
if time.monotonic() >= deadline:
|
|
333
|
+
break
|
|
334
|
+
sleep_for = min(cadence, max(0.0, deadline - time.monotonic()))
|
|
335
|
+
time.sleep(sleep_for)
|
|
336
|
+
if time.monotonic() >= deadline:
|
|
337
|
+
break
|
|
338
|
+
return SourceCollection(
|
|
339
|
+
source_id=source.source_id,
|
|
340
|
+
kind=ObservabilitySourceKind.PROBE,
|
|
341
|
+
ok=True,
|
|
342
|
+
note=f"{len(samples)} probe sample(s)",
|
|
343
|
+
latency_ms=round(latency_ms_total, 3),
|
|
344
|
+
samples=tuple(samples),
|
|
345
|
+
)
|
|
346
|
+
except Exception as exc:
|
|
347
|
+
return _fail(
|
|
348
|
+
source,
|
|
349
|
+
f"collection failed: {type(exc).__name__}: {exc}",
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _collect_metrics(
|
|
354
|
+
source: MetricsSource, timeout_s: float, cadence: float, deadline: float
|
|
355
|
+
) -> SourceCollection:
|
|
356
|
+
latency_ms_total = 0.0
|
|
357
|
+
samples: list[ObservabilitySample] = []
|
|
358
|
+
note = ""
|
|
359
|
+
try:
|
|
360
|
+
runner = _default_metrics_runner(timeout_s)
|
|
361
|
+
while True:
|
|
362
|
+
value, scrape_note, latency = runner(source.endpoint, source.metric)
|
|
363
|
+
latency_ms_total += latency
|
|
364
|
+
samples.append(ObservabilitySample.now(value))
|
|
365
|
+
if value is None and scrape_note:
|
|
366
|
+
note = scrape_note
|
|
367
|
+
if time.monotonic() >= deadline:
|
|
368
|
+
break
|
|
369
|
+
sleep_for = min(cadence, max(0.0, deadline - time.monotonic()))
|
|
370
|
+
time.sleep(sleep_for)
|
|
371
|
+
if time.monotonic() >= deadline:
|
|
372
|
+
break
|
|
373
|
+
return SourceCollection(
|
|
374
|
+
source_id=source.source_id,
|
|
375
|
+
kind=ObservabilitySourceKind.METRICS,
|
|
376
|
+
ok=note == "" or bool(samples),
|
|
377
|
+
note=note or f"{len(samples)} metric sample(s)",
|
|
378
|
+
latency_ms=round(latency_ms_total, 3),
|
|
379
|
+
samples=tuple(samples),
|
|
380
|
+
)
|
|
381
|
+
except Exception as exc:
|
|
382
|
+
return _fail(source, f"collection failed: {type(exc).__name__}: {exc}")
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Observation engine — structured event log for experiment runs (ADR-0020).
|
|
2
|
+
|
|
3
|
+
An ``Observation`` is a timestamped, typed event that occurred during an
|
|
4
|
+
experiment — fault injected, probe measured, recovery attempted, threshold
|
|
5
|
+
breached, etc. The ``ObservationLog`` is an append-only, in-memory store
|
|
6
|
+
that can be serialized to JSON for persistence or display.
|
|
7
|
+
|
|
8
|
+
Observations are the raw material that the evaluation system consumes.
|
|
9
|
+
They are never mutated after creation.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from enum import StrEnum
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
18
|
+
|
|
19
|
+
from mayhem.domain.common import utc_now
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ObservationKind(StrEnum):
|
|
23
|
+
"""Categories of observations that can occur during an experiment."""
|
|
24
|
+
|
|
25
|
+
FAULT_INJECTED = "fault.injected"
|
|
26
|
+
FAULT_UNDONE = "fault.undone"
|
|
27
|
+
PROBE_MEASURED = "probe.measured"
|
|
28
|
+
THRESHOLD_BREACHED = "threshold.breached"
|
|
29
|
+
RECOVERY_ATTEMPTED = "recovery.attempted"
|
|
30
|
+
RECOVERY_SUCCEEDED = "recovery.succeeded"
|
|
31
|
+
RECOVERY_FAILED = "recovery.failed"
|
|
32
|
+
STEP_STARTED = "step.started"
|
|
33
|
+
STEP_COMPLETED = "step.completed"
|
|
34
|
+
STEP_FAILED = "step.failed"
|
|
35
|
+
ANOMALY_DETECTED = "anomaly.detected"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Observation(BaseModel):
|
|
39
|
+
"""A single immutable observation event."""
|
|
40
|
+
|
|
41
|
+
model_config = ConfigDict(frozen=True)
|
|
42
|
+
|
|
43
|
+
kind: ObservationKind
|
|
44
|
+
run_id: str
|
|
45
|
+
timestamp: str = Field(default_factory=lambda: utc_now().isoformat())
|
|
46
|
+
source: str = "" # e.g. "executor", "janitor", "probe_runner"
|
|
47
|
+
data: dict[str, Any] = Field(default_factory=dict)
|
|
48
|
+
runtime_identity: str | None = None # canonical identity key (ADR-M1-1/1-3)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class ObservationLog:
|
|
52
|
+
"""Append-only observation log with query helpers.
|
|
53
|
+
|
|
54
|
+
The log lives in memory for the duration of a run. Periodic snapshots
|
|
55
|
+
can be flushed to SQLite or JSON for durability.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def __init__(self) -> None:
|
|
59
|
+
self._observations: list[Observation] = []
|
|
60
|
+
|
|
61
|
+
def record(self, observation: Observation) -> None:
|
|
62
|
+
"""Append an observation to the log."""
|
|
63
|
+
self._observations.append(observation)
|
|
64
|
+
|
|
65
|
+
def emit(
|
|
66
|
+
self,
|
|
67
|
+
kind: ObservationKind,
|
|
68
|
+
run_id: str,
|
|
69
|
+
*,
|
|
70
|
+
source: str = "",
|
|
71
|
+
**data: Any,
|
|
72
|
+
) -> None:
|
|
73
|
+
"""Shorthand: create and record in one call."""
|
|
74
|
+
self.record(Observation(kind=kind, run_id=run_id, source=source, data=data))
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def observations(self) -> tuple[Observation, ...]:
|
|
78
|
+
return tuple(self._observations)
|
|
79
|
+
|
|
80
|
+
def for_run(self, run_id: str) -> tuple[Observation, ...]:
|
|
81
|
+
"""All observations for a specific run."""
|
|
82
|
+
return tuple(o for o in self._observations if o.run_id == run_id)
|
|
83
|
+
|
|
84
|
+
def for_kind(
|
|
85
|
+
self, kind: ObservationKind, *, run_id: str | None = None
|
|
86
|
+
) -> tuple[Observation, ...]:
|
|
87
|
+
"""All observations of a given kind, optionally filtered by run."""
|
|
88
|
+
result = (o for o in self._observations if o.kind == kind)
|
|
89
|
+
if run_id is not None:
|
|
90
|
+
result = (o for o in result if o.run_id == run_id)
|
|
91
|
+
return tuple(result)
|
|
92
|
+
|
|
93
|
+
def anomalies_for_run(self, run_id: str) -> tuple[Observation, ...]:
|
|
94
|
+
"""All anomaly observations for a run."""
|
|
95
|
+
return self.for_kind(ObservationKind.ANOMALY_DETECTED, run_id=run_id)
|
|
96
|
+
|
|
97
|
+
def snapshot(self) -> tuple[dict[str, Any], ...]:
|
|
98
|
+
"""Serialize all observations to JSON-compatible dicts."""
|
|
99
|
+
return tuple(o.model_dump(mode="json") for o in self._observations)
|
|
100
|
+
|
|
101
|
+
def __len__(self) -> int:
|
|
102
|
+
return len(self._observations)
|