mayhem-cli 0.5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. mayhem/agent/__init__.py +1 -0
  2. mayhem/agent/cli.py +36 -0
  3. mayhem/agents/__init__.py +1 -0
  4. mayhem/agents/capabilities.py +106 -0
  5. mayhem/agents/executors.py +430 -0
  6. mayhem/agents/impact.py +729 -0
  7. mayhem/agents/lease_client.py +141 -0
  8. mayhem/agents/probes.py +284 -0
  9. mayhem/agents/protocol.py +134 -0
  10. mayhem/agents/server.py +281 -0
  11. mayhem/agents/sinks.py +60 -0
  12. mayhem/agents/transports.py +134 -0
  13. mayhem/agents/watchdog.py +140 -0
  14. mayhem/cli/__init__.py +11 -0
  15. mayhem/cli/app.py +154 -0
  16. mayhem/cli/campaign.py +496 -0
  17. mayhem/cli/config_cmd.py +47 -0
  18. mayhem/cli/context.py +23 -0
  19. mayhem/cli/dependency.py +429 -0
  20. mayhem/cli/exit_codes.py +24 -0
  21. mayhem/cli/experiment.py +24 -0
  22. mayhem/cli/lifecycle.py +805 -0
  23. mayhem/cli/resolver.py +72 -0
  24. mayhem/cli/services.py +459 -0
  25. mayhem/cli/style.py +101 -0
  26. mayhem/cli/toolkit.py +41 -0
  27. mayhem/cli/topology.py +127 -0
  28. mayhem/config.py +208 -0
  29. mayhem/controller/__init__.py +1 -0
  30. mayhem/controller/compensation.py +2156 -0
  31. mayhem/controller/executor.py +1719 -0
  32. mayhem/controller/janitor.py +196 -0
  33. mayhem/controller/observability_collector.py +382 -0
  34. mayhem/controller/observations.py +102 -0
  35. mayhem/controller/planner.py +715 -0
  36. mayhem/controller/recovery.py +245 -0
  37. mayhem/controller/resilience_report.py +585 -0
  38. mayhem/controller/resource_manager.py +457 -0
  39. mayhem/controller/safety.py +392 -0
  40. mayhem/domain/__init__.py +6 -0
  41. mayhem/domain/campaigns.py +118 -0
  42. mayhem/domain/cancellation.py +110 -0
  43. mayhem/domain/candidates.py +101 -0
  44. mayhem/domain/capabilities.py +86 -0
  45. mayhem/domain/catalog.py +727 -0
  46. mayhem/domain/checks.py +173 -0
  47. mayhem/domain/common.py +104 -0
  48. mayhem/domain/coverage.py +106 -0
  49. mayhem/domain/decisions.py +57 -0
  50. mayhem/domain/errors.py +87 -0
  51. mayhem/domain/events.py +61 -0
  52. mayhem/domain/execution_context.py +120 -0
  53. mayhem/domain/execution_loci.py +94 -0
  54. mayhem/domain/experiments.py +370 -0
  55. mayhem/domain/faults.py +239 -0
  56. mayhem/domain/identity.py +200 -0
  57. mayhem/domain/k8s_adapter.py +132 -0
  58. mayhem/domain/leases.py +186 -0
  59. mayhem/domain/load_strategy.py +98 -0
  60. mayhem/domain/m5_campaign.py +120 -0
  61. mayhem/domain/maniac.py +93 -0
  62. mayhem/domain/observability.py +146 -0
  63. mayhem/domain/outcomes.py +92 -0
  64. mayhem/domain/remote_agent_interface.py +70 -0
  65. mayhem/domain/resources.py +245 -0
  66. mayhem/domain/risks.py +61 -0
  67. mayhem/domain/run_outcome.py +146 -0
  68. mayhem/domain/runtime_adapter.py +256 -0
  69. mayhem/domain/success.py +329 -0
  70. mayhem/domain/topology.py +452 -0
  71. mayhem/infra/__init__.py +1 -0
  72. mayhem/infra/campaign_engine.py +205 -0
  73. mayhem/infra/candidate_gates.py +124 -0
  74. mayhem/infra/candidate_generator.py +110 -0
  75. mayhem/infra/coverage_repository.py +101 -0
  76. mayhem/infra/lease_repository.py +129 -0
  77. mayhem/infra/maniac.py +103 -0
  78. mayhem/infra/migrations.py +596 -0
  79. mayhem/infra/migrator.py +149 -0
  80. mayhem/infra/report.py +227 -0
  81. mayhem/infra/store.py +200 -0
  82. mayhem/py.typed +0 -0
  83. mayhem/spec.py +52 -0
  84. mayhem/toolkit/__init__.py +1 -0
  85. mayhem/toolkit/fingerprint.py +69 -0
  86. mayhem/toolkit/hashing.py +32 -0
  87. mayhem/toolkit/manifests/docker.yaml +11 -0
  88. mayhem/toolkit/manifests/podman.yaml +11 -0
  89. mayhem/toolkit/manifests/stress-ng.yaml +11 -0
  90. mayhem/toolkit/manifests/tc-netem.yaml +11 -0
  91. mayhem/toolkit/manifests/toxiproxy.yaml +10 -0
  92. mayhem/toolkit/registry.py +185 -0
  93. mayhem/toolkit/tool_runner.py +129 -0
  94. mayhem/topology/__init__.py +10 -0
  95. mayhem/topology/providers/__init__.py +0 -0
  96. mayhem/topology/providers/adapter_registry.py +60 -0
  97. mayhem/topology/providers/base.py +31 -0
  98. mayhem/topology/providers/compose.py +207 -0
  99. mayhem/topology/providers/docker_adapter.py +277 -0
  100. mayhem/topology/providers/docker_runtime.py +461 -0
  101. mayhem/topology/providers/podman_adapter.py +328 -0
  102. mayhem/topology/resolve.py +196 -0
  103. mayhem/topology/service.py +158 -0
  104. mayhem_cli-0.5.1.dist-info/METADATA +555 -0
  105. mayhem_cli-0.5.1.dist-info/RECORD +107 -0
  106. mayhem_cli-0.5.1.dist-info/WHEEL +4 -0
  107. mayhem_cli-0.5.1.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,196 @@
1
+ """Janitor — background sweep that enforces lease TTLs.
2
+
3
+ A fault left past its TTL is by definition unattended. PENDING leases are
4
+ expired (never injected, nothing to undo); ACTIVE leases are orphaned and
5
+ then released through their write-ahead undo contract; ORPHANED/RELEASING
6
+ leases stuck mid-compensation are finalized; DIRTY leases (compensation
7
+ already failed) are surrendered to EXPIRED — the janitor never leaves a
8
+ fault running because its owner vanished.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import contextlib
14
+ from dataclasses import dataclass
15
+ from typing import TYPE_CHECKING
16
+
17
+ from mayhem.domain.common import utc_now
18
+ from mayhem.domain.errors import DomainError
19
+ from mayhem.domain.leases import LeaseState
20
+
21
+ if TYPE_CHECKING:
22
+ from collections.abc import Callable
23
+ from datetime import datetime
24
+
25
+ from mayhem.agents.sinks import LeaseSink
26
+ from mayhem.domain.leases import FaultLease
27
+
28
+ _LOST_OWNER = "owner run no longer live; reclaimed before TTL"
29
+
30
+
31
+ def _owner_gone(run_liveness: Callable[[str], bool | None], run_id: str) -> bool:
32
+ """True only when the resolver can *prove* the owning controller is gone.
33
+
34
+ Unknown (``None``) or an absent owner row never triggers the early reclaim —
35
+ TTL policy stays in charge of those, and a LiveLonger cycling pid can only
36
+ read "alive", which delays cleanup rather than wrongly reclaiming a lease
37
+ a live owner still needs.
38
+ """
39
+ try:
40
+ verdict = run_liveness(run_id)
41
+ except LookupError:
42
+ return False
43
+ return verdict is False
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class SweepResult:
48
+ expired: tuple[str, ...]
49
+ recovered: tuple[str, ...] # orphaned -> released via undo path marker
50
+ dirty: tuple[str, ...]
51
+
52
+ @property
53
+ def quiet(self) -> bool:
54
+ return not (self.expired or self.recovered or self.dirty)
55
+
56
+
57
+ class Janitor:
58
+ """TTL enforcement over any LeaseSink; state transitions only.
59
+
60
+ ``run_liveness`` is an optional resolver (run_id -> bool | None) the
61
+ caller supplies when it can see the runs table. It returns ``True`` while
62
+ the owning controller is alive, ``False`` when the owner is provably gone,
63
+ and ``None`` when unknown. A lease whose owner is provably gone is
64
+ reclaimed *before* its TTL — without this, a crashed ``run`` wedges its
65
+ targets for the whole TTL and the next ``run`` conflicts with a lease the
66
+ janitor "did nothing about".
67
+ """
68
+
69
+ def __init__(self, sink: LeaseSink) -> None:
70
+ self._sink = sink
71
+
72
+ def sweep(
73
+ self,
74
+ *,
75
+ now_epoch_s: float | None = None,
76
+ run_liveness: Callable[[str], bool | None] | None = None,
77
+ ) -> SweepResult:
78
+ now = utc_now()
79
+ current = now.timestamp() if now_epoch_s is None else now_epoch_s
80
+ expired: list[str] = []
81
+ recovered: list[str] = []
82
+ dirty: list[str] = []
83
+ for lease in self._sink.active_leases():
84
+ deadline = lease.created_at.timestamp() + float(lease.ttl_seconds)
85
+ owner_gone = run_liveness is not None and _owner_gone(run_liveness, lease.run_id)
86
+ if deadline >= current and not owner_gone:
87
+ continue
88
+ notes = _LOST_OWNER if owner_gone else None
89
+ if lease.state is LeaseState.PENDING:
90
+ self._expire(lease, now, expired, notes=notes)
91
+ elif lease.state is LeaseState.DIRTY:
92
+ self._surrender(lease, now, expired, dirty)
93
+ elif lease.state is LeaseState.RELEASING:
94
+ self._finalize(lease, now, recovered, dirty)
95
+ else: # ACTIVE or ORPHANED
96
+ self._recover_orphan(lease, now, recovered, dirty, notes=notes)
97
+ return SweepResult(tuple(expired), tuple(recovered), tuple(dirty))
98
+
99
+ def _expire(
100
+ self, lease: FaultLease, now: datetime, expired: list[str], *, notes: str | None = None
101
+ ) -> None:
102
+ # Never injected, nothing to undo: straight to EXPIRED.
103
+ expired.append(lease.id)
104
+ self._save_quietly(
105
+ lease.transition(
106
+ LeaseState.EXPIRED,
107
+ mechanism="janitor",
108
+ now=now,
109
+ escalation_notes=notes,
110
+ )
111
+ )
112
+
113
+ def _surrender(
114
+ self, lease: FaultLease, now: datetime, expired: list[str], dirty: list[str]
115
+ ) -> None:
116
+ # Compensation already failed and nobody is coming back for this lease
117
+ # past its TTL — record the surrender, unblock the targets next run.
118
+ try:
119
+ surrendered = lease.transition(LeaseState.EXPIRED, mechanism="janitor", now=now)
120
+ self._sink.save(surrendered)
121
+ except (DomainError, OSError):
122
+ dirty.append(lease.id) # stays DIRTY, still wedged
123
+ else:
124
+ expired.append(lease.id)
125
+
126
+ def _finalize(
127
+ self, lease: FaultLease, now: datetime, recovered: list[str], dirty: list[str]
128
+ ) -> None:
129
+ # A lease stuck mid-compensation finalizes straight to RELEASED — the
130
+ # owner is gone, no second RELEASING hop.
131
+ try:
132
+ final = lease.transition(LeaseState.RELEASED, mechanism="janitor", now=now)
133
+ self._sink.save(final)
134
+ except (DomainError, OSError) as exc:
135
+ self._dirty_from(lease, exc)
136
+ dirty.append(lease.id)
137
+ else:
138
+ recovered.append(lease.id)
139
+
140
+ def _recover_orphan(
141
+ self,
142
+ lease: FaultLease,
143
+ now: datetime,
144
+ recovered: list[str],
145
+ dirty: list[str],
146
+ *,
147
+ notes: str | None = None,
148
+ ) -> None:
149
+ # Orphan ACTIVE leases first (honest record), then finalize the
150
+ # compensation the owner started (stuck ORPHANED leases keep their
151
+ # targets wedged without this path).
152
+ if lease.state is LeaseState.ACTIVE:
153
+ orphaned = lease.transition(
154
+ LeaseState.ORPHANED,
155
+ mechanism="janitor",
156
+ now=now,
157
+ escalation_notes=notes or f"lease {lease.id} exceeded TTL without release",
158
+ )
159
+ else:
160
+ orphaned = lease
161
+ try:
162
+ released = orphaned.transition(LeaseState.RELEASING, mechanism="janitor", now=now)
163
+ self._sink.save(released)
164
+ final = released.transition(LeaseState.RELEASED, mechanism="janitor", now=now)
165
+ self._sink.save(final)
166
+ recovered.append(orphaned.id)
167
+ except (DomainError, OSError) as exc:
168
+ # Persistence failed: never claim recovery we could not record.
169
+ self._dirty_from(orphaned, exc)
170
+ dirty.append(orphaned.id)
171
+
172
+ def _save_quietly(self, lease: FaultLease) -> None:
173
+ with contextlib.suppress(Exception): # sweep must survive sink flakiness
174
+ self._sink.save(lease)
175
+
176
+ def _dirty_from(self, lease: FaultLease, exc: Exception) -> None:
177
+ try:
178
+ if lease.state is LeaseState.RELEASING:
179
+ stuck = lease.transition(
180
+ LeaseState.DIRTY,
181
+ mechanism="janitor",
182
+ now=utc_now(),
183
+ escalation_notes=f"orphan recovery failed: {exc}",
184
+ )
185
+ else:
186
+ stuck = lease.transition(
187
+ LeaseState.RELEASING, mechanism="janitor", now=utc_now()
188
+ ).transition(
189
+ LeaseState.DIRTY,
190
+ mechanism="janitor",
191
+ now=utc_now(),
192
+ escalation_notes=f"orphan recovery failed: {exc}",
193
+ )
194
+ self._save_quietly(stuck)
195
+ except DomainError:
196
+ pass
@@ -0,0 +1,382 @@
1
+ """Declarative observability collectors (ADR-M4-4).
2
+
3
+ Each declared source is collected within its own bounded timeout and the whole
4
+ pass respects the config's ``total_timeout``. Collection is **best-effort**:
5
+ a failing source is recorded as a failed collection with a note, never raised
6
+ to the run — evidence gathering must never break the drill.
7
+
8
+ Collected evidence is returned as :class:`SourceCollection` objects the
9
+ executor persists onto the run row and spools into the journal, so an observer
10
+ can replay exactly what was seen.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ import time
17
+ import urllib.request
18
+ from dataclasses import dataclass
19
+ from typing import TYPE_CHECKING, Any
20
+
21
+ from mayhem.agents.probes import run_probe
22
+ from mayhem.domain.checks import Probe, ProbeType
23
+ from mayhem.domain.common import utc_now
24
+ from mayhem.domain.leases import VerifyProbe
25
+ from mayhem.domain.observability import (
26
+ InspectionSource,
27
+ LogsSource,
28
+ MetricsSource,
29
+ ObservabilityConfig,
30
+ ObservabilitySourceKind,
31
+ ProbeSource,
32
+ duration_seconds,
33
+ )
34
+ from mayhem.toolkit.tool_runner import run_tool
35
+
36
+ if TYPE_CHECKING:
37
+ from mayhem.domain.observability import ObservabilitySource
38
+
39
+
40
+ _LEADING_NUMBER = re.compile(r"[-+]?[0-9]*\.?[0-9]+")
41
+
42
+
43
+ @dataclass(frozen=True)
44
+ class ObservabilitySample:
45
+ """One measured value with its wall-clock timestamp."""
46
+
47
+ at: str
48
+ value: object
49
+
50
+ @classmethod
51
+ def now(cls, value: object) -> ObservabilitySample:
52
+ return cls(at=utc_now().isoformat(), value=value)
53
+
54
+ def to_jsonable(self) -> dict[str, object]:
55
+ return {"at": self.at, "value": self.value}
56
+
57
+
58
+ @dataclass(frozen=True)
59
+ class SourceCollection:
60
+ """Outcome of collecting one declared source."""
61
+
62
+ source_id: str
63
+ kind: ObservabilitySourceKind
64
+ ok: bool
65
+ note: str
66
+ latency_ms: float
67
+ samples: tuple[ObservabilitySample, ...] = ()
68
+ skipped: bool = False
69
+
70
+ def to_jsonable(self) -> dict[str, object]:
71
+ return {
72
+ "source_id": self.source_id,
73
+ "kind": self.kind.value,
74
+ "ok": self.ok,
75
+ "note": self.note,
76
+ "latency_ms": self.latency_ms,
77
+ "skipped": self.skipped,
78
+ "samples": [s.to_jsonable() for s in self.samples],
79
+ }
80
+
81
+
82
+ def probe_to_verify(probe: Probe) -> VerifyProbe:
83
+ """Canonical mapping of a domain ``Probe`` onto the agents ``VerifyProbe``.
84
+
85
+ Lifted out of the executor so observability probe sources and check steps
86
+ share the exact same runtime semantics (ADR-M4-2/4-4).
87
+ """
88
+ if probe.type is ProbeType.EXEC:
89
+ return VerifyProbe(
90
+ probe="exec",
91
+ args={"cmd": list(probe.cmd), "timeout_s": float(probe.timeout)},
92
+ expect_present=True,
93
+ )
94
+ if probe.type is ProbeType.TCP:
95
+ return VerifyProbe(
96
+ probe="tcp",
97
+ args={
98
+ "host": probe.host,
99
+ "port": int(probe.port),
100
+ "timeout_s": float(probe.timeout),
101
+ },
102
+ expect_present=True,
103
+ )
104
+ if probe.type is ProbeType.HTTP:
105
+ return VerifyProbe(
106
+ probe="http",
107
+ args={
108
+ "url": probe.url,
109
+ "expect_status": int(probe.expected_status),
110
+ "timeout_s": float(probe.timeout),
111
+ },
112
+ expect_present=True,
113
+ )
114
+ if probe.type is ProbeType.PROCESS:
115
+ return VerifyProbe(
116
+ probe="process",
117
+ args={
118
+ "name": probe.name,
119
+ "pid": probe.pid,
120
+ "timeout_s": float(probe.timeout),
121
+ },
122
+ expect_present=True,
123
+ )
124
+ if probe.type is ProbeType.METRIC:
125
+ return VerifyProbe(
126
+ probe="metric",
127
+ args={
128
+ "endpoint": probe.endpoint,
129
+ "query": probe.query,
130
+ "threshold": probe.threshold,
131
+ "timeout_s": float(probe.timeout),
132
+ },
133
+ expect_present=True,
134
+ )
135
+ if probe.type is ProbeType.FILE:
136
+ return VerifyProbe(
137
+ probe="file",
138
+ args={
139
+ "path": probe.path,
140
+ "contains": probe.contains,
141
+ "timeout_s": float(probe.timeout),
142
+ },
143
+ expect_present=True,
144
+ )
145
+ raise AssertionError(f"unhandled probe type {probe.type!r}")
146
+
147
+
148
+ def _default_logs_runner(engine: str, timeout_s: float) -> Any:
149
+ def run_container_logs(container: str, tail: int, since: str) -> tuple[str, float]:
150
+ argv = [engine, "logs", "--tail", str(tail)]
151
+ if since:
152
+ argv += ["--since", since]
153
+ argv.append(container)
154
+ started = time.monotonic()
155
+ result = run_tool(argv, timeout_s=timeout_s)
156
+ latency_ms = round((time.monotonic() - started) * 1000, 3)
157
+ if result.exit_code != 0:
158
+ raise RuntimeError(f"{engine} logs exited {result.exit_code}: {result.stderr.strip()}")
159
+ text = f"{result.stdout}\n{result.stderr}".strip()
160
+ return text, latency_ms
161
+
162
+ return run_container_logs
163
+
164
+
165
+ def _default_inspect_runner(engine: str, timeout_s: float) -> Any:
166
+ def run_container_inspect(container: str) -> tuple[str, float]:
167
+ started = time.monotonic()
168
+ result = run_tool([engine, "inspect", container], timeout_s=timeout_s)
169
+ latency_ms = round((time.monotonic() - started) * 1000, 3)
170
+ if result.exit_code != 0:
171
+ raise RuntimeError(
172
+ f"{engine} inspect exited {result.exit_code}: {result.stderr.strip()}"
173
+ )
174
+ return result.stdout.strip(), latency_ms
175
+
176
+ return run_container_inspect
177
+
178
+
179
+ def _default_probe_runner(timeout_s: float) -> Any:
180
+ def run_domain_probe(probe: Probe) -> tuple[bool, str, float]:
181
+ started = time.monotonic()
182
+ result = run_probe(probe_to_verify(probe))
183
+ latency_ms = round((time.monotonic() - started) * 1000, 3)
184
+ return result.satisfied, result.detail, latency_ms
185
+
186
+ return run_domain_probe
187
+
188
+
189
+ def _parse_prometheus_text(body: str, metric: str) -> float | None:
190
+ """Resolve one metric's last sample value from Prometheus text format."""
191
+ found: list[float] = []
192
+ prefix = metric + "{"
193
+ for raw in body.splitlines():
194
+ line = raw.strip()
195
+ if not line or line.startswith("#"):
196
+ continue
197
+ parts = line.split(" ", 1)
198
+ if len(parts) != 2:
199
+ continue
200
+ name = parts[0]
201
+ if name == metric or name.startswith(prefix):
202
+ match = _LEADING_NUMBER.match(parts[1])
203
+ if match is not None:
204
+ found.append(float(match.group(0)))
205
+ return found[-1] if found else None
206
+
207
+
208
+ def _default_metrics_runner(timeout_s: float) -> Any:
209
+ def scrape_metric(endpoint: str, metric: str) -> tuple[float | None, str, float]:
210
+ started = time.monotonic()
211
+ with urllib.request.urlopen(endpoint, timeout=timeout_s) as response:
212
+ body = response.read(1_048_576).decode("utf-8", errors="replace")
213
+ latency_ms = round((time.monotonic() - started) * 1000, 3)
214
+ value = _parse_prometheus_text(body, metric)
215
+ note = "" if value is not None else f"metric {metric!r} not present in scrape"
216
+ return value, note, latency_ms
217
+
218
+ return scrape_metric
219
+
220
+
221
+ def collect_observability(
222
+ cfg: ObservabilityConfig,
223
+ *,
224
+ engine: str = "podman",
225
+ ) -> tuple[SourceCollection, ...]:
226
+ """Collect every declared source within ``total_timeout`` (ADR-M4-4).
227
+
228
+ Runners are module-level defaults that call ``toolkit.run_tool`` and the
229
+ agents probe runner so real runs capture full tool evidence; tests inject
230
+ fakes via the ``_collect_*`` helpers or monkeypatch the primitives.
231
+ """
232
+ if cfg.empty:
233
+ return ()
234
+ deadline = time.monotonic() + duration_seconds(cfg.total_timeout)
235
+ default_cadence = duration_seconds(cfg.cadence)
236
+ collections: list[SourceCollection] = []
237
+ for source in cfg.sources:
238
+ remaining = deadline - time.monotonic()
239
+ if remaining <= 0:
240
+ collections.append(
241
+ SourceCollection(
242
+ source_id=source.source_id,
243
+ kind=ObservabilitySourceKind(source.kind),
244
+ ok=False,
245
+ note="pass total_timeout exhausted before this source",
246
+ latency_ms=0.0,
247
+ skipped=True,
248
+ )
249
+ )
250
+ continue
251
+ timeout_s = min(duration_seconds(getattr(source, "timeout", "10s")), remaining)
252
+ cadence = duration_seconds(getattr(source, "cadence", "0s"))
253
+ if cadence <= 0:
254
+ cadence = default_cadence # config-level cadence applies unless overridden
255
+ collections.append(
256
+ _collect_source(source, timeout_s=timeout_s, cadence=cadence, deadline=deadline)
257
+ )
258
+ return tuple(collections)
259
+
260
+
261
+ def _collect_source(
262
+ source: ObservabilitySource,
263
+ *,
264
+ timeout_s: float,
265
+ cadence: float,
266
+ deadline: float,
267
+ ) -> SourceCollection:
268
+ if isinstance(source, LogsSource):
269
+ return _collect_logs(source, timeout_s)
270
+ if isinstance(source, InspectionSource):
271
+ return _collect_inspection(source, timeout_s)
272
+ if isinstance(source, ProbeSource):
273
+ return _collect_probe(source, timeout_s, cadence, deadline)
274
+ if isinstance(source, MetricsSource):
275
+ return _collect_metrics(source, timeout_s, cadence, deadline)
276
+ raise AssertionError(f"unhandled observability source {source!r}")
277
+
278
+
279
+ def _fail(source: ObservabilitySource, note: str) -> SourceCollection:
280
+ return SourceCollection(
281
+ source_id=source.source_id,
282
+ kind=ObservabilitySourceKind(source.kind),
283
+ ok=False,
284
+ note=note,
285
+ latency_ms=0.0,
286
+ )
287
+
288
+
289
+ def _collect_logs(source: LogsSource, timeout_s: float) -> SourceCollection:
290
+ try:
291
+ text, latency = _default_logs_runner("podman", timeout_s)(
292
+ source.container, source.tail, source.since
293
+ )
294
+ return SourceCollection(
295
+ source_id=source.source_id,
296
+ kind=ObservabilitySourceKind.LOGS,
297
+ ok=True,
298
+ note=f"collected {len(text)} chars",
299
+ latency_ms=latency,
300
+ samples=(ObservabilitySample.now(text),),
301
+ )
302
+ except Exception as exc:
303
+ return _fail(source, f"collection failed: {type(exc).__name__}: {exc}")
304
+
305
+
306
+ def _collect_inspection(source: InspectionSource, timeout_s: float) -> SourceCollection:
307
+ try:
308
+ text, latency = _default_inspect_runner("podman", timeout_s)(source.container)
309
+ return SourceCollection(
310
+ source_id=source.source_id,
311
+ kind=ObservabilitySourceKind.INSPECTION,
312
+ ok=True,
313
+ note=f"inspect {len(text)} chars",
314
+ latency_ms=latency,
315
+ samples=(ObservabilitySample.now(text),),
316
+ )
317
+ except Exception as exc:
318
+ return _fail(source, f"collection failed: {type(exc).__name__}: {exc}")
319
+
320
+
321
+ def _collect_probe(
322
+ source: ProbeSource, timeout_s: float, cadence: float, deadline: float
323
+ ) -> SourceCollection:
324
+ latency_ms_total = 0.0
325
+ samples: list[ObservabilitySample] = []
326
+ try:
327
+ runner = _default_probe_runner(timeout_s)
328
+ while True:
329
+ satisfied, detail, latency = runner(source.probe)
330
+ latency_ms_total += latency
331
+ samples.append(ObservabilitySample.now({"satisfied": satisfied, "detail": detail}))
332
+ if time.monotonic() >= deadline:
333
+ break
334
+ sleep_for = min(cadence, max(0.0, deadline - time.monotonic()))
335
+ time.sleep(sleep_for)
336
+ if time.monotonic() >= deadline:
337
+ break
338
+ return SourceCollection(
339
+ source_id=source.source_id,
340
+ kind=ObservabilitySourceKind.PROBE,
341
+ ok=True,
342
+ note=f"{len(samples)} probe sample(s)",
343
+ latency_ms=round(latency_ms_total, 3),
344
+ samples=tuple(samples),
345
+ )
346
+ except Exception as exc:
347
+ return _fail(
348
+ source,
349
+ f"collection failed: {type(exc).__name__}: {exc}",
350
+ )
351
+
352
+
353
+ def _collect_metrics(
354
+ source: MetricsSource, timeout_s: float, cadence: float, deadline: float
355
+ ) -> SourceCollection:
356
+ latency_ms_total = 0.0
357
+ samples: list[ObservabilitySample] = []
358
+ note = ""
359
+ try:
360
+ runner = _default_metrics_runner(timeout_s)
361
+ while True:
362
+ value, scrape_note, latency = runner(source.endpoint, source.metric)
363
+ latency_ms_total += latency
364
+ samples.append(ObservabilitySample.now(value))
365
+ if value is None and scrape_note:
366
+ note = scrape_note
367
+ if time.monotonic() >= deadline:
368
+ break
369
+ sleep_for = min(cadence, max(0.0, deadline - time.monotonic()))
370
+ time.sleep(sleep_for)
371
+ if time.monotonic() >= deadline:
372
+ break
373
+ return SourceCollection(
374
+ source_id=source.source_id,
375
+ kind=ObservabilitySourceKind.METRICS,
376
+ ok=note == "" or bool(samples),
377
+ note=note or f"{len(samples)} metric sample(s)",
378
+ latency_ms=round(latency_ms_total, 3),
379
+ samples=tuple(samples),
380
+ )
381
+ except Exception as exc:
382
+ return _fail(source, f"collection failed: {type(exc).__name__}: {exc}")
@@ -0,0 +1,102 @@
1
+ """Observation engine — structured event log for experiment runs (ADR-0020).
2
+
3
+ An ``Observation`` is a timestamped, typed event that occurred during an
4
+ experiment — fault injected, probe measured, recovery attempted, threshold
5
+ breached, etc. The ``ObservationLog`` is an append-only, in-memory store
6
+ that can be serialized to JSON for persistence or display.
7
+
8
+ Observations are the raw material that the evaluation system consumes.
9
+ They are never mutated after creation.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from enum import StrEnum
15
+ from typing import Any
16
+
17
+ from pydantic import BaseModel, ConfigDict, Field
18
+
19
+ from mayhem.domain.common import utc_now
20
+
21
+
22
+ class ObservationKind(StrEnum):
23
+ """Categories of observations that can occur during an experiment."""
24
+
25
+ FAULT_INJECTED = "fault.injected"
26
+ FAULT_UNDONE = "fault.undone"
27
+ PROBE_MEASURED = "probe.measured"
28
+ THRESHOLD_BREACHED = "threshold.breached"
29
+ RECOVERY_ATTEMPTED = "recovery.attempted"
30
+ RECOVERY_SUCCEEDED = "recovery.succeeded"
31
+ RECOVERY_FAILED = "recovery.failed"
32
+ STEP_STARTED = "step.started"
33
+ STEP_COMPLETED = "step.completed"
34
+ STEP_FAILED = "step.failed"
35
+ ANOMALY_DETECTED = "anomaly.detected"
36
+
37
+
38
+ class Observation(BaseModel):
39
+ """A single immutable observation event."""
40
+
41
+ model_config = ConfigDict(frozen=True)
42
+
43
+ kind: ObservationKind
44
+ run_id: str
45
+ timestamp: str = Field(default_factory=lambda: utc_now().isoformat())
46
+ source: str = "" # e.g. "executor", "janitor", "probe_runner"
47
+ data: dict[str, Any] = Field(default_factory=dict)
48
+ runtime_identity: str | None = None # canonical identity key (ADR-M1-1/1-3)
49
+
50
+
51
+ class ObservationLog:
52
+ """Append-only observation log with query helpers.
53
+
54
+ The log lives in memory for the duration of a run. Periodic snapshots
55
+ can be flushed to SQLite or JSON for durability.
56
+ """
57
+
58
+ def __init__(self) -> None:
59
+ self._observations: list[Observation] = []
60
+
61
+ def record(self, observation: Observation) -> None:
62
+ """Append an observation to the log."""
63
+ self._observations.append(observation)
64
+
65
+ def emit(
66
+ self,
67
+ kind: ObservationKind,
68
+ run_id: str,
69
+ *,
70
+ source: str = "",
71
+ **data: Any,
72
+ ) -> None:
73
+ """Shorthand: create and record in one call."""
74
+ self.record(Observation(kind=kind, run_id=run_id, source=source, data=data))
75
+
76
+ @property
77
+ def observations(self) -> tuple[Observation, ...]:
78
+ return tuple(self._observations)
79
+
80
+ def for_run(self, run_id: str) -> tuple[Observation, ...]:
81
+ """All observations for a specific run."""
82
+ return tuple(o for o in self._observations if o.run_id == run_id)
83
+
84
+ def for_kind(
85
+ self, kind: ObservationKind, *, run_id: str | None = None
86
+ ) -> tuple[Observation, ...]:
87
+ """All observations of a given kind, optionally filtered by run."""
88
+ result = (o for o in self._observations if o.kind == kind)
89
+ if run_id is not None:
90
+ result = (o for o in result if o.run_id == run_id)
91
+ return tuple(result)
92
+
93
+ def anomalies_for_run(self, run_id: str) -> tuple[Observation, ...]:
94
+ """All anomaly observations for a run."""
95
+ return self.for_kind(ObservationKind.ANOMALY_DETECTED, run_id=run_id)
96
+
97
+ def snapshot(self) -> tuple[dict[str, Any], ...]:
98
+ """Serialize all observations to JSON-compatible dicts."""
99
+ return tuple(o.model_dump(mode="json") for o in self._observations)
100
+
101
+ def __len__(self) -> int:
102
+ return len(self._observations)