shadowbox 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- shadowbox/__init__.py +3 -0
- shadowbox/api.py +136 -0
- shadowbox/cards.py +59 -0
- shadowbox/cli.py +222 -0
- shadowbox/compare.py +103 -0
- shadowbox/data/__init__.py +1 -0
- shadowbox/data/cards/cache-poison.yaml +7 -0
- shadowbox/data/cards/db-down.yaml +7 -0
- shadowbox/data/cards/latency-500ms.yaml +8 -0
- shadowbox/data/cards/queue-overflow.yaml +6 -0
- shadowbox/data/cards/slow-dependency.yaml +8 -0
- shadowbox/data/cards/traffic-10x.yaml +6 -0
- shadowbox/data/cards/zone-loss.yaml +8 -0
- shadowbox/data/example/docker-compose.yaml +13 -0
- shadowbox/data/example/model.yaml +33 -0
- shadowbox/data/example/scenarios/db-failure.yaml +12 -0
- shadowbox/dsl.py +94 -0
- shadowbox/engine.py +220 -0
- shadowbox/errors.py +38 -0
- shadowbox/importers/__init__.py +3 -0
- shadowbox/importers/compose.py +109 -0
- shadowbox/metrics.py +49 -0
- shadowbox/model.py +74 -0
- shadowbox/report.py +48 -0
- shadowbox/store.py +85 -0
- shadowbox-0.3.0.dist-info/METADATA +130 -0
- shadowbox-0.3.0.dist-info/RECORD +29 -0
- shadowbox-0.3.0.dist-info/WHEEL +4 -0
- shadowbox-0.3.0.dist-info/entry_points.txt +2 -0
shadowbox/dsl.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Safe YAML loading + structural validation for M0 (no engine)."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import yaml
|
|
7
|
+
from pydantic import ValidationError
|
|
8
|
+
|
|
9
|
+
from shadowbox.errors import (
|
|
10
|
+
CycleError,
|
|
11
|
+
RefError,
|
|
12
|
+
SchemaError,
|
|
13
|
+
TooLargeError,
|
|
14
|
+
UnsafeYamlError,
|
|
15
|
+
)
|
|
16
|
+
from shadowbox.model import MAX_COMPONENTS, Scenario, SystemModel
|
|
17
|
+
|
|
18
|
+
_UNSAFE_TAGS = ("!!python/", "!python/")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _read_text(path: Path) -> str:
|
|
22
|
+
try:
|
|
23
|
+
return path.read_text(encoding="utf-8")
|
|
24
|
+
except OSError as exc:
|
|
25
|
+
raise SchemaError(f"cannot read {path}: {exc}") from exc
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _safe_load_yaml(path: Path) -> Any:
|
|
29
|
+
raw = _read_text(path)
|
|
30
|
+
for tag in _UNSAFE_TAGS:
|
|
31
|
+
if tag in raw:
|
|
32
|
+
raise UnsafeYamlError(f"{path}: unsafe YAML tag {tag!r} rejected")
|
|
33
|
+
try:
|
|
34
|
+
return yaml.safe_load(raw)
|
|
35
|
+
except yaml.YAMLError as exc:
|
|
36
|
+
raise SchemaError(f"{path}: invalid YAML: {exc}") from exc
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _check_cycles(model: SystemModel) -> None:
|
|
40
|
+
adjacency: dict[str, list[str]] = {c.id: [] for c in model.components}
|
|
41
|
+
for conn in model.connections:
|
|
42
|
+
adjacency[conn.from_].append(conn.to)
|
|
43
|
+
visiting: set[str] = set()
|
|
44
|
+
visited: set[str] = set()
|
|
45
|
+
|
|
46
|
+
def visit(node: str, stack: list[str]) -> None:
|
|
47
|
+
if node in visiting:
|
|
48
|
+
raise CycleError(f"cycle detected: {' -> '.join([*stack, node])}")
|
|
49
|
+
if node in visited:
|
|
50
|
+
return
|
|
51
|
+
visiting.add(node)
|
|
52
|
+
for nxt in sorted(adjacency[node]):
|
|
53
|
+
visit(nxt, [*stack, node])
|
|
54
|
+
visiting.remove(node)
|
|
55
|
+
visited.add(node)
|
|
56
|
+
|
|
57
|
+
for node in sorted(adjacency):
|
|
58
|
+
visit(node, [])
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def load_model(path: Path) -> SystemModel:
|
|
62
|
+
data = _safe_load_yaml(path)
|
|
63
|
+
if not isinstance(data, dict):
|
|
64
|
+
raise SchemaError(f"{path}: top-level mapping required")
|
|
65
|
+
try:
|
|
66
|
+
model = SystemModel.model_validate(data)
|
|
67
|
+
except ValidationError as exc:
|
|
68
|
+
raise SchemaError(f"{path}: {exc}") from exc
|
|
69
|
+
if len(model.components) > MAX_COMPONENTS:
|
|
70
|
+
raise TooLargeError(f"{path}: {len(model.components)} components > {MAX_COMPONENTS}")
|
|
71
|
+
known = {c.id for c in model.components}
|
|
72
|
+
for conn in model.connections:
|
|
73
|
+
if conn.from_ not in known or conn.to not in known:
|
|
74
|
+
raise RefError(f"{path}: unknown connection endpoint {conn.from_!r} -> {conn.to!r}")
|
|
75
|
+
_check_cycles(model)
|
|
76
|
+
return model
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def load_scenario(path: Path, model: SystemModel) -> Scenario:
|
|
80
|
+
data = _safe_load_yaml(path)
|
|
81
|
+
if not isinstance(data, dict):
|
|
82
|
+
raise SchemaError(f"{path}: top-level mapping required")
|
|
83
|
+
payload = data.get("scenario", data)
|
|
84
|
+
try:
|
|
85
|
+
scenario = Scenario.model_validate(payload)
|
|
86
|
+
except ValidationError as exc:
|
|
87
|
+
raise SchemaError(f"{path}: {exc}") from exc
|
|
88
|
+
known = {c.id for c in model.components}
|
|
89
|
+
for fault in scenario.faults:
|
|
90
|
+
if fault.target not in known:
|
|
91
|
+
raise RefError(f"{path}: unknown fault target {fault.target!r}")
|
|
92
|
+
if fault.start_s + fault.duration_s > scenario.duration_s:
|
|
93
|
+
raise SchemaError(f"{path}: fault exceeds scenario duration")
|
|
94
|
+
return scenario
|
shadowbox/engine.py
ADDED
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
"""Deterministic discrete-event engine (M1).
|
|
2
|
+
|
|
3
|
+
Virtual time, 1ms resolution, single isolated RNG. No wall-clock reads.
|
|
4
|
+
Only event *counts* are kept; full event logs are deferred to M5.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import heapq
|
|
10
|
+
import random
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
|
|
13
|
+
from shadowbox.errors import TooLargeError
|
|
14
|
+
from shadowbox.model import Component, Fault, Scenario, SystemModel
|
|
15
|
+
|
|
16
|
+
MS_PER_S = 1000
|
|
17
|
+
MAX_REQUESTS = 200_000
|
|
18
|
+
MAX_SAMPLE = 500 # stored request summaries; full event log deferred to M5
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class ComponentStats:
|
|
23
|
+
busy_time_ms: int = 0
|
|
24
|
+
max_queue_depth: int = 0
|
|
25
|
+
timeouts: int = 0
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class _Runtime:
|
|
30
|
+
busy_until: list[int] = field(default_factory=list) # min-heap of slot releases
|
|
31
|
+
waiter_starts: list[int] = field(default_factory=list) # min-heap of queued starts
|
|
32
|
+
stats: ComponentStats = field(default_factory=ComponentStats)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class RequestSample:
|
|
37
|
+
correlation_id: str
|
|
38
|
+
ok: bool
|
|
39
|
+
latency_ms: int
|
|
40
|
+
failed_at: str | None
|
|
41
|
+
timed_out: bool
|
|
42
|
+
arrival_ms: int
|
|
43
|
+
finish_ms: int
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class SimulationResult:
|
|
48
|
+
total: int
|
|
49
|
+
succeeded: int
|
|
50
|
+
failed: int
|
|
51
|
+
timeouts: int
|
|
52
|
+
latencies_ms: list[int] # end-to-end, successes only
|
|
53
|
+
cascade_depth: int # max failing-hop index over failed requests
|
|
54
|
+
events_processed: int
|
|
55
|
+
component_stats: dict[str, ComponentStats]
|
|
56
|
+
sample: list[RequestSample] = field(default_factory=list) # first MAX_SAMPLE
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def build_path(model: SystemModel) -> list[Component]:
|
|
60
|
+
"""Depth-first traversal from entry points, each component visited once."""
|
|
61
|
+
by_id = {c.id: c for c in model.components}
|
|
62
|
+
incoming = {c.id: 0 for c in model.components}
|
|
63
|
+
adjacency: dict[str, list[str]] = {c.id: [] for c in model.components}
|
|
64
|
+
for conn in model.connections:
|
|
65
|
+
adjacency[conn.from_].append(conn.to)
|
|
66
|
+
incoming[conn.to] += 1
|
|
67
|
+
for neighbours in adjacency.values():
|
|
68
|
+
neighbours.sort()
|
|
69
|
+
path: list[Component] = []
|
|
70
|
+
seen: set[str] = set()
|
|
71
|
+
|
|
72
|
+
def visit(node: str) -> None:
|
|
73
|
+
if node in seen:
|
|
74
|
+
return
|
|
75
|
+
seen.add(node)
|
|
76
|
+
path.append(by_id[node])
|
|
77
|
+
for nxt in adjacency[node]:
|
|
78
|
+
visit(nxt)
|
|
79
|
+
|
|
80
|
+
for entry in sorted(n for n, deg in incoming.items() if deg == 0):
|
|
81
|
+
visit(entry)
|
|
82
|
+
return path
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _window_ms(fault: Fault) -> tuple[int, int]:
|
|
86
|
+
start = fault.start_s * MS_PER_S
|
|
87
|
+
return (start, start + fault.duration_s * MS_PER_S)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _effective_capacity(comp: Component, faults: list[Fault]) -> int:
|
|
91
|
+
capacity = comp.capacity
|
|
92
|
+
for fault in faults:
|
|
93
|
+
if fault.type == "resource-saturation":
|
|
94
|
+
return 1
|
|
95
|
+
if fault.type == "capacity-reduction":
|
|
96
|
+
capacity = min(capacity, max(1, int(comp.capacity * fault.factor)))
|
|
97
|
+
return capacity
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _process(
|
|
101
|
+
comp: Component,
|
|
102
|
+
arrival_ms: int,
|
|
103
|
+
runtime: _Runtime,
|
|
104
|
+
faults: list[Fault],
|
|
105
|
+
rng: random.Random,
|
|
106
|
+
) -> tuple[bool, int, bool, int]:
|
|
107
|
+
"""Run one component visit. Returns (ok, end_ms, timed_out, events)."""
|
|
108
|
+
events = 4 # sent, received, processing-started, processing-completed
|
|
109
|
+
for fault in faults:
|
|
110
|
+
events += 1 # failure-injected observed
|
|
111
|
+
if fault.type == "unavailable":
|
|
112
|
+
return (False, arrival_ms, False, events)
|
|
113
|
+
if fault.type in ("error-rate", "packet-loss") and rng.random() < fault.probability:
|
|
114
|
+
return (False, arrival_ms, False, events)
|
|
115
|
+
extra_ms = sum(f.extra_ms for f in faults if f.type == "latency")
|
|
116
|
+
jitter = rng.randint(0, comp.latency_ms.jitter_ms) if comp.latency_ms.jitter_ms else 0
|
|
117
|
+
service_ms = comp.latency_ms.base + jitter + extra_ms
|
|
118
|
+
capacity = _effective_capacity(comp, faults)
|
|
119
|
+
while runtime.busy_until and runtime.busy_until[0] <= arrival_ms:
|
|
120
|
+
heapq.heappop(runtime.busy_until)
|
|
121
|
+
if len(runtime.busy_until) < capacity:
|
|
122
|
+
start_ms = arrival_ms
|
|
123
|
+
else:
|
|
124
|
+
while runtime.waiter_starts and runtime.waiter_starts[0] <= arrival_ms:
|
|
125
|
+
heapq.heappop(runtime.waiter_starts)
|
|
126
|
+
if len(runtime.waiter_starts) >= comp.queue_size:
|
|
127
|
+
return (False, arrival_ms, False, events) # queue-full drop
|
|
128
|
+
start_ms = heapq.heappop(runtime.busy_until) # consume the earliest slot
|
|
129
|
+
heapq.heappush(runtime.waiter_starts, start_ms)
|
|
130
|
+
depth_now = len(runtime.waiter_starts)
|
|
131
|
+
runtime.stats.max_queue_depth = max(runtime.stats.max_queue_depth, depth_now)
|
|
132
|
+
wait_ms = start_ms - arrival_ms
|
|
133
|
+
if wait_ms >= comp.timeout_ms:
|
|
134
|
+
runtime.stats.timeouts += 1
|
|
135
|
+
return (False, arrival_ms + comp.timeout_ms, True, events) # gave up waiting
|
|
136
|
+
if service_ms > comp.timeout_ms - wait_ms:
|
|
137
|
+
busy_until = start_ms + (comp.timeout_ms - wait_ms)
|
|
138
|
+
runtime.stats.busy_time_ms += busy_until - start_ms
|
|
139
|
+
runtime.stats.timeouts += 1
|
|
140
|
+
heapq.heappush(runtime.busy_until, busy_until)
|
|
141
|
+
return (False, busy_until, True, events)
|
|
142
|
+
busy_until = start_ms + service_ms
|
|
143
|
+
runtime.stats.busy_time_ms += service_ms
|
|
144
|
+
heapq.heappush(runtime.busy_until, busy_until)
|
|
145
|
+
return (True, busy_until, False, events)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
@dataclass
|
|
149
|
+
class _RequestOutcome:
|
|
150
|
+
ok: bool
|
|
151
|
+
latency_ms: int
|
|
152
|
+
failed_at: str | None
|
|
153
|
+
timed_out: bool
|
|
154
|
+
finish_ms: int
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def simulate(model: SystemModel, scenario: Scenario, seed: int) -> SimulationResult:
|
|
158
|
+
"""Run the scenario deterministically; same inputs always give same outputs."""
|
|
159
|
+
total = scenario.workload.rate_rps * scenario.duration_s
|
|
160
|
+
if total > MAX_REQUESTS:
|
|
161
|
+
raise TooLargeError(f"{total} requests exceed cap of {MAX_REQUESTS}")
|
|
162
|
+
path = build_path(model)
|
|
163
|
+
windows: dict[str, list[tuple[int, int, Fault]]] = {}
|
|
164
|
+
for fault in scenario.faults:
|
|
165
|
+
start, end = _window_ms(fault)
|
|
166
|
+
windows.setdefault(fault.target, []).append((start, end, fault))
|
|
167
|
+
runtimes = {c.id: _Runtime() for c in model.components}
|
|
168
|
+
rng = random.Random(seed)
|
|
169
|
+
interval_ms = MS_PER_S / scenario.workload.rate_rps
|
|
170
|
+
latencies: list[int] = []
|
|
171
|
+
sample: list[RequestSample] = []
|
|
172
|
+
succeeded = 0
|
|
173
|
+
failed = 0
|
|
174
|
+
timeouts = 0
|
|
175
|
+
cascade_depth = 0
|
|
176
|
+
events = 0
|
|
177
|
+
for i in range(total):
|
|
178
|
+
arrival_ms = int(i * interval_ms)
|
|
179
|
+
cursor_ms = arrival_ms
|
|
180
|
+
events += 2 # request-created, request-completed
|
|
181
|
+
outcome = _RequestOutcome(True, 0, None, False, arrival_ms)
|
|
182
|
+
for depth, comp in enumerate(path):
|
|
183
|
+
active = windows.get(comp.id, [])
|
|
184
|
+
faults = [f for (start, end, f) in active if start <= cursor_ms < end]
|
|
185
|
+
ok, end_ms, timed_out, delta = _process(comp, cursor_ms, runtimes[comp.id], faults, rng)
|
|
186
|
+
events += delta
|
|
187
|
+
if not ok:
|
|
188
|
+
failed += 1
|
|
189
|
+
cascade_depth = max(cascade_depth, depth)
|
|
190
|
+
timeouts += 1 if timed_out else 0
|
|
191
|
+
outcome = _RequestOutcome(False, 0, comp.id, timed_out, end_ms)
|
|
192
|
+
break
|
|
193
|
+
cursor_ms = end_ms
|
|
194
|
+
else:
|
|
195
|
+
succeeded += 1
|
|
196
|
+
latencies.append(cursor_ms - arrival_ms)
|
|
197
|
+
outcome = _RequestOutcome(True, cursor_ms - arrival_ms, None, False, cursor_ms)
|
|
198
|
+
if len(sample) < MAX_SAMPLE:
|
|
199
|
+
sample.append(
|
|
200
|
+
RequestSample(
|
|
201
|
+
correlation_id=f"req-{i:06d}",
|
|
202
|
+
ok=outcome.ok,
|
|
203
|
+
latency_ms=outcome.latency_ms,
|
|
204
|
+
failed_at=outcome.failed_at,
|
|
205
|
+
timed_out=outcome.timed_out,
|
|
206
|
+
arrival_ms=arrival_ms,
|
|
207
|
+
finish_ms=outcome.finish_ms,
|
|
208
|
+
)
|
|
209
|
+
)
|
|
210
|
+
return SimulationResult(
|
|
211
|
+
total=total,
|
|
212
|
+
succeeded=succeeded,
|
|
213
|
+
failed=failed,
|
|
214
|
+
timeouts=timeouts,
|
|
215
|
+
latencies_ms=latencies,
|
|
216
|
+
cascade_depth=cascade_depth,
|
|
217
|
+
events_processed=events,
|
|
218
|
+
component_stats={cid: rt.stats for cid, rt in runtimes.items()},
|
|
219
|
+
sample=sample,
|
|
220
|
+
)
|
shadowbox/errors.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Stable error codes for model/scenario validation (M0 contract)."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class ShadowBoxError(Exception):
|
|
5
|
+
"""Base error with a stable machine-readable code."""
|
|
6
|
+
|
|
7
|
+
code: str = "E_UNKNOWN"
|
|
8
|
+
|
|
9
|
+
def __init__(self, message: str) -> None:
|
|
10
|
+
super().__init__(message)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class RefError(ShadowBoxError):
|
|
14
|
+
code = "E_REF"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class CycleError(ShadowBoxError):
|
|
18
|
+
code = "E_CYCLE"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class SchemaError(ShadowBoxError):
|
|
22
|
+
code = "E_SCHEMA"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class TooLargeError(ShadowBoxError):
|
|
26
|
+
code = "E_TOO_LARGE"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class UnsafeYamlError(ShadowBoxError):
|
|
30
|
+
code = "E_UNSAFE_YAML"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class SeedMismatchWarning(ShadowBoxError):
|
|
34
|
+
code = "E_SEED_MISMATCH"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class ExistsError(ShadowBoxError):
|
|
38
|
+
code = "E_EXISTS"
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Import a docker-compose file into a ShadowBox model (M2).
|
|
2
|
+
|
|
3
|
+
Every performance field is an estimated default, never a measurement:
|
|
4
|
+
the caller must calibrate before trusting any simulation output.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
import yaml
|
|
11
|
+
|
|
12
|
+
from shadowbox.errors import SchemaError, UnsafeYamlError
|
|
13
|
+
from shadowbox.model import SystemModel
|
|
14
|
+
|
|
15
|
+
_DB_HINTS = ("postgres", "mysql", "mariadb", "mongo", "sqlite", "cockroach", "cassandra")
|
|
16
|
+
_CACHE_HINTS = ("redis", "memcached", "valkey", "cache")
|
|
17
|
+
|
|
18
|
+
_DEFAULTS: dict[str, dict[str, Any]] = {
|
|
19
|
+
"service": {
|
|
20
|
+
"capacity": 20,
|
|
21
|
+
"latency_ms": {"base": 20, "jitter_ms": 5},
|
|
22
|
+
"timeout_ms": 1000,
|
|
23
|
+
"queue_size": 100,
|
|
24
|
+
"queue_policy": "drop",
|
|
25
|
+
},
|
|
26
|
+
"database": {
|
|
27
|
+
"capacity": 10,
|
|
28
|
+
"latency_ms": {"base": 5, "jitter_ms": 1},
|
|
29
|
+
"timeout_ms": 500,
|
|
30
|
+
"queue_size": 50,
|
|
31
|
+
"queue_policy": "fifo",
|
|
32
|
+
},
|
|
33
|
+
"cache": {
|
|
34
|
+
"capacity": 50,
|
|
35
|
+
"latency_ms": {"base": 2, "jitter_ms": 1},
|
|
36
|
+
"timeout_ms": 200,
|
|
37
|
+
"queue_size": 200,
|
|
38
|
+
"queue_policy": "drop",
|
|
39
|
+
},
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _infer_type(service: str, image: str) -> str:
|
|
44
|
+
lowered = image.lower()
|
|
45
|
+
if any(hint in lowered for hint in _DB_HINTS):
|
|
46
|
+
return "database"
|
|
47
|
+
if any(hint in lowered for hint in _CACHE_HINTS):
|
|
48
|
+
return "cache"
|
|
49
|
+
return "service"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _depends_on(spec: Any) -> list[str]:
|
|
53
|
+
if spec is None:
|
|
54
|
+
return []
|
|
55
|
+
if isinstance(spec, list):
|
|
56
|
+
return [str(item) for item in spec]
|
|
57
|
+
if isinstance(spec, dict):
|
|
58
|
+
return sorted(str(key) for key in spec)
|
|
59
|
+
return []
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _links(spec: Any) -> list[str]:
|
|
63
|
+
if not isinstance(spec, list):
|
|
64
|
+
return []
|
|
65
|
+
return [str(item).split(":")[0] for item in spec]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def import_compose(path: Path) -> tuple[SystemModel, list[str]]:
|
|
69
|
+
"""Parse compose file; returns (model, warnings). All metrics are estimates."""
|
|
70
|
+
try:
|
|
71
|
+
raw_text = path.read_text(encoding="utf-8")
|
|
72
|
+
except OSError as exc:
|
|
73
|
+
raise SchemaError(f"cannot read {path}: {exc}") from exc
|
|
74
|
+
if "!!python/" in raw_text or "!python/" in raw_text:
|
|
75
|
+
raise UnsafeYamlError(f"{path}: unsafe YAML tag rejected")
|
|
76
|
+
try:
|
|
77
|
+
data = yaml.safe_load(raw_text)
|
|
78
|
+
except yaml.YAMLError as exc:
|
|
79
|
+
raise SchemaError(f"{path}: invalid YAML: {exc}") from exc
|
|
80
|
+
if not isinstance(data, dict) or not isinstance(data.get("services"), dict):
|
|
81
|
+
raise SchemaError(f"{path}: top-level `services` mapping required")
|
|
82
|
+
services: dict[str, Any] = data["services"]
|
|
83
|
+
if not services:
|
|
84
|
+
raise SchemaError(f"{path}: no services defined")
|
|
85
|
+
components: list[dict[str, Any]] = []
|
|
86
|
+
connections: list[dict[str, Any]] = []
|
|
87
|
+
warnings: list[str] = []
|
|
88
|
+
for name in sorted(services):
|
|
89
|
+
spec = services[name] if isinstance(services[name], dict) else {}
|
|
90
|
+
image = str(spec.get("image", ""))
|
|
91
|
+
kind = _infer_type(name, image)
|
|
92
|
+
fields = dict(_DEFAULTS[kind])
|
|
93
|
+
components.append({"id": name, "type": kind, **fields})
|
|
94
|
+
shown_image = image if image else "(none)"
|
|
95
|
+
warnings.append(f"{name!r}: {kind} inferred from {shown_image}; fields estimated")
|
|
96
|
+
for dep in sorted(set(_depends_on(spec.get("depends_on")) + _links(spec.get("links")))):
|
|
97
|
+
connections.append({"from": name, "to": dep})
|
|
98
|
+
known_names = set(services)
|
|
99
|
+
kept: list[dict[str, Any]] = []
|
|
100
|
+
for conn in connections:
|
|
101
|
+
if conn["to"] not in known_names:
|
|
102
|
+
warnings.append(f"dropped {conn['from']} -> {conn['to']} (outside compose)")
|
|
103
|
+
continue
|
|
104
|
+
kept.append(conn)
|
|
105
|
+
try:
|
|
106
|
+
model = SystemModel.model_validate({"components": components, "connections": kept})
|
|
107
|
+
except Exception as exc:
|
|
108
|
+
raise SchemaError(f"{path}: cannot build model: {exc}") from exc
|
|
109
|
+
return (model, warnings)
|
shadowbox/metrics.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Metric aggregation over a finished simulation (exact percentiles)."""
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
|
|
5
|
+
from shadowbox.engine import MS_PER_S, SimulationResult
|
|
6
|
+
from shadowbox.model import SystemModel
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def percentile(sorted_values: list[int], pct: float) -> float:
|
|
10
|
+
"""Nearest-rank percentile; empty input yields 0.0 (no successful requests)."""
|
|
11
|
+
if not sorted_values:
|
|
12
|
+
return 0.0
|
|
13
|
+
rank = max(1, math.ceil((pct / 100.0) * len(sorted_values)))
|
|
14
|
+
return float(sorted_values[rank - 1])
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def summarize(
|
|
18
|
+
model: SystemModel,
|
|
19
|
+
result: SimulationResult,
|
|
20
|
+
duration_s: int,
|
|
21
|
+
) -> dict[str, object]:
|
|
22
|
+
"""Aggregate counts, exact latency percentiles, and per-component stats."""
|
|
23
|
+
ordered = sorted(result.latencies_ms)
|
|
24
|
+
by_id = {c.id: c for c in model.components}
|
|
25
|
+
components = {}
|
|
26
|
+
for cid in sorted(result.component_stats):
|
|
27
|
+
stats = result.component_stats[cid]
|
|
28
|
+
capacity = by_id[cid].capacity
|
|
29
|
+
components[cid] = {
|
|
30
|
+
"utilization": stats.busy_time_ms / (capacity * duration_s * MS_PER_S),
|
|
31
|
+
"queue_depth": stats.max_queue_depth,
|
|
32
|
+
"timeouts": stats.timeouts,
|
|
33
|
+
}
|
|
34
|
+
total = result.total
|
|
35
|
+
return {
|
|
36
|
+
"throughput_rps": result.succeeded / duration_s if duration_s else 0.0,
|
|
37
|
+
"error_rate": result.failed / total if total else 0.0,
|
|
38
|
+
"latency_ms": {
|
|
39
|
+
"p50": percentile(ordered, 50),
|
|
40
|
+
"p90": percentile(ordered, 90),
|
|
41
|
+
"p95": percentile(ordered, 95),
|
|
42
|
+
"p99": percentile(ordered, 99),
|
|
43
|
+
},
|
|
44
|
+
"timeouts": result.timeouts,
|
|
45
|
+
"retries": 0, # auto-retry disabled in MVP; scenario max_retries lands in M3
|
|
46
|
+
"cascade_depth": result.cascade_depth,
|
|
47
|
+
"requests": {"total": total, "succeeded": result.succeeded, "failed": result.failed},
|
|
48
|
+
"components": components,
|
|
49
|
+
}
|
shadowbox/model.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Pydantic domain model for M0 (structure only, no simulation)."""
|
|
2
|
+
|
|
3
|
+
from typing import Literal
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel, Field, field_validator
|
|
6
|
+
|
|
7
|
+
QueuePolicy = Literal["drop", "fifo"]
|
|
8
|
+
ConnectionKind = Literal["sync"]
|
|
9
|
+
|
|
10
|
+
MAX_COMPONENTS = 200
|
|
11
|
+
MAX_EVENTS = 500_000
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class LatencyMs(BaseModel):
|
|
15
|
+
base: int = Field(ge=0)
|
|
16
|
+
jitter_ms: int = Field(default=0, ge=0)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class Component(BaseModel):
|
|
20
|
+
id: str = Field(min_length=1)
|
|
21
|
+
type: Literal["service", "database", "cache"]
|
|
22
|
+
capacity: int = Field(gt=0)
|
|
23
|
+
latency_ms: LatencyMs
|
|
24
|
+
timeout_ms: int = Field(gt=0)
|
|
25
|
+
queue_size: int = Field(gt=0)
|
|
26
|
+
queue_policy: QueuePolicy
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Connection(BaseModel):
|
|
30
|
+
from_: str = Field(alias="from")
|
|
31
|
+
to: str
|
|
32
|
+
kind: ConnectionKind = "sync"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class SystemModel(BaseModel):
|
|
36
|
+
components: list[Component] = Field(min_length=1)
|
|
37
|
+
connections: list[Connection] = Field(default_factory=list)
|
|
38
|
+
|
|
39
|
+
@field_validator("components")
|
|
40
|
+
@classmethod
|
|
41
|
+
def unique_ids(cls, items: list[Component]) -> list[Component]:
|
|
42
|
+
seen = {c.id for c in items}
|
|
43
|
+
if len(seen) != len(items):
|
|
44
|
+
raise ValueError("duplicate component id")
|
|
45
|
+
return items
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class Fault(BaseModel):
|
|
49
|
+
target: str = Field(min_length=1)
|
|
50
|
+
type: Literal[
|
|
51
|
+
"unavailable",
|
|
52
|
+
"latency",
|
|
53
|
+
"error-rate",
|
|
54
|
+
"capacity-reduction",
|
|
55
|
+
"packet-loss",
|
|
56
|
+
"resource-saturation",
|
|
57
|
+
]
|
|
58
|
+
start_s: int = Field(ge=0)
|
|
59
|
+
duration_s: int = Field(ge=0)
|
|
60
|
+
extra_ms: int = Field(default=0, ge=0)
|
|
61
|
+
probability: float = Field(default=0.0, ge=0.0, le=1.0)
|
|
62
|
+
factor: float = Field(default=1.0, gt=0.0, le=1.0)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class Workload(BaseModel):
|
|
66
|
+
rate_rps: int = Field(gt=0)
|
|
67
|
+
arrival: Literal["deterministic"] = "deterministic"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class Scenario(BaseModel):
|
|
71
|
+
name: str = Field(min_length=1)
|
|
72
|
+
duration_s: int = Field(gt=0)
|
|
73
|
+
workload: Workload
|
|
74
|
+
faults: list[Fault] = Field(default_factory=list)
|
shadowbox/report.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""Report envelope: metrics plus honesty metadata and a reproducibility hash."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
ENGINE_VERSION = "shadowbox-m1/0.1.0"
|
|
8
|
+
|
|
9
|
+
ASSUMPTIONS = [
|
|
10
|
+
"sync calls only",
|
|
11
|
+
"depth-first traversal, each component visited once per request",
|
|
12
|
+
"deterministic uniform arrivals",
|
|
13
|
+
"no auto-retry (max_retries unsupported in M1)",
|
|
14
|
+
"no GC pause or infrastructure noise model",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
LIMITATIONS = [
|
|
18
|
+
"uncalibrated latencies: outputs are what-if illustrations, not predictions",
|
|
19
|
+
"queueing is per-component FIFO with fixed capacity; no backpressure protocol",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def canonical_hash(payload: dict[str, Any]) -> str:
|
|
24
|
+
"""SHA-256 over canonical JSON (sorted keys, compact separators)."""
|
|
25
|
+
raw = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
|
26
|
+
return hashlib.sha256(raw).hexdigest()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def build_report(
|
|
30
|
+
model: dict[str, Any],
|
|
31
|
+
scenario: dict[str, Any],
|
|
32
|
+
seed: int,
|
|
33
|
+
metrics: dict[str, Any],
|
|
34
|
+
events_processed: int,
|
|
35
|
+
) -> dict[str, Any]:
|
|
36
|
+
"""Envelope the metrics; the hash covers model+scenario+seed+metrics only."""
|
|
37
|
+
payload = {"model": model, "scenario": scenario, "seed": seed, "metrics": metrics}
|
|
38
|
+
return {
|
|
39
|
+
"engine": ENGINE_VERSION,
|
|
40
|
+
"seed": seed,
|
|
41
|
+
"metrics": metrics,
|
|
42
|
+
"metrics_hash": canonical_hash(payload),
|
|
43
|
+
"assumptions": ASSUMPTIONS,
|
|
44
|
+
"limitations": LIMITATIONS,
|
|
45
|
+
"confidence": "low",
|
|
46
|
+
"calibration_source": "none",
|
|
47
|
+
"events_processed": events_processed,
|
|
48
|
+
}
|