admitperf 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- admitperf/__init__.py +43 -0
- admitperf/cli.py +282 -0
- admitperf/comparison.py +326 -0
- admitperf/core/__init__.py +29 -0
- admitperf/core/decision.py +47 -0
- admitperf/core/log.py +101 -0
- admitperf/core/policy.py +231 -0
- admitperf/core/reasons.py +33 -0
- admitperf/core/signal.py +94 -0
- admitperf/core/signals.py +87 -0
- admitperf/core/verdict.py +18 -0
- admitperf/dashboard/.streamlit/config.toml +12 -0
- admitperf/dashboard/__init__.py +0 -0
- admitperf/dashboard/app.py +770 -0
- admitperf/dashboard/assets/icon.png +0 -0
- admitperf/dashboard/assets/logo.png +0 -0
- admitperf/dashboard/theme.py +157 -0
- admitperf/demo.py +274 -0
- admitperf/discover.py +85 -0
- admitperf/experiment.py +42 -0
- admitperf/item.py +22 -0
- admitperf/items.py +188 -0
- admitperf/policies/__init__.py +18 -0
- admitperf/policies/dual_gate.py +36 -0
- admitperf/policies/kv_threshold.py +35 -0
- admitperf/policies/no_admission.py +27 -0
- admitperf/policies/queue_depth.py +35 -0
- admitperf/policy_runs.py +31 -0
- admitperf/report.py +340 -0
- admitperf/status.py +19 -0
- admitperf/terminology.py +207 -0
- admitperf/watch.py +107 -0
- admitperf-0.0.1.dist-info/METADATA +380 -0
- admitperf-0.0.1.dist-info/RECORD +37 -0
- admitperf-0.0.1.dist-info/WHEEL +4 -0
- admitperf-0.0.1.dist-info/entry_points.txt +2 -0
- admitperf-0.0.1.dist-info/licenses/LICENSE +21 -0
admitperf/__init__.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""AdmitPerf — standardized admission control for LLM inference.
|
|
2
|
+
|
|
3
|
+
Bring your own infra. Your gateway already has raw metrics; AdmitPerf turns them
|
|
4
|
+
into signals, policies decide on signals, and every decision is recorded in a form
|
|
5
|
+
a report can compare across deployments.
|
|
6
|
+
|
|
7
|
+
from admitperf import Policy
|
|
8
|
+
from admitperf.core.signals import KV_PRESSURE
|
|
9
|
+
|
|
10
|
+
class KvWall(Policy):
|
|
11
|
+
name = "kv_wall"
|
|
12
|
+
|
|
13
|
+
def decide(self, metrics):
|
|
14
|
+
if KV_PRESSURE.read(metrics) >= self.threshold:
|
|
15
|
+
return self.reject("kv_pressure")
|
|
16
|
+
return self.admit()
|
|
17
|
+
|
|
18
|
+
`admitperf.core` is what runs in your request path and imports nothing but the
|
|
19
|
+
standard library. Everything else is a consumer of what it writes.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
25
|
+
|
|
26
|
+
from admitperf.core import REASONS, Decision, Log, Policy, Signal, Verdict
|
|
27
|
+
|
|
28
|
+
try:
|
|
29
|
+
# Read from installed metadata rather than restating it. Two copies of a version
|
|
30
|
+
# number is one too many, and the stale one is always the one a report quotes.
|
|
31
|
+
__version__ = version("admitperf")
|
|
32
|
+
except PackageNotFoundError: # a source tree with no install
|
|
33
|
+
__version__ = "0.0.0+unknown"
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"REASONS",
|
|
37
|
+
"Decision",
|
|
38
|
+
"Log",
|
|
39
|
+
"Policy",
|
|
40
|
+
"Signal",
|
|
41
|
+
"Verdict",
|
|
42
|
+
"__version__",
|
|
43
|
+
]
|
admitperf/cli.py
ADDED
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
"""The CLI. Three verbs, and none of them is in a request path.
|
|
2
|
+
|
|
3
|
+
admitperf demo try it, no GPU needed
|
|
4
|
+
admitperf watch http://host:8000/metrics --for 1h -o trace.jsonl
|
|
5
|
+
admitperf report trace.jsonl
|
|
6
|
+
admitperf compare baseline.jsonl with-policy.jsonl
|
|
7
|
+
admitperf dashboard
|
|
8
|
+
|
|
9
|
+
`admitperf.core` never fetches anything, because a network round trip has no place
|
|
10
|
+
on an admission decision. These commands run out of band, so `watch` scraping a
|
|
11
|
+
metrics endpoint for you is a convenience rather than a contradiction.
|
|
12
|
+
|
|
13
|
+
Nothing here provisions, serves, or generates load. That is your infra's job.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
import click
|
|
22
|
+
|
|
23
|
+
from admitperf import __version__
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _seconds(text: str) -> float:
|
|
27
|
+
"""`90`, `30s`, `15m`, `2h` — because `--for 3600` is how you mean an hour and
|
|
28
|
+
type a mistake."""
|
|
29
|
+
units = {"s": 1, "m": 60, "h": 3600}
|
|
30
|
+
if text and text[-1] in units:
|
|
31
|
+
return float(text[:-1]) * units[text[-1]]
|
|
32
|
+
return float(text)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@click.group()
|
|
36
|
+
@click.version_option(__version__)
|
|
37
|
+
def main() -> None:
|
|
38
|
+
"""Standardized admission control for LLM inference."""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@main.command()
|
|
42
|
+
@click.argument("url")
|
|
43
|
+
@click.option("--for", "duration", default="60s", show_default=True, help="90s · 15m · 2h")
|
|
44
|
+
@click.option("--every", default=1.0, show_default=True, help="seconds between scrapes")
|
|
45
|
+
@click.option("-o", "--out", default="trace.jsonl", show_default=True)
|
|
46
|
+
def watch(url: str, duration: str, every: float, out: str) -> None:
|
|
47
|
+
"""Record what your cluster is doing. No code in your request path.
|
|
48
|
+
|
|
49
|
+
\b
|
|
50
|
+
admitperf watch http://localhost:8000/metrics --for 1h -o trace.jsonl
|
|
51
|
+
|
|
52
|
+
Answers the question no surveyed paper answers — could a policy have fired
|
|
53
|
+
here, and which signal actually moved — before you change any code.
|
|
54
|
+
"""
|
|
55
|
+
from admitperf.watch import Watch
|
|
56
|
+
|
|
57
|
+
seconds = _seconds(duration)
|
|
58
|
+
click.echo(f"watching {url} every {every:g}s for {seconds:g}s -> {out}")
|
|
59
|
+
w = Watch(url, out, interval=every).run(seconds)
|
|
60
|
+
click.echo(f"{w.samples} samples, {w.failures} failed scrape(s)")
|
|
61
|
+
if w.samples == 0:
|
|
62
|
+
raise SystemExit(f"nothing was recorded. Is {url} reachable and serving Prometheus text?")
|
|
63
|
+
click.echo(f"\n admitperf report {out}")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@main.command()
|
|
67
|
+
@click.argument("log", type=click.Path(exists=True, dir_okay=False))
|
|
68
|
+
@click.option(
|
|
69
|
+
"--check",
|
|
70
|
+
is_flag=True,
|
|
71
|
+
help="exit non-zero unless the log can support a claim about the policy",
|
|
72
|
+
)
|
|
73
|
+
def report(log: str, check: bool) -> None:
|
|
74
|
+
"""The finding from a log.
|
|
75
|
+
|
|
76
|
+
\b
|
|
77
|
+
admitperf report trace.jsonl
|
|
78
|
+
admitperf report decisions.jsonl --check # for CI
|
|
79
|
+
"""
|
|
80
|
+
from admitperf.report import Report
|
|
81
|
+
|
|
82
|
+
r = Report.from_log(log)
|
|
83
|
+
click.echo(r.text())
|
|
84
|
+
if check and r.verdict() != "LIVE":
|
|
85
|
+
raise SystemExit(f"verdict is {r.verdict()}: this log cannot support a claim")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@main.command("experiments")
|
|
89
|
+
def list_experiments() -> None:
|
|
90
|
+
"""Every experiment found under this directory, and its arms."""
|
|
91
|
+
from admitperf.discover import experiments
|
|
92
|
+
|
|
93
|
+
found = experiments()
|
|
94
|
+
if not found:
|
|
95
|
+
click.echo("no logs here. `admitperf demo --repeats 5` makes some.")
|
|
96
|
+
return
|
|
97
|
+
for name, exp in sorted(found.items()):
|
|
98
|
+
click.echo(f"\n{name} ({exp.runs} run(s))")
|
|
99
|
+
if exp.notes:
|
|
100
|
+
click.echo(f" {exp.notes}")
|
|
101
|
+
for runs in exp.policies.values():
|
|
102
|
+
click.echo(f" {runs.label}")
|
|
103
|
+
if exp.baseline is None:
|
|
104
|
+
click.echo(" !! no baseline, so nothing can be compared against")
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@main.command()
|
|
108
|
+
@click.option("--repeats", default=1, show_default=True, help="runs per policy")
|
|
109
|
+
@click.option(
|
|
110
|
+
"-o",
|
|
111
|
+
"--out",
|
|
112
|
+
default="experiments",
|
|
113
|
+
show_default=True,
|
|
114
|
+
help="logs land in <out>/<experiment>/<policy>/r<n>.jsonl",
|
|
115
|
+
)
|
|
116
|
+
def demo(repeats: int, out: str) -> None:
|
|
117
|
+
"""Try the whole thing in one command. No GPU, no cluster, no config.
|
|
118
|
+
|
|
119
|
+
\b
|
|
120
|
+
admitperf demo
|
|
121
|
+
admitperf demo --repeats 5
|
|
122
|
+
|
|
123
|
+
Drives identical traffic through a simulated engine three times — admitting
|
|
124
|
+
everything, with a KV-pressure threshold, and with a queue bound derived from the
|
|
125
|
+
SLO — then compares the reports.
|
|
126
|
+
|
|
127
|
+
A demo, not an experiment runner: it takes no engine URL and no workload, because
|
|
128
|
+
AdmitPerf does not generate load. Against a real cluster your own load generator
|
|
129
|
+
drives traffic and `admitperf watch` records it.
|
|
130
|
+
"""
|
|
131
|
+
from admitperf.demo import main as run_demo
|
|
132
|
+
|
|
133
|
+
run_demo(repeats=repeats, out=out, echo=click.echo)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
@main.command()
|
|
137
|
+
@click.argument("baseline", type=click.Path(exists=True), required=False)
|
|
138
|
+
@click.argument("policy", type=click.Path(exists=True), required=False)
|
|
139
|
+
@click.option(
|
|
140
|
+
"--experiment",
|
|
141
|
+
"-e",
|
|
142
|
+
help="compare every policy in a named experiment against its baseline",
|
|
143
|
+
)
|
|
144
|
+
@click.option(
|
|
145
|
+
"--check",
|
|
146
|
+
is_flag=True,
|
|
147
|
+
help="exit non-zero unless the comparison is trustworthy: the policy fired, both "
|
|
148
|
+
"runs faced the same load, and outcomes were recorded",
|
|
149
|
+
)
|
|
150
|
+
def compare(baseline: str | None, policy: str | None, experiment: str | None, check: bool) -> None:
|
|
151
|
+
"""What did the policy buy?
|
|
152
|
+
|
|
153
|
+
\b
|
|
154
|
+
admitperf compare --experiment demo-overload-2.5x # every arm vs its baseline
|
|
155
|
+
admitperf compare baseline.jsonl with-policy.jsonl # or two paths directly
|
|
156
|
+
|
|
157
|
+
The question the package exists to answer. Four checks come before any number,
|
|
158
|
+
because a comparison between a policy that never fired and a baseline is two
|
|
159
|
+
measurements of the same configuration.
|
|
160
|
+
"""
|
|
161
|
+
from admitperf.comparison import Comparison
|
|
162
|
+
from admitperf.discover import find
|
|
163
|
+
from admitperf.report import Report
|
|
164
|
+
|
|
165
|
+
if experiment:
|
|
166
|
+
exp = find(experiment)
|
|
167
|
+
if exp.baseline is None:
|
|
168
|
+
raise SystemExit(
|
|
169
|
+
f"experiment {experiment!r} has no baseline, so there is nothing to "
|
|
170
|
+
"compare against. Run a policy with `baseline = True` — NoAdmission is one."
|
|
171
|
+
)
|
|
172
|
+
if not exp.candidates:
|
|
173
|
+
raise SystemExit(f"experiment {experiment!r} has only a baseline")
|
|
174
|
+
if exp.notes:
|
|
175
|
+
click.echo(f"{experiment} — {exp.notes}\n")
|
|
176
|
+
failed = False
|
|
177
|
+
base = [Report.from_log(p) for p in exp.baseline.logs]
|
|
178
|
+
for candidate in exp.candidates:
|
|
179
|
+
click.echo(f"\n### {exp.baseline.label} vs {candidate.label}")
|
|
180
|
+
c = Comparison(base, [Report.from_log(p) for p in candidate.logs])
|
|
181
|
+
click.echo(c.text())
|
|
182
|
+
failed = failed or not (c.fired() and c.comparable_load() and c.measurable())
|
|
183
|
+
if check and failed:
|
|
184
|
+
raise SystemExit("at least one pair cannot support a claim")
|
|
185
|
+
return
|
|
186
|
+
|
|
187
|
+
if not (baseline and policy):
|
|
188
|
+
raise SystemExit("give two logs, or --experiment NAME. `admitperf dashboard` lists them.")
|
|
189
|
+
c = Comparison.from_logs(baseline, policy)
|
|
190
|
+
click.echo(c.text())
|
|
191
|
+
if check and not (c.fired() and c.comparable_load() and c.measurable()):
|
|
192
|
+
raise SystemExit("this pair cannot support a claim — see CAN YOU TRUST THIS above")
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _free_port(start: int, tries: int = 20) -> int:
|
|
196
|
+
"""The first free port at or above `start`.
|
|
197
|
+
|
|
198
|
+
Bound on all interfaces and WITHOUT SO_REUSEADDR, because that is what Streamlit
|
|
199
|
+
does. A probe against 127.0.0.1 with SO_REUSEADDR set reports a port as free while
|
|
200
|
+
Streamlit then fails to bind it — which is exactly how "Port 8501 is not available"
|
|
201
|
+
became a dead end instead of a retry.
|
|
202
|
+
"""
|
|
203
|
+
import socket
|
|
204
|
+
|
|
205
|
+
for offset in range(tries):
|
|
206
|
+
candidate = start + offset
|
|
207
|
+
with socket.socket() as s:
|
|
208
|
+
try:
|
|
209
|
+
s.bind(("", candidate))
|
|
210
|
+
except OSError:
|
|
211
|
+
continue
|
|
212
|
+
return candidate
|
|
213
|
+
raise SystemExit(f"no free port between {start} and {start + tries}")
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
@main.command()
|
|
217
|
+
@click.option("--port", default=8501, show_default=True, help="or the next one free")
|
|
218
|
+
def dashboard(port: int) -> None:
|
|
219
|
+
"""Open the dashboard: logs, findings, and what a policy bought.
|
|
220
|
+
|
|
221
|
+
\b
|
|
222
|
+
admitperf demo --repeats 5 # make some logs first
|
|
223
|
+
admitperf dashboard
|
|
224
|
+
|
|
225
|
+
It discovers logs under the current directory, so run it where your logs are.
|
|
226
|
+
"""
|
|
227
|
+
import subprocess
|
|
228
|
+
|
|
229
|
+
try:
|
|
230
|
+
import streamlit # noqa: F401
|
|
231
|
+
except ImportError:
|
|
232
|
+
raise SystemExit("the dashboard is an extra: pip install 'admitperf[dashboard]'") from None
|
|
233
|
+
|
|
234
|
+
chosen = _free_port(port)
|
|
235
|
+
if chosen != port:
|
|
236
|
+
# A stale Streamlit from another checkout holding the default port should not
|
|
237
|
+
# stop you looking at a report.
|
|
238
|
+
click.echo(f"port {port} is busy, using {chosen}")
|
|
239
|
+
click.echo(f" http://localhost:{chosen}\n")
|
|
240
|
+
app = Path(__file__).parent / "dashboard" / "app.py"
|
|
241
|
+
subprocess.run(
|
|
242
|
+
[
|
|
243
|
+
sys.executable,
|
|
244
|
+
"-m",
|
|
245
|
+
"streamlit",
|
|
246
|
+
"run",
|
|
247
|
+
str(app),
|
|
248
|
+
"--server.port",
|
|
249
|
+
str(chosen),
|
|
250
|
+
"--server.headless",
|
|
251
|
+
"true",
|
|
252
|
+
],
|
|
253
|
+
check=False,
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
@main.command()
|
|
258
|
+
def policies() -> None:
|
|
259
|
+
"""The policies that ship with AdmitPerf."""
|
|
260
|
+
from admitperf import policies as shipped
|
|
261
|
+
|
|
262
|
+
for name in shipped.__all__:
|
|
263
|
+
cls = getattr(shipped, name)
|
|
264
|
+
doc = (cls.__doc__ or "").strip().splitlines()[0]
|
|
265
|
+
click.echo(f"{cls.name:<16} {doc}")
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
@main.command()
|
|
269
|
+
def signals() -> None:
|
|
270
|
+
"""Every signal, and the metrics it reads.
|
|
271
|
+
|
|
272
|
+
Run this first: it tells you what AdmitPerf can already read from your stack,
|
|
273
|
+
and what to export (or map) for the rest.
|
|
274
|
+
"""
|
|
275
|
+
from admitperf.core.signals import ALL
|
|
276
|
+
|
|
277
|
+
for s in ALL:
|
|
278
|
+
srcs = ", ".join(x if isinstance(x, str) else f"{x.__name__}()" for x in s.sources)
|
|
279
|
+
rng = f"[{s.lo:g}, {s.hi:g}]" if s.hi is not None else f">= {s.lo:g}"
|
|
280
|
+
click.echo(f"{s.name:<18} {rng:<12} {srcs}")
|
|
281
|
+
click.echo(f"{'':<18} {s.help}")
|
|
282
|
+
click.echo()
|
admitperf/comparison.py
ADDED
|
@@ -0,0 +1,326 @@
|
|
|
1
|
+
"""Two logs, side by side: what did the policy buy?"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from admitperf.report import THIN, WIDTH, Report
|
|
9
|
+
|
|
10
|
+
RULE = "=" * WIDTH
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _reports(target: str | Path) -> list[Report]:
|
|
14
|
+
"""One log, or every log in a directory.
|
|
15
|
+
|
|
16
|
+
A directory is how a policy gets more than one run. Sorted by name so `r1, r2, r10` is at
|
|
17
|
+
least stable between invocations, even though it is not numeric order — the
|
|
18
|
+
comparison does not care about sequence, only that the same set is read twice.
|
|
19
|
+
"""
|
|
20
|
+
path = Path(target)
|
|
21
|
+
if path.is_dir():
|
|
22
|
+
logs = sorted(path.glob("*.jsonl"))
|
|
23
|
+
if not logs:
|
|
24
|
+
raise ValueError(f"no .jsonl logs in {path}")
|
|
25
|
+
return [Report.from_log(p) for p in logs]
|
|
26
|
+
return [Report.from_log(path)]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Comparison:
|
|
30
|
+
"""A baseline run against a policy run.
|
|
31
|
+
|
|
32
|
+
This is the question the whole package exists to answer, and the one the survey
|
|
33
|
+
found sixteen papers answering incomparably. It is also the easiest thing to get
|
|
34
|
+
wrong, so three checks come before any number:
|
|
35
|
+
|
|
36
|
+
1. **Did the policy fire?** If it refused nothing, the two runs are the same
|
|
37
|
+
configuration measured twice and any difference between them is noise.
|
|
38
|
+
2. **Did both runs face the same load?** A policy run against half the traffic
|
|
39
|
+
will look wonderful. Offered counts are reported side by side so a mismatch
|
|
40
|
+
is visible rather than buried.
|
|
41
|
+
3. **Can cost be measured at all?** Without outcomes there is no latency and no
|
|
42
|
+
goodput, so the comparison can only report what was refused — which it says
|
|
43
|
+
rather than implying more.
|
|
44
|
+
|
|
45
|
+
Goodput divides by requests OFFERED, never by admitted. Dividing by admitted
|
|
46
|
+
rewards a policy for refusing more rather than for refusing better: refuse 95% of
|
|
47
|
+
traffic, serve the rest perfectly, and the number reads 1.00.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(
|
|
51
|
+
self,
|
|
52
|
+
baseline: Report | list[Report],
|
|
53
|
+
policy: Report | list[Report],
|
|
54
|
+
) -> None:
|
|
55
|
+
self.baselines = [baseline] if isinstance(baseline, Report) else list(baseline)
|
|
56
|
+
self.policies = [policy] if isinstance(policy, Report) else list(policy)
|
|
57
|
+
if not self.baselines or not self.policies:
|
|
58
|
+
raise ValueError("a comparison needs at least one run on each side")
|
|
59
|
+
#: The representative run of each side, for the parts that do not vary between
|
|
60
|
+
#: repeats: which policy, its parameters, whether it fired.
|
|
61
|
+
self.baseline = self.baselines[0]
|
|
62
|
+
self.policy = self.policies[0]
|
|
63
|
+
|
|
64
|
+
@classmethod
|
|
65
|
+
def from_logs(cls, baseline: str | Path, policy: str | Path) -> Comparison:
|
|
66
|
+
"""Each side is a log file, or a DIRECTORY holding every run of one policy.
|
|
67
|
+
|
|
68
|
+
A directory is how you get an error bar. One run of each has none, and a
|
|
69
|
+
gap smaller than the spread between repeats is not a result — so the tool has
|
|
70
|
+
to be able to read more than one.
|
|
71
|
+
"""
|
|
72
|
+
return cls(_reports(baseline), _reports(policy))
|
|
73
|
+
|
|
74
|
+
# -- repeats -----------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def repeats(self) -> int:
|
|
78
|
+
"""The smaller of the two, since that is what the comparison is limited by."""
|
|
79
|
+
return min(len(self.baselines), len(self.policies))
|
|
80
|
+
|
|
81
|
+
@staticmethod
|
|
82
|
+
def _spread(values: list[float]) -> tuple[float, float, float]:
|
|
83
|
+
"""Median, min, max. Median rather than mean because one pathological repeat
|
|
84
|
+
— a stalled scrape, a noisy neighbour — should not move the headline."""
|
|
85
|
+
ordered = sorted(values)
|
|
86
|
+
return ordered[len(ordered) // 2], ordered[0], ordered[-1]
|
|
87
|
+
|
|
88
|
+
def _arm(self, reports: list[Report], metric) -> tuple[float, float, float] | None:
|
|
89
|
+
values = [v for r in reports if (v := metric(r)) is not None]
|
|
90
|
+
return self._spread(values) if values else None
|
|
91
|
+
|
|
92
|
+
def separated(self) -> bool | None:
|
|
93
|
+
"""Whether the two policies are further apart than their own runs are.
|
|
94
|
+
|
|
95
|
+
None when there is only one run each, because then the question cannot be
|
|
96
|
+
asked — and answering it anyway is how a difference inside the noise gets
|
|
97
|
+
published as a finding.
|
|
98
|
+
"""
|
|
99
|
+
if self.repeats < 2:
|
|
100
|
+
return None
|
|
101
|
+
base = self._arm(self.baselines, self._goodput)
|
|
102
|
+
pol = self._arm(self.policies, self._goodput)
|
|
103
|
+
if base is None or pol is None:
|
|
104
|
+
return None
|
|
105
|
+
# No overlap between the two policies' observed ranges.
|
|
106
|
+
return base[2] < pol[1] or pol[2] < base[1]
|
|
107
|
+
|
|
108
|
+
# -- the three checks --------------------------------------------------
|
|
109
|
+
|
|
110
|
+
def fired(self) -> bool:
|
|
111
|
+
"""Every repeat must have fired. One that did not is a different experiment,
|
|
112
|
+
and averaging it in hides that."""
|
|
113
|
+
return all(r.refused > 0 for r in self.policies)
|
|
114
|
+
|
|
115
|
+
def comparable_load(self) -> bool:
|
|
116
|
+
"""Within 10%, across every run on both sides. Two runs offered materially
|
|
117
|
+
different traffic are not a comparison, however similar the configuration."""
|
|
118
|
+
counts = [len(r.decisions) for r in self.baselines + self.policies]
|
|
119
|
+
if not all(counts):
|
|
120
|
+
return False
|
|
121
|
+
return (max(counts) - min(counts)) / max(counts) <= 0.10
|
|
122
|
+
|
|
123
|
+
def measurable(self) -> bool:
|
|
124
|
+
return all(r.outcomes for r in self.baselines + self.policies)
|
|
125
|
+
|
|
126
|
+
# -- the numbers -------------------------------------------------------
|
|
127
|
+
|
|
128
|
+
@staticmethod
|
|
129
|
+
def _ttft(report: Report) -> dict[str, float] | None:
|
|
130
|
+
"""Latency of requests that were admitted and returned."""
|
|
131
|
+
values = sorted(
|
|
132
|
+
v for r in report.outcomes if (v := r["outcome"].get("ttft_ms")) is not None
|
|
133
|
+
)
|
|
134
|
+
if not values:
|
|
135
|
+
return None
|
|
136
|
+
return {
|
|
137
|
+
"p50": values[len(values) // 2],
|
|
138
|
+
"p95": values[min(len(values) - 1, math.ceil(0.95 * len(values)) - 1)],
|
|
139
|
+
"max": values[-1],
|
|
140
|
+
"n": len(values),
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
@staticmethod
|
|
144
|
+
def _goodput(report: Report) -> float | None:
|
|
145
|
+
"""Met-SLO over OFFERED. A refusal counts as a miss, which is the point."""
|
|
146
|
+
offered = len(report.decisions)
|
|
147
|
+
if not offered or not report.outcomes:
|
|
148
|
+
return None
|
|
149
|
+
met = sum(1 for r in report.outcomes if r["outcome"].get("ok"))
|
|
150
|
+
return met / offered
|
|
151
|
+
|
|
152
|
+
def refused_share(self) -> float:
|
|
153
|
+
shares = [r.refused / len(r.decisions) for r in self.policies if r.decisions]
|
|
154
|
+
return self._spread(shares)[0] if shares else 0.0
|
|
155
|
+
|
|
156
|
+
# -- the page ----------------------------------------------------------
|
|
157
|
+
|
|
158
|
+
def text(self) -> str:
|
|
159
|
+
lines = [RULE, " AdmitPerf — what did the policy buy?", RULE, ""]
|
|
160
|
+
lines += self._headline()
|
|
161
|
+
lines += ["", THIN, "", " SIDE BY SIDE", ""]
|
|
162
|
+
lines += self._table()
|
|
163
|
+
lines += ["", THIN, "", " CAN YOU TRUST THIS?", ""]
|
|
164
|
+
lines += self._trust()
|
|
165
|
+
lines.append(RULE)
|
|
166
|
+
return "\n".join(lines)
|
|
167
|
+
|
|
168
|
+
def _headline(self) -> list[str]:
|
|
169
|
+
if not self.fired():
|
|
170
|
+
return [
|
|
171
|
+
" NO FINDING — the policy never fired.",
|
|
172
|
+
"",
|
|
173
|
+
f" {self.policy.policy or 'the policy'} refused 0 of "
|
|
174
|
+
f"{len(self.policy.decisions)} requests, so these two runs are the same",
|
|
175
|
+
" configuration measured twice. Any difference between the numbers below",
|
|
176
|
+
" is noise, and a latency improvement here would be a false result.",
|
|
177
|
+
"",
|
|
178
|
+
" The fix is usually the LOAD, not the policy: concurrency is arrival",
|
|
179
|
+
" rate times request duration, so short requests cannot fill a cache at",
|
|
180
|
+
" any rate. `admitperf report` on the policy log shows which signal did",
|
|
181
|
+
" move, if any.",
|
|
182
|
+
]
|
|
183
|
+
|
|
184
|
+
out = [
|
|
185
|
+
f" The policy refused {self.policy.refused} of {len(self.policy.decisions)} "
|
|
186
|
+
f"requests ({self.refused_share():.1%}).",
|
|
187
|
+
"",
|
|
188
|
+
]
|
|
189
|
+
if not self.measurable():
|
|
190
|
+
out += [
|
|
191
|
+
" What that bought cannot be measured: no outcomes were recorded, so",
|
|
192
|
+
" there is no latency and no goodput. Call policy.outcome(...) after you",
|
|
193
|
+
" forward a request and run this again.",
|
|
194
|
+
]
|
|
195
|
+
return out
|
|
196
|
+
|
|
197
|
+
p95 = lambda r: t["p95"] if (t := self._ttft(r)) else None # noqa: E731
|
|
198
|
+
base_t = self._arm(self.baselines, p95)
|
|
199
|
+
pol_t = self._arm(self.policies, p95)
|
|
200
|
+
base_g = self._arm(self.baselines, self._goodput)
|
|
201
|
+
pol_g = self._arm(self.policies, self._goodput)
|
|
202
|
+
|
|
203
|
+
def band(s: tuple[float, float, float], fmt: str) -> str:
|
|
204
|
+
"""Median, and the observed range when there is more than one run. A bare
|
|
205
|
+
number from one run reads as more certain than it is."""
|
|
206
|
+
median, lo, hi = s
|
|
207
|
+
if self.repeats < 2 or lo == hi:
|
|
208
|
+
return fmt.format(median)
|
|
209
|
+
return f"{fmt.format(median)} ({fmt.format(lo)}-{fmt.format(hi)})"
|
|
210
|
+
|
|
211
|
+
if base_t and pol_t and base_t[0]:
|
|
212
|
+
factor = base_t[0] / pol_t[0] if pol_t[0] else float("inf")
|
|
213
|
+
direction = "lower" if factor > 1 else "HIGHER"
|
|
214
|
+
out.append(
|
|
215
|
+
f" p95 TTFT of served requests is {abs(factor):.2f}x {direction}: "
|
|
216
|
+
f"{band(base_t, '{:.0f}')}ms -> {band(pol_t, '{:.0f}')}ms"
|
|
217
|
+
)
|
|
218
|
+
if base_g and pol_g:
|
|
219
|
+
verb = "up" if pol_g[0] > base_g[0] else "DOWN"
|
|
220
|
+
out.append(
|
|
221
|
+
f" goodput is {verb}: {band(base_g, '{:.3f}')} -> "
|
|
222
|
+
f"{band(pol_g, '{:.3f}')} (of offered)"
|
|
223
|
+
)
|
|
224
|
+
if pol_g[0] < base_g[0]:
|
|
225
|
+
out += [
|
|
226
|
+
"",
|
|
227
|
+
" Goodput fell, so the policy refused requests the cluster could have",
|
|
228
|
+
" served. Faster tails bought at that price are not a win.",
|
|
229
|
+
]
|
|
230
|
+
|
|
231
|
+
sep = self.separated()
|
|
232
|
+
if sep is False:
|
|
233
|
+
out += [
|
|
234
|
+
"",
|
|
235
|
+
f" BUT the two overlap across {self.repeats} repeats, so this gap is",
|
|
236
|
+
" inside the run-to-run noise. It is not a result yet — more repeats, or a",
|
|
237
|
+
" larger effect.",
|
|
238
|
+
]
|
|
239
|
+
return out
|
|
240
|
+
|
|
241
|
+
def _table(self) -> list[str]:
|
|
242
|
+
def med(reports: list[Report], metric, fmt: str = "{:.0f}") -> str:
|
|
243
|
+
s = self._arm(reports, metric)
|
|
244
|
+
return "--" if s is None else fmt.format(s[0])
|
|
245
|
+
|
|
246
|
+
p50 = lambda r: t["p50"] if (t := self._ttft(r)) else None # noqa: E731
|
|
247
|
+
p95 = lambda r: t["p95"] if (t := self._ttft(r)) else None # noqa: E731
|
|
248
|
+
|
|
249
|
+
rows = [
|
|
250
|
+
("policy", self.baseline.policy or "-", self.policy.policy or "-"),
|
|
251
|
+
("repeats", str(len(self.baselines)), str(len(self.policies))),
|
|
252
|
+
(
|
|
253
|
+
"offered",
|
|
254
|
+
med(self.baselines, lambda r: float(len(r.decisions))),
|
|
255
|
+
med(self.policies, lambda r: float(len(r.decisions))),
|
|
256
|
+
),
|
|
257
|
+
(
|
|
258
|
+
"refused",
|
|
259
|
+
med(self.baselines, lambda r: float(r.refused)),
|
|
260
|
+
med(self.policies, lambda r: float(r.refused)),
|
|
261
|
+
),
|
|
262
|
+
("p50 TTFT ms", med(self.baselines, p50), med(self.policies, p50)),
|
|
263
|
+
("p95 TTFT ms", med(self.baselines, p95), med(self.policies, p95)),
|
|
264
|
+
(
|
|
265
|
+
"goodput",
|
|
266
|
+
med(self.baselines, self._goodput, "{:.3f}"),
|
|
267
|
+
med(self.policies, self._goodput, "{:.3f}"),
|
|
268
|
+
),
|
|
269
|
+
]
|
|
270
|
+
label_row = "median of repeats" if self.repeats > 1 else ""
|
|
271
|
+
out = [f" {label_row:<16} {'baseline':>14} {'with policy':>14}", ""]
|
|
272
|
+
out += [f" {label:<16} {a:>14} {b:>14}" for label, a, b in rows]
|
|
273
|
+
|
|
274
|
+
# Signals, because "the policy changed the cluster" is checkable rather than
|
|
275
|
+
# assumed: KV pressure that did not move means the refusals bought nothing.
|
|
276
|
+
shared = [n for n in self.policy.signals() if n in self.baseline.signals()]
|
|
277
|
+
if shared:
|
|
278
|
+
out += ["", f" {'signal max':<16} {'baseline':>14} {'with policy':>14}", ""]
|
|
279
|
+
for name in shared:
|
|
280
|
+
top = lambda r, n=name: ( # noqa: E731
|
|
281
|
+
s["max"] if (s := r.signal_range(n)) else None
|
|
282
|
+
)
|
|
283
|
+
out.append(
|
|
284
|
+
f" {name:<16} {med(self.baselines, top, '{:.3g}'):>14} "
|
|
285
|
+
f"{med(self.policies, top, '{:.3g}'):>14}"
|
|
286
|
+
)
|
|
287
|
+
return out
|
|
288
|
+
|
|
289
|
+
def _trust(self) -> list[str]:
|
|
290
|
+
out = []
|
|
291
|
+
mark = {True: "ok ", False: "NO "}
|
|
292
|
+
out.append(f" {mark[self.fired()]} the policy fired at all")
|
|
293
|
+
out.append(f" {mark[self.comparable_load()]} both were offered the same load")
|
|
294
|
+
if not self.comparable_load():
|
|
295
|
+
out.append(
|
|
296
|
+
f" {len(self.baseline.decisions)} vs {len(self.policy.decisions)} "
|
|
297
|
+
"requests — a policy run against less traffic will look better than it is"
|
|
298
|
+
)
|
|
299
|
+
out.append(f" {mark[self.measurable()]} outcomes recorded, so cost is measurable")
|
|
300
|
+
sep = self.separated()
|
|
301
|
+
if sep is None:
|
|
302
|
+
out.append("-- only one run per policy, so there is no error bar")
|
|
303
|
+
else:
|
|
304
|
+
out.append(f" {mark[sep]} the two separate across {self.repeats} repeats")
|
|
305
|
+
for name, reports in (("baseline", self.baselines), ("policy", self.policies)):
|
|
306
|
+
for i, r in enumerate(reports, 1):
|
|
307
|
+
for fault in r.fault_text():
|
|
308
|
+
tag = f"{name}" if len(reports) == 1 else f"{name} r{i}"
|
|
309
|
+
out.append(f" !! {tag}:{fault.strip()}")
|
|
310
|
+
|
|
311
|
+
ok = self.fired() and self.comparable_load() and self.measurable()
|
|
312
|
+
if ok and sep:
|
|
313
|
+
out += [
|
|
314
|
+
"",
|
|
315
|
+
" All four hold, so the difference above is the policy and it is larger",
|
|
316
|
+
" than the noise between repeats. This is a result you can quote.",
|
|
317
|
+
]
|
|
318
|
+
elif ok and sep is None:
|
|
319
|
+
out += [
|
|
320
|
+
"",
|
|
321
|
+
" The first three hold, but one run of each has no error bar. Run the pair",
|
|
322
|
+
" again — a gap smaller than the spread between repeats is not a result:",
|
|
323
|
+
"",
|
|
324
|
+
" admitperf demo --repeats 5",
|
|
325
|
+
]
|
|
326
|
+
return out
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""The contract: metrics in, signals derived, policies decide.
|
|
2
|
+
|
|
3
|
+
This subpackage is the part that runs in your request path, so it holds the line
|
|
4
|
+
that makes AdmitPerf safe to install: no third-party imports, no sockets, no
|
|
5
|
+
subprocesses, no filesystem beyond the log you asked for. A test enforces each.
|
|
6
|
+
|
|
7
|
+
Everything else in the package — the CLI, the watcher, the report, the dashboard —
|
|
8
|
+
is a consumer of what this produces and may do as it likes.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from admitperf.core.decision import Decision
|
|
14
|
+
from admitperf.core.log import Log
|
|
15
|
+
from admitperf.core.policy import Policy
|
|
16
|
+
from admitperf.core.reasons import ADMITTED, REASONS
|
|
17
|
+
from admitperf.core.signal import Signal, Source
|
|
18
|
+
from admitperf.core.verdict import Verdict
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"ADMITTED",
|
|
22
|
+
"REASONS",
|
|
23
|
+
"Decision",
|
|
24
|
+
"Log",
|
|
25
|
+
"Policy",
|
|
26
|
+
"Signal",
|
|
27
|
+
"Source",
|
|
28
|
+
"Verdict",
|
|
29
|
+
]
|