admitperf 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
admitperf/__init__.py ADDED
@@ -0,0 +1,43 @@
1
+ """AdmitPerf — standardized admission control for LLM inference.
2
+
3
+ Bring your own infra. Your gateway already has raw metrics; AdmitPerf turns them
4
+ into signals, policies decide on signals, and every decision is recorded in a form
5
+ a report can compare across deployments.
6
+
7
+ from admitperf import Policy
8
+ from admitperf.core.signals import KV_PRESSURE
9
+
10
+ class KvWall(Policy):
11
+ name = "kv_wall"
12
+
13
+ def decide(self, metrics):
14
+ if KV_PRESSURE.read(metrics) >= self.threshold:
15
+ return self.reject("kv_pressure")
16
+ return self.admit()
17
+
18
+ `admitperf.core` is what runs in your request path and imports nothing but the
19
+ standard library. Everything else is a consumer of what it writes.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from importlib.metadata import PackageNotFoundError, version
25
+
26
+ from admitperf.core import REASONS, Decision, Log, Policy, Signal, Verdict
27
+
28
+ try:
29
+ # Read from installed metadata rather than restating it. Two copies of a version
30
+ # number is one too many, and the stale one is always the one a report quotes.
31
+ __version__ = version("admitperf")
32
+ except PackageNotFoundError: # a source tree with no install
33
+ __version__ = "0.0.0+unknown"
34
+
35
+ __all__ = [
36
+ "REASONS",
37
+ "Decision",
38
+ "Log",
39
+ "Policy",
40
+ "Signal",
41
+ "Verdict",
42
+ "__version__",
43
+ ]
admitperf/cli.py ADDED
@@ -0,0 +1,282 @@
1
+ """The CLI. Three verbs, and none of them is in a request path.
2
+
3
+ admitperf demo try it, no GPU needed
4
+ admitperf watch http://host:8000/metrics --for 1h -o trace.jsonl
5
+ admitperf report trace.jsonl
6
+ admitperf compare baseline.jsonl with-policy.jsonl
7
+ admitperf dashboard
8
+
9
+ `admitperf.core` never fetches anything, because a network round trip has no place
10
+ on an admission decision. These commands run out of band, so `watch` scraping a
11
+ metrics endpoint for you is a convenience rather than a contradiction.
12
+
13
+ Nothing here provisions, serves, or generates load. That is your infra's job.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import sys
19
+ from pathlib import Path
20
+
21
+ import click
22
+
23
+ from admitperf import __version__
24
+
25
+
26
+ def _seconds(text: str) -> float:
27
+ """`90`, `30s`, `15m`, `2h` — because `--for 3600` is how you mean an hour and
28
+ type a mistake."""
29
+ units = {"s": 1, "m": 60, "h": 3600}
30
+ if text and text[-1] in units:
31
+ return float(text[:-1]) * units[text[-1]]
32
+ return float(text)
33
+
34
+
35
+ @click.group()
36
+ @click.version_option(__version__)
37
+ def main() -> None:
38
+ """Standardized admission control for LLM inference."""
39
+
40
+
41
+ @main.command()
42
+ @click.argument("url")
43
+ @click.option("--for", "duration", default="60s", show_default=True, help="90s · 15m · 2h")
44
+ @click.option("--every", default=1.0, show_default=True, help="seconds between scrapes")
45
+ @click.option("-o", "--out", default="trace.jsonl", show_default=True)
46
+ def watch(url: str, duration: str, every: float, out: str) -> None:
47
+ """Record what your cluster is doing. No code in your request path.
48
+
49
+ \b
50
+ admitperf watch http://localhost:8000/metrics --for 1h -o trace.jsonl
51
+
52
+ Answers the question no surveyed paper answers — could a policy have fired
53
+ here, and which signal actually moved — before you change any code.
54
+ """
55
+ from admitperf.watch import Watch
56
+
57
+ seconds = _seconds(duration)
58
+ click.echo(f"watching {url} every {every:g}s for {seconds:g}s -> {out}")
59
+ w = Watch(url, out, interval=every).run(seconds)
60
+ click.echo(f"{w.samples} samples, {w.failures} failed scrape(s)")
61
+ if w.samples == 0:
62
+ raise SystemExit(f"nothing was recorded. Is {url} reachable and serving Prometheus text?")
63
+ click.echo(f"\n admitperf report {out}")
64
+
65
+
66
+ @main.command()
67
+ @click.argument("log", type=click.Path(exists=True, dir_okay=False))
68
+ @click.option(
69
+ "--check",
70
+ is_flag=True,
71
+ help="exit non-zero unless the log can support a claim about the policy",
72
+ )
73
+ def report(log: str, check: bool) -> None:
74
+ """The finding from a log.
75
+
76
+ \b
77
+ admitperf report trace.jsonl
78
+ admitperf report decisions.jsonl --check # for CI
79
+ """
80
+ from admitperf.report import Report
81
+
82
+ r = Report.from_log(log)
83
+ click.echo(r.text())
84
+ if check and r.verdict() != "LIVE":
85
+ raise SystemExit(f"verdict is {r.verdict()}: this log cannot support a claim")
86
+
87
+
88
+ @main.command("experiments")
89
+ def list_experiments() -> None:
90
+ """Every experiment found under this directory, and its arms."""
91
+ from admitperf.discover import experiments
92
+
93
+ found = experiments()
94
+ if not found:
95
+ click.echo("no logs here. `admitperf demo --repeats 5` makes some.")
96
+ return
97
+ for name, exp in sorted(found.items()):
98
+ click.echo(f"\n{name} ({exp.runs} run(s))")
99
+ if exp.notes:
100
+ click.echo(f" {exp.notes}")
101
+ for runs in exp.policies.values():
102
+ click.echo(f" {runs.label}")
103
+ if exp.baseline is None:
104
+ click.echo(" !! no baseline, so nothing can be compared against")
105
+
106
+
107
+ @main.command()
108
+ @click.option("--repeats", default=1, show_default=True, help="runs per policy")
109
+ @click.option(
110
+ "-o",
111
+ "--out",
112
+ default="experiments",
113
+ show_default=True,
114
+ help="logs land in <out>/<experiment>/<policy>/r<n>.jsonl",
115
+ )
116
+ def demo(repeats: int, out: str) -> None:
117
+ """Try the whole thing in one command. No GPU, no cluster, no config.
118
+
119
+ \b
120
+ admitperf demo
121
+ admitperf demo --repeats 5
122
+
123
+ Drives identical traffic through a simulated engine three times — admitting
124
+ everything, with a KV-pressure threshold, and with a queue bound derived from the
125
+ SLO — then compares the reports.
126
+
127
+ A demo, not an experiment runner: it takes no engine URL and no workload, because
128
+ AdmitPerf does not generate load. Against a real cluster your own load generator
129
+ drives traffic and `admitperf watch` records it.
130
+ """
131
+ from admitperf.demo import main as run_demo
132
+
133
+ run_demo(repeats=repeats, out=out, echo=click.echo)
134
+
135
+
136
+ @main.command()
137
+ @click.argument("baseline", type=click.Path(exists=True), required=False)
138
+ @click.argument("policy", type=click.Path(exists=True), required=False)
139
+ @click.option(
140
+ "--experiment",
141
+ "-e",
142
+ help="compare every policy in a named experiment against its baseline",
143
+ )
144
+ @click.option(
145
+ "--check",
146
+ is_flag=True,
147
+ help="exit non-zero unless the comparison is trustworthy: the policy fired, both "
148
+ "runs faced the same load, and outcomes were recorded",
149
+ )
150
+ def compare(baseline: str | None, policy: str | None, experiment: str | None, check: bool) -> None:
151
+ """What did the policy buy?
152
+
153
+ \b
154
+ admitperf compare --experiment demo-overload-2.5x # every arm vs its baseline
155
+ admitperf compare baseline.jsonl with-policy.jsonl # or two paths directly
156
+
157
+ The question the package exists to answer. Four checks come before any number,
158
+ because a comparison between a policy that never fired and a baseline is two
159
+ measurements of the same configuration.
160
+ """
161
+ from admitperf.comparison import Comparison
162
+ from admitperf.discover import find
163
+ from admitperf.report import Report
164
+
165
+ if experiment:
166
+ exp = find(experiment)
167
+ if exp.baseline is None:
168
+ raise SystemExit(
169
+ f"experiment {experiment!r} has no baseline, so there is nothing to "
170
+ "compare against. Run a policy with `baseline = True` — NoAdmission is one."
171
+ )
172
+ if not exp.candidates:
173
+ raise SystemExit(f"experiment {experiment!r} has only a baseline")
174
+ if exp.notes:
175
+ click.echo(f"{experiment} — {exp.notes}\n")
176
+ failed = False
177
+ base = [Report.from_log(p) for p in exp.baseline.logs]
178
+ for candidate in exp.candidates:
179
+ click.echo(f"\n### {exp.baseline.label} vs {candidate.label}")
180
+ c = Comparison(base, [Report.from_log(p) for p in candidate.logs])
181
+ click.echo(c.text())
182
+ failed = failed or not (c.fired() and c.comparable_load() and c.measurable())
183
+ if check and failed:
184
+ raise SystemExit("at least one pair cannot support a claim")
185
+ return
186
+
187
+ if not (baseline and policy):
188
+ raise SystemExit("give two logs, or --experiment NAME. `admitperf dashboard` lists them.")
189
+ c = Comparison.from_logs(baseline, policy)
190
+ click.echo(c.text())
191
+ if check and not (c.fired() and c.comparable_load() and c.measurable()):
192
+ raise SystemExit("this pair cannot support a claim — see CAN YOU TRUST THIS above")
193
+
194
+
195
+ def _free_port(start: int, tries: int = 20) -> int:
196
+ """The first free port at or above `start`.
197
+
198
+ Bound on all interfaces and WITHOUT SO_REUSEADDR, because that is what Streamlit
199
+ does. A probe against 127.0.0.1 with SO_REUSEADDR set reports a port as free while
200
+ Streamlit then fails to bind it — which is exactly how "Port 8501 is not available"
201
+ became a dead end instead of a retry.
202
+ """
203
+ import socket
204
+
205
+ for offset in range(tries):
206
+ candidate = start + offset
207
+ with socket.socket() as s:
208
+ try:
209
+ s.bind(("", candidate))
210
+ except OSError:
211
+ continue
212
+ return candidate
213
+ raise SystemExit(f"no free port between {start} and {start + tries}")
214
+
215
+
216
+ @main.command()
217
+ @click.option("--port", default=8501, show_default=True, help="or the next one free")
218
+ def dashboard(port: int) -> None:
219
+ """Open the dashboard: logs, findings, and what a policy bought.
220
+
221
+ \b
222
+ admitperf demo --repeats 5 # make some logs first
223
+ admitperf dashboard
224
+
225
+ It discovers logs under the current directory, so run it where your logs are.
226
+ """
227
+ import subprocess
228
+
229
+ try:
230
+ import streamlit # noqa: F401
231
+ except ImportError:
232
+ raise SystemExit("the dashboard is an extra: pip install 'admitperf[dashboard]'") from None
233
+
234
+ chosen = _free_port(port)
235
+ if chosen != port:
236
+ # A stale Streamlit from another checkout holding the default port should not
237
+ # stop you looking at a report.
238
+ click.echo(f"port {port} is busy, using {chosen}")
239
+ click.echo(f" http://localhost:{chosen}\n")
240
+ app = Path(__file__).parent / "dashboard" / "app.py"
241
+ subprocess.run(
242
+ [
243
+ sys.executable,
244
+ "-m",
245
+ "streamlit",
246
+ "run",
247
+ str(app),
248
+ "--server.port",
249
+ str(chosen),
250
+ "--server.headless",
251
+ "true",
252
+ ],
253
+ check=False,
254
+ )
255
+
256
+
257
+ @main.command()
258
+ def policies() -> None:
259
+ """The policies that ship with AdmitPerf."""
260
+ from admitperf import policies as shipped
261
+
262
+ for name in shipped.__all__:
263
+ cls = getattr(shipped, name)
264
+ doc = (cls.__doc__ or "").strip().splitlines()[0]
265
+ click.echo(f"{cls.name:<16} {doc}")
266
+
267
+
268
+ @main.command()
269
+ def signals() -> None:
270
+ """Every signal, and the metrics it reads.
271
+
272
+ Run this first: it tells you what AdmitPerf can already read from your stack,
273
+ and what to export (or map) for the rest.
274
+ """
275
+ from admitperf.core.signals import ALL
276
+
277
+ for s in ALL:
278
+ srcs = ", ".join(x if isinstance(x, str) else f"{x.__name__}()" for x in s.sources)
279
+ rng = f"[{s.lo:g}, {s.hi:g}]" if s.hi is not None else f">= {s.lo:g}"
280
+ click.echo(f"{s.name:<18} {rng:<12} {srcs}")
281
+ click.echo(f"{'':<18} {s.help}")
282
+ click.echo()
@@ -0,0 +1,326 @@
1
+ """Two logs, side by side: what did the policy buy?"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from pathlib import Path
7
+
8
+ from admitperf.report import THIN, WIDTH, Report
9
+
10
+ RULE = "=" * WIDTH
11
+
12
+
13
+ def _reports(target: str | Path) -> list[Report]:
14
+ """One log, or every log in a directory.
15
+
16
+ A directory is how a policy gets more than one run. Sorted by name so `r1, r2, r10` is at
17
+ least stable between invocations, even though it is not numeric order — the
18
+ comparison does not care about sequence, only that the same set is read twice.
19
+ """
20
+ path = Path(target)
21
+ if path.is_dir():
22
+ logs = sorted(path.glob("*.jsonl"))
23
+ if not logs:
24
+ raise ValueError(f"no .jsonl logs in {path}")
25
+ return [Report.from_log(p) for p in logs]
26
+ return [Report.from_log(path)]
27
+
28
+
29
+ class Comparison:
30
+ """A baseline run against a policy run.
31
+
32
+ This is the question the whole package exists to answer, and the one the survey
33
+ found sixteen papers answering incomparably. It is also the easiest thing to get
34
+ wrong, so three checks come before any number:
35
+
36
+ 1. **Did the policy fire?** If it refused nothing, the two runs are the same
37
+ configuration measured twice and any difference between them is noise.
38
+ 2. **Did both runs face the same load?** A policy run against half the traffic
39
+ will look wonderful. Offered counts are reported side by side so a mismatch
40
+ is visible rather than buried.
41
+ 3. **Can cost be measured at all?** Without outcomes there is no latency and no
42
+ goodput, so the comparison can only report what was refused — which it says
43
+ rather than implying more.
44
+
45
+ Goodput divides by requests OFFERED, never by admitted. Dividing by admitted
46
+ rewards a policy for refusing more rather than for refusing better: refuse 95% of
47
+ traffic, serve the rest perfectly, and the number reads 1.00.
48
+ """
49
+
50
+ def __init__(
51
+ self,
52
+ baseline: Report | list[Report],
53
+ policy: Report | list[Report],
54
+ ) -> None:
55
+ self.baselines = [baseline] if isinstance(baseline, Report) else list(baseline)
56
+ self.policies = [policy] if isinstance(policy, Report) else list(policy)
57
+ if not self.baselines or not self.policies:
58
+ raise ValueError("a comparison needs at least one run on each side")
59
+ #: The representative run of each side, for the parts that do not vary between
60
+ #: repeats: which policy, its parameters, whether it fired.
61
+ self.baseline = self.baselines[0]
62
+ self.policy = self.policies[0]
63
+
64
+ @classmethod
65
+ def from_logs(cls, baseline: str | Path, policy: str | Path) -> Comparison:
66
+ """Each side is a log file, or a DIRECTORY holding every run of one policy.
67
+
68
+ A directory is how you get an error bar. One run of each has none, and a
69
+ gap smaller than the spread between repeats is not a result — so the tool has
70
+ to be able to read more than one.
71
+ """
72
+ return cls(_reports(baseline), _reports(policy))
73
+
74
+ # -- repeats -----------------------------------------------------------
75
+
76
+ @property
77
+ def repeats(self) -> int:
78
+ """The smaller of the two, since that is what the comparison is limited by."""
79
+ return min(len(self.baselines), len(self.policies))
80
+
81
+ @staticmethod
82
+ def _spread(values: list[float]) -> tuple[float, float, float]:
83
+ """Median, min, max. Median rather than mean because one pathological repeat
84
+ — a stalled scrape, a noisy neighbour — should not move the headline."""
85
+ ordered = sorted(values)
86
+ return ordered[len(ordered) // 2], ordered[0], ordered[-1]
87
+
88
+ def _arm(self, reports: list[Report], metric) -> tuple[float, float, float] | None:
89
+ values = [v for r in reports if (v := metric(r)) is not None]
90
+ return self._spread(values) if values else None
91
+
92
+ def separated(self) -> bool | None:
93
+ """Whether the two policies are further apart than their own runs are.
94
+
95
+ None when there is only one run each, because then the question cannot be
96
+ asked — and answering it anyway is how a difference inside the noise gets
97
+ published as a finding.
98
+ """
99
+ if self.repeats < 2:
100
+ return None
101
+ base = self._arm(self.baselines, self._goodput)
102
+ pol = self._arm(self.policies, self._goodput)
103
+ if base is None or pol is None:
104
+ return None
105
+ # No overlap between the two policies' observed ranges.
106
+ return base[2] < pol[1] or pol[2] < base[1]
107
+
108
+ # -- the three checks --------------------------------------------------
109
+
110
+ def fired(self) -> bool:
111
+ """Every repeat must have fired. One that did not is a different experiment,
112
+ and averaging it in hides that."""
113
+ return all(r.refused > 0 for r in self.policies)
114
+
115
+ def comparable_load(self) -> bool:
116
+ """Within 10%, across every run on both sides. Two runs offered materially
117
+ different traffic are not a comparison, however similar the configuration."""
118
+ counts = [len(r.decisions) for r in self.baselines + self.policies]
119
+ if not all(counts):
120
+ return False
121
+ return (max(counts) - min(counts)) / max(counts) <= 0.10
122
+
123
+ def measurable(self) -> bool:
124
+ return all(r.outcomes for r in self.baselines + self.policies)
125
+
126
+ # -- the numbers -------------------------------------------------------
127
+
128
+ @staticmethod
129
+ def _ttft(report: Report) -> dict[str, float] | None:
130
+ """Latency of requests that were admitted and returned."""
131
+ values = sorted(
132
+ v for r in report.outcomes if (v := r["outcome"].get("ttft_ms")) is not None
133
+ )
134
+ if not values:
135
+ return None
136
+ return {
137
+ "p50": values[len(values) // 2],
138
+ "p95": values[min(len(values) - 1, math.ceil(0.95 * len(values)) - 1)],
139
+ "max": values[-1],
140
+ "n": len(values),
141
+ }
142
+
143
+ @staticmethod
144
+ def _goodput(report: Report) -> float | None:
145
+ """Met-SLO over OFFERED. A refusal counts as a miss, which is the point."""
146
+ offered = len(report.decisions)
147
+ if not offered or not report.outcomes:
148
+ return None
149
+ met = sum(1 for r in report.outcomes if r["outcome"].get("ok"))
150
+ return met / offered
151
+
152
+ def refused_share(self) -> float:
153
+ shares = [r.refused / len(r.decisions) for r in self.policies if r.decisions]
154
+ return self._spread(shares)[0] if shares else 0.0
155
+
156
+ # -- the page ----------------------------------------------------------
157
+
158
+ def text(self) -> str:
159
+ lines = [RULE, " AdmitPerf — what did the policy buy?", RULE, ""]
160
+ lines += self._headline()
161
+ lines += ["", THIN, "", " SIDE BY SIDE", ""]
162
+ lines += self._table()
163
+ lines += ["", THIN, "", " CAN YOU TRUST THIS?", ""]
164
+ lines += self._trust()
165
+ lines.append(RULE)
166
+ return "\n".join(lines)
167
+
168
+ def _headline(self) -> list[str]:
169
+ if not self.fired():
170
+ return [
171
+ " NO FINDING — the policy never fired.",
172
+ "",
173
+ f" {self.policy.policy or 'the policy'} refused 0 of "
174
+ f"{len(self.policy.decisions)} requests, so these two runs are the same",
175
+ " configuration measured twice. Any difference between the numbers below",
176
+ " is noise, and a latency improvement here would be a false result.",
177
+ "",
178
+ " The fix is usually the LOAD, not the policy: concurrency is arrival",
179
+ " rate times request duration, so short requests cannot fill a cache at",
180
+ " any rate. `admitperf report` on the policy log shows which signal did",
181
+ " move, if any.",
182
+ ]
183
+
184
+ out = [
185
+ f" The policy refused {self.policy.refused} of {len(self.policy.decisions)} "
186
+ f"requests ({self.refused_share():.1%}).",
187
+ "",
188
+ ]
189
+ if not self.measurable():
190
+ out += [
191
+ " What that bought cannot be measured: no outcomes were recorded, so",
192
+ " there is no latency and no goodput. Call policy.outcome(...) after you",
193
+ " forward a request and run this again.",
194
+ ]
195
+ return out
196
+
197
+ p95 = lambda r: t["p95"] if (t := self._ttft(r)) else None # noqa: E731
198
+ base_t = self._arm(self.baselines, p95)
199
+ pol_t = self._arm(self.policies, p95)
200
+ base_g = self._arm(self.baselines, self._goodput)
201
+ pol_g = self._arm(self.policies, self._goodput)
202
+
203
+ def band(s: tuple[float, float, float], fmt: str) -> str:
204
+ """Median, and the observed range when there is more than one run. A bare
205
+ number from one run reads as more certain than it is."""
206
+ median, lo, hi = s
207
+ if self.repeats < 2 or lo == hi:
208
+ return fmt.format(median)
209
+ return f"{fmt.format(median)} ({fmt.format(lo)}-{fmt.format(hi)})"
210
+
211
+ if base_t and pol_t and base_t[0]:
212
+ factor = base_t[0] / pol_t[0] if pol_t[0] else float("inf")
213
+ direction = "lower" if factor > 1 else "HIGHER"
214
+ out.append(
215
+ f" p95 TTFT of served requests is {abs(factor):.2f}x {direction}: "
216
+ f"{band(base_t, '{:.0f}')}ms -> {band(pol_t, '{:.0f}')}ms"
217
+ )
218
+ if base_g and pol_g:
219
+ verb = "up" if pol_g[0] > base_g[0] else "DOWN"
220
+ out.append(
221
+ f" goodput is {verb}: {band(base_g, '{:.3f}')} -> "
222
+ f"{band(pol_g, '{:.3f}')} (of offered)"
223
+ )
224
+ if pol_g[0] < base_g[0]:
225
+ out += [
226
+ "",
227
+ " Goodput fell, so the policy refused requests the cluster could have",
228
+ " served. Faster tails bought at that price are not a win.",
229
+ ]
230
+
231
+ sep = self.separated()
232
+ if sep is False:
233
+ out += [
234
+ "",
235
+ f" BUT the two overlap across {self.repeats} repeats, so this gap is",
236
+ " inside the run-to-run noise. It is not a result yet — more repeats, or a",
237
+ " larger effect.",
238
+ ]
239
+ return out
240
+
241
+ def _table(self) -> list[str]:
242
+ def med(reports: list[Report], metric, fmt: str = "{:.0f}") -> str:
243
+ s = self._arm(reports, metric)
244
+ return "--" if s is None else fmt.format(s[0])
245
+
246
+ p50 = lambda r: t["p50"] if (t := self._ttft(r)) else None # noqa: E731
247
+ p95 = lambda r: t["p95"] if (t := self._ttft(r)) else None # noqa: E731
248
+
249
+ rows = [
250
+ ("policy", self.baseline.policy or "-", self.policy.policy or "-"),
251
+ ("repeats", str(len(self.baselines)), str(len(self.policies))),
252
+ (
253
+ "offered",
254
+ med(self.baselines, lambda r: float(len(r.decisions))),
255
+ med(self.policies, lambda r: float(len(r.decisions))),
256
+ ),
257
+ (
258
+ "refused",
259
+ med(self.baselines, lambda r: float(r.refused)),
260
+ med(self.policies, lambda r: float(r.refused)),
261
+ ),
262
+ ("p50 TTFT ms", med(self.baselines, p50), med(self.policies, p50)),
263
+ ("p95 TTFT ms", med(self.baselines, p95), med(self.policies, p95)),
264
+ (
265
+ "goodput",
266
+ med(self.baselines, self._goodput, "{:.3f}"),
267
+ med(self.policies, self._goodput, "{:.3f}"),
268
+ ),
269
+ ]
270
+ label_row = "median of repeats" if self.repeats > 1 else ""
271
+ out = [f" {label_row:<16} {'baseline':>14} {'with policy':>14}", ""]
272
+ out += [f" {label:<16} {a:>14} {b:>14}" for label, a, b in rows]
273
+
274
+ # Signals, because "the policy changed the cluster" is checkable rather than
275
+ # assumed: KV pressure that did not move means the refusals bought nothing.
276
+ shared = [n for n in self.policy.signals() if n in self.baseline.signals()]
277
+ if shared:
278
+ out += ["", f" {'signal max':<16} {'baseline':>14} {'with policy':>14}", ""]
279
+ for name in shared:
280
+ top = lambda r, n=name: ( # noqa: E731
281
+ s["max"] if (s := r.signal_range(n)) else None
282
+ )
283
+ out.append(
284
+ f" {name:<16} {med(self.baselines, top, '{:.3g}'):>14} "
285
+ f"{med(self.policies, top, '{:.3g}'):>14}"
286
+ )
287
+ return out
288
+
289
+ def _trust(self) -> list[str]:
290
+ out = []
291
+ mark = {True: "ok ", False: "NO "}
292
+ out.append(f" {mark[self.fired()]} the policy fired at all")
293
+ out.append(f" {mark[self.comparable_load()]} both were offered the same load")
294
+ if not self.comparable_load():
295
+ out.append(
296
+ f" {len(self.baseline.decisions)} vs {len(self.policy.decisions)} "
297
+ "requests — a policy run against less traffic will look better than it is"
298
+ )
299
+ out.append(f" {mark[self.measurable()]} outcomes recorded, so cost is measurable")
300
+ sep = self.separated()
301
+ if sep is None:
302
+ out.append("-- only one run per policy, so there is no error bar")
303
+ else:
304
+ out.append(f" {mark[sep]} the two separate across {self.repeats} repeats")
305
+ for name, reports in (("baseline", self.baselines), ("policy", self.policies)):
306
+ for i, r in enumerate(reports, 1):
307
+ for fault in r.fault_text():
308
+ tag = f"{name}" if len(reports) == 1 else f"{name} r{i}"
309
+ out.append(f" !! {tag}:{fault.strip()}")
310
+
311
+ ok = self.fired() and self.comparable_load() and self.measurable()
312
+ if ok and sep:
313
+ out += [
314
+ "",
315
+ " All four hold, so the difference above is the policy and it is larger",
316
+ " than the noise between repeats. This is a result you can quote.",
317
+ ]
318
+ elif ok and sep is None:
319
+ out += [
320
+ "",
321
+ " The first three hold, but one run of each has no error bar. Run the pair",
322
+ " again — a gap smaller than the spread between repeats is not a result:",
323
+ "",
324
+ " admitperf demo --repeats 5",
325
+ ]
326
+ return out
@@ -0,0 +1,29 @@
1
+ """The contract: metrics in, signals derived, policies decide.
2
+
3
+ This subpackage is the part that runs in your request path, so it holds the line
4
+ that makes AdmitPerf safe to install: no third-party imports, no sockets, no
5
+ subprocesses, no filesystem beyond the log you asked for. A test enforces each.
6
+
7
+ Everything else in the package — the CLI, the watcher, the report, the dashboard —
8
+ is a consumer of what this produces and may do as it likes.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from admitperf.core.decision import Decision
14
+ from admitperf.core.log import Log
15
+ from admitperf.core.policy import Policy
16
+ from admitperf.core.reasons import ADMITTED, REASONS
17
+ from admitperf.core.signal import Signal, Source
18
+ from admitperf.core.verdict import Verdict
19
+
20
+ __all__ = [
21
+ "ADMITTED",
22
+ "REASONS",
23
+ "Decision",
24
+ "Log",
25
+ "Policy",
26
+ "Signal",
27
+ "Source",
28
+ "Verdict",
29
+ ]