admitperf 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (168) hide show
  1. admitperf-0.0.1/.claude/skills/README.md +51 -0
  2. admitperf-0.0.1/.claude/skills/admitperf-debt/SKILL.md +45 -0
  3. admitperf-0.0.1/.claude/skills/admitperf-issue/SKILL.md +62 -0
  4. admitperf-0.0.1/.claude/skills/admitperf-policy/SKILL.md +58 -0
  5. admitperf-0.0.1/.claude/skills/admitperf-pr/SKILL.md +50 -0
  6. admitperf-0.0.1/.claude/skills/admitperf-release/SKILL.md +54 -0
  7. admitperf-0.0.1/.claude/skills/admitperf-review/SKILL.md +47 -0
  8. admitperf-0.0.1/.claude/skills/admitperf-run/SKILL.md +65 -0
  9. admitperf-0.0.1/.claude/skills/admitperf-spec/SKILL.md +60 -0
  10. admitperf-0.0.1/.env.example +167 -0
  11. admitperf-0.0.1/.github/ISSUE_TEMPLATE/bug_report.yml +42 -0
  12. admitperf-0.0.1/.github/ISSUE_TEMPLATE/config.yml +14 -0
  13. admitperf-0.0.1/.github/ISSUE_TEMPLATE/documentation.yml +19 -0
  14. admitperf-0.0.1/.github/ISSUE_TEMPLATE/policy_port.yml +36 -0
  15. admitperf-0.0.1/.github/pull_request_template.md +45 -0
  16. admitperf-0.0.1/.github/workflows/ci.yml +98 -0
  17. admitperf-0.0.1/.github/workflows/release.yml +165 -0
  18. admitperf-0.0.1/.gitignore +113 -0
  19. admitperf-0.0.1/.pre-commit-config.yaml +35 -0
  20. admitperf-0.0.1/.python-version +1 -0
  21. admitperf-0.0.1/CONTRIBUTING.md +40 -0
  22. admitperf-0.0.1/LICENSE +21 -0
  23. admitperf-0.0.1/Makefile +17 -0
  24. admitperf-0.0.1/PKG-INFO +380 -0
  25. admitperf-0.0.1/README.md +337 -0
  26. admitperf-0.0.1/assets/banner.svg +26 -0
  27. admitperf-0.0.1/assets/icon.png +0 -0
  28. admitperf-0.0.1/assets/icon.svg +18 -0
  29. admitperf-0.0.1/assets/logo.png +0 -0
  30. admitperf-0.0.1/assets/logo.svg +16 -0
  31. admitperf-0.0.1/client_app/open-webui.yaml +48 -0
  32. admitperf-0.0.1/infra/README.md +28 -0
  33. admitperf-0.0.1/infra/__init__.py +0 -0
  34. admitperf-0.0.1/infra/config/cluster.yaml +88 -0
  35. admitperf-0.0.1/infra/docs/ARCHITECTURE.md +214 -0
  36. admitperf-0.0.1/infra/docs/COMMANDS.md +329 -0
  37. admitperf-0.0.1/infra/docs/DATAFLOW.md +425 -0
  38. admitperf-0.0.1/infra/docs/README.md +28 -0
  39. admitperf-0.0.1/infra/docs/RUNBOOK.md +483 -0
  40. admitperf-0.0.1/infra/docs/diagrams/e2e.png +0 -0
  41. admitperf-0.0.1/infra/docs/diagrams/kernels.png +0 -0
  42. admitperf-0.0.1/infra/docs/diagrams/kv.png +0 -0
  43. admitperf-0.0.1/infra/docs/diagrams/system.png +0 -0
  44. admitperf-0.0.1/infra/docs/gateway.md +33 -0
  45. admitperf-0.0.1/infra/docs/gotchas.md +42 -0
  46. admitperf-0.0.1/infra/docs/infrastructure.md +166 -0
  47. admitperf-0.0.1/infra/docs/kernels.md +219 -0
  48. admitperf-0.0.1/infra/docs/kvbus.md +157 -0
  49. admitperf-0.0.1/infra/docs/observability.md +102 -0
  50. admitperf-0.0.1/infra/docs/router.md +153 -0
  51. admitperf-0.0.1/infra/docs/topology.md +173 -0
  52. admitperf-0.0.1/infra/gateway/__init__.py +1 -0
  53. admitperf-0.0.1/infra/gateway/admission.py +88 -0
  54. admitperf-0.0.1/infra/gateway/harness.py +378 -0
  55. admitperf-0.0.1/infra/gateway/metrics.py +256 -0
  56. admitperf-0.0.1/infra/gateway/queue.py +18 -0
  57. admitperf-0.0.1/infra/gateway/repl.py +672 -0
  58. admitperf-0.0.1/infra/gateway/serve.py +169 -0
  59. admitperf-0.0.1/infra/gateway/types.py +147 -0
  60. admitperf-0.0.1/infra/k8s-config/gateway/httproute.yaml +15 -0
  61. admitperf-0.0.1/infra/k8s-config/gateway/orch-serve.yaml +70 -0
  62. admitperf-0.0.1/infra/k8s-config/hami/hami-decode.yaml +25 -0
  63. admitperf-0.0.1/infra/k8s-config/hami/hami-lambda.yaml +202 -0
  64. admitperf-0.0.1/infra/k8s-config/hami/hami-prefill.yaml +25 -0
  65. admitperf-0.0.1/infra/k8s-config/mooncake/mooncake.yaml +21 -0
  66. admitperf-0.0.1/infra/k8s-config/observability/dashboards/cluster.json +483 -0
  67. admitperf-0.0.1/infra/k8s-config/observability/dashboards/errors.json +518 -0
  68. admitperf-0.0.1/infra/k8s-config/observability/dashboards/gateway.json +298 -0
  69. admitperf-0.0.1/infra/k8s-config/observability/dashboards/hami.json +251 -0
  70. admitperf-0.0.1/infra/k8s-config/observability/dashboards/keda.json +288 -0
  71. admitperf-0.0.1/infra/k8s-config/observability/dashboards/mooncake.json +303 -0
  72. admitperf-0.0.1/infra/k8s-config/observability/dashboards/overview.json +480 -0
  73. admitperf-0.0.1/infra/k8s-config/observability/dashboards/replicas.json +504 -0
  74. admitperf-0.0.1/infra/k8s-config/observability/dashboards/router.json +519 -0
  75. admitperf-0.0.1/infra/k8s-config/observability/dashboards/vllm.json +495 -0
  76. admitperf-0.0.1/infra/k8s-config/observability/dcgm-exporter.yaml +58 -0
  77. admitperf-0.0.1/infra/k8s-config/observability/grafana-values.yaml +21 -0
  78. admitperf-0.0.1/infra/k8s-config/observability/prometheus-values.yaml +27 -0
  79. admitperf-0.0.1/infra/k8s-config/router/inferencepool.yaml +31 -0
  80. admitperf-0.0.1/infra/k8s-config/router/keda-decode.yaml +16 -0
  81. admitperf-0.0.1/infra/k8s-config/router/keda-prefill.yaml +16 -0
  82. admitperf-0.0.1/infra/kernels/__init__.py +0 -0
  83. admitperf-0.0.1/infra/kernels/engineering.py +116 -0
  84. admitperf-0.0.1/infra/observability/__init__.py +0 -0
  85. admitperf-0.0.1/infra/observability/dashboards.py +541 -0
  86. admitperf-0.0.1/infra/providers/__init__.py +0 -0
  87. admitperf-0.0.1/infra/providers/modal_app.py +177 -0
  88. admitperf-0.0.1/infra/providers/modal_provider.py +175 -0
  89. admitperf-0.0.1/infra/render.py +777 -0
  90. admitperf-0.0.1/infra/requirements.txt +1 -0
  91. admitperf-0.0.1/infra/router/__init__.py +1 -0
  92. admitperf-0.0.1/infra/router/kv_eviction.py +163 -0
  93. admitperf-0.0.1/infra/router/kvbus.py +97 -0
  94. admitperf-0.0.1/infra/router/mooncake.py +79 -0
  95. admitperf-0.0.1/infra/router/nccl.py +5 -0
  96. admitperf-0.0.1/infra/router/nixl.py +5 -0
  97. admitperf-0.0.1/infra/router/overflow.py +266 -0
  98. admitperf-0.0.1/infra/router/planner.py +35 -0
  99. admitperf-0.0.1/infra/router/pools.py +301 -0
  100. admitperf-0.0.1/infra/router/router.py +290 -0
  101. admitperf-0.0.1/infra/router/trace.py +29 -0
  102. admitperf-0.0.1/infra/router/warmup.py +116 -0
  103. admitperf-0.0.1/infra/setup/_lambda_only.sh +13 -0
  104. admitperf-0.0.1/infra/setup/day2_observability.sh +67 -0
  105. admitperf-0.0.1/infra/setup/lambda_apply_slices.sh +47 -0
  106. admitperf-0.0.1/infra/setup/lambda_cluster.sh +9 -0
  107. admitperf-0.0.1/infra/setup/lambda_k3s_hami.sh +92 -0
  108. admitperf-0.0.1/infra/setup/lambda_setup.sh +27 -0
  109. admitperf-0.0.1/infra/setup/lambda_sliced.sh +79 -0
  110. admitperf-0.0.1/infra/setup/lambda_vllm.sh +30 -0
  111. admitperf-0.0.1/infra/setup/overflow_smoke.py +97 -0
  112. admitperf-0.0.1/infra/setup/smoke_lambda.sh +21 -0
  113. admitperf-0.0.1/infra/setup/smoke_overflow.sh +16 -0
  114. admitperf-0.0.1/infra/setup/smoke_sliced.sh +70 -0
  115. admitperf-0.0.1/infra/setup/ssh.sh +26 -0
  116. admitperf-0.0.1/infra/setup/sync_to_lambda.sh +30 -0
  117. admitperf-0.0.1/infra/traces/.gitkeep +0 -0
  118. admitperf-0.0.1/infra/traces/requests.jsonl +45 -0
  119. admitperf-0.0.1/infra/traces/shared_prefix.jsonl +50 -0
  120. admitperf-0.0.1/infra/traces/unique_prefix.jsonl +80 -0
  121. admitperf-0.0.1/pyproject.toml +115 -0
  122. admitperf-0.0.1/src/admitperf/__init__.py +43 -0
  123. admitperf-0.0.1/src/admitperf/cli.py +282 -0
  124. admitperf-0.0.1/src/admitperf/comparison.py +326 -0
  125. admitperf-0.0.1/src/admitperf/core/__init__.py +29 -0
  126. admitperf-0.0.1/src/admitperf/core/decision.py +47 -0
  127. admitperf-0.0.1/src/admitperf/core/log.py +101 -0
  128. admitperf-0.0.1/src/admitperf/core/policy.py +231 -0
  129. admitperf-0.0.1/src/admitperf/core/reasons.py +33 -0
  130. admitperf-0.0.1/src/admitperf/core/signal.py +94 -0
  131. admitperf-0.0.1/src/admitperf/core/signals.py +87 -0
  132. admitperf-0.0.1/src/admitperf/core/verdict.py +18 -0
  133. admitperf-0.0.1/src/admitperf/dashboard/.streamlit/config.toml +12 -0
  134. admitperf-0.0.1/src/admitperf/dashboard/__init__.py +0 -0
  135. admitperf-0.0.1/src/admitperf/dashboard/app.py +770 -0
  136. admitperf-0.0.1/src/admitperf/dashboard/assets/icon.png +0 -0
  137. admitperf-0.0.1/src/admitperf/dashboard/assets/logo.png +0 -0
  138. admitperf-0.0.1/src/admitperf/dashboard/theme.py +157 -0
  139. admitperf-0.0.1/src/admitperf/demo.py +274 -0
  140. admitperf-0.0.1/src/admitperf/discover.py +85 -0
  141. admitperf-0.0.1/src/admitperf/experiment.py +42 -0
  142. admitperf-0.0.1/src/admitperf/item.py +22 -0
  143. admitperf-0.0.1/src/admitperf/items.py +188 -0
  144. admitperf-0.0.1/src/admitperf/policies/__init__.py +18 -0
  145. admitperf-0.0.1/src/admitperf/policies/dual_gate.py +36 -0
  146. admitperf-0.0.1/src/admitperf/policies/kv_threshold.py +35 -0
  147. admitperf-0.0.1/src/admitperf/policies/no_admission.py +27 -0
  148. admitperf-0.0.1/src/admitperf/policies/queue_depth.py +35 -0
  149. admitperf-0.0.1/src/admitperf/policy_runs.py +31 -0
  150. admitperf-0.0.1/src/admitperf/report.py +340 -0
  151. admitperf-0.0.1/src/admitperf/status.py +19 -0
  152. admitperf-0.0.1/src/admitperf/terminology.py +207 -0
  153. admitperf-0.0.1/src/admitperf/watch.py +107 -0
  154. admitperf-0.0.1/tests/__init__.py +0 -0
  155. admitperf-0.0.1/tests/test_cli.py +166 -0
  156. admitperf-0.0.1/tests/test_comparison.py +310 -0
  157. admitperf-0.0.1/tests/test_dashboard.py +188 -0
  158. admitperf-0.0.1/tests/test_decision.py +60 -0
  159. admitperf-0.0.1/tests/test_example.py +71 -0
  160. admitperf-0.0.1/tests/test_layering.py +186 -0
  161. admitperf-0.0.1/tests/test_log.py +67 -0
  162. admitperf-0.0.1/tests/test_policies.py +50 -0
  163. admitperf-0.0.1/tests/test_policy.py +222 -0
  164. admitperf-0.0.1/tests/test_render_topology.py +569 -0
  165. admitperf-0.0.1/tests/test_report.py +165 -0
  166. admitperf-0.0.1/tests/test_signal.py +113 -0
  167. admitperf-0.0.1/tests/test_watch.py +99 -0
  168. admitperf-0.0.1/uv.lock +3426 -0
@@ -0,0 +1,51 @@
1
+ # Project skills
2
+
3
+ Eight skills that encode how this project is worked on. They load automatically for anyone
4
+ using Claude Code in this repository — no setup, because they are versioned alongside the
5
+ code they describe.
6
+
7
+ | Skill | Use it when |
8
+ |---|---|
9
+ | [`admitperf-run`](admitperf-run/SKILL.md) | Running an experiment, or reading what one produced |
10
+ | [`admitperf-policy`](admitperf-policy/SKILL.md) | Porting a published admission policy |
11
+ | [`admitperf-review`](admitperf-review/SKILL.md) | Reviewing a change, adversarially |
12
+ | [`admitperf-pr`](admitperf-pr/SKILL.md) | Opening a pull request |
13
+ | [`admitperf-issue`](admitperf-issue/SKILL.md) | Filing a bug, a policy port, or a docs problem |
14
+ | [`admitperf-debt`](admitperf-debt/SKILL.md) | Auditing claims against the bundles behind them |
15
+ | [`admitperf-spec`](admitperf-spec/SKILL.md) | Planning docs, the board, and what is actually done |
16
+ | [`admitperf-release`](admitperf-release/SKILL.md) | Cutting a release |
17
+
18
+ ## Why these exist
19
+
20
+ They are not style guides. Each encodes a mistake this project actually made.
21
+
22
+ - **`admitperf-run`** exists because of `experiments/half-capacity-headroom`:
23
+ `waiting_requests` identically zero across all three repeats, `running_requests` peaking
24
+ at 6-8 against a cap of 10. The engine was never saturated, so every signal was flat —
25
+ not just the one under test. That run was treated as banked evidence for weeks. It
26
+ isolates nothing.
27
+ - **`admitperf-policy`** front-loads the Class A / Class B question, because a Class B
28
+ policy cannot sit behind this API at all and discovering that mid-port wastes the port.
29
+ - **`admitperf-debt`** exists because a figure was published that its bundles did not
30
+ support, and because `docs/status.md` has disagreed with what the software can actually
31
+ demonstrate.
32
+ - **`admitperf-pr`** says to check the gate's *exit codes*: piping `ruff` into `tail`
33
+ returns tail's status, and a formatting failure can be committed while the gate reports
34
+ clean. It also says to run the gate in a clean venv, because CI has already failed on an
35
+ extra the working venv had accumulated by hand.
36
+ - **`admitperf-issue`** asks for the signal range before anything else, because "the policy
37
+ admitted everything" and "the policy never saw pressure" look identical from outside.
38
+ - **`admitperf-spec`** exists because the plan and the board have disagreed: a direction was
39
+ recorded as banked while the run behind it isolated nothing.
40
+
41
+ The thread running through all eight: **a number is not a result until you know the signal
42
+ moved.** A policy that never saw its threshold did not perform badly — it never ran, and it
43
+ reports the same numbers as no policy at all. Telling those apart is the entire point of
44
+ this repository, so it is the first thing every skill checks.
45
+
46
+ ## Adding one
47
+
48
+ A skill earns its place by encoding something that went wrong, or a rule a newcomer would
49
+ otherwise learn by breaking. Keep it short, make it concrete, and cite the real incident —
50
+ the specifics are what make a rule memorable and what let a reader judge whether it still
51
+ applies.
@@ -0,0 +1,45 @@
1
+ ---
2
+ name: admitperf-debt
3
+ description: >-
4
+ Audit what AdmitPerf claims against what its bundles prove — numbers without a bundle,
5
+ policies never exercised, docs describing code that moved. Use when asked about technical
6
+ debt, what is unproven or over-claimed, or before a release or a paper submission.
7
+ ---
8
+
9
+ # Auditing claims against proof
10
+
11
+ ## What this looks for
12
+
13
+ **A claim the software cannot demonstrate in one command.** Reviewers read the repo; a gap
14
+ between `docs/status.md` and what runs is the risk that matters here.
15
+
16
+ ## The audit
17
+
18
+ 1. **Every number in a doc names a bundle.** Walk the docs, list the figures, and for each
19
+ one find the run it came from. A number with no bundle is either unverified or lost, and
20
+ both are reported the same way: as unproven.
21
+ 2. **Every cited bundle names its commit.** Traceable only to a version string that never
22
+ changes is not traceable.
23
+ 3. **Every registered policy has a run where its signal crossed its threshold.** A policy
24
+ never exercised under pressure is a policy with no evidence behind it, however many
25
+ tests it passes.
26
+ 4. **Every claimed capability has a command that shows it.** If showing it needs three
27
+ steps and local knowledge, it is not demonstrable.
28
+ 5. **Docs against the tree.** This repo has moved its package layout more than once;
29
+ anything citing a path is suspect until checked.
30
+
31
+ ## Known debts, so they are not rediscovered
32
+
33
+ - **A published figure its bundles did not support.** The reason this audit exists.
34
+ - **`half-capacity-headroom` isolates nothing.** Engine never saturated, every signal flat.
35
+ Any claim resting on it is unsupported.
36
+ - **KV pressure and queue depth have never been compared in the same deployment.**
37
+ `which-policy-when` is specified for it and has not run.
38
+ - **`infra/` is vendored and unlinted.** Its lint debt is upstream's; ours is that we have
39
+ not decided what happens to the gateway inside it.
40
+
41
+ ## How to report
42
+
43
+ Per item: the claim, where it is made, what would prove it, and whether that evidence
44
+ exists. **An item that cannot be proven is reported as not-shown, never quietly dropped** —
45
+ dropping it is how an over-claim survives an audit that was meant to catch it.
@@ -0,0 +1,62 @@
1
+ ---
2
+ name: admitperf-issue
3
+ description: >-
4
+ File an AdmitPerf issue that is actionable on arrival — which template, the evidence to
5
+ gather before writing it, and the labels. Use when asked to file an issue, open a bug,
6
+ track something, or when a problem is found that will not be fixed in the current change.
7
+ ---
8
+
9
+ # Filing an issue
10
+
11
+ ## Gather the evidence first
12
+
13
+ For anything about a run, three things answer most of what would otherwise be asked, and
14
+ each round trip of asking costs a day:
15
+
16
+ 1. **`manifest.json`** — what ran, against what, at which commit.
17
+ 2. **The signal range from `summary.json`.** Usually *the* answer. A policy whose signal
18
+ never approached its threshold did not misbehave; it never fired, and it reports the same
19
+ numbers as no policy at all.
20
+ 3. **The command**, and the config it ran against.
21
+
22
+ If the signal range explains the behaviour, say so in the issue rather than filing it as a
23
+ bug. "The policy admitted everything" and "the policy never saw pressure" look identical
24
+ from outside and are completely different problems.
25
+
26
+ ## Which template
27
+
28
+ | Template | For |
29
+ |---|---|
30
+ | **Bug report** | Something behaves differently from what it says it does |
31
+ | **Policy port** | A published policy that should be available here |
32
+ | **Documentation** | Wrong, missing, or misleading — misleading is the worst, the reader leaves confident |
33
+
34
+ Blank issues are off on purpose. The templates ask for what we would otherwise have to
35
+ chase.
36
+
37
+ ## Labels
38
+
39
+ `policy` · `infra` · `measurement` · `blocked-on-gpu` · `bug` · `documentation` ·
40
+ `enhancement`
41
+
42
+ **`blocked-on-gpu` matters most.** It separates what can be done on a laptop today from what
43
+ waits on a rented cluster. Without it the board looks full while everything on it is
44
+ unstartable.
45
+
46
+ ## Writing it
47
+
48
+ **One issue, one thing.** An issue that contains three problems gets closed when one is
49
+ fixed.
50
+
51
+ **Say what you expected, not just what happened.** *"It admitted everything"* is a
52
+ description. *"It admitted everything, and I expected refusals once KV passed 0.9"* is a
53
+ specification, and someone can tell whether it is fixed.
54
+
55
+ **Attach it to the milestone** if it belongs in the current one. An issue with no milestone
56
+ is invisible to anyone planning.
57
+
58
+ ## Before filing
59
+
60
+ A problem found mid-change that you are about to fix does not need an issue. A problem found
61
+ mid-change that you are **not** going to fix does — otherwise it survives as a comment in a
62
+ diff nobody reads again.
@@ -0,0 +1,58 @@
1
+ ---
2
+ name: admitperf-policy
3
+ description: >-
4
+ Port a published admission policy into AdmitPerf — the Class A/B test that comes first,
5
+ the frozen interface, declaring what signal it needs, and recording deviations from the
6
+ paper. Use when adding a policy, porting one from a paper, or reviewing a port.
7
+ ---
8
+
9
+ # Porting a policy
10
+
11
+ ## Class A or Class B — decide before writing anything
12
+
13
+ **Class A** is a pure function of `(request, observable fleet state) → decision`, running in
14
+ front of an unmodified engine and reading its metrics. That is all this API accepts.
15
+
16
+ **Class B** changes how batches are *formed inside* the engine. Porting one means
17
+ maintaining a fork of thousands of lines of scheduler internals. FairBatching, FastServe
18
+ and CONCUR are Class B and do not fit here — saying precisely why is itself the
19
+ Portability axis the survey needs.
20
+
21
+ Getting this wrong costs the whole port, and it is answerable from the paper's abstract.
22
+
23
+ ## The interface is frozen
24
+
25
+ ```python
26
+ def decide(self, req: Request, state: SystemState) -> Decision
27
+ ```
28
+
29
+ `ADMIT`, `DEFER` (with `retry_after_ms`), or `REJECT` (with a reason). Deterministic: the
30
+ same inputs and internal state must give the same decision, or experiments stop being
31
+ reproducible.
32
+
33
+ Anything the policy needs that `SystemState` lacks arrives as an **optional** field. Never
34
+ change `Decision` or `decide()`.
35
+
36
+ ## Declare what it reads
37
+
38
+ A policy states its signal in the same vocabulary as the survey's axes — unit, setting,
39
+ objective, signal, portability. Two things fall out for free:
40
+
41
+ - The harness can **refuse to run** a policy against an engine that cannot supply its
42
+ signal, before any provisioning cost.
43
+ - The report can show the **observed range** of that signal, which is what tells a reader
44
+ the policy was live.
45
+
46
+ ## Write the deviation list while porting, not after
47
+
48
+ Anything that cannot be reproduced faithfully: a simulator-only assumption, a metric the
49
+ engine does not expose, a constant the paper never states. Written during the port it is a
50
+ record; written afterwards it is a reconstruction, and the details that mattered are gone.
51
+
52
+ ## Before calling it done
53
+
54
+ - A test that the policy refuses *something* under pressure. A policy that admits
55
+ everything passes most tests.
56
+ - A run where its signal actually crossed its threshold. Otherwise the port is unexercised
57
+ and you have no evidence it works.
58
+ - Its entry in `docs/policies.md`, with the deviation list.
@@ -0,0 +1,50 @@
1
+ ---
2
+ name: admitperf-pr
3
+ description: >-
4
+ Open a pull request against AdmitPerf — the gate to run, what the body must explain, and
5
+ the branch rules. Use when asked to open or raise a PR, or when finishing a change that
6
+ is ready to land.
7
+ ---
8
+
9
+ # Opening a PR
10
+
11
+ ## Run the gate, and check exit codes
12
+
13
+ ```bash
14
+ ruff check . && ruff format --check . && pytest -q
15
+ ```
16
+
17
+ **Check the exit code, not the output.** Piping either into `tail` returns tail's status,
18
+ and a failure then looks clean. This is how a formatting failure gets committed.
19
+
20
+ Run it in a **clean environment**, not your working venv. A narrower install than CI uses
21
+ collects tests and then fails to import them — which has already shipped here once, when
22
+ `.[dev]` passed locally only because the venv had accumulated `pandas` and `streamlit` by
23
+ hand. Use `pip install -e '.[all]'` in a fresh venv, which is what CI does.
24
+
25
+ ## Branch and title
26
+
27
+ Branch from `main`. Never commit to it directly.
28
+
29
+ Conventional-commit title, because release notes are assembled from these:
30
+ `feat(policies):` · `fix(bench):` · `refactor:` · `docs:` · `ci:`
31
+
32
+ ## The body
33
+
34
+ **What changed, and why.** The *why* is the part worth writing — a change that explains its
35
+ reasoning is understandable in six months. If you fixed something, describe the broken
36
+ behaviour from outside: the symptom someone would have reported.
37
+
38
+ **How it was verified.** Not "tests pass". What you actually drove — a run you watched, a
39
+ bundle you read, the decision log you checked. This project has shipped a figure its
40
+ bundles did not support and a run where every signal was flat. Both passed their tests.
41
+
42
+ **Anything deliberately left undone.** Known gaps and follow-ups. A limitation stated here
43
+ is a known limitation; the same limitation unstated is a bug waiting to be rediscovered.
44
+
45
+ ## Before marking ready
46
+
47
+ - Any number in the diff names the bundle it came from.
48
+ - A new policy is Class A and declares its signal.
49
+ - Version untouched — releases are cut by tag, never by editing a version by hand.
50
+ - Linked to its issue and milestone, so the board reflects reality.
@@ -0,0 +1,54 @@
1
+ ---
2
+ name: admitperf-release
3
+ description: >-
4
+ Cut an AdmitPerf release — the tag-driven flow, the TestPyPI round trip, the trusted
5
+ publisher step only a human can do, and what cannot be undone. Use when asked to cut a
6
+ release, tag a version, or prepare one.
7
+ ---
8
+
9
+ # Cutting a release
10
+
11
+ ## What cannot be undone
12
+
13
+ **A published version number can never be reused.** Not on PyPI, not on TestPyPI. If `0.1.0`
14
+ publishes broken, the fix is `0.1.1` and `0.1.0` stays broken forever. Everything below
15
+ exists because of that.
16
+
17
+ ## Before anything
18
+
19
+ **Trusted Publishing must be configured first, by a human, in a browser.** The workflow
20
+ stores no token; it mints one per run over OIDC. Register at
21
+ `pypi.org/manage/account/publishing/` with owner `harshuljain13`, repo `AdmitPerf`,
22
+ workflow `release.yml`, environment `pypi` — and the same on TestPyPI.
23
+
24
+ A first-time package that is not registered on **both** indexes half-publishes: TestPyPI
25
+ succeeds, PyPI fails, and that version is now burned.
26
+
27
+ ## The flow
28
+
29
+ ```bash
30
+ # bump the version in pyproject.toml and admitperf/__init__.py — they must match
31
+ git commit -am "release: 0.1.0"
32
+ git tag v0.1.0
33
+ git push origin v0.1.0
34
+ ```
35
+
36
+ A tag publishes. A branch never does.
37
+
38
+ The workflow then goes **TestPyPI → install back out of TestPyPI → smoke-test → PyPI**. That
39
+ round trip is the only thing that catches a missing dependency or a bad pin, because it is a
40
+ real index resolve. A local path install cannot fail that way, which is why local success
41
+ proves nothing here.
42
+
43
+ ## Rehearse first
44
+
45
+ `workflow_dispatch` with `target: testpypi` runs the whole thing without cutting a tag. Do
46
+ this for any release that changes packaging, adds a dependency, or moves a module.
47
+
48
+ ## Before tagging
49
+
50
+ - CI green on `main`, including the wheel job that installs into a clean venv and imports
51
+ the public API. That job is what catches a module missing from the package manifest; the
52
+ editable install cannot, because it sees the working tree.
53
+ - `docs/status.md` matches what ships.
54
+ - No number in the docs lacks a bundle. A release is a bad time to discover one.
@@ -0,0 +1,47 @@
1
+ ---
2
+ name: admitperf-review
3
+ description: >-
4
+ Adversarial review for AdmitPerf changes, tuned to the defects this repository actually
5
+ produces — results from runs where the signal never moved, and claims the bundles do not
6
+ support. Use when reviewing a PR or a diff, or before marking anything done.
7
+ ---
8
+
9
+ # Reviewing a change
10
+
11
+ ## The defect this repo produces
12
+
13
+ Not broken code. **A number that looks like a result and is not one.**
14
+
15
+ The shape: a run completes, a table is produced, the table is cited — and the signal the
16
+ policy reads never left its floor, so the policy never acted and the table says nothing
17
+ about it. This has happened: `half-capacity-headroom`, `waiting_requests` identically zero
18
+ across three repeats, treated as banked evidence.
19
+
20
+ So on any change that produces or cites a number, ask first: **could this number have come
21
+ from a run where nothing was saturated?** If yes, that is the review finding.
22
+
23
+ ## Checks, in order of what they catch
24
+
25
+ 1. **Does a cited number name its bundle, and does that bundle name its commit?** A number
26
+ traceable only to a version string that never changes is not traceable.
27
+ 2. **Is the comparison within one deployment?** Pooling across hardware reports the
28
+ machine, not the policy.
29
+ 3. **Repeats and spread?** A single run has no error bar. A gap inside noise is not a gap.
30
+ 4. **Does a new policy actually refuse anything** in its tests? One that admits everything
31
+ passes most of them.
32
+ 5. **Class A?** A policy that needs engine changes does not fit behind this API, and a
33
+ review is the last cheap place to notice.
34
+ 6. **Does `docs/status.md` still match what the code can demonstrate in one command?** The
35
+ gap between claim and proof is the risk that matters for this project.
36
+
37
+ ## On vendored code
38
+
39
+ `infra/` is the vendored serving cluster, excluded from ruff. Do not reformat it — that
40
+ makes every future diff against upstream unreadable. Changes there should be minimal and
41
+ should say why they could not be made upstream.
42
+
43
+ ## What a good review comment looks like
44
+
45
+ Name the failing input, not the style. *"If `max_num_seqs` is below what KV allows, this
46
+ measures the scheduler rather than the cache"* is reviewable. *"Consider refactoring"* is
47
+ not.
@@ -0,0 +1,65 @@
1
+ ---
2
+ name: admitperf-run
3
+ description: >-
4
+ Run an AdmitPerf experiment and read what it produced — the saturation check that comes
5
+ before any comparison, what a bundle contains, and the one question that decides whether
6
+ a result means anything. Use when running a benchmark, reading results, comparing
7
+ policies, or interpreting a report.
8
+ ---
9
+
10
+ # Running an experiment, and reading one
11
+
12
+ ## Ask this first, every time
13
+
14
+ **Did the signal move?**
15
+
16
+ A policy that never reached its threshold did not perform badly. It never ran, and it
17
+ reports the same numbers as having no policy at all. Until you know the signal's observed
18
+ range, a results table tells you nothing.
19
+
20
+ This is not hypothetical. `experiments/half-capacity-headroom` recorded `waiting_requests`
21
+ identically **zero** across all three repeats, with `running_requests` peaking at 6-8
22
+ against a cap of 10. The engine was never saturated, so *every* signal was flat — not just
23
+ the one under test. The run was treated as banked evidence for weeks. It isolates nothing.
24
+
25
+ ## Before comparing anything
26
+
27
+ 1. **Establish capacity.** `bench capacity` finds what the deployment can actually serve.
28
+ Comparing policies below that ceiling compares nothing.
29
+ 2. **Check the load is capacity-relative.** "15 rps" is meaningless alone; "15 rps at 46%
30
+ of ceiling" is a number someone else can reproduce.
31
+ 3. **Confirm the engine queues internally.** Load must back up *inside* vLLM, not in front
32
+ of it. If `max_concurrent_inputs` is below the engine's `max_num_seqs`, requests wait at
33
+ the platform and the admission signal stays flat while latency climbs.
34
+
35
+ ## What a bundle contains
36
+
37
+ ```
38
+ results/<timestamp>-<experiment>/<policy>-r<n>/
39
+ ├── manifest.json what ran, against what, with which settings and commit
40
+ ├── summary.json the numbers, each tagged with where it came from
41
+ ├── decisions.jsonl every admit/defer/reject, with the state it was decided on
42
+ └── outcomes.jsonl per-request TTFT, inter-token gaps, deadline verdict
43
+ ```
44
+
45
+ `decisions.jsonl` is the one that matters. A verdict alone cannot be interpreted after the
46
+ fact; the state beside it can.
47
+
48
+ ## Reading a result
49
+
50
+ - **Identical numbers across policies** is a finding, not a failure — but only once you can
51
+ say whether it is because the policies agreed or because none of them fired.
52
+ - **Never pool runs across deployments.** A table mixing an A10G row with an A100 row
53
+ reports the machine, not the policy. `same-policies-across-gpus` exists to enforce this.
54
+ - **Repeats and spread, or no claim.** A single run has no error bar, and a difference
55
+ inside run-to-run noise is not a difference.
56
+ - **Reject and defer are outcomes**, not losses. Goodput-under-admission counts them; raw
57
+ throughput hides them.
58
+
59
+ ## Do not
60
+
61
+ - Quote latency from a freshly started worker. The engine reports ready minutes before it
62
+ serves, while it downloads weights and compiles graphs.
63
+ - Report a win without naming the regime it holds in. The question this project exists to
64
+ answer is *which signal wins under which workload*, and an unqualified winner answers a
65
+ question nobody asked.
@@ -0,0 +1,60 @@
1
+ ---
2
+ name: admitperf-spec
3
+ description: >-
4
+ Work with AdmitPerf's planning docs and the issue board — which one is authoritative,
5
+ reconciling a claim against the tree before calling anything done, and recording
6
+ deviations. Use when asked what is done, about phases or tasks, or when closing work.
7
+ ---
8
+
9
+ # Planning, and what is actually done
10
+
11
+ ## Two places track work. Only one is authoritative.
12
+
13
+ - **GitHub issues and the milestone** — the source of truth for status. What is open, what
14
+ is blocked, what shipped.
15
+ - **`.spec-dev/`** — local planning, deliberately not in git. Requirements, design
16
+ decisions, task breakdowns. It explains *why*; it does not record *whether*.
17
+
18
+ When they disagree, the board wins and the plan gets updated. A plan that claims work is
19
+ done is how this project once recorded a direction as "banked" while the run behind it
20
+ isolated nothing.
21
+
22
+ ## `.spec-dev/` contents
23
+
24
+ ```
25
+ requirements.md the thesis, the goals, what would make it wrong
26
+ spec.md the active design decisions
27
+ spec-paperB-stale.md superseded, kept because its findings are still true
28
+ tasks.md phases, each one sitting
29
+ ```
30
+
31
+ **A stale plan is marked stale, not deleted.** `spec-paperB-stale.md` carries the
32
+ half-capacity finding, which is still the most important thing in that file.
33
+
34
+ ## Before calling anything done
35
+
36
+ **Reconcile against the tree, not against the plan.** The question is never "did I write
37
+ this down as finished" but "can I show it in one command".
38
+
39
+ - For a policy: a run where its signal crossed its threshold. Tests passing is not that.
40
+ - For a measurement: the bundle, and the commit that produced it.
41
+ - For a claim in a doc: the command that demonstrates it.
42
+
43
+ Anything that cannot be shown is reported as **not-shown**, never quietly dropped. Dropping
44
+ it is how an over-claim survives.
45
+
46
+ ## Recording a deviation
47
+
48
+ When the work diverges from the plan — and it will — write the deviation where the plan is,
49
+ with the reason, at the time. Afterwards it is a reconstruction and the detail that mattered
50
+ is already gone.
51
+
52
+ The deviations worth recording are the ones where something turned out to be impossible or
53
+ unnecessary, not the ones where an estimate was wrong.
54
+
55
+ ## Closing
56
+
57
+ An issue closes when its stated "done when" is true, not when the code was written. Several
58
+ issues on this board have a "done when" that requires a live cluster; those cannot close
59
+ from a laptop, and marking them done because the code exists is the failure mode the
60
+ `blocked-on-gpu` label was created to prevent.
@@ -0,0 +1,167 @@
1
+ # Optional. Nothing here is needed to run AdmitPerf against a local engine, or
2
+ # to deploy an ungated model to Modal after `modal setup`.
3
+ #
4
+ # cp .env.example .env
5
+ #
6
+ # `.env` is read by every `admitperf` command, including the ones the dashboard
7
+ # runs for you. Anything already exported wins, so a one-off
8
+ # `HF_TOKEN=... admitperf infra up` still overrides the file.
9
+ #
10
+ # This is a superset of the lab's own env file plus AdmitPerf's. Every name
11
+ # below is read somewhere in this repository — the list was taken from the code
12
+ # rather than copied, and the file that reads each one is named. A variable the
13
+ # code does not read is worse than a missing one: it looks configured.
14
+ #
15
+ # The cluster reads its settings from the ENVIRONMENT, not from
16
+ # infra/config/cluster.yaml. The YAML records intent and generates manifests;
17
+ # these are what a running gateway actually sees. Where the two disagree, the
18
+ # environment wins and nothing warns you. See #15.
19
+
20
+
21
+ # ===========================================================================
22
+ # GPU box access infra/render.py, infra/setup/*.sh
23
+ # ===========================================================================
24
+ # Addresses and keys live HERE, not in infra/config/cluster.yaml, which is
25
+ # committed. An IP is a fact about today; the topology is not. Re-renting a box is
26
+ # an edit to this file rather than a commit, and a private key path is never
27
+ # committed at all.
28
+ #
29
+ # Resolved per host by name: host `gpu-a` reads LAMBDA_HOST_GPU_A, falling back to
30
+ # LAMBDA only when it is the ONLY host. With two boxes LAMBDA is ambiguous and is
31
+ # refused — applying one address to both would point every URL at one machine
32
+ # while the plan claimed two, which is a one-worker run reported as two.
33
+ #
34
+ # `render.py --plan` works with these unset. `--env` and bring-up refuse, naming
35
+ # the variable they want.
36
+ # LAMBDA=ubuntu@YOUR_LAMBDA_IP # shorthand, single host only
37
+ # LAMBDA_HOST_GPU_A=ubuntu@YOUR_LAMBDA_IP
38
+ # LAMBDA_HOST_GPU_B=ubuntu@YOUR_SECOND_IP
39
+ # LAMBDA_SSH_KEY=$HOME/.ssh/YOUR_LAMBDA_KEY # one key for every host
40
+ # LAMBDA_SSH_KEY_GPU_B=$HOME/.ssh/OTHER_KEY # per-host override
41
+
42
+
43
+ # ===========================================================================
44
+ # Models and topology infra/gateway/serve.py, router/
45
+ # ===========================================================================
46
+ # READ THIS BEFORE CHANGING THE MODELS. Two traps live here, and both of them
47
+ # produce a run that looks fine and proves nothing.
48
+ #
49
+ # 1. TEXT_MODEL and VISION_MODEL being DIFFERENT, with LAB_SPLIT=capability,
50
+ # makes the two workers non-interchangeable. Placement then becomes a
51
+ # capability lookup — a vision request has exactly one destination — and the
52
+ # admission question disappears, because there is no choice left to make.
53
+ # The project's design (infra/config/cluster.yaml) calls for ONE model on
54
+ # both workers so placement stays a load decision. The defaults below are
55
+ # the lab's and do not match that. Pick one deliberately.
56
+ #
57
+ # 2. A 3B model will not fill KV. These defaults are 3B because the lab ran on
58
+ # one small GPU. Admission never bites, every signal stays flat, and the run
59
+ # reproduces experiments/half-capacity-headroom by construction. If you keep
60
+ # them, you must also cap max_num_seqs by hand and declare the cap artificial.
61
+ #
62
+ # export TEXT_MODEL=Qwen/Qwen2.5-3B-Instruct
63
+ # export VISION_MODEL=Qwen/Qwen2.5-VL-3B-Instruct
64
+ # export LOCAL_MODEL=Qwen/Qwen2.5-3B-Instruct
65
+
66
+ # phase = prefill/decode disaggregation. capability = text worker + vision
67
+ # worker, which is trap 1 above.
68
+ # export LAB_TOPOLOGY=disaggregated
69
+ # export LAB_SPLIT=capability
70
+
71
+
72
+ # ===========================================================================
73
+ # Endpoints infra/gateway/, infra/router/
74
+ # ===========================================================================
75
+ # PREFILL_URLS, DECODE_URLS, LAB_TOPOLOGY, LAB_SPLIT, KV_BACKEND and LOCAL_MODEL
76
+ # are RENDERED by `infra.render --env` from the topology. Setting them by hand
77
+ # overrides the config silently, which is how a two-pool plan ends up serving
78
+ # from one. Prefer re-rendering.
79
+ # ===========================================================================
80
+ # export LOCAL_BASE_URL=http://127.0.0.1:8000/v1
81
+ # export PREFILL_URLS=http://127.0.0.1:8000
82
+ # export DECODE_URLS=http://127.0.0.1:8001 # infra/setup/lambda_sliced.sh
83
+ # export ORCH_URL=http://127.0.0.1:8080/v1
84
+
85
+ # Where the gateway itself binds. infra/gateway/serve.py
86
+ # export SERVE_HOST=0.0.0.0
87
+ # export SERVE_PORT=8080
88
+
89
+
90
+ # ===========================================================================
91
+ # KV transport infra/router/kvbus.py
92
+ # ===========================================================================
93
+ # mooncake is the only backend that does anything. nccl and nixl are
94
+ # deliberate no-op stubs — four lines that return. Setting either makes hops
95
+ # vanish from Mooncake's /hops while requests still succeed, so ZERO HOPS IN
96
+ # THE DASHBOARD DOES NOT PROVE THE HOP DID NOT HAPPEN. Check this first.
97
+ # export KV_BACKEND=mooncake
98
+ # export MOONCAKE_URL=http://127.0.0.1:50051
99
+
100
+
101
+ # ===========================================================================
102
+ # Load generation NOT VENDORED YET — see #27
103
+ # ===========================================================================
104
+ # The lab's load generators (locustfile.py, crew_flood.py) live in its app/
105
+ # directory and were not copied into this repository. LOCUST_HOST is listed for
106
+ # completeness; nothing here reads it today.
107
+ # export LOCUST_HOST=http://127.0.0.1:8080
108
+
109
+
110
+ # ===========================================================================
111
+ # Overflow — paid capacity for when our own cluster refuses
112
+ # infra/router/overflow.py
113
+ # ===========================================================================
114
+ # OVERFLOW_MAX_REQS is a LIFETIME counter on the gateway process, not a rate
115
+ # limit. After this many overflows the path turns itself off and requests fall
116
+ # back to the local refusal. That is a deliberate spend guard, and it is also a
117
+ # measurement hazard: a run can report "overflow did not help" when what
118
+ # actually happened is that overflow stopped being available. See #26.
119
+ #
120
+ # A 4B model answering in place of a 72B is a relief valve, not an equivalent
121
+ # service. Overflowed responses are not comparable to local ones and must not
122
+ # be pooled with them when reporting quality.
123
+ #
124
+ # Unrecognised names here are ignored silently and the gateway reports
125
+ # "overflow off" — a typo looks like a configuration choice.
126
+ # RENDERED FROM THE CONFIG, except the key. `python -m infra.render --env` emits
127
+ # OVERFLOW_BACKEND, OVERFLOW_BASE_URL, OVERFLOW_MODEL, OVERFLOW_MAX_REQS and
128
+ # OVERFLOW_MAX_TOKENS from infra/config/cluster.yaml, so they are not typed in two
129
+ # places where they would drift. Set them below only to override a rendered run.
130
+ #
131
+ # The key is the one secret in that block, and it belongs here:
132
+ # export OVERFLOW_API_KEY= # unset = overflow off
133
+ #
134
+ # export OVERFLOW_BACKEND=superlinked
135
+ # export OVERFLOW_BASE_URL=https://api.superlinked.com/v1
136
+ # export OVERFLOW_MODEL=Qwen/Qwen3.5-4B
137
+ # export OVERFLOW_MAX_REQS=20
138
+ # export OVERFLOW_MAX_TOKENS=64
139
+ # export OVERFLOW_MAX_REQS=20
140
+
141
+
142
+ # ===========================================================================
143
+ # Trace — one row per request infra/gateway/serve.py
144
+ # ===========================================================================
145
+ # The report is generated from this file. It is also where the signal value
146
+ # behind each admission decision will be recorded (#11).
147
+ # export TRACE_PATH=traces/requests.jsonl
148
+
149
+
150
+ # ===========================================================================
151
+ # AdmitPerf's own settings
152
+ # ===========================================================================
153
+ # Override the config a command loads. src/admitperf/core/config.py
154
+ # ADMITPERF_CONFIG=src/admitperf/experiments/my-run/config.yaml
155
+
156
+ # Gated weights only (Llama, Gemma). Forwarded to the container at deploy time
157
+ # as a Modal secret; never written into an image layer.
158
+ # HF_TOKEN=hf_xxx
159
+
160
+ # Alternative to the above, if you would rather keep the token in Modal:
161
+ # modal secret create huggingface HF_TOKEN=hf_xxx
162
+ # ADMITPERF_HF_SECRET=huggingface
163
+
164
+ # Modal auth, if you cannot run `modal setup` interactively (CI). The Modal
165
+ # provider is not on the cluster path — see decision D1.
166
+ # MODAL_TOKEN_ID=ak-...
167
+ # MODAL_TOKEN_SECRET=as-...