admitperf 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- admitperf-0.0.1/.claude/skills/README.md +51 -0
- admitperf-0.0.1/.claude/skills/admitperf-debt/SKILL.md +45 -0
- admitperf-0.0.1/.claude/skills/admitperf-issue/SKILL.md +62 -0
- admitperf-0.0.1/.claude/skills/admitperf-policy/SKILL.md +58 -0
- admitperf-0.0.1/.claude/skills/admitperf-pr/SKILL.md +50 -0
- admitperf-0.0.1/.claude/skills/admitperf-release/SKILL.md +54 -0
- admitperf-0.0.1/.claude/skills/admitperf-review/SKILL.md +47 -0
- admitperf-0.0.1/.claude/skills/admitperf-run/SKILL.md +65 -0
- admitperf-0.0.1/.claude/skills/admitperf-spec/SKILL.md +60 -0
- admitperf-0.0.1/.env.example +167 -0
- admitperf-0.0.1/.github/ISSUE_TEMPLATE/bug_report.yml +42 -0
- admitperf-0.0.1/.github/ISSUE_TEMPLATE/config.yml +14 -0
- admitperf-0.0.1/.github/ISSUE_TEMPLATE/documentation.yml +19 -0
- admitperf-0.0.1/.github/ISSUE_TEMPLATE/policy_port.yml +36 -0
- admitperf-0.0.1/.github/pull_request_template.md +45 -0
- admitperf-0.0.1/.github/workflows/ci.yml +98 -0
- admitperf-0.0.1/.github/workflows/release.yml +165 -0
- admitperf-0.0.1/.gitignore +113 -0
- admitperf-0.0.1/.pre-commit-config.yaml +35 -0
- admitperf-0.0.1/.python-version +1 -0
- admitperf-0.0.1/CONTRIBUTING.md +40 -0
- admitperf-0.0.1/LICENSE +21 -0
- admitperf-0.0.1/Makefile +17 -0
- admitperf-0.0.1/PKG-INFO +380 -0
- admitperf-0.0.1/README.md +337 -0
- admitperf-0.0.1/assets/banner.svg +26 -0
- admitperf-0.0.1/assets/icon.png +0 -0
- admitperf-0.0.1/assets/icon.svg +18 -0
- admitperf-0.0.1/assets/logo.png +0 -0
- admitperf-0.0.1/assets/logo.svg +16 -0
- admitperf-0.0.1/client_app/open-webui.yaml +48 -0
- admitperf-0.0.1/infra/README.md +28 -0
- admitperf-0.0.1/infra/__init__.py +0 -0
- admitperf-0.0.1/infra/config/cluster.yaml +88 -0
- admitperf-0.0.1/infra/docs/ARCHITECTURE.md +214 -0
- admitperf-0.0.1/infra/docs/COMMANDS.md +329 -0
- admitperf-0.0.1/infra/docs/DATAFLOW.md +425 -0
- admitperf-0.0.1/infra/docs/README.md +28 -0
- admitperf-0.0.1/infra/docs/RUNBOOK.md +483 -0
- admitperf-0.0.1/infra/docs/diagrams/e2e.png +0 -0
- admitperf-0.0.1/infra/docs/diagrams/kernels.png +0 -0
- admitperf-0.0.1/infra/docs/diagrams/kv.png +0 -0
- admitperf-0.0.1/infra/docs/diagrams/system.png +0 -0
- admitperf-0.0.1/infra/docs/gateway.md +33 -0
- admitperf-0.0.1/infra/docs/gotchas.md +42 -0
- admitperf-0.0.1/infra/docs/infrastructure.md +166 -0
- admitperf-0.0.1/infra/docs/kernels.md +219 -0
- admitperf-0.0.1/infra/docs/kvbus.md +157 -0
- admitperf-0.0.1/infra/docs/observability.md +102 -0
- admitperf-0.0.1/infra/docs/router.md +153 -0
- admitperf-0.0.1/infra/docs/topology.md +173 -0
- admitperf-0.0.1/infra/gateway/__init__.py +1 -0
- admitperf-0.0.1/infra/gateway/admission.py +88 -0
- admitperf-0.0.1/infra/gateway/harness.py +378 -0
- admitperf-0.0.1/infra/gateway/metrics.py +256 -0
- admitperf-0.0.1/infra/gateway/queue.py +18 -0
- admitperf-0.0.1/infra/gateway/repl.py +672 -0
- admitperf-0.0.1/infra/gateway/serve.py +169 -0
- admitperf-0.0.1/infra/gateway/types.py +147 -0
- admitperf-0.0.1/infra/k8s-config/gateway/httproute.yaml +15 -0
- admitperf-0.0.1/infra/k8s-config/gateway/orch-serve.yaml +70 -0
- admitperf-0.0.1/infra/k8s-config/hami/hami-decode.yaml +25 -0
- admitperf-0.0.1/infra/k8s-config/hami/hami-lambda.yaml +202 -0
- admitperf-0.0.1/infra/k8s-config/hami/hami-prefill.yaml +25 -0
- admitperf-0.0.1/infra/k8s-config/mooncake/mooncake.yaml +21 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/cluster.json +483 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/errors.json +518 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/gateway.json +298 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/hami.json +251 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/keda.json +288 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/mooncake.json +303 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/overview.json +480 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/replicas.json +504 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/router.json +519 -0
- admitperf-0.0.1/infra/k8s-config/observability/dashboards/vllm.json +495 -0
- admitperf-0.0.1/infra/k8s-config/observability/dcgm-exporter.yaml +58 -0
- admitperf-0.0.1/infra/k8s-config/observability/grafana-values.yaml +21 -0
- admitperf-0.0.1/infra/k8s-config/observability/prometheus-values.yaml +27 -0
- admitperf-0.0.1/infra/k8s-config/router/inferencepool.yaml +31 -0
- admitperf-0.0.1/infra/k8s-config/router/keda-decode.yaml +16 -0
- admitperf-0.0.1/infra/k8s-config/router/keda-prefill.yaml +16 -0
- admitperf-0.0.1/infra/kernels/__init__.py +0 -0
- admitperf-0.0.1/infra/kernels/engineering.py +116 -0
- admitperf-0.0.1/infra/observability/__init__.py +0 -0
- admitperf-0.0.1/infra/observability/dashboards.py +541 -0
- admitperf-0.0.1/infra/providers/__init__.py +0 -0
- admitperf-0.0.1/infra/providers/modal_app.py +177 -0
- admitperf-0.0.1/infra/providers/modal_provider.py +175 -0
- admitperf-0.0.1/infra/render.py +777 -0
- admitperf-0.0.1/infra/requirements.txt +1 -0
- admitperf-0.0.1/infra/router/__init__.py +1 -0
- admitperf-0.0.1/infra/router/kv_eviction.py +163 -0
- admitperf-0.0.1/infra/router/kvbus.py +97 -0
- admitperf-0.0.1/infra/router/mooncake.py +79 -0
- admitperf-0.0.1/infra/router/nccl.py +5 -0
- admitperf-0.0.1/infra/router/nixl.py +5 -0
- admitperf-0.0.1/infra/router/overflow.py +266 -0
- admitperf-0.0.1/infra/router/planner.py +35 -0
- admitperf-0.0.1/infra/router/pools.py +301 -0
- admitperf-0.0.1/infra/router/router.py +290 -0
- admitperf-0.0.1/infra/router/trace.py +29 -0
- admitperf-0.0.1/infra/router/warmup.py +116 -0
- admitperf-0.0.1/infra/setup/_lambda_only.sh +13 -0
- admitperf-0.0.1/infra/setup/day2_observability.sh +67 -0
- admitperf-0.0.1/infra/setup/lambda_apply_slices.sh +47 -0
- admitperf-0.0.1/infra/setup/lambda_cluster.sh +9 -0
- admitperf-0.0.1/infra/setup/lambda_k3s_hami.sh +92 -0
- admitperf-0.0.1/infra/setup/lambda_setup.sh +27 -0
- admitperf-0.0.1/infra/setup/lambda_sliced.sh +79 -0
- admitperf-0.0.1/infra/setup/lambda_vllm.sh +30 -0
- admitperf-0.0.1/infra/setup/overflow_smoke.py +97 -0
- admitperf-0.0.1/infra/setup/smoke_lambda.sh +21 -0
- admitperf-0.0.1/infra/setup/smoke_overflow.sh +16 -0
- admitperf-0.0.1/infra/setup/smoke_sliced.sh +70 -0
- admitperf-0.0.1/infra/setup/ssh.sh +26 -0
- admitperf-0.0.1/infra/setup/sync_to_lambda.sh +30 -0
- admitperf-0.0.1/infra/traces/.gitkeep +0 -0
- admitperf-0.0.1/infra/traces/requests.jsonl +45 -0
- admitperf-0.0.1/infra/traces/shared_prefix.jsonl +50 -0
- admitperf-0.0.1/infra/traces/unique_prefix.jsonl +80 -0
- admitperf-0.0.1/pyproject.toml +115 -0
- admitperf-0.0.1/src/admitperf/__init__.py +43 -0
- admitperf-0.0.1/src/admitperf/cli.py +282 -0
- admitperf-0.0.1/src/admitperf/comparison.py +326 -0
- admitperf-0.0.1/src/admitperf/core/__init__.py +29 -0
- admitperf-0.0.1/src/admitperf/core/decision.py +47 -0
- admitperf-0.0.1/src/admitperf/core/log.py +101 -0
- admitperf-0.0.1/src/admitperf/core/policy.py +231 -0
- admitperf-0.0.1/src/admitperf/core/reasons.py +33 -0
- admitperf-0.0.1/src/admitperf/core/signal.py +94 -0
- admitperf-0.0.1/src/admitperf/core/signals.py +87 -0
- admitperf-0.0.1/src/admitperf/core/verdict.py +18 -0
- admitperf-0.0.1/src/admitperf/dashboard/.streamlit/config.toml +12 -0
- admitperf-0.0.1/src/admitperf/dashboard/__init__.py +0 -0
- admitperf-0.0.1/src/admitperf/dashboard/app.py +770 -0
- admitperf-0.0.1/src/admitperf/dashboard/assets/icon.png +0 -0
- admitperf-0.0.1/src/admitperf/dashboard/assets/logo.png +0 -0
- admitperf-0.0.1/src/admitperf/dashboard/theme.py +157 -0
- admitperf-0.0.1/src/admitperf/demo.py +274 -0
- admitperf-0.0.1/src/admitperf/discover.py +85 -0
- admitperf-0.0.1/src/admitperf/experiment.py +42 -0
- admitperf-0.0.1/src/admitperf/item.py +22 -0
- admitperf-0.0.1/src/admitperf/items.py +188 -0
- admitperf-0.0.1/src/admitperf/policies/__init__.py +18 -0
- admitperf-0.0.1/src/admitperf/policies/dual_gate.py +36 -0
- admitperf-0.0.1/src/admitperf/policies/kv_threshold.py +35 -0
- admitperf-0.0.1/src/admitperf/policies/no_admission.py +27 -0
- admitperf-0.0.1/src/admitperf/policies/queue_depth.py +35 -0
- admitperf-0.0.1/src/admitperf/policy_runs.py +31 -0
- admitperf-0.0.1/src/admitperf/report.py +340 -0
- admitperf-0.0.1/src/admitperf/status.py +19 -0
- admitperf-0.0.1/src/admitperf/terminology.py +207 -0
- admitperf-0.0.1/src/admitperf/watch.py +107 -0
- admitperf-0.0.1/tests/__init__.py +0 -0
- admitperf-0.0.1/tests/test_cli.py +166 -0
- admitperf-0.0.1/tests/test_comparison.py +310 -0
- admitperf-0.0.1/tests/test_dashboard.py +188 -0
- admitperf-0.0.1/tests/test_decision.py +60 -0
- admitperf-0.0.1/tests/test_example.py +71 -0
- admitperf-0.0.1/tests/test_layering.py +186 -0
- admitperf-0.0.1/tests/test_log.py +67 -0
- admitperf-0.0.1/tests/test_policies.py +50 -0
- admitperf-0.0.1/tests/test_policy.py +222 -0
- admitperf-0.0.1/tests/test_render_topology.py +569 -0
- admitperf-0.0.1/tests/test_report.py +165 -0
- admitperf-0.0.1/tests/test_signal.py +113 -0
- admitperf-0.0.1/tests/test_watch.py +99 -0
- admitperf-0.0.1/uv.lock +3426 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# Project skills
|
|
2
|
+
|
|
3
|
+
Eight skills that encode how this project is worked on. They load automatically for anyone
|
|
4
|
+
using Claude Code in this repository — no setup, because they are versioned alongside the
|
|
5
|
+
code they describe.
|
|
6
|
+
|
|
7
|
+
| Skill | Use it when |
|
|
8
|
+
|---|---|
|
|
9
|
+
| [`admitperf-run`](admitperf-run/SKILL.md) | Running an experiment, or reading what one produced |
|
|
10
|
+
| [`admitperf-policy`](admitperf-policy/SKILL.md) | Porting a published admission policy |
|
|
11
|
+
| [`admitperf-review`](admitperf-review/SKILL.md) | Reviewing a change, adversarially |
|
|
12
|
+
| [`admitperf-pr`](admitperf-pr/SKILL.md) | Opening a pull request |
|
|
13
|
+
| [`admitperf-issue`](admitperf-issue/SKILL.md) | Filing a bug, a policy port, or a docs problem |
|
|
14
|
+
| [`admitperf-debt`](admitperf-debt/SKILL.md) | Auditing claims against the bundles behind them |
|
|
15
|
+
| [`admitperf-spec`](admitperf-spec/SKILL.md) | Planning docs, the board, and what is actually done |
|
|
16
|
+
| [`admitperf-release`](admitperf-release/SKILL.md) | Cutting a release |
|
|
17
|
+
|
|
18
|
+
## Why these exist
|
|
19
|
+
|
|
20
|
+
They are not style guides. Each encodes a mistake this project actually made.
|
|
21
|
+
|
|
22
|
+
- **`admitperf-run`** exists because of `experiments/half-capacity-headroom`:
|
|
23
|
+
`waiting_requests` identically zero across all three repeats, `running_requests` peaking
|
|
24
|
+
at 6-8 against a cap of 10. The engine was never saturated, so every signal was flat —
|
|
25
|
+
not just the one under test. That run was treated as banked evidence for weeks. It
|
|
26
|
+
isolates nothing.
|
|
27
|
+
- **`admitperf-policy`** front-loads the Class A / Class B question, because a Class B
|
|
28
|
+
policy cannot sit behind this API at all and discovering that mid-port wastes the port.
|
|
29
|
+
- **`admitperf-debt`** exists because a figure was published that its bundles did not
|
|
30
|
+
support, and because `docs/status.md` has disagreed with what the software can actually
|
|
31
|
+
demonstrate.
|
|
32
|
+
- **`admitperf-pr`** says to check the gate's *exit codes*: piping `ruff` into `tail`
|
|
33
|
+
returns tail's status, and a formatting failure can be committed while the gate reports
|
|
34
|
+
clean. It also says to run the gate in a clean venv, because CI has already failed on an
|
|
35
|
+
extra the working venv had accumulated by hand.
|
|
36
|
+
- **`admitperf-issue`** asks for the signal range before anything else, because "the policy
|
|
37
|
+
admitted everything" and "the policy never saw pressure" look identical from outside.
|
|
38
|
+
- **`admitperf-spec`** exists because the plan and the board have disagreed: a direction was
|
|
39
|
+
recorded as banked while the run behind it isolated nothing.
|
|
40
|
+
|
|
41
|
+
The thread running through all eight: **a number is not a result until you know the signal
|
|
42
|
+
moved.** A policy that never saw its threshold did not perform badly — it never ran, and it
|
|
43
|
+
reports the same numbers as no policy at all. Telling those apart is the entire point of
|
|
44
|
+
this repository, so it is the first thing every skill checks.
|
|
45
|
+
|
|
46
|
+
## Adding one
|
|
47
|
+
|
|
48
|
+
A skill earns its place by encoding something that went wrong, or a rule a newcomer would
|
|
49
|
+
otherwise learn by breaking. Keep it short, make it concrete, and cite the real incident —
|
|
50
|
+
the specifics are what make a rule memorable and what let a reader judge whether it still
|
|
51
|
+
applies.
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-debt
|
|
3
|
+
description: >-
|
|
4
|
+
Audit what AdmitPerf claims against what its bundles prove — numbers without a bundle,
|
|
5
|
+
policies never exercised, docs describing code that moved. Use when asked about technical
|
|
6
|
+
debt, what is unproven or over-claimed, or before a release or a paper submission.
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# Auditing claims against proof
|
|
10
|
+
|
|
11
|
+
## What this looks for
|
|
12
|
+
|
|
13
|
+
**A claim the software cannot demonstrate in one command.** Reviewers read the repo; a gap
|
|
14
|
+
between `docs/status.md` and what runs is the risk that matters here.
|
|
15
|
+
|
|
16
|
+
## The audit
|
|
17
|
+
|
|
18
|
+
1. **Every number in a doc names a bundle.** Walk the docs, list the figures, and for each
|
|
19
|
+
one find the run it came from. A number with no bundle is either unverified or lost, and
|
|
20
|
+
both are reported the same way: as unproven.
|
|
21
|
+
2. **Every cited bundle names its commit.** Traceable only to a version string that never
|
|
22
|
+
changes is not traceable.
|
|
23
|
+
3. **Every registered policy has a run where its signal crossed its threshold.** A policy
|
|
24
|
+
never exercised under pressure is a policy with no evidence behind it, however many
|
|
25
|
+
tests it passes.
|
|
26
|
+
4. **Every claimed capability has a command that shows it.** If showing it needs three
|
|
27
|
+
steps and local knowledge, it is not demonstrable.
|
|
28
|
+
5. **Docs against the tree.** This repo has moved its package layout more than once;
|
|
29
|
+
anything citing a path is suspect until checked.
|
|
30
|
+
|
|
31
|
+
## Known debts, so they are not rediscovered
|
|
32
|
+
|
|
33
|
+
- **A published figure its bundles did not support.** The reason this audit exists.
|
|
34
|
+
- **`half-capacity-headroom` isolates nothing.** Engine never saturated, every signal flat.
|
|
35
|
+
Any claim resting on it is unsupported.
|
|
36
|
+
- **KV pressure and queue depth have never been compared in the same deployment.**
|
|
37
|
+
`which-policy-when` is specified for it and has not run.
|
|
38
|
+
- **`infra/` is vendored and unlinted.** Its lint debt is upstream's; ours is that we have
|
|
39
|
+
not decided what happens to the gateway inside it.
|
|
40
|
+
|
|
41
|
+
## How to report
|
|
42
|
+
|
|
43
|
+
Per item: the claim, where it is made, what would prove it, and whether that evidence
|
|
44
|
+
exists. **An item that cannot be proven is reported as not-shown, never quietly dropped** —
|
|
45
|
+
dropping it is how an over-claim survives an audit that was meant to catch it.
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-issue
|
|
3
|
+
description: >-
|
|
4
|
+
File an AdmitPerf issue that is actionable on arrival — which template, the evidence to
|
|
5
|
+
gather before writing it, and the labels. Use when asked to file an issue, open a bug,
|
|
6
|
+
track something, or when a problem is found that will not be fixed in the current change.
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# Filing an issue
|
|
10
|
+
|
|
11
|
+
## Gather the evidence first
|
|
12
|
+
|
|
13
|
+
For anything about a run, three things answer most of what would otherwise be asked, and
|
|
14
|
+
each round trip of asking costs a day:
|
|
15
|
+
|
|
16
|
+
1. **`manifest.json`** — what ran, against what, at which commit.
|
|
17
|
+
2. **The signal range from `summary.json`.** Usually *the* answer. A policy whose signal
|
|
18
|
+
never approached its threshold did not misbehave; it never fired, and it reports the same
|
|
19
|
+
numbers as no policy at all.
|
|
20
|
+
3. **The command**, and the config it ran against.
|
|
21
|
+
|
|
22
|
+
If the signal range explains the behaviour, say so in the issue rather than filing it as a
|
|
23
|
+
bug. "The policy admitted everything" and "the policy never saw pressure" look identical
|
|
24
|
+
from outside and are completely different problems.
|
|
25
|
+
|
|
26
|
+
## Which template
|
|
27
|
+
|
|
28
|
+
| Template | For |
|
|
29
|
+
|---|---|
|
|
30
|
+
| **Bug report** | Something behaves differently from what it says it does |
|
|
31
|
+
| **Policy port** | A published policy that should be available here |
|
|
32
|
+
| **Documentation** | Wrong, missing, or misleading — misleading is the worst, the reader leaves confident |
|
|
33
|
+
|
|
34
|
+
Blank issues are off on purpose. The templates ask for what we would otherwise have to
|
|
35
|
+
chase.
|
|
36
|
+
|
|
37
|
+
## Labels
|
|
38
|
+
|
|
39
|
+
`policy` · `infra` · `measurement` · `blocked-on-gpu` · `bug` · `documentation` ·
|
|
40
|
+
`enhancement`
|
|
41
|
+
|
|
42
|
+
**`blocked-on-gpu` matters most.** It separates what can be done on a laptop today from what
|
|
43
|
+
waits on a rented cluster. Without it the board looks full while everything on it is
|
|
44
|
+
unstartable.
|
|
45
|
+
|
|
46
|
+
## Writing it
|
|
47
|
+
|
|
48
|
+
**One issue, one thing.** An issue that contains three problems gets closed when one is
|
|
49
|
+
fixed.
|
|
50
|
+
|
|
51
|
+
**Say what you expected, not just what happened.** *"It admitted everything"* is a
|
|
52
|
+
description. *"It admitted everything, and I expected refusals once KV passed 0.9"* is a
|
|
53
|
+
specification, and someone can tell whether it is fixed.
|
|
54
|
+
|
|
55
|
+
**Attach it to the milestone** if it belongs in the current one. An issue with no milestone
|
|
56
|
+
is invisible to anyone planning.
|
|
57
|
+
|
|
58
|
+
## Before filing
|
|
59
|
+
|
|
60
|
+
A problem found mid-change that you are about to fix does not need an issue. A problem found
|
|
61
|
+
mid-change that you are **not** going to fix does — otherwise it survives as a comment in a
|
|
62
|
+
diff nobody reads again.
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-policy
|
|
3
|
+
description: >-
|
|
4
|
+
Port a published admission policy into AdmitPerf — the Class A/B test that comes first,
|
|
5
|
+
the frozen interface, declaring what signal it needs, and recording deviations from the
|
|
6
|
+
paper. Use when adding a policy, porting one from a paper, or reviewing a port.
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# Porting a policy
|
|
10
|
+
|
|
11
|
+
## Class A or Class B — decide before writing anything
|
|
12
|
+
|
|
13
|
+
**Class A** is a pure function of `(request, observable fleet state) → decision`, running in
|
|
14
|
+
front of an unmodified engine and reading its metrics. That is all this API accepts.
|
|
15
|
+
|
|
16
|
+
**Class B** changes how batches are *formed inside* the engine. Porting one means
|
|
17
|
+
maintaining a fork of thousands of lines of scheduler internals. FairBatching, FastServe
|
|
18
|
+
and CONCUR are Class B and do not fit here — saying precisely why is itself the
|
|
19
|
+
Portability axis the survey needs.
|
|
20
|
+
|
|
21
|
+
Getting this wrong costs the whole port, and it is answerable from the paper's abstract.
|
|
22
|
+
|
|
23
|
+
## The interface is frozen
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
def decide(self, req: Request, state: SystemState) -> Decision
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
`ADMIT`, `DEFER` (with `retry_after_ms`), or `REJECT` (with a reason). Deterministic: the
|
|
30
|
+
same inputs and internal state must give the same decision, or experiments stop being
|
|
31
|
+
reproducible.
|
|
32
|
+
|
|
33
|
+
Anything the policy needs that `SystemState` lacks arrives as an **optional** field. Never
|
|
34
|
+
change `Decision` or `decide()`.
|
|
35
|
+
|
|
36
|
+
## Declare what it reads
|
|
37
|
+
|
|
38
|
+
A policy states its signal in the same vocabulary as the survey's axes — unit, setting,
|
|
39
|
+
objective, signal, portability. Two things fall out for free:
|
|
40
|
+
|
|
41
|
+
- The harness can **refuse to run** a policy against an engine that cannot supply its
|
|
42
|
+
signal, before any provisioning cost.
|
|
43
|
+
- The report can show the **observed range** of that signal, which is what tells a reader
|
|
44
|
+
the policy was live.
|
|
45
|
+
|
|
46
|
+
## Write the deviation list while porting, not after
|
|
47
|
+
|
|
48
|
+
Anything that cannot be reproduced faithfully: a simulator-only assumption, a metric the
|
|
49
|
+
engine does not expose, a constant the paper never states. Written during the port it is a
|
|
50
|
+
record; written afterwards it is a reconstruction, and the details that mattered are gone.
|
|
51
|
+
|
|
52
|
+
## Before calling it done
|
|
53
|
+
|
|
54
|
+
- A test that the policy refuses *something* under pressure. A policy that admits
|
|
55
|
+
everything passes most tests.
|
|
56
|
+
- A run where its signal actually crossed its threshold. Otherwise the port is unexercised
|
|
57
|
+
and you have no evidence it works.
|
|
58
|
+
- Its entry in `docs/policies.md`, with the deviation list.
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-pr
|
|
3
|
+
description: >-
|
|
4
|
+
Open a pull request against AdmitPerf — the gate to run, what the body must explain, and
|
|
5
|
+
the branch rules. Use when asked to open or raise a PR, or when finishing a change that
|
|
6
|
+
is ready to land.
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# Opening a PR
|
|
10
|
+
|
|
11
|
+
## Run the gate, and check exit codes
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
ruff check . && ruff format --check . && pytest -q
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
**Check the exit code, not the output.** Piping either into `tail` returns tail's status,
|
|
18
|
+
and a failure then looks clean. This is how a formatting failure gets committed.
|
|
19
|
+
|
|
20
|
+
Run it in a **clean environment**, not your working venv. A narrower install than CI uses
|
|
21
|
+
collects tests and then fails to import them — which has already shipped here once, when
|
|
22
|
+
`.[dev]` passed locally only because the venv had accumulated `pandas` and `streamlit` by
|
|
23
|
+
hand. Use `pip install -e '.[all]'` in a fresh venv, which is what CI does.
|
|
24
|
+
|
|
25
|
+
## Branch and title
|
|
26
|
+
|
|
27
|
+
Branch from `main`. Never commit to it directly.
|
|
28
|
+
|
|
29
|
+
Conventional-commit title, because release notes are assembled from these:
|
|
30
|
+
`feat(policies):` · `fix(bench):` · `refactor:` · `docs:` · `ci:`
|
|
31
|
+
|
|
32
|
+
## The body
|
|
33
|
+
|
|
34
|
+
**What changed, and why.** The *why* is the part worth writing — a change that explains its
|
|
35
|
+
reasoning is understandable in six months. If you fixed something, describe the broken
|
|
36
|
+
behaviour from outside: the symptom someone would have reported.
|
|
37
|
+
|
|
38
|
+
**How it was verified.** Not "tests pass". What you actually drove — a run you watched, a
|
|
39
|
+
bundle you read, the decision log you checked. This project has shipped a figure its
|
|
40
|
+
bundles did not support and a run where every signal was flat. Both passed their tests.
|
|
41
|
+
|
|
42
|
+
**Anything deliberately left undone.** Known gaps and follow-ups. A limitation stated here
|
|
43
|
+
is a known limitation; the same limitation unstated is a bug waiting to be rediscovered.
|
|
44
|
+
|
|
45
|
+
## Before marking ready
|
|
46
|
+
|
|
47
|
+
- Any number in the diff names the bundle it came from.
|
|
48
|
+
- A new policy is Class A and declares its signal.
|
|
49
|
+
- Version untouched — releases are cut by tag, never by editing a version by hand.
|
|
50
|
+
- Linked to its issue and milestone, so the board reflects reality.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-release
|
|
3
|
+
description: >-
|
|
4
|
+
Cut an AdmitPerf release — the tag-driven flow, the TestPyPI round trip, the trusted
|
|
5
|
+
publisher step only a human can do, and what cannot be undone. Use when asked to cut a
|
|
6
|
+
release, tag a version, or prepare one.
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# Cutting a release
|
|
10
|
+
|
|
11
|
+
## What cannot be undone
|
|
12
|
+
|
|
13
|
+
**A published version number can never be reused.** Not on PyPI, not on TestPyPI. If `0.1.0`
|
|
14
|
+
publishes broken, the fix is `0.1.1` and `0.1.0` stays broken forever. Everything below
|
|
15
|
+
exists because of that.
|
|
16
|
+
|
|
17
|
+
## Before anything
|
|
18
|
+
|
|
19
|
+
**Trusted Publishing must be configured first, by a human, in a browser.** The workflow
|
|
20
|
+
stores no token; it mints one per run over OIDC. Register at
|
|
21
|
+
`pypi.org/manage/account/publishing/` with owner `harshuljain13`, repo `AdmitPerf`,
|
|
22
|
+
workflow `release.yml`, environment `pypi` — and the same on TestPyPI.
|
|
23
|
+
|
|
24
|
+
A first-time package that is not registered on **both** indexes half-publishes: TestPyPI
|
|
25
|
+
succeeds, PyPI fails, and that version is now burned.
|
|
26
|
+
|
|
27
|
+
## The flow
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
# bump the version in pyproject.toml and admitperf/__init__.py — they must match
|
|
31
|
+
git commit -am "release: 0.1.0"
|
|
32
|
+
git tag v0.1.0
|
|
33
|
+
git push origin v0.1.0
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
A tag publishes. A branch never does.
|
|
37
|
+
|
|
38
|
+
The workflow then goes **TestPyPI → install back out of TestPyPI → smoke-test → PyPI**. That
|
|
39
|
+
round trip is the only thing that catches a missing dependency or a bad pin, because it is a
|
|
40
|
+
real index resolve. A local path install cannot fail that way, which is why local success
|
|
41
|
+
proves nothing here.
|
|
42
|
+
|
|
43
|
+
## Rehearse first
|
|
44
|
+
|
|
45
|
+
`workflow_dispatch` with `target: testpypi` runs the whole thing without cutting a tag. Do
|
|
46
|
+
this for any release that changes packaging, adds a dependency, or moves a module.
|
|
47
|
+
|
|
48
|
+
## Before tagging
|
|
49
|
+
|
|
50
|
+
- CI green on `main`, including the wheel job that installs into a clean venv and imports
|
|
51
|
+
the public API. That job is what catches a module missing from the package manifest; the
|
|
52
|
+
editable install cannot, because it sees the working tree.
|
|
53
|
+
- `docs/status.md` matches what ships.
|
|
54
|
+
- No number in the docs lacks a bundle. A release is a bad time to discover one.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-review
|
|
3
|
+
description: >-
|
|
4
|
+
Adversarial review for AdmitPerf changes, tuned to the defects this repository actually
|
|
5
|
+
produces — results from runs where the signal never moved, and claims the bundles do not
|
|
6
|
+
support. Use when reviewing a PR or a diff, or before marking anything done.
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# Reviewing a change
|
|
10
|
+
|
|
11
|
+
## The defect this repo produces
|
|
12
|
+
|
|
13
|
+
Not broken code. **A number that looks like a result and is not one.**
|
|
14
|
+
|
|
15
|
+
The shape: a run completes, a table is produced, the table is cited — and the signal the
|
|
16
|
+
policy reads never left its floor, so the policy never acted and the table says nothing
|
|
17
|
+
about it. This has happened: `half-capacity-headroom`, `waiting_requests` identically zero
|
|
18
|
+
across three repeats, treated as banked evidence.
|
|
19
|
+
|
|
20
|
+
So on any change that produces or cites a number, ask first: **could this number have come
|
|
21
|
+
from a run where nothing was saturated?** If yes, that is the review finding.
|
|
22
|
+
|
|
23
|
+
## Checks, in order of what they catch
|
|
24
|
+
|
|
25
|
+
1. **Does a cited number name its bundle, and does that bundle name its commit?** A number
|
|
26
|
+
traceable only to a version string that never changes is not traceable.
|
|
27
|
+
2. **Is the comparison within one deployment?** Pooling across hardware reports the
|
|
28
|
+
machine, not the policy.
|
|
29
|
+
3. **Repeats and spread?** A single run has no error bar. A gap inside noise is not a gap.
|
|
30
|
+
4. **Does a new policy actually refuse anything** in its tests? One that admits everything
|
|
31
|
+
passes most of them.
|
|
32
|
+
5. **Class A?** A policy that needs engine changes does not fit behind this API, and a
|
|
33
|
+
review is the last cheap place to notice.
|
|
34
|
+
6. **Does `docs/status.md` still match what the code can demonstrate in one command?** The
|
|
35
|
+
gap between claim and proof is the risk that matters for this project.
|
|
36
|
+
|
|
37
|
+
## On vendored code
|
|
38
|
+
|
|
39
|
+
`infra/` is the vendored serving cluster, excluded from ruff. Do not reformat it — that
|
|
40
|
+
makes every future diff against upstream unreadable. Changes there should be minimal and
|
|
41
|
+
should say why they could not be made upstream.
|
|
42
|
+
|
|
43
|
+
## What a good review comment looks like
|
|
44
|
+
|
|
45
|
+
Name the failing input, not the style. *"If `max_num_seqs` is below what KV allows, this
|
|
46
|
+
measures the scheduler rather than the cache"* is reviewable. *"Consider refactoring"* is
|
|
47
|
+
not.
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-run
|
|
3
|
+
description: >-
|
|
4
|
+
Run an AdmitPerf experiment and read what it produced — the saturation check that comes
|
|
5
|
+
before any comparison, what a bundle contains, and the one question that decides whether
|
|
6
|
+
a result means anything. Use when running a benchmark, reading results, comparing
|
|
7
|
+
policies, or interpreting a report.
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# Running an experiment, and reading one
|
|
11
|
+
|
|
12
|
+
## Ask this first, every time
|
|
13
|
+
|
|
14
|
+
**Did the signal move?**
|
|
15
|
+
|
|
16
|
+
A policy that never reached its threshold did not perform badly. It never ran, and it
|
|
17
|
+
reports the same numbers as having no policy at all. Until you know the signal's observed
|
|
18
|
+
range, a results table tells you nothing.
|
|
19
|
+
|
|
20
|
+
This is not hypothetical. `experiments/half-capacity-headroom` recorded `waiting_requests`
|
|
21
|
+
identically **zero** across all three repeats, with `running_requests` peaking at 6-8
|
|
22
|
+
against a cap of 10. The engine was never saturated, so *every* signal was flat — not just
|
|
23
|
+
the one under test. The run was treated as banked evidence for weeks. It isolates nothing.
|
|
24
|
+
|
|
25
|
+
## Before comparing anything
|
|
26
|
+
|
|
27
|
+
1. **Establish capacity.** `bench capacity` finds what the deployment can actually serve.
|
|
28
|
+
Comparing policies below that ceiling compares nothing.
|
|
29
|
+
2. **Check the load is capacity-relative.** "15 rps" is meaningless alone; "15 rps at 46%
|
|
30
|
+
of ceiling" is a number someone else can reproduce.
|
|
31
|
+
3. **Confirm the engine queues internally.** Load must back up *inside* vLLM, not in front
|
|
32
|
+
of it. If `max_concurrent_inputs` is below the engine's `max_num_seqs`, requests wait at
|
|
33
|
+
the platform and the admission signal stays flat while latency climbs.
|
|
34
|
+
|
|
35
|
+
## What a bundle contains
|
|
36
|
+
|
|
37
|
+
```
|
|
38
|
+
results/<timestamp>-<experiment>/<policy>-r<n>/
|
|
39
|
+
├── manifest.json what ran, against what, with which settings and commit
|
|
40
|
+
├── summary.json the numbers, each tagged with where it came from
|
|
41
|
+
├── decisions.jsonl every admit/defer/reject, with the state it was decided on
|
|
42
|
+
└── outcomes.jsonl per-request TTFT, inter-token gaps, deadline verdict
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`decisions.jsonl` is the one that matters. A verdict alone cannot be interpreted after the
|
|
46
|
+
fact; the state beside it can.
|
|
47
|
+
|
|
48
|
+
## Reading a result
|
|
49
|
+
|
|
50
|
+
- **Identical numbers across policies** is a finding, not a failure — but only once you can
|
|
51
|
+
say whether it is because the policies agreed or because none of them fired.
|
|
52
|
+
- **Never pool runs across deployments.** A table mixing an A10G row with an A100 row
|
|
53
|
+
reports the machine, not the policy. `same-policies-across-gpus` exists to enforce this.
|
|
54
|
+
- **Repeats and spread, or no claim.** A single run has no error bar, and a difference
|
|
55
|
+
inside run-to-run noise is not a difference.
|
|
56
|
+
- **Reject and defer are outcomes**, not losses. Goodput-under-admission counts them; raw
|
|
57
|
+
throughput hides them.
|
|
58
|
+
|
|
59
|
+
## Do not
|
|
60
|
+
|
|
61
|
+
- Quote latency from a freshly started worker. The engine reports ready minutes before it
|
|
62
|
+
serves, while it downloads weights and compiles graphs.
|
|
63
|
+
- Report a win without naming the regime it holds in. The question this project exists to
|
|
64
|
+
answer is *which signal wins under which workload*, and an unqualified winner answers a
|
|
65
|
+
question nobody asked.
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: admitperf-spec
|
|
3
|
+
description: >-
|
|
4
|
+
Work with AdmitPerf's planning docs and the issue board — which one is authoritative,
|
|
5
|
+
reconciling a claim against the tree before calling anything done, and recording
|
|
6
|
+
deviations. Use when asked what is done, about phases or tasks, or when closing work.
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# Planning, and what is actually done
|
|
10
|
+
|
|
11
|
+
## Two places track work. Only one is authoritative.
|
|
12
|
+
|
|
13
|
+
- **GitHub issues and the milestone** — the source of truth for status. What is open, what
|
|
14
|
+
is blocked, what shipped.
|
|
15
|
+
- **`.spec-dev/`** — local planning, deliberately not in git. Requirements, design
|
|
16
|
+
decisions, task breakdowns. It explains *why*; it does not record *whether*.
|
|
17
|
+
|
|
18
|
+
When they disagree, the board wins and the plan gets updated. A plan that claims work is
|
|
19
|
+
done is how this project once recorded a direction as "banked" while the run behind it
|
|
20
|
+
isolated nothing.
|
|
21
|
+
|
|
22
|
+
## `.spec-dev/` contents
|
|
23
|
+
|
|
24
|
+
```
|
|
25
|
+
requirements.md the thesis, the goals, what would make it wrong
|
|
26
|
+
spec.md the active design decisions
|
|
27
|
+
spec-paperB-stale.md superseded, kept because its findings are still true
|
|
28
|
+
tasks.md phases, each one sitting
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
**A stale plan is marked stale, not deleted.** `spec-paperB-stale.md` carries the
|
|
32
|
+
half-capacity finding, which is still the most important thing in that file.
|
|
33
|
+
|
|
34
|
+
## Before calling anything done
|
|
35
|
+
|
|
36
|
+
**Reconcile against the tree, not against the plan.** The question is never "did I write
|
|
37
|
+
this down as finished" but "can I show it in one command".
|
|
38
|
+
|
|
39
|
+
- For a policy: a run where its signal crossed its threshold. Tests passing is not that.
|
|
40
|
+
- For a measurement: the bundle, and the commit that produced it.
|
|
41
|
+
- For a claim in a doc: the command that demonstrates it.
|
|
42
|
+
|
|
43
|
+
Anything that cannot be shown is reported as **not-shown**, never quietly dropped. Dropping
|
|
44
|
+
it is how an over-claim survives.
|
|
45
|
+
|
|
46
|
+
## Recording a deviation
|
|
47
|
+
|
|
48
|
+
When the work diverges from the plan — and it will — write the deviation where the plan is,
|
|
49
|
+
with the reason, at the time. Afterwards it is a reconstruction and the detail that mattered
|
|
50
|
+
is already gone.
|
|
51
|
+
|
|
52
|
+
The deviations worth recording are the ones where something turned out to be impossible or
|
|
53
|
+
unnecessary, not the ones where an estimate was wrong.
|
|
54
|
+
|
|
55
|
+
## Closing
|
|
56
|
+
|
|
57
|
+
An issue closes when its stated "done when" is true, not when the code was written. Several
|
|
58
|
+
issues on this board have a "done when" that requires a live cluster; those cannot close
|
|
59
|
+
from a laptop, and marking them done because the code exists is the failure mode the
|
|
60
|
+
`blocked-on-gpu` label was created to prevent.
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
# Optional. Nothing here is needed to run AdmitPerf against a local engine, or
|
|
2
|
+
# to deploy an ungated model to Modal after `modal setup`.
|
|
3
|
+
#
|
|
4
|
+
# cp .env.example .env
|
|
5
|
+
#
|
|
6
|
+
# `.env` is read by every `admitperf` command, including the ones the dashboard
|
|
7
|
+
# runs for you. Anything already exported wins, so a one-off
|
|
8
|
+
# `HF_TOKEN=... admitperf infra up` still overrides the file.
|
|
9
|
+
#
|
|
10
|
+
# This is a superset of the lab's own env file plus AdmitPerf's. Every name
|
|
11
|
+
# below is read somewhere in this repository — the list was taken from the code
|
|
12
|
+
# rather than copied, and the file that reads each one is named. A variable the
|
|
13
|
+
# code does not read is worse than a missing one: it looks configured.
|
|
14
|
+
#
|
|
15
|
+
# The cluster reads its settings from the ENVIRONMENT, not from
|
|
16
|
+
# infra/config/cluster.yaml. The YAML records intent and generates manifests;
|
|
17
|
+
# these are what a running gateway actually sees. Where the two disagree, the
|
|
18
|
+
# environment wins and nothing warns you. See #15.
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# ===========================================================================
|
|
22
|
+
# GPU box access infra/render.py, infra/setup/*.sh
|
|
23
|
+
# ===========================================================================
|
|
24
|
+
# Addresses and keys live HERE, not in infra/config/cluster.yaml, which is
|
|
25
|
+
# committed. An IP is a fact about today; the topology is not. Re-renting a box is
|
|
26
|
+
# an edit to this file rather than a commit, and a private key path is never
|
|
27
|
+
# committed at all.
|
|
28
|
+
#
|
|
29
|
+
# Resolved per host by name: host `gpu-a` reads LAMBDA_HOST_GPU_A, falling back to
|
|
30
|
+
# LAMBDA only when it is the ONLY host. With two boxes LAMBDA is ambiguous and is
|
|
31
|
+
# refused — applying one address to both would point every URL at one machine
|
|
32
|
+
# while the plan claimed two, which is a one-worker run reported as two.
|
|
33
|
+
#
|
|
34
|
+
# `render.py --plan` works with these unset. `--env` and bring-up refuse, naming
|
|
35
|
+
# the variable they want.
|
|
36
|
+
# LAMBDA=ubuntu@YOUR_LAMBDA_IP # shorthand, single host only
|
|
37
|
+
# LAMBDA_HOST_GPU_A=ubuntu@YOUR_LAMBDA_IP
|
|
38
|
+
# LAMBDA_HOST_GPU_B=ubuntu@YOUR_SECOND_IP
|
|
39
|
+
# LAMBDA_SSH_KEY=$HOME/.ssh/YOUR_LAMBDA_KEY # one key for every host
|
|
40
|
+
# LAMBDA_SSH_KEY_GPU_B=$HOME/.ssh/OTHER_KEY # per-host override
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# ===========================================================================
|
|
44
|
+
# Models and topology infra/gateway/serve.py, router/
|
|
45
|
+
# ===========================================================================
|
|
46
|
+
# READ THIS BEFORE CHANGING THE MODELS. Two traps live here, and both of them
|
|
47
|
+
# produce a run that looks fine and proves nothing.
|
|
48
|
+
#
|
|
49
|
+
# 1. TEXT_MODEL and VISION_MODEL being DIFFERENT, with LAB_SPLIT=capability,
|
|
50
|
+
# makes the two workers non-interchangeable. Placement then becomes a
|
|
51
|
+
# capability lookup — a vision request has exactly one destination — and the
|
|
52
|
+
# admission question disappears, because there is no choice left to make.
|
|
53
|
+
# The project's design (infra/config/cluster.yaml) calls for ONE model on
|
|
54
|
+
# both workers so placement stays a load decision. The defaults below are
|
|
55
|
+
# the lab's and do not match that. Pick one deliberately.
|
|
56
|
+
#
|
|
57
|
+
# 2. A 3B model will not fill KV. These defaults are 3B because the lab ran on
|
|
58
|
+
# one small GPU. Admission never bites, every signal stays flat, and the run
|
|
59
|
+
# reproduces experiments/half-capacity-headroom by construction. If you keep
|
|
60
|
+
# them, you must also cap max_num_seqs by hand and declare the cap artificial.
|
|
61
|
+
#
|
|
62
|
+
# export TEXT_MODEL=Qwen/Qwen2.5-3B-Instruct
|
|
63
|
+
# export VISION_MODEL=Qwen/Qwen2.5-VL-3B-Instruct
|
|
64
|
+
# export LOCAL_MODEL=Qwen/Qwen2.5-3B-Instruct
|
|
65
|
+
|
|
66
|
+
# phase = prefill/decode disaggregation. capability = text worker + vision
|
|
67
|
+
# worker, which is trap 1 above.
|
|
68
|
+
# export LAB_TOPOLOGY=disaggregated
|
|
69
|
+
# export LAB_SPLIT=capability
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ===========================================================================
|
|
73
|
+
# Endpoints infra/gateway/, infra/router/
|
|
74
|
+
# ===========================================================================
|
|
75
|
+
# PREFILL_URLS, DECODE_URLS, LAB_TOPOLOGY, LAB_SPLIT, KV_BACKEND and LOCAL_MODEL
|
|
76
|
+
# are RENDERED by `infra.render --env` from the topology. Setting them by hand
|
|
77
|
+
# overrides the config silently, which is how a two-pool plan ends up serving
|
|
78
|
+
# from one. Prefer re-rendering.
|
|
79
|
+
# ===========================================================================
|
|
80
|
+
# export LOCAL_BASE_URL=http://127.0.0.1:8000/v1
|
|
81
|
+
# export PREFILL_URLS=http://127.0.0.1:8000
|
|
82
|
+
# export DECODE_URLS=http://127.0.0.1:8001 # infra/setup/lambda_sliced.sh
|
|
83
|
+
# export ORCH_URL=http://127.0.0.1:8080/v1
|
|
84
|
+
|
|
85
|
+
# Where the gateway itself binds. infra/gateway/serve.py
|
|
86
|
+
# export SERVE_HOST=0.0.0.0
|
|
87
|
+
# export SERVE_PORT=8080
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# ===========================================================================
|
|
91
|
+
# KV transport infra/router/kvbus.py
|
|
92
|
+
# ===========================================================================
|
|
93
|
+
# mooncake is the only backend that does anything. nccl and nixl are
|
|
94
|
+
# deliberate no-op stubs — four lines that return. Setting either makes hops
|
|
95
|
+
# vanish from Mooncake's /hops while requests still succeed, so ZERO HOPS IN
|
|
96
|
+
# THE DASHBOARD DOES NOT PROVE THE HOP DID NOT HAPPEN. Check this first.
|
|
97
|
+
# export KV_BACKEND=mooncake
|
|
98
|
+
# export MOONCAKE_URL=http://127.0.0.1:50051
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# ===========================================================================
|
|
102
|
+
# Load generation NOT VENDORED YET — see #27
|
|
103
|
+
# ===========================================================================
|
|
104
|
+
# The lab's load generators (locustfile.py, crew_flood.py) live in its app/
|
|
105
|
+
# directory and were not copied into this repository. LOCUST_HOST is listed for
|
|
106
|
+
# completeness; nothing here reads it today.
|
|
107
|
+
# export LOCUST_HOST=http://127.0.0.1:8080
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
# ===========================================================================
|
|
111
|
+
# Overflow — paid capacity for when our own cluster refuses
|
|
112
|
+
# infra/router/overflow.py
|
|
113
|
+
# ===========================================================================
|
|
114
|
+
# OVERFLOW_MAX_REQS is a LIFETIME counter on the gateway process, not a rate
|
|
115
|
+
# limit. After this many overflows the path turns itself off and requests fall
|
|
116
|
+
# back to the local refusal. That is a deliberate spend guard, and it is also a
|
|
117
|
+
# measurement hazard: a run can report "overflow did not help" when what
|
|
118
|
+
# actually happened is that overflow stopped being available. See #26.
|
|
119
|
+
#
|
|
120
|
+
# A 4B model answering in place of a 72B is a relief valve, not an equivalent
|
|
121
|
+
# service. Overflowed responses are not comparable to local ones and must not
|
|
122
|
+
# be pooled with them when reporting quality.
|
|
123
|
+
#
|
|
124
|
+
# Unrecognised names here are ignored silently and the gateway reports
|
|
125
|
+
# "overflow off" — a typo looks like a configuration choice.
|
|
126
|
+
# RENDERED FROM THE CONFIG, except the key. `python -m infra.render --env` emits
|
|
127
|
+
# OVERFLOW_BACKEND, OVERFLOW_BASE_URL, OVERFLOW_MODEL, OVERFLOW_MAX_REQS and
|
|
128
|
+
# OVERFLOW_MAX_TOKENS from infra/config/cluster.yaml, so they are not typed in two
|
|
129
|
+
# places where they would drift. Set them below only to override a rendered run.
|
|
130
|
+
#
|
|
131
|
+
# The key is the one secret in that block, and it belongs here:
|
|
132
|
+
# export OVERFLOW_API_KEY= # unset = overflow off
|
|
133
|
+
#
|
|
134
|
+
# export OVERFLOW_BACKEND=superlinked
|
|
135
|
+
# export OVERFLOW_BASE_URL=https://api.superlinked.com/v1
|
|
136
|
+
# export OVERFLOW_MODEL=Qwen/Qwen3.5-4B
|
|
137
|
+
# export OVERFLOW_MAX_REQS=20
|
|
138
|
+
# export OVERFLOW_MAX_TOKENS=64
|
|
139
|
+
# export OVERFLOW_MAX_REQS=20
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
# ===========================================================================
|
|
143
|
+
# Trace — one row per request infra/gateway/serve.py
|
|
144
|
+
# ===========================================================================
|
|
145
|
+
# The report is generated from this file. It is also where the signal value
|
|
146
|
+
# behind each admission decision will be recorded (#11).
|
|
147
|
+
# export TRACE_PATH=traces/requests.jsonl
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
# ===========================================================================
|
|
151
|
+
# AdmitPerf's own settings
|
|
152
|
+
# ===========================================================================
|
|
153
|
+
# Override the config a command loads. src/admitperf/core/config.py
|
|
154
|
+
# ADMITPERF_CONFIG=src/admitperf/experiments/my-run/config.yaml
|
|
155
|
+
|
|
156
|
+
# Gated weights only (Llama, Gemma). Forwarded to the container at deploy time
|
|
157
|
+
# as a Modal secret; never written into an image layer.
|
|
158
|
+
# HF_TOKEN=hf_xxx
|
|
159
|
+
|
|
160
|
+
# Alternative to the above, if you would rather keep the token in Modal:
|
|
161
|
+
# modal secret create huggingface HF_TOKEN=hf_xxx
|
|
162
|
+
# ADMITPERF_HF_SECRET=huggingface
|
|
163
|
+
|
|
164
|
+
# Modal auth, if you cannot run `modal setup` interactively (CI). The Modal
|
|
165
|
+
# provider is not on the cluster path — see decision D1.
|
|
166
|
+
# MODAL_TOKEN_ID=ak-...
|
|
167
|
+
# MODAL_TOKEN_SECRET=as-...
|