interp-engine 1.0.1__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interp_engine-1.2.0/.gitignore +36 -0
- interp_engine-1.2.0/PKG-INFO +140 -0
- interp_engine-1.2.0/README.md +109 -0
- interp_engine-1.2.0/benchmarks/README.md +349 -0
- interp_engine-1.2.0/benchmarks/bench_spec.py +516 -0
- interp_engine-1.2.0/benchmarks/cells.py +163 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/probe.py +19 -62
- interp_engine-1.2.0/benchmarks/publish.py +391 -0
- interp_engine-1.2.0/benchmarks/report_bench.py +509 -0
- interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__eager.json +175 -0
- interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-cudagraph.json +135 -0
- interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark-cudagraph.json +106 -0
- interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark.json +180 -0
- interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-freeze.json +197 -0
- interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm.json +180 -0
- interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__eager.json → interp_engine-1.2.0/benchmarks/results/gemma-2-2b__eager.json +48 -63
- interp_engine-1.2.0/benchmarks/results/gemma-2-2b__vllm-cudagraph.json +131 -0
- interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__vllm.json → interp_engine-1.2.0/benchmarks/results/gemma-2-2b__vllm-freeze.json +56 -69
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/gemma-2-2b__vllm.json +53 -67
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/llama-3.1-8b__eager.json +48 -63
- interp_engine-1.2.0/benchmarks/results/llama-3.1-8b__vllm-cudagraph.json +131 -0
- interp_engine-1.0.1/benchmarks/results/llama-3.1-8b__eager-sdpa.json → interp_engine-1.2.0/benchmarks/results/llama-3.1-8b__vllm-freeze.json +63 -73
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/llama-3.1-8b__vllm.json +53 -67
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/qwen3-4b__eager.json +49 -64
- interp_engine-1.2.0/benchmarks/results/qwen3-4b__vllm-cudagraph.json +131 -0
- interp_engine-1.2.0/benchmarks/results/qwen3-4b__vllm-freeze.json +177 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/qwen3-4b__vllm.json +53 -67
- interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__eager.json +171 -0
- interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__vllm-cudagraph.json +131 -0
- interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__vllm-freeze.json +177 -0
- interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__vllm.json +176 -0
- interp_engine-1.2.0/benchmarks/results-latest.md +254 -0
- interp_engine-1.2.0/benchmarks/run_all.sh +209 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/run_bench.py +113 -16
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/workloads.py +112 -61
- interp_engine-1.2.0/docs/AGENT_INTEGRATION.md +274 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/docs/ARCHITECTURE_QUIRKS.md +294 -170
- interp_engine-1.2.0/docs/COMPATIBILITY.md +70 -0
- interp_engine-1.2.0/docs/ENGINE_HOOK_MAPPINGS.md +596 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/docs/GRADIENTS.md +28 -3
- interp_engine-1.2.0/docs/INTERNALS.md +121 -0
- interp_engine-1.2.0/docs/PERFORMANCE.md +247 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/docs/PORTING.md +19 -6
- interp_engine-1.2.0/docs/README.md +21 -0
- interp_engine-1.2.0/docs/SUPPORTED_POINTS.md +119 -0
- interp_engine-1.2.0/docs/USAGE.md +398 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/__init__.py +44 -2
- interp_engine-1.2.0/interp_engine/_loop.py +174 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/address.py +2 -2
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/arch.py +181 -36
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/attn_config.py +1 -1
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/autograd_support.py +50 -5
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/capture.py +326 -22
- interp_engine-1.2.0/interp_engine/dispatch.py +204 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/facts.py +855 -33
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/hooks.py +75 -10
- interp_engine-1.2.0/interp_engine/lens.py +224 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/load.py +29 -19
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/mappers.py +185 -8
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/model.py +188 -55
- interp_engine-1.2.0/interp_engine/moe_routing.py +81 -0
- interp_engine-1.2.0/interp_engine/points.py +580 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/protocol.py +55 -2
- interp_engine-1.2.0/interp_engine/residual_basis.py +534 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/select.py +14 -0
- interp_engine-1.2.0/interp_engine/steer.py +608 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/steer_specs.py +31 -7
- interp_engine-1.2.0/interp_engine/sync.py +235 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/tokenize.py +19 -3
- interp_engine-1.2.0/interp_engine/vllm_backend.py +2562 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/__init__.py +262 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/_demux.py +265 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/_hooks.py +215 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/_payload.py +168 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/_tree.py +787 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/attn.py +291 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/capture.py +178 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/freeze.py +1785 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/graphs.py +114 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/lens/__init__.py +56 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/lens/intervene.py +142 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/lens/readout.py +443 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/lens/unembed.py +300 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/mhc.py +715 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/native.py +95 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/requests.py +702 -0
- interp_engine-1.2.0/interp_engine/vllm_capture/steering.py +173 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/vllm_plugin.py +65 -0
- interp_engine-1.2.0/pyproject.toml +245 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/harness.py +22 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/model_expectations.yaml +3 -5
- interp_engine-1.2.0/tests/synthetic_families.py +408 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_autograd_support.py +63 -0
- interp_engine-1.2.0/tests/test_bench_workloads.py +44 -0
- interp_engine-1.2.0/tests/test_capability_refusals.py +186 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_chat_templates.py +33 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_core.py +8 -8
- interp_engine-1.2.0/tests/test_doc_code_fences.py +135 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_eager_autograd.py +54 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_facts.py +122 -1
- interp_engine-1.2.0/tests/test_family_points.py +430 -0
- interp_engine-1.2.0/tests/test_freeze_dsv4_gpu.py +124 -0
- interp_engine-1.2.0/tests/test_freeze_parity_gpu.py +253 -0
- interp_engine-1.2.0/tests/test_freeze_set.py +1009 -0
- interp_engine-1.2.0/tests/test_freeze_warmup.py +138 -0
- interp_engine-1.2.0/tests/test_hook_call_conventions.py +163 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_layer_kinds.py +92 -2
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_load.py +124 -4
- interp_engine-1.2.0/tests/test_logit_transform.py +201 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_mappers.py +143 -6
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_mlp_internals.py +61 -3
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_model_expectations.py +2 -4
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_moe.py +258 -3
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_new_models_gpu.py +97 -5
- interp_engine-1.2.0/tests/test_normalized_hook.py +303 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_packaging.py +30 -1
- interp_engine-1.2.0/tests/test_per_layer_attn_dims.py +434 -0
- interp_engine-1.2.0/tests/test_points_registry.py +330 -0
- interp_engine-1.2.0/tests/test_published_benchmarks.py +166 -0
- interp_engine-1.2.0/tests/test_qk_norm.py +382 -0
- interp_engine-1.2.0/tests/test_release.py +211 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_resid_mid.py +113 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_residual_basis.py +70 -3
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_select.py +33 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_small_models_gpu.py +7 -1
- interp_engine-1.2.0/tests/test_steer_context.py +173 -0
- interp_engine-1.2.0/tests/test_steer_math_parity.py +278 -0
- interp_engine-1.2.0/tests/test_sync_loop.py +145 -0
- interp_engine-1.2.0/tests/test_sync_parity.py +202 -0
- interp_engine-1.2.0/tests/test_unified_free_functions.py +114 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_capture_gpu.py +94 -4
- interp_engine-1.2.0/tests/test_vllm_capture_scales.py +168 -0
- interp_engine-1.2.0/tests/test_vllm_graph_path.py +135 -0
- interp_engine-1.2.0/tests/test_vllm_graphs_on_gpu.py +212 -0
- interp_engine-1.2.0/tests/test_vllm_hook_availability.py +259 -0
- interp_engine-1.2.0/tests/test_vllm_hyper_connections.py +1374 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_kv_isolation.py +20 -10
- interp_engine-1.2.0/tests/test_vllm_new_points.py +746 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_wire_grammar.py +48 -0
- interp_engine-1.2.0/tests/test_worker_lens_capture_readout.py +392 -0
- interp_engine-1.2.0/tests/test_worker_lens_readout.py +200 -0
- interp_engine-1.0.1/.gitignore +0 -33
- interp_engine-1.0.1/AGENTS.md +0 -54
- interp_engine-1.0.1/CLAUDE.md +0 -1
- interp_engine-1.0.1/Makefile +0 -12
- interp_engine-1.0.1/PKG-INFO +0 -210
- interp_engine-1.0.1/README.md +0 -179
- interp_engine-1.0.1/benchmarks/README.md +0 -257
- interp_engine-1.0.1/benchmarks/bench_spec.py +0 -213
- interp_engine-1.0.1/benchmarks/report_bench.py +0 -547
- interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__eager-sdpa.json +0 -187
- interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__vllm-cudagraph.json +0 -190
- interp_engine-1.0.1/benchmarks/results/gemma-2-2b__eager-sdpa.json +0 -187
- interp_engine-1.0.1/benchmarks/results/gemma-2-2b__eager.json +0 -186
- interp_engine-1.0.1/benchmarks/results/gemma-2-2b__vllm-cudagraph.json +0 -190
- interp_engine-1.0.1/benchmarks/results/gemma-3-1b__eager-sdpa.json +0 -187
- interp_engine-1.0.1/benchmarks/results/gemma-3-1b__eager.json +0 -186
- interp_engine-1.0.1/benchmarks/results/gemma-3-1b__vllm-cudagraph.json +0 -190
- interp_engine-1.0.1/benchmarks/results/gemma-3-1b__vllm.json +0 -190
- interp_engine-1.0.1/benchmarks/results/llama-3.1-8b__vllm-cudagraph.json +0 -190
- interp_engine-1.0.1/benchmarks/results/qwen3-4b__eager-sdpa.json +0 -187
- interp_engine-1.0.1/benchmarks/results/qwen3-4b__vllm-cudagraph.json +0 -190
- interp_engine-1.0.1/benchmarks/results-latest.md +0 -329
- interp_engine-1.0.1/benchmarks/run_all.sh +0 -149
- interp_engine-1.0.1/docs/ENGINE_HOOK_MAPPINGS.md +0 -212
- interp_engine-1.0.1/docs/PERFORMANCE.md +0 -108
- interp_engine-1.0.1/interp_engine/lens.py +0 -124
- interp_engine-1.0.1/interp_engine/points.py +0 -390
- interp_engine-1.0.1/interp_engine/residual_basis.py +0 -320
- interp_engine-1.0.1/interp_engine/steer.py +0 -359
- interp_engine-1.0.1/interp_engine/vllm_backend.py +0 -1223
- interp_engine-1.0.1/interp_engine/vllm_capture.py +0 -2038
- interp_engine-1.0.1/pyproject.toml +0 -171
- interp_engine-1.0.1/tests/synthetic_families.py +0 -153
- interp_engine-1.0.1/tests/test_family_points.py +0 -249
- interp_engine-1.0.1/tests/test_per_layer_attn_dims.py +0 -214
- interp_engine-1.0.1/tests/test_points_registry.py +0 -165
- interp_engine-1.0.1/tests/test_qk_norm.py +0 -199
- interp_engine-1.0.1/uv.lock +0 -4516
- {interp_engine-1.0.1 → interp_engine-1.2.0}/LICENSE +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/__init__.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/attn_scores.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/chat_compose.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/chat_conventions.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/cuda_preflight.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/conftest.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_address.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_config_tripwire.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_probs_indexing.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_scores.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_z_gqa.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_capture_addressing.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_chat_compose.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_cuda_preflight.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_gated_attn_out.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_head_contributions.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_multimodal_arch.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_no_chat_template.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_parity_gpt2.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_protocol.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_qkv_layout.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_reasoning_spans.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_sandwich_norms.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_sliding_window_attn.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_unresolved_families.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_only_families.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_plugin.py +0 -0
- {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vocabulary_boundary.py +0 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Secrets. Gated-model tests and the comparison sweep both read HF_TOKEN, so a real token lives
|
|
2
|
+
# here on any box that runs them.
|
|
3
|
+
.env
|
|
4
|
+
.env.*
|
|
5
|
+
|
|
6
|
+
# Python caches
|
|
7
|
+
__pycache__/
|
|
8
|
+
*__pycache__
|
|
9
|
+
*.pyc
|
|
10
|
+
*.pyo
|
|
11
|
+
|
|
12
|
+
# Virtual envs. Each project in this repo has its own: interp-engine at the root, the validator in
|
|
13
|
+
# validator/, and the per-engine venvs the sweep builds (.venv-tlens, .venv-vllm, ...).
|
|
14
|
+
.venv
|
|
15
|
+
.venv/
|
|
16
|
+
.venv-*/
|
|
17
|
+
venv/
|
|
18
|
+
venv.bak/
|
|
19
|
+
|
|
20
|
+
# Build / packaging artifacts
|
|
21
|
+
build/
|
|
22
|
+
dist/
|
|
23
|
+
*.egg-info/
|
|
24
|
+
.eggs/
|
|
25
|
+
|
|
26
|
+
# Tooling caches
|
|
27
|
+
.pytest_cache/
|
|
28
|
+
.ruff_cache/
|
|
29
|
+
.coverage
|
|
30
|
+
.mypy_cache/
|
|
31
|
+
|
|
32
|
+
# Local CI venv (see .github/workflows/engine-tests.yml)
|
|
33
|
+
.venv-ci/
|
|
34
|
+
|
|
35
|
+
plans/
|
|
36
|
+
*.plan.md
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: interp-engine
|
|
3
|
+
Version: 1.2.0
|
|
4
|
+
Summary: A fast, standardized interpretability engine that supports most modern models and architectures. Powers Neuronpedia.
|
|
5
|
+
Project-URL: Homepage, https://github.com/decoderesearch/interp-engine
|
|
6
|
+
Project-URL: Repository, https://github.com/decoderesearch/interp-engine
|
|
7
|
+
Project-URL: Issues, https://github.com/decoderesearch/interp-engine/issues
|
|
8
|
+
License-Expression: Apache-2.0
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Python: <3.14,>=3.11
|
|
11
|
+
Requires-Dist: einops
|
|
12
|
+
Requires-Dist: numpy>=1.24
|
|
13
|
+
Requires-Dist: torch>=1.10
|
|
14
|
+
Requires-Dist: transformers>=4.57.1
|
|
15
|
+
Provides-Extra: awq
|
|
16
|
+
Requires-Dist: accelerate>=1.0; extra == 'awq'
|
|
17
|
+
Requires-Dist: gptqmodel>=5.0; extra == 'awq'
|
|
18
|
+
Provides-Extra: dev
|
|
19
|
+
Requires-Dist: pyright<1.2,>=1.1.411; extra == 'dev'
|
|
20
|
+
Requires-Dist: pytest<9,>=8.3.1; extra == 'dev'
|
|
21
|
+
Requires-Dist: pyyaml>=6; extra == 'dev'
|
|
22
|
+
Requires-Dist: ruff<0.17,>=0.16.2; extra == 'dev'
|
|
23
|
+
Provides-Extra: parity
|
|
24
|
+
Requires-Dist: transformer-lens>=3.0; extra == 'parity'
|
|
25
|
+
Provides-Extra: quant
|
|
26
|
+
Requires-Dist: accelerate>=1.0; extra == 'quant'
|
|
27
|
+
Requires-Dist: kernels<0.17.0,>=0.15.2; extra == 'quant'
|
|
28
|
+
Provides-Extra: vllm
|
|
29
|
+
Requires-Dist: vllm>=0.25.1; (sys_platform == 'linux') and extra == 'vllm'
|
|
30
|
+
Description-Content-Type: text/markdown
|
|
31
|
+
|
|
32
|
+
# interp-engine
|
|
33
|
+
|
|
34
|
+
<p align="center">
|
|
35
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
|
|
36
|
+
</p>
|
|
37
|
+
<p align="center">
|
|
38
|
+
🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
|
|
39
|
+
</p>
|
|
40
|
+
|
|
41
|
+
`interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference (circuit tracing, j-lens, steering, NLAs, etc) and is checked for accuracy against HF Transformers and other engines.
|
|
42
|
+
|
|
43
|
+
<p align="center">
|
|
44
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-freeze on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="620">
|
|
45
|
+
</p>
|
|
46
|
+
<p align="center">
|
|
47
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="480">
|
|
48
|
+
</p>
|
|
49
|
+
|
|
50
|
+
This repo contains:
|
|
51
|
+
|
|
52
|
+
1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
|
|
53
|
+
2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
|
|
54
|
+
|
|
55
|
+
## Installation
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install 'interp-engine[vllm]' # preferred install: includes vLLM support (CUDA required)
|
|
59
|
+
pip install interp-engine # eager backend only
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## Simple Usage
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from interp_engine import Address, load_model, run_with_cache
|
|
66
|
+
|
|
67
|
+
model = load_model("Qwen/Qwen3-8B") # vLLM default, use backend='eager' to override
|
|
68
|
+
point = Address("resid_post", 10) # or string: "resid_post.10"
|
|
69
|
+
cache = run_with_cache(model, model.to_tokens("Hello, world"), [point])
|
|
70
|
+
cache[point] # [batch, pos, ...]
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### AI Agents
|
|
74
|
+
|
|
75
|
+
Add "use interp-engine" to your prompt and let your agent figure it out - everything is fully documented in this repo and open source.
|
|
76
|
+
|
|
77
|
+
## Supported Points ("Addresses")
|
|
78
|
+
|
|
79
|
+
`interp-engine` supports 34 standardized points ("Addresses") across architectures: every one of them on the eager backend, 28 of them on vLLM. Check [interp-engine.org](https://interp-engine.org) for the "cheat sheet", or [SUPPORTED_POINTS.md](docs/SUPPORTED_POINTS.md) for a markdown version with the per-backend detail.
|
|
80
|
+
|
|
81
|
+
## Performance / Speed
|
|
82
|
+
|
|
83
|
+
vLLM gives `interp-engine` high throughput via concurrency, and **graph freeze** adds CUDA-graph replay on top of that _without giving up capture or steering_. Every column below is capture-capable.
|
|
84
|
+
|
|
85
|
+
<!-- THROUGHPUT:START -->
|
|
86
|
+
|
|
87
|
+
<!-- Generated by `python -m benchmarks.report_bench`. Do not edit: rerun the sweep. -->
|
|
88
|
+
|
|
89
|
+
Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
|
|
90
|
+
|
|
91
|
+
One stream (tok/s):
|
|
92
|
+
|
|
93
|
+
| model | eager | vLLM | vLLM + graph freeze |
|
|
94
|
+
| ------------------------ | ----- | ---------- | ------------------- |
|
|
95
|
+
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
|
|
96
|
+
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
|
|
97
|
+
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
|
|
98
|
+
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
|
|
99
|
+
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
|
|
100
|
+
|
|
101
|
+
8 concurrent requests (aggregate tok/s):
|
|
102
|
+
|
|
103
|
+
| model | eager | vLLM | vLLM + graph freeze |
|
|
104
|
+
| ------------------------ | ----- | ----------- | ------------------- |
|
|
105
|
+
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
|
|
106
|
+
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
|
|
107
|
+
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
|
|
108
|
+
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
|
|
109
|
+
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
|
|
110
|
+
|
|
111
|
+
<!-- THROUGHPUT:END -->
|
|
112
|
+
|
|
113
|
+
Every multiplier is against eager on the same workload, and is the ratio of the _unrounded_ figures — tok/s is printed whole at 10 and above, so dividing two cells by hand can differ in the last place. Eager's generation loop is synchronous underneath, so eight requests serialize rather than batch — which is why it earns almost nothing from the second table and why the multipliers there are so much larger. `deepseek-v4-flash-0731` is block-quantized FP8 on both backends rather than bf16, so read its row as two backends serving the same quantized weights; it also keeps a tenth, because at 3 tok/s that digit is worth several percent. [`benchmarks/results-latest.md`](benchmarks/results-latest.md) has every figure at full precision.
|
|
114
|
+
|
|
115
|
+
**Graph freeze** is opt-in, via `freeze_points`. Ordinary hooks cannot survive CUDA graphs — replay never calls the Python forward a hook is attached to — so the tap becomes a preallocated buffer plus a `copy_` that graph capture records and replay re-executes. Steering rides the same wrap, so additive, orthogonal, projection-cap and j-lens steer/ablate/swap all keep working under graphs, position masks included.
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
model = load_model("Qwen/Qwen3-8B", freeze_points="auto") # resid_post at every layer, under graphs
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
The trade is that a frozen engine serves _the set it froze_ rather than any point on request: `"auto"` covers `resid_post` at every layer on a conventional trunk, an explicit list covers whatever you name, and anything outside the set is refused rather than quietly returned empty. Omit `freeze_points` for today's hooked vLLM, which still serves every point. `qwen3.8-27b` is a hybrid trunk, so its freeze row runs prefill eagerly and keeps the decode graphs — freeze pins that mode, because breakable graphs turn `torch.compile` off and vLLM's mixed prefill-decode capture then miscomputes prefill on a gated-delta trunk. On a hyper-connection trunk such as `deepseek-v4-flash-0731`, whose block carries four parallel residual streams, `"auto"` freezes `resid_streams` instead — the whole stack per layer, which is four times the width and so four times the buffer, and is what that row's freeze cell prices. See [PERFORMANCE.md](docs/PERFORMANCE.md) for the full trade-off, and [`benchmarks/results-latest.md`](benchmarks/results-latest.md) for the sweep, including capture, steering and lens latencies.
|
|
122
|
+
|
|
123
|
+
## Correctness
|
|
124
|
+
|
|
125
|
+
We verify correctness in two main ways:
|
|
126
|
+
|
|
127
|
+
1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
|
|
128
|
+
2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
|
|
129
|
+
|
|
130
|
+
## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
|
|
131
|
+
|
|
132
|
+
Of course, software is easy to make these days. But as smart as as AIs are these days, there are benefits to leveraging a pre-written engine:
|
|
133
|
+
|
|
134
|
+
1. **Speed**: Get performance without sacrificing correctness.
|
|
135
|
+
1. **Standardization + Verification**: Eliminate ambiguity when referring to points, plus a full test suite included.
|
|
136
|
+
1. **Faster Dev / Fewer Tokens Used**: You could spend ten million tokens and have your AI write, test, and make production-ready an interpretability engine. Or you could just `pip install interp-engine[vllm]`.
|
|
137
|
+
|
|
138
|
+
## License
|
|
139
|
+
|
|
140
|
+
Apache 2.0
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
# interp-engine
|
|
2
|
+
|
|
3
|
+
<p align="center">
|
|
4
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
|
|
5
|
+
</p>
|
|
6
|
+
<p align="center">
|
|
7
|
+
🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
|
|
8
|
+
</p>
|
|
9
|
+
|
|
10
|
+
`interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference (circuit tracing, j-lens, steering, NLAs, etc) and is checked for accuracy against HF Transformers and other engines.
|
|
11
|
+
|
|
12
|
+
<p align="center">
|
|
13
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-freeze on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="620">
|
|
14
|
+
</p>
|
|
15
|
+
<p align="center">
|
|
16
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="480">
|
|
17
|
+
</p>
|
|
18
|
+
|
|
19
|
+
This repo contains:
|
|
20
|
+
|
|
21
|
+
1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
|
|
22
|
+
2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
|
|
23
|
+
|
|
24
|
+
## Installation
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install 'interp-engine[vllm]' # preferred install: includes vLLM support (CUDA required)
|
|
28
|
+
pip install interp-engine # eager backend only
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Simple Usage
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from interp_engine import Address, load_model, run_with_cache
|
|
35
|
+
|
|
36
|
+
model = load_model("Qwen/Qwen3-8B") # vLLM default, use backend='eager' to override
|
|
37
|
+
point = Address("resid_post", 10) # or string: "resid_post.10"
|
|
38
|
+
cache = run_with_cache(model, model.to_tokens("Hello, world"), [point])
|
|
39
|
+
cache[point] # [batch, pos, ...]
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
### AI Agents
|
|
43
|
+
|
|
44
|
+
Add "use interp-engine" to your prompt and let your agent figure it out - everything is fully documented in this repo and open source.
|
|
45
|
+
|
|
46
|
+
## Supported Points ("Addresses")
|
|
47
|
+
|
|
48
|
+
`interp-engine` supports 34 standardized points ("Addresses") across architectures: every one of them on the eager backend, 28 of them on vLLM. Check [interp-engine.org](https://interp-engine.org) for the "cheat sheet", or [SUPPORTED_POINTS.md](docs/SUPPORTED_POINTS.md) for a markdown version with the per-backend detail.
|
|
49
|
+
|
|
50
|
+
## Performance / Speed
|
|
51
|
+
|
|
52
|
+
vLLM gives `interp-engine` high throughput via concurrency, and **graph freeze** adds CUDA-graph replay on top of that _without giving up capture or steering_. Every column below is capture-capable.
|
|
53
|
+
|
|
54
|
+
<!-- THROUGHPUT:START -->
|
|
55
|
+
|
|
56
|
+
<!-- Generated by `python -m benchmarks.report_bench`. Do not edit: rerun the sweep. -->
|
|
57
|
+
|
|
58
|
+
Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
|
|
59
|
+
|
|
60
|
+
One stream (tok/s):
|
|
61
|
+
|
|
62
|
+
| model | eager | vLLM | vLLM + graph freeze |
|
|
63
|
+
| ------------------------ | ----- | ---------- | ------------------- |
|
|
64
|
+
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
|
|
65
|
+
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
|
|
66
|
+
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
|
|
67
|
+
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
|
|
68
|
+
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
|
|
69
|
+
|
|
70
|
+
8 concurrent requests (aggregate tok/s):
|
|
71
|
+
|
|
72
|
+
| model | eager | vLLM | vLLM + graph freeze |
|
|
73
|
+
| ------------------------ | ----- | ----------- | ------------------- |
|
|
74
|
+
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
|
|
75
|
+
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
|
|
76
|
+
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
|
|
77
|
+
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
|
|
78
|
+
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
|
|
79
|
+
|
|
80
|
+
<!-- THROUGHPUT:END -->
|
|
81
|
+
|
|
82
|
+
Every multiplier is against eager on the same workload, and is the ratio of the _unrounded_ figures — tok/s is printed whole at 10 and above, so dividing two cells by hand can differ in the last place. Eager's generation loop is synchronous underneath, so eight requests serialize rather than batch — which is why it earns almost nothing from the second table and why the multipliers there are so much larger. `deepseek-v4-flash-0731` is block-quantized FP8 on both backends rather than bf16, so read its row as two backends serving the same quantized weights; it also keeps a tenth, because at 3 tok/s that digit is worth several percent. [`benchmarks/results-latest.md`](benchmarks/results-latest.md) has every figure at full precision.
|
|
83
|
+
|
|
84
|
+
**Graph freeze** is opt-in, via `freeze_points`. Ordinary hooks cannot survive CUDA graphs — replay never calls the Python forward a hook is attached to — so the tap becomes a preallocated buffer plus a `copy_` that graph capture records and replay re-executes. Steering rides the same wrap, so additive, orthogonal, projection-cap and j-lens steer/ablate/swap all keep working under graphs, position masks included.
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
model = load_model("Qwen/Qwen3-8B", freeze_points="auto") # resid_post at every layer, under graphs
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The trade is that a frozen engine serves _the set it froze_ rather than any point on request: `"auto"` covers `resid_post` at every layer on a conventional trunk, an explicit list covers whatever you name, and anything outside the set is refused rather than quietly returned empty. Omit `freeze_points` for today's hooked vLLM, which still serves every point. `qwen3.8-27b` is a hybrid trunk, so its freeze row runs prefill eagerly and keeps the decode graphs — freeze pins that mode, because breakable graphs turn `torch.compile` off and vLLM's mixed prefill-decode capture then miscomputes prefill on a gated-delta trunk. On a hyper-connection trunk such as `deepseek-v4-flash-0731`, whose block carries four parallel residual streams, `"auto"` freezes `resid_streams` instead — the whole stack per layer, which is four times the width and so four times the buffer, and is what that row's freeze cell prices. See [PERFORMANCE.md](docs/PERFORMANCE.md) for the full trade-off, and [`benchmarks/results-latest.md`](benchmarks/results-latest.md) for the sweep, including capture, steering and lens latencies.
|
|
91
|
+
|
|
92
|
+
## Correctness
|
|
93
|
+
|
|
94
|
+
We verify correctness in two main ways:
|
|
95
|
+
|
|
96
|
+
1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
|
|
97
|
+
2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
|
|
98
|
+
|
|
99
|
+
## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
|
|
100
|
+
|
|
101
|
+
Of course, software is easy to make these days. But as smart as as AIs are these days, there are benefits to leveraging a pre-written engine:
|
|
102
|
+
|
|
103
|
+
1. **Speed**: Get performance without sacrificing correctness.
|
|
104
|
+
1. **Standardization + Verification**: Eliminate ambiguity when referring to points, plus a full test suite included.
|
|
105
|
+
1. **Faster Dev / Fewer Tokens Used**: You could spend ten million tokens and have your AI write, test, and make production-ready an interpretability engine. Or you could just `pip install interp-engine[vllm]`.
|
|
106
|
+
|
|
107
|
+
## License
|
|
108
|
+
|
|
109
|
+
Apache 2.0
|
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
# Speed benchmarks
|
|
2
|
+
|
|
3
|
+
How fast the two backends are at the things the engine does: generating, capturing activations,
|
|
4
|
+
steering, and the lens read-out. Latest numbers are in [`results-latest.md`](results-latest.md).
|
|
5
|
+
|
|
6
|
+
Every workload is written against the shared `InterpModel` protocol, so the same harness code drives
|
|
7
|
+
both backends and a difference in the numbers is a difference in the backend rather than in the
|
|
8
|
+
measurement.
|
|
9
|
+
|
|
10
|
+
`lens_topk` is the single deliberate exception, and it selects on `hasattr` rather than on a backend
|
|
11
|
+
name: it benchmarks the route each backend's *serving* code actually takes to the same answer, because
|
|
12
|
+
for that one read-out the two are genuinely different code (see below).
|
|
13
|
+
|
|
14
|
+
## Requirements
|
|
15
|
+
|
|
16
|
+
- An interpreter with this checkout installed and, for the vLLM variants, the `vllm` extra:
|
|
17
|
+
`uv sync --extra vllm` in the repo root, or any venv that already has `interp-engine[vllm]`.
|
|
18
|
+
Without vLLM you can still run `--variants eager`.
|
|
19
|
+
- The `quant` extra (`uv sync --extra quant`) for the quantized rows, which today means
|
|
20
|
+
`deepseek-v4-flash-0731`. It brings `accelerate` and `kernels`, both of which transformers *requires* to
|
|
21
|
+
load and run a block-quantized FP8 checkpoint rather than merely preferring: without them the eager
|
|
22
|
+
cell fails at load with `Loading an FP8 quantized model requires accelerate`, or at the first
|
|
23
|
+
forward with `finegrained-fp8 kernel requires the kernels package`. The vLLM variants of the same
|
|
24
|
+
model are unaffected — vLLM has its own kernels — so it is easy to read the missing extra as the
|
|
25
|
+
eager backend not supporting the model.
|
|
26
|
+
- A CUDA GPU. The workload sizes below assume something in the 24-32 GiB class, which covers the
|
|
27
|
+
spec's first three models. `qwen3.8-27b` and `deepseek-v4-flash-0731` do not fit that class — 52 GiB and
|
|
28
|
+
149 GiB of weights — so the full sweep needs a 180 GB card. Nothing has to be passed for that: a
|
|
29
|
+
sweep with no `--models` runs what the card can hold and names what it dropped, and each model
|
|
30
|
+
carries the memory fraction and engine arguments its own weights need
|
|
31
|
+
(`ModelSpec.min_gpu_gib`, `.gpu_memory_utilization`, `.extra_vllm_kwargs`).
|
|
32
|
+
- `HF_TOKEN` for gated repos (Gemma, Llama). Put it in a gitignored `.env` at the repo root:
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
HF_TOKEN=hf_...
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
The runner reads that file when `HF_TOKEN` is not already in the environment. This matters more
|
|
39
|
+
than it looks: without a token the eager backend quietly succeeds from your local HF cache while
|
|
40
|
+
vLLM dies with a 401 several minutes into engine bring-up.
|
|
41
|
+
- Nothing else on the GPU. vLLM reserves `gpu_memory_utilization` of the *whole card* up front, so
|
|
42
|
+
another process holding a few GiB can starve the largest model, and anything sharing the card moves
|
|
43
|
+
the timings.
|
|
44
|
+
|
|
45
|
+
## Running it
|
|
46
|
+
|
|
47
|
+
The whole sweep — every model in the spec, every backend variant, every workload:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
bash benchmarks/run_all.sh
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
A subset, which is what you want while iterating:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
bash benchmarks/run_all.sh --models gemma-2-2b --variants eager,vllm
|
|
57
|
+
bash benchmarks/run_all.sh --workloads generate,capture_mid --no-report
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
One cell directly, which is what the sweep loops over:
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
python -m benchmarks.run_bench --model gemma-2-2b --variant vllm
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
See what is defined without loading anything:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
python -m benchmarks.run_bench --list
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Point it at a different interpreter with `BENCH_PYTHON` or `--python`:
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
BENCH_PYTHON=../apps/inference/.venv/bin/python bash benchmarks/run_all.sh
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Each cell writes `benchmarks/results/<model>__<variant>.json`; `report_bench.py` renders those into
|
|
79
|
+
`results-latest.md`. Regenerate the report without re-measuring:
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
python -m benchmarks.report_bench
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
`run_all.sh` re-runs cells it already has, so a plain rerun replaces stale numbers rather than
|
|
86
|
+
mixing them with fresh ones. Pass `--skip-existing` to resume an interrupted sweep instead.
|
|
87
|
+
|
|
88
|
+
## Where the numbers get published
|
|
89
|
+
|
|
90
|
+
Three files carry these measurements, and one command writes all three. `report_bench` renders the
|
|
91
|
+
full record into `results-latest.md` and then calls `publish`, which rewrites:
|
|
92
|
+
|
|
93
|
+
| target | what it gets |
|
|
94
|
+
| --- | --- |
|
|
95
|
+
| `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and graph freeze, every model |
|
|
96
|
+
| `visualizer-web/data/benchmarks.generated.ts` | the same figures for the card behind the site's **Fast** claim, each row that ran differently carrying a footnote saying how |
|
|
97
|
+
|
|
98
|
+
Both were transcribed by hand until this existed, and both had drifted -- one carried percentages the
|
|
99
|
+
other had dropped, and two of the card's multipliers were ratios of already-rounded figures. So:
|
|
100
|
+
**never edit a number in either.** Re-render from the cells on disk, which needs no GPU:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
python -m benchmarks.publish # rewrite both
|
|
104
|
+
python -m benchmarks.publish --check # exit 1 if either has drifted, and name it
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
`tests/test_published_benchmarks.py` runs that check over the committed cells, so a stale copy is a
|
|
108
|
+
red suite rather than a claim nobody re-read. The visualizer's chatbot answers out of a bundle holding
|
|
109
|
+
the README verbatim, so a publish that changed the README also wants `make viz-knowledge` -- the
|
|
110
|
+
command says so when it happens.
|
|
111
|
+
|
|
112
|
+
Two things `publish` decides for itself, rather than from a list someone maintains. It refuses to
|
|
113
|
+
write at all when the cells disagree about the GPU or the dtype, because those tables print one shared
|
|
114
|
+
conditions line. And it footnotes a card row when the sweep gave that model anything of its own -- its
|
|
115
|
+
own memory fraction, its own engine arguments, its own capture point -- so that the conditions line
|
|
116
|
+
keeps covering the rows it claims to. `deepseek-v4-flash-0731` earns a footnote for all four reasons.
|
|
117
|
+
The card dropped such a row until the footnote existed, which is why the largest freeze win in the
|
|
118
|
+
sweep was for a while the one figure the site did not show; a row is still dropped, but only when it is
|
|
119
|
+
missing a baseline figure and so has no multiplier to print. A scratch sweep of ad-hoc models should
|
|
120
|
+
pass `--no-publish` to `report_bench` rather than publish rows nobody deployed.
|
|
121
|
+
|
|
122
|
+
## Benchmarking your own model
|
|
123
|
+
|
|
124
|
+
Any model `interp_engine.load_model` can load works, with no edit to the spec:
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
python -m benchmarks.run_bench --hf-id mistralai/Mistral-7B-v0.1 --variant eager
|
|
128
|
+
python -m benchmarks.run_bench --hf-id mistralai/Mistral-7B-v0.1 --variant vllm
|
|
129
|
+
python -m benchmarks.report_bench
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
`--family` and `--params` are optional labels for the report's model table. The row key is derived
|
|
133
|
+
from the repo id, so `mistralai/Mistral-7B-v0.1` becomes `mistral-7b-v0.1` and sorts after the
|
|
134
|
+
spec's own models.
|
|
135
|
+
|
|
136
|
+
Three things to check for a model that is much larger than the ones in the spec:
|
|
137
|
+
|
|
138
|
+
- **vLLM memory.** `GPU_MEMORY_UTILIZATION` in `bench_spec.py` is `0.8`, below vLLM's own `0.9`
|
|
139
|
+
default, because `worker_lens_readout` needs roughly twice the vocab-logits in worker scratch (it
|
|
140
|
+
takes a `logsumexp` over them) and at `0.9` there is none left — `lens_topk` dies with a worker-side
|
|
141
|
+
CUDA OOM. It is uniform across models on purpose, since a per-model fraction measures each model
|
|
142
|
+
under a different reservation, and the report says so where one is used. A model whose weights do
|
|
143
|
+
not fit inside that fraction can declare its own `ModelSpec.gpu_memory_utilization` — as
|
|
144
|
+
`deepseek-v4-flash-0731` does, at 0.95, because 149 GiB does not fit in what 0.8 of a 180 GB card reserves
|
|
145
|
+
— and `--gpu-memory-utilization` on either script still overrides both.
|
|
146
|
+
- **Context length.** `MAX_MODEL_LEN` is capped at 2048, because vLLM refuses to boot unless its KV
|
|
147
|
+
pool can hold one request at the model's advertised context and these checkpoints advertise
|
|
148
|
+
32k-131k. It must stay above the longest `prompt_tokens + max_new_tokens` in `WORKLOADS`.
|
|
149
|
+
- **Prefix caching** is forced **off** for the vLLM variants, though the engine defaults it on. Every
|
|
150
|
+
workload issues the same prompt for each repeat, so with caching on the second and third would be
|
|
151
|
+
served from the KV cache and the reported median would be a cache hit rather than the work — and
|
|
152
|
+
the eager column, which has no such cache, would stop measuring the same thing. What caching is
|
|
153
|
+
worth is priced separately in `docs/PERFORMANCE.md`.
|
|
154
|
+
|
|
155
|
+
To add a model permanently, append a `ModelSpec` to `MODELS` in `bench_spec.py`. `run_all.sh` reads
|
|
156
|
+
that list, so nothing else needs changing.
|
|
157
|
+
|
|
158
|
+
## What the workloads are
|
|
159
|
+
|
|
160
|
+
| workload | what it does | what it isolates |
|
|
161
|
+
| --- | --- | --- |
|
|
162
|
+
| `generate` | 512-token prompt, 128 new tokens, greedy, one request | single-stream decode rate and time to first token |
|
|
163
|
+
| `generate_x8` | the same request 8x concurrently | batching: vLLM batches, eager serves one at a time |
|
|
164
|
+
| `capture_mid` | `resid_post` at the middle layer over a 512-token prompt | prefill plus a single point's transport |
|
|
165
|
+
| `capture_all` | `resid_post` at every layer, same prompt | transport cost, since the forward is identical to `capture_mid` |
|
|
166
|
+
| `capture_gen` | generate 32 tokens capturing `resid_post` | decode-time capture |
|
|
167
|
+
| `steer` | `capture_gen` again with an add-steering vector | steering, since nothing else differs |
|
|
168
|
+
| `lens_topk` | 512 rows read out to top-10 ids, the way lens serving does it | the read-out alone, with no forward attached |
|
|
169
|
+
|
|
170
|
+
The four capture and steering workloads address `resid_post` on every model that has one. A model
|
|
171
|
+
that does not can name a substitute in `ModelSpec.capture_point`, and the report lists the rows that
|
|
172
|
+
did. `deepseek-v4-flash-0731` is the case: its blocks carry four parallel residual streams
|
|
173
|
+
(hyper-connections), so `resid_post` names four tensors rather than one and the engine refuses it
|
|
174
|
+
instead of silently picking the first. That row uses `mlp_out`, which keeps both properties the
|
|
175
|
+
tables depend on — one `d_model`-wide row per position per layer, so the transport figures still
|
|
176
|
+
compare, and a plain module output, so steering has something to write to.
|
|
177
|
+
|
|
178
|
+
Pairs are deliberate. `capture_mid` and `capture_all` share a prompt length and differ only in how
|
|
179
|
+
many points come back, so the difference between them is transport rather than compute. `steer` and
|
|
180
|
+
`capture_gen` are identical apart from the spec, so the difference is what steering costs. `generate`
|
|
181
|
+
and `generate_x8` differ only in concurrency.
|
|
182
|
+
|
|
183
|
+
`lens_topk` is the one place the two backends run different code. `VLLMModel` has
|
|
184
|
+
`decode_residuals_topk`, which does the norm, unembed and `topk` in the worker and returns
|
|
185
|
+
`[rows, 10]`; that is what `apps/inference` calls, and it is not part of the `InterpModel` protocol
|
|
186
|
+
because the eager backend needs no such thing — it has no boundary to keep a large tensor away from,
|
|
187
|
+
so it decodes and reduces in process. The unreduced `decode_residuals` is deliberately **not**
|
|
188
|
+
benchmarked: it returns `[rows, vocab]`, which on vLLM is hundreds of MiB crossing a process boundary
|
|
189
|
+
per call, and nothing serves a lens that way. Each `lens_topk` cell checks its ids against the full
|
|
190
|
+
read-out, so a speedup that returned different tokens would be recorded as an error rather than a win.
|
|
191
|
+
|
|
192
|
+
Prompts are normalized to a **token count**, not a character count: the same passage is 20% more
|
|
193
|
+
tokens under one tokenizer than another, and prefill cost scales with tokens. Every model therefore
|
|
194
|
+
does the same amount of work.
|
|
195
|
+
|
|
196
|
+
## The backend variants
|
|
197
|
+
|
|
198
|
+
The report names these by what they are; `--variant` and the result filenames use the short key.
|
|
199
|
+
|
|
200
|
+
| variant | `--variant` | what it is |
|
|
201
|
+
| --- | --- | --- |
|
|
202
|
+
| interp-engine eager | `eager` | raw HF forward; `attn_implementation="eager"`, which is what the engine sets |
|
|
203
|
+
| interp-engine vllm | `vllm` | vLLM with `enforce_eager=True` — CUDA graphs and inductor compile off |
|
|
204
|
+
| vllm (vanilla) | `vllm-cudagraph` | vLLM left at its own defaults, graphs and compile on |
|
|
205
|
+
| interp-engine vllm freeze | `vllm-freeze` | breakable graphs with `resid_post` freeze wraps at every layer, plus one write tap mid-stack |
|
|
206
|
+
|
|
207
|
+
`bench_spec.VARIANTS` also carries two speculative-decoding variants that exist on one checkpoint
|
|
208
|
+
only, and are not part of these tables: `report_bench.EXCLUDED` says why, and `--variant` still
|
|
209
|
+
measures them.
|
|
210
|
+
|
|
211
|
+
The third exists to price a default the engine chose for capture's sake, which is the most useful
|
|
212
|
+
thing a speed benchmark of this library can say: `VLLMModel` defaults `enforce_eager=True`, because
|
|
213
|
+
CUDA-graph replay does not re-execute the Python forward and so never fires a
|
|
214
|
+
`register_forward_hook`.
|
|
215
|
+
|
|
216
|
+
The capture workloads are **run** on `vllm-cudagraph` rather than skipped, so the report shows what a
|
|
217
|
+
capture actually returns under replay instead of asserting the outcome. A capture that comes back
|
|
218
|
+
with no points, or with fewer rows than the prompt had tokens, is recorded as `unsupported` with the
|
|
219
|
+
shape it got, and the report renders that cell as `n/a`.
|
|
220
|
+
|
|
221
|
+
### `vllm-freeze`, and what its `steer` cell needs
|
|
222
|
+
|
|
223
|
+
The fourth is the answer to the third: freeze copies activations in and out of buffers the graph
|
|
224
|
+
already refers to, so a replay serves capture and steering without a Python forward. It is the path
|
|
225
|
+
`freeze_points="auto"` takes in production, and the row exists to price it against the
|
|
226
|
+
`enforce_eager=True` column capture would otherwise have to use.
|
|
227
|
+
|
|
228
|
+
`"auto"` installs **reads** only, and a steering op needs a write tap to land in, so this row priced
|
|
229
|
+
half the feature and reported the other half as `n/a` -- with a message that blamed graph replay for
|
|
230
|
+
it, which is the thing freeze exists to work around. It now also passes `freeze_writes`, whose value
|
|
231
|
+
is the sentinel `run_bench.STEER_WRITES` rather than a site: the `steer` workload writes mid-stack,
|
|
232
|
+
that layer differs per model, and a freeze write is a `load_model` argument, so it has to be resolved
|
|
233
|
+
from the config before a model exists to ask.
|
|
234
|
+
|
|
235
|
+
`VariantSpec.models` restricts the row to the checkpoints freeze has been shown correct on, so a model
|
|
236
|
+
missing from it renders `--` rather than a number nobody checked.
|
|
237
|
+
|
|
238
|
+
One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk freezes
|
|
239
|
+
`resid_streams` — the whole stack of four parallel streams per layer — and a frozen engine serves the
|
|
240
|
+
set it froze, so `deepseek-v4-flash-0731` cannot be asked for the `mlp_out` its other columns capture.
|
|
241
|
+
That row's freeze cell therefore prices the stack where every other cell prices one row, declared in
|
|
242
|
+
`ModelSpec.freeze_capture_point`, stated by the report under *Where a row differs*, and carried onto
|
|
243
|
+
the visualizer's card as one line of that row's footnote.
|
|
244
|
+
|
|
245
|
+
## Reading the numbers honestly
|
|
246
|
+
|
|
247
|
+
- Each figure is the **median of the measured repeats** (`repeats` per workload in `bench_spec.py`),
|
|
248
|
+
after one unmeasured warmup run. The warmup is load-bearing: the first call of any workload pays a
|
|
249
|
+
lazy import, the allocator's first growth to working size, and on vLLM the first `collective_rpc`
|
|
250
|
+
round trip.
|
|
251
|
+
- **Every model asks for bfloat16**, pinned in `ModelSpec.dtype` rather than left at `"auto"`, which
|
|
252
|
+
is the checkpoint's own precision and does not resolve equally on both backends: eager honors a
|
|
253
|
+
float32 checkpoint while vLLM downcasts it. Each cell records what its backend resolved to, so an
|
|
254
|
+
added model can be checked. On a quantized checkpoint this is the **compute** dtype and not a
|
|
255
|
+
request to expand the weights: `deepseek-v4-flash-0731` stays block-quantized FP8 under both backends
|
|
256
|
+
(transformers dequantizes only when asked with `dequantize=True`, which nothing here passes), so
|
|
257
|
+
those rows compare two backends serving the same quantized weights.
|
|
258
|
+
- **Decode throughput** excludes the first token, whose cost is the prefill already reported as time
|
|
259
|
+
to first token. Dividing all tokens by the total would blend prefill into the decode figure.
|
|
260
|
+
- **`capture_gen` is not the same algorithm on both backends.** vLLM captures during decode; eager
|
|
261
|
+
generates and then re-runs one forward over prompt plus generated tokens (documented at
|
|
262
|
+
`EagerModel.capture_generation`), so eager pays an extra prefill that vLLM does not. The numbers
|
|
263
|
+
are still the right comparison — this is what each backend does when you call the method — but the
|
|
264
|
+
gap is not purely kernel speed.
|
|
265
|
+
- **`generate_x8` on eager is expected to look flat.** Its generation loop is synchronous
|
|
266
|
+
underneath, so awaiting it does not yield to the event loop and the eight requests serialize. That
|
|
267
|
+
is a correct result for that backend, not a harness artifact.
|
|
268
|
+
|
|
269
|
+
## Files
|
|
270
|
+
|
|
271
|
+
| file | what it is |
|
|
272
|
+
| --- | --- |
|
|
273
|
+
| `bench_spec.py` | models, variants, workloads, prompt normalization — plain data, imports no torch |
|
|
274
|
+
| `probe.py` | the environment stamp and the timing primitives |
|
|
275
|
+
| `workloads.py` | the timed operations, written against the protocol only |
|
|
276
|
+
| `run_bench.py` | runs one `(model, variant)` cell, writes JSON |
|
|
277
|
+
| `cells.py` | what a directory of JSON cells says — row order, and "not measured" against "measured as zero" |
|
|
278
|
+
| `report_bench.py` | JSON cells to `results-latest.md`, then `publish` |
|
|
279
|
+
| `publish.py` | the three published columns to the root README's tables and the visualizer's card |
|
|
280
|
+
| `run_all.sh` | the sweep loop, one process per cell |
|
|
281
|
+
|
|
282
|
+
One process per cell is a requirement, not tidiness: vLLM reserves its memory fraction of the whole
|
|
283
|
+
card during bring-up and keeps its KV cache in a worker subprocess that a dropped Python reference
|
|
284
|
+
does not reap, so two cells in one interpreter would have the second fighting the first for free
|
|
285
|
+
memory.
|
|
286
|
+
|
|
287
|
+
This directory is not part of the installed package — `pyproject.toml` ships only `interp_engine`.
|
|
288
|
+
It is dev tooling, run from a checkout.
|
|
289
|
+
|
|
290
|
+
## Troubleshooting
|
|
291
|
+
|
|
292
|
+
**`FileNotFoundError: 'ninja'` inside `EngineCore`, minutes into a vLLM load.** vLLM's flashinfer
|
|
293
|
+
sampler JIT-compiles a CUDA extension on first use and shells out to `ninja` and `nvcc`. `run_bench`
|
|
294
|
+
puts the venv's script directory and `$CUDA_HOME/bin` on `PATH` for you; if your CUDA toolkit is
|
|
295
|
+
somewhere unusual, set `CUDA_HOME`.
|
|
296
|
+
|
|
297
|
+
**`CUDA_ERROR_LAUNCH_FAILED` during `DeepGEMM warmup`, after the weights have loaded.** Preceded by a
|
|
298
|
+
wall of `Assertion failed: ... smxx_layout.cuh:131, condition: (values[j] & 0x807fffffu) == 0`. The
|
|
299
|
+
warmup precompiles DeepGEMM's FP8 kernels by calling them on synthetic scales, and on a UE8M0-scaled
|
|
300
|
+
checkpoint those carry mantissa bits, which is precisely what that assertion rejects. A device-side
|
|
301
|
+
assertion takes the CUDA context with it, so the engine dies at startup having spent ten minutes
|
|
302
|
+
loading weights. `run_bench` sets `VLLM_DEEP_GEMM_WARMUP=skip` for you and records it in the cell's
|
|
303
|
+
environment stamp; the real forwards pass correctly formed scales and keep DeepGEMM. The cost is a
|
|
304
|
+
slower `warmup_s` on the FP8 rows, since that first compile moves into the first forward — which is
|
|
305
|
+
outside every measured median.
|
|
306
|
+
|
|
307
|
+
**An eager cell on `deepseek-v4-flash-0731` fails at load with `Loading an FP8 quantized model requires
|
|
308
|
+
accelerate`, or at the first forward with `finegrained-fp8 kernel requires the kernels package`.**
|
|
309
|
+
The `quant` extra is missing from that interpreter — see Requirements. Both are hard requirements of
|
|
310
|
+
transformers' FP8 path, and only the eager variants use it, so the vLLM cells of the same model pass
|
|
311
|
+
and the row reads as an eager-backend limitation rather than a missing wheel.
|
|
312
|
+
|
|
313
|
+
**`GatedRepoError: 401` for a model that loads fine on eager.** No `HF_TOKEN` — see Requirements.
|
|
314
|
+
Eager found the weights in your local HF cache; vLLM resolves the safetensors index through the hub.
|
|
315
|
+
|
|
316
|
+
**A cell reports `unsupported` for every capture workload.** Expected on `vllm-cudagraph`, and the
|
|
317
|
+
point of that variant. On `vllm` it means something turned CUDA graphs back on.
|
|
318
|
+
|
|
319
|
+
**vLLM OOMs or sizes a tiny KV cache.** Something else is on the GPU, or the previous cell's worker
|
|
320
|
+
has not exited. `run_all.sh` waits for free VRAM between cells and kills stragglers after a failure;
|
|
321
|
+
if you are running cells by hand, check `nvidia-smi` first.
|
|
322
|
+
|
|
323
|
+
**`lens_topk` fails with a CUDA OOM inside the worker while every other workload passes.** Not a
|
|
324
|
+
sweep problem: the read-out allocates the full `[rows, vocab]` logits and then a `logsumexp`
|
|
325
|
+
intermediate of the same size, so it needs roughly twice the vocab-logits free *inside* vLLM's
|
|
326
|
+
reservation — about 0.5 GiB for a 256k-vocab model at 512 rows. Lower
|
|
327
|
+
`--gpu-memory-utilization`. Worth knowing outside the benchmark too, since this is the recommended
|
|
328
|
+
lens path on vLLM.
|
|
329
|
+
|
|
330
|
+
**A cell hangs.** `run_all.sh` caps each cell at `BENCH_TIMEOUT_S` (default 1800) and kills it,
|
|
331
|
+
recording the failure and moving on, so a wedged engine cannot stall the sweep.
|
|
332
|
+
|
|
333
|
+
**An eager cell warns `You have loaded an FP8 model on CPU`, then takes an hour to load.** The eager
|
|
334
|
+
backend loads through transformers and then calls `.to(device)`, which stages the whole checkpoint in
|
|
335
|
+
host RAM first. That is unremarkable at 5 GiB and pathological at 149. A model this size should
|
|
336
|
+
declare `extra_eager_kwargs={"device_map": "cuda"}`, as `deepseek-v4-flash-0731` does, so transformers places
|
|
337
|
+
each shard on the card as it reads it; `load_model` drops `device` whenever a `device_map` is given,
|
|
338
|
+
so the two do not fight. It stays per-model rather than becoming the default because it changes what
|
|
339
|
+
`construct_s` measures.
|
|
340
|
+
|
|
341
|
+
**An eager cell reports single-digit tok/s.** Usually the model ran on the CPU — though not always:
|
|
342
|
+
`deepseek-v4-flash-0731` is genuinely single-digit on eager, because a 291B MoE dispatches 256 experts per
|
|
343
|
+
layer through Python. Check `load.device` in the cell's JSON before assuming. For the CPU case,
|
|
344
|
+
`load_model` only runs its device-selection ladder for `backend="auto"`; with an explicit
|
|
345
|
+
`backend="eager"` the `device` argument is passed straight through, and `EagerModel` skips its
|
|
346
|
+
`.to(device)` when that is None, so the model stays where transformers loaded it — on the CPU, with
|
|
347
|
+
no error and no warning. The harness passes `device="cuda"` for exactly this reason and now refuses to
|
|
348
|
+
report a cell that landed on the CPU. If you call `load_model(..., backend="eager")` yourself, pass a
|
|
349
|
+
device.
|