interp-engine 1.2.9__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {interp_engine-1.2.9 → interp_engine-1.3.0}/PKG-INFO +26 -23
- {interp_engine-1.2.9 → interp_engine-1.3.0}/README.md +25 -22
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/README.md +15 -15
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/bench_spec.py +44 -42
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/cells.py +6 -6
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/probe_lens_stream.py +16 -12
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/publish.py +4 -4
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/report_bench.py +7 -7
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-cudagraph.json +2 -4
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark-cudagraph.json +1 -3
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark.json +0 -1
- interp_engine-1.2.9/benchmarks/results/deepseek-v4-flash-0731__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/deepseek-v4-flash-0731__vllm-static.json +8 -8
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm.json +0 -1
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.9/benchmarks/results/gemma-2-2b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/gemma-2-2b__vllm-static.json +6 -6
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__vllm.json +0 -1
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.9/benchmarks/results/llama-3.1-8b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/llama-3.1-8b__vllm-static.json +6 -6
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__vllm.json +0 -1
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.9/benchmarks/results/qwen3-4b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/qwen3-4b__vllm-static.json +6 -6
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__vllm.json +0 -1
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.9/benchmarks/results/qwen3.8-27b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/qwen3.8-27b__vllm-static.json +6 -6
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__vllm.json +0 -1
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results-latest.md +14 -14
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/run_bench.py +15 -10
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/workloads.py +13 -13
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/AGENT_INTEGRATION.md +42 -3
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/PERFORMANCE.md +54 -47
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/__init__.py +2 -1
- interp_engine-1.3.0/interp_engine/load.py +216 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/model.py +2 -2
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/protocol.py +7 -7
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/sync.py +4 -4
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_backend.py +341 -244
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/__init__.py +25 -25
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/_hooks.py +2 -2
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/graphs.py +5 -4
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/readout.py +22 -22
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/mhc.py +1 -1
- interp_engine-1.2.9/interp_engine/vllm_capture/freeze.py → interp_engine-1.3.0/interp_engine/vllm_capture/static.py +217 -214
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_plugin.py +25 -25
- {interp_engine-1.2.9 → interp_engine-1.3.0}/pyproject.toml +2 -2
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_bench_workloads.py +14 -14
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_layer_kinds.py +1 -1
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_load.py +46 -10
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_published_benchmarks.py +8 -8
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_qk_norm.py +1 -1
- interp_engine-1.2.9/tests/test_freeze_dsv4_gpu.py → interp_engine-1.3.0/tests/test_static_dsv4_gpu.py +17 -16
- interp_engine-1.2.9/tests/test_freeze_parity_gpu.py → interp_engine-1.3.0/tests/test_static_parity_gpu.py +51 -48
- interp_engine-1.2.9/tests/test_freeze_set.py → interp_engine-1.3.0/tests/test_static_set.py +113 -113
- interp_engine-1.2.9/tests/test_freeze_warmup.py → interp_engine-1.3.0/tests/test_static_warmup.py +17 -17
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_graph_path.py +2 -1
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_graphs_on_gpu.py +26 -31
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_hook_availability.py +57 -32
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_hyper_connections.py +97 -97
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_worker_lens_capture_readout.py +11 -11
- interp_engine-1.2.9/interp_engine/load.py +0 -152
- {interp_engine-1.2.9 → interp_engine-1.3.0}/.gitignore +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/LICENSE +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/__init__.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/probe.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__eager.json +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__eager.json +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__eager.json +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__eager.json +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__eager.json +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/benchmarks/run_all.sh +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/ARCHITECTURE_QUIRKS.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/COMPATIBILITY.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/ENGINE_HOOK_MAPPINGS.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/GRADIENTS.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/INTERNALS.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/PORTING.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/README.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/SUPPORTED_POINTS.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/docs/USAGE.md +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/_loop.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/address.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/arch.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/attn_config.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/attn_scores.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/autograd_support.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/capture.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/chat_compose.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/chat_conventions.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/chat_formatters.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/cuda_preflight.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/dispatch.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/facts.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/hooks.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/lens.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/mappers.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/moe_routing.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/points.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/residual_basis.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/select.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/steer.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/steer_specs.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/tokenize.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/_demux.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/_payload.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/_tree.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/attn.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/capture.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/__init__.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/intervene.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/unembed.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/native.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/requests.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/interp_engine/vllm_capture/steering.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/conftest.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/harness.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/model_expectations.yaml +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/synthetic_families.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_address.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_attn_config_tripwire.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_attn_probs_indexing.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_attn_scores.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_attn_z_gqa.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_autograd_support.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_capability_refusals.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_capture_addressing.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_chat_compose.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_chat_formatters.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_chat_templates.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_core.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_cuda_preflight.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_doc_code_fences.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_eager_autograd.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_facts.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_family_points.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_gated_attn_out.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_head_contributions.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_hook_call_conventions.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_logit_transform.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_mappers.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_mlp_internals.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_model_expectations.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_moe.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_multimodal_arch.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_new_models_gpu.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_no_chat_template.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_normalized_hook.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_packaging.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_parity_gpt2.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_per_layer_attn_dims.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_points_registry.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_protocol.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_qkv_layout.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_reasoning_spans.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_release.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_resid_mid.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_residual_basis.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_sandwich_norms.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_select.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_sliding_window_attn.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_small_models_gpu.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_steer_context.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_steer_math_parity.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_sync_loop.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_sync_parity.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_unified_free_functions.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_unresolved_families.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_capture_gpu.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_capture_scales.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_engine_loop.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_kv_isolation.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_new_points.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_only_families.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_plugin.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vllm_wire_grammar.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_vocabulary_boundary.py +0 -0
- {interp_engine-1.2.9 → interp_engine-1.3.0}/tests/test_worker_lens_readout.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: interp-engine
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: A fast, standardized interpretability engine that supports most modern models and architectures. Powers Neuronpedia.
|
|
5
5
|
Project-URL: Homepage, https://github.com/decoderesearch/interp-engine
|
|
6
6
|
Project-URL: Repository, https://github.com/decoderesearch/interp-engine
|
|
@@ -48,7 +48,7 @@ Description-Content-Type: text/markdown
|
|
|
48
48
|
`interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference and is checked for accuracy against HF Transformers and other engines.
|
|
49
49
|
|
|
50
50
|
<p align="center">
|
|
51
|
-
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-
|
|
51
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-static on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="100%">
|
|
52
52
|
</p>
|
|
53
53
|
<p align="center">
|
|
54
54
|
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="100%">
|
|
@@ -71,13 +71,16 @@ pip install interp-engine # eager backend only
|
|
|
71
71
|
```python
|
|
72
72
|
from interp_engine import Address, load_model, run_with_cache
|
|
73
73
|
|
|
74
|
-
# VLLM
|
|
75
|
-
model = load_model("Qwen/Qwen3-8B")
|
|
74
|
+
# VLLM (default): low VRAM, medium speed, every point, chosen per request
|
|
75
|
+
model = load_model("Qwen/Qwen3-8B")
|
|
76
76
|
|
|
77
|
-
# VLLM-
|
|
78
|
-
# model = load_model("Qwen/Qwen3-8B",
|
|
77
|
+
# VLLM-STATIC: high VRAM, high speed, only the points you declare (default resid_post)
|
|
78
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-static")
|
|
79
79
|
|
|
80
|
-
#
|
|
80
|
+
# VLLM-GENERATE: fastest, generation only -- no capture, no steering
|
|
81
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-generate")
|
|
82
|
+
|
|
83
|
+
# EAGER: low VRAM, low speed
|
|
81
84
|
# model = load_model("Qwen/Qwen3-8B", backend="eager")
|
|
82
85
|
|
|
83
86
|
point = Address("resid_post", 10) # or string: "resid_post.10"
|
|
@@ -95,7 +98,7 @@ Add "use interp-engine" to your prompt and let your agent figure it out - everyt
|
|
|
95
98
|
|
|
96
99
|
## Performance / Speed
|
|
97
100
|
|
|
98
|
-
vLLM gives `interp-engine` high throughput via concurrency, and
|
|
101
|
+
vLLM gives `interp-engine` high throughput via concurrency, and `backend="vllm-static"` adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
|
|
99
102
|
|
|
100
103
|
<!-- THROUGHPUT:START -->
|
|
101
104
|
|
|
@@ -105,27 +108,27 @@ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
|
|
|
105
108
|
|
|
106
109
|
One stream (tok/s):
|
|
107
110
|
|
|
108
|
-
| model | eager | vLLM | vLLM +
|
|
109
|
-
| ------------------------ | ----- | ---------- |
|
|
110
|
-
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)**
|
|
111
|
-
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)**
|
|
112
|
-
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)**
|
|
113
|
-
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)**
|
|
114
|
-
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)**
|
|
111
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
112
|
+
| ------------------------ | ----- | ---------- | ------------------ |
|
|
113
|
+
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
|
|
114
|
+
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
|
|
115
|
+
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
|
|
116
|
+
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
|
|
117
|
+
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
|
|
115
118
|
|
|
116
119
|
8 concurrent requests (aggregate tok/s):
|
|
117
120
|
|
|
118
|
-
| model | eager | vLLM | vLLM +
|
|
119
|
-
| ------------------------ | ----- | ----------- |
|
|
120
|
-
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)**
|
|
121
|
-
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)**
|
|
122
|
-
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)**
|
|
123
|
-
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)**
|
|
124
|
-
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)**
|
|
121
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
122
|
+
| ------------------------ | ----- | ----------- | ------------------ |
|
|
123
|
+
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
|
|
124
|
+
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
|
|
125
|
+
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
|
|
126
|
+
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
|
|
127
|
+
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
|
|
125
128
|
|
|
126
129
|
<!-- THROUGHPUT:END -->
|
|
127
130
|
|
|
128
|
-
|
|
131
|
+
`backend="vllm-static"` is opt-in, and serves only the tap set it declared — `static_points="auto"` by default, or a list you name. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; [benchmarks/results-latest.md](benchmarks/results-latest.md) has every figure at full precision, including capture, steering and lens latencies; [benchmarks/README.md](benchmarks/README.md) has how the tables above are rounded.
|
|
129
132
|
|
|
130
133
|
## Correctness
|
|
131
134
|
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
`interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference and is checked for accuracy against HF Transformers and other engines.
|
|
18
18
|
|
|
19
19
|
<p align="center">
|
|
20
|
-
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-
|
|
20
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-static on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="100%">
|
|
21
21
|
</p>
|
|
22
22
|
<p align="center">
|
|
23
23
|
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="100%">
|
|
@@ -40,13 +40,16 @@ pip install interp-engine # eager backend only
|
|
|
40
40
|
```python
|
|
41
41
|
from interp_engine import Address, load_model, run_with_cache
|
|
42
42
|
|
|
43
|
-
# VLLM
|
|
44
|
-
model = load_model("Qwen/Qwen3-8B")
|
|
43
|
+
# VLLM (default): low VRAM, medium speed, every point, chosen per request
|
|
44
|
+
model = load_model("Qwen/Qwen3-8B")
|
|
45
45
|
|
|
46
|
-
# VLLM-
|
|
47
|
-
# model = load_model("Qwen/Qwen3-8B",
|
|
46
|
+
# VLLM-STATIC: high VRAM, high speed, only the points you declare (default resid_post)
|
|
47
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-static")
|
|
48
48
|
|
|
49
|
-
#
|
|
49
|
+
# VLLM-GENERATE: fastest, generation only -- no capture, no steering
|
|
50
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-generate")
|
|
51
|
+
|
|
52
|
+
# EAGER: low VRAM, low speed
|
|
50
53
|
# model = load_model("Qwen/Qwen3-8B", backend="eager")
|
|
51
54
|
|
|
52
55
|
point = Address("resid_post", 10) # or string: "resid_post.10"
|
|
@@ -64,7 +67,7 @@ Add "use interp-engine" to your prompt and let your agent figure it out - everyt
|
|
|
64
67
|
|
|
65
68
|
## Performance / Speed
|
|
66
69
|
|
|
67
|
-
vLLM gives `interp-engine` high throughput via concurrency, and
|
|
70
|
+
vLLM gives `interp-engine` high throughput via concurrency, and `backend="vllm-static"` adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
|
|
68
71
|
|
|
69
72
|
<!-- THROUGHPUT:START -->
|
|
70
73
|
|
|
@@ -74,27 +77,27 @@ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
|
|
|
74
77
|
|
|
75
78
|
One stream (tok/s):
|
|
76
79
|
|
|
77
|
-
| model | eager | vLLM | vLLM +
|
|
78
|
-
| ------------------------ | ----- | ---------- |
|
|
79
|
-
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)**
|
|
80
|
-
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)**
|
|
81
|
-
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)**
|
|
82
|
-
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)**
|
|
83
|
-
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)**
|
|
80
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
81
|
+
| ------------------------ | ----- | ---------- | ------------------ |
|
|
82
|
+
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
|
|
83
|
+
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
|
|
84
|
+
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
|
|
85
|
+
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
|
|
86
|
+
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
|
|
84
87
|
|
|
85
88
|
8 concurrent requests (aggregate tok/s):
|
|
86
89
|
|
|
87
|
-
| model | eager | vLLM | vLLM +
|
|
88
|
-
| ------------------------ | ----- | ----------- |
|
|
89
|
-
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)**
|
|
90
|
-
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)**
|
|
91
|
-
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)**
|
|
92
|
-
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)**
|
|
93
|
-
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)**
|
|
90
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
91
|
+
| ------------------------ | ----- | ----------- | ------------------ |
|
|
92
|
+
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
|
|
93
|
+
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
|
|
94
|
+
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
|
|
95
|
+
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
|
|
96
|
+
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
|
|
94
97
|
|
|
95
98
|
<!-- THROUGHPUT:END -->
|
|
96
99
|
|
|
97
|
-
|
|
100
|
+
`backend="vllm-static"` is opt-in, and serves only the tap set it declared — `static_points="auto"` by default, or a list you name. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; [benchmarks/results-latest.md](benchmarks/results-latest.md) has every figure at full precision, including capture, steering and lens latencies; [benchmarks/README.md](benchmarks/README.md) has how the tables above are rounded.
|
|
98
101
|
|
|
99
102
|
## Correctness
|
|
100
103
|
|
|
@@ -92,7 +92,7 @@ full record into `results-latest.md` and then calls `publish`, which rewrites:
|
|
|
92
92
|
|
|
93
93
|
| target | what it gets |
|
|
94
94
|
| --- | --- |
|
|
95
|
-
| `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and
|
|
95
|
+
| `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and static taps, every model |
|
|
96
96
|
| `visualizer-web/data/benchmarks.generated.ts` | the same figures for the card behind the site's **Fast** claim, each row that ran differently carrying a footnote saying how |
|
|
97
97
|
|
|
98
98
|
Both were transcribed by hand until this existed, and both had drifted -- one carried percentages the
|
|
@@ -129,7 +129,7 @@ write at all when the cells disagree about the GPU or the dtype, because those t
|
|
|
129
129
|
conditions line. And it footnotes a card row when the sweep gave that model anything of its own -- its
|
|
130
130
|
own memory fraction, its own engine arguments, its own capture point -- so that the conditions line
|
|
131
131
|
keeps covering the rows it claims to. `deepseek-v4-flash-0731` earns a footnote for all four reasons.
|
|
132
|
-
The card dropped such a row until the footnote existed, which is why the largest
|
|
132
|
+
The card dropped such a row until the footnote existed, which is why the largest static win in the
|
|
133
133
|
sweep was for a while the one figure the site did not show; a row is still dropped, but only when it is
|
|
134
134
|
missing a baseline figure and so has no multiplier to print. A scratch sweep of ad-hoc models should
|
|
135
135
|
pass `--no-publish` to `report_bench` rather than publish rows nobody deployed.
|
|
@@ -217,7 +217,7 @@ The report names these by what they are; `--variant` and the result filenames us
|
|
|
217
217
|
| interp-engine eager | `eager` | raw HF forward; `attn_implementation="eager"`, which is what the engine sets |
|
|
218
218
|
| interp-engine vllm | `vllm` | vLLM with `enforce_eager=True` — CUDA graphs and inductor compile off |
|
|
219
219
|
| vllm (vanilla) | `vllm-cudagraph` | vLLM left at its own defaults, graphs and compile on |
|
|
220
|
-
| interp-engine vllm
|
|
220
|
+
| interp-engine vllm static | `vllm-static` | breakable graphs with `resid_post` static wraps at every layer, plus one write tap mid-stack |
|
|
221
221
|
|
|
222
222
|
`bench_spec.VARIANTS` also carries two speculative-decoding variants that exist on one checkpoint
|
|
223
223
|
only, and are not part of these tables: `report_bench.EXCLUDED` says why, and `--variant` still
|
|
@@ -233,31 +233,31 @@ capture actually returns under replay instead of asserting the outcome. A captur
|
|
|
233
233
|
with no points, or with fewer rows than the prompt had tokens, is recorded as `unsupported` with the
|
|
234
234
|
shape it got, and the report renders that cell as `n/a`.
|
|
235
235
|
|
|
236
|
-
### `vllm-
|
|
236
|
+
### `vllm-static`, and what its `steer` cell needs
|
|
237
237
|
|
|
238
|
-
The fourth is the answer to the third:
|
|
238
|
+
The fourth is the answer to the third: static copies activations in and out of buffers the graph
|
|
239
239
|
already refers to, so a replay serves capture and steering without a Python forward. It is the path
|
|
240
|
-
`
|
|
240
|
+
`static_points="auto"` takes in production, and the row exists to price it against the
|
|
241
241
|
`enforce_eager=True` column capture would otherwise have to use.
|
|
242
242
|
|
|
243
243
|
`"auto"` once installed **reads** only, and a steering op needs a write tap to land in, so this row
|
|
244
244
|
priced half the feature and reported the other half as `n/a` -- with a message that blamed graph
|
|
245
|
-
replay for it, which is the thing
|
|
246
|
-
cell would be a number either way, and this row still passes `
|
|
245
|
+
replay for it, which is the thing static exists to work around. Auto now covers both halves, so the
|
|
246
|
+
cell would be a number either way, and this row still passes `static_writes` on purpose: an explicit
|
|
247
247
|
list *narrows* what auto would install, to the one mid-stack site the `steer` workload actually
|
|
248
248
|
writes. A row that priced a write buffer at every layer would not be comparable with the ones beside
|
|
249
249
|
it, which is the whole job of the column. Its value is the sentinel `run_bench.STEER_WRITES` rather
|
|
250
|
-
than a site, because that layer differs per model and a
|
|
250
|
+
than a site, because that layer differs per model and a static write is a `load_model` argument, so
|
|
251
251
|
it has to be resolved from the config before a model exists to ask.
|
|
252
252
|
|
|
253
|
-
`VariantSpec.models` restricts the row to the checkpoints
|
|
253
|
+
`VariantSpec.models` restricts the row to the checkpoints static has been shown correct on, so a model
|
|
254
254
|
missing from it renders `--` rather than a number nobody checked.
|
|
255
255
|
|
|
256
|
-
One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk
|
|
257
|
-
`resid_streams` — the whole stack of four parallel streams per layer — and a
|
|
258
|
-
set it
|
|
259
|
-
That row's
|
|
260
|
-
`ModelSpec.
|
|
256
|
+
One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk declares
|
|
257
|
+
`resid_streams` — the whole stack of four parallel streams per layer — and a static engine serves the
|
|
258
|
+
set it declared, so `deepseek-v4-flash-0731` cannot be asked for the `mlp_out` its other columns capture.
|
|
259
|
+
That row's static cell therefore prices the stack where every other cell prices one row, declared in
|
|
260
|
+
`ModelSpec.static_capture_point`, stated by the report under *Where a row differs*, and carried onto
|
|
261
261
|
the visualizer's card as one line of that row's footnote.
|
|
262
262
|
|
|
263
263
|
## Reading the numbers honestly
|
|
@@ -50,21 +50,21 @@ class ModelSpec:
|
|
|
50
50
|
sweep's shared budget, which is all but the largest.
|
|
51
51
|
|
|
52
52
|
Separate from :attr:`extra_vllm_kwargs` because that is a fact about the weights and applies to
|
|
53
|
-
every vLLM column, so putting a
|
|
53
|
+
every vLLM column, so putting a static-only budget there would change three cells that have
|
|
54
54
|
already been measured and are comparable as they stand. The overrides here are recorded in the cell
|
|
55
55
|
and stated by the report, so the one cell that ran on a different budget says so."""
|
|
56
|
-
|
|
57
|
-
"""The point those workloads address instead when the variant under test
|
|
56
|
+
static_capture_point: str | None = None
|
|
57
|
+
"""The point those workloads address instead when the variant under test declared a set, for a
|
|
58
58
|
checkpoint where the two cannot be the same point. ``None`` means :attr:`capture_point` serves both.
|
|
59
59
|
|
|
60
|
-
A
|
|
61
|
-
name a point the
|
|
62
|
-
``
|
|
60
|
+
A static engine serves *the set it declared* and refuses anything outside it, so the workload has to
|
|
61
|
+
name a point the static set covers. On a hyper-connection trunk it cannot be the same one:
|
|
62
|
+
``static_points="auto"`` resolves to ``resid_streams`` -- the whole stack -- while the other columns
|
|
63
63
|
need one ``d_model`` row to stay comparable with the rest of the table. Declared here rather than
|
|
64
64
|
inferred at run time so the cell records it, ``report_bench`` states it under *Where a row differs*,
|
|
65
65
|
and ``cells.nonuniform`` carries it onto the visualizer's card as one line of that row's footnote.
|
|
66
66
|
|
|
67
|
-
Only the
|
|
67
|
+
Only the static cell records it, so read it through ``cells.row_spec`` rather than off a cell.
|
|
68
68
|
"""
|
|
69
69
|
extra_eager_kwargs: dict[str, object] = field(default_factory=dict)
|
|
70
70
|
"""``load_model`` arguments the eager variants need for this checkpoint, merged under the
|
|
@@ -138,15 +138,15 @@ MODELS: tuple[ModelSpec, ...] = (
|
|
|
138
138
|
# bytes, which would price this row's transport as though DeepSeek moved four times the
|
|
139
139
|
# activations to answer the same question.
|
|
140
140
|
capture_point="mlp_out",
|
|
141
|
-
# ...except under
|
|
141
|
+
# ...except under static, which serves the set it declared: `"auto"` on this trunk declares
|
|
142
142
|
# `resid_streams`, so `mlp_out` is outside the set and would be refused. That column therefore
|
|
143
143
|
# prices the stack rather than a row, which is declared as a row exception instead of hidden.
|
|
144
|
-
|
|
145
|
-
# The
|
|
144
|
+
static_capture_point="resid_streams",
|
|
145
|
+
# The static column is the one configuration here that does not fit the shared budget, and both
|
|
146
146
|
# of these are why. 149 GiB of weights at 0.95 of a 178 GiB card leave about 15 GiB for the
|
|
147
|
-
# activation peak, the
|
|
147
|
+
# activation peak, the static buffers, the KV pool and the graph pool together.
|
|
148
148
|
#
|
|
149
|
-
# A
|
|
149
|
+
# A static buffer is allocated per batched row, so at the 4096 `max_num_batched_tokens` static
|
|
150
150
|
# lowered vLLM's default to, `resid_streams` costs 4096 rows x 4 streams x 4096 wide x 2 bytes
|
|
151
151
|
# x 43 layers = 5.8 GiB -- and vLLM then sized the KV pool from what was left and refused to
|
|
152
152
|
# start: "No available memory for the cache blocks". 1024 rows costs a quarter of that and is
|
|
@@ -157,7 +157,7 @@ MODELS: tuple[ModelSpec, ...] = (
|
|
|
157
157
|
# sizes it will actually replay keeps the rest of that memory, and the startup time, unspent.
|
|
158
158
|
# Neither changes what the workloads measure: both sizes they use are still captured.
|
|
159
159
|
per_variant_vllm_kwargs={
|
|
160
|
-
"vllm-
|
|
160
|
+
"vllm-static": {
|
|
161
161
|
"max_num_batched_tokens": 1024,
|
|
162
162
|
"compilation_config": {"cudagraph_capture_sizes": [1, 2, 4, 8, 16, 32]},
|
|
163
163
|
}
|
|
@@ -215,12 +215,17 @@ class VariantSpec:
|
|
|
215
215
|
|
|
216
216
|
|
|
217
217
|
#: One of these three exists to price a default that the engine chose for capture's sake, which is the
|
|
218
|
-
#: most useful thing a speed benchmark of this library can say. vLLM capture
|
|
219
|
-
#:
|
|
220
|
-
#:
|
|
221
|
-
#: ``vllm-cudagraph`` -- vLLM left at its own defaults, hence the
|
|
222
|
-
#: it measures what ours costs. The capture workloads run there too
|
|
223
|
-
#: report can show *what* a capture returns under replay instead of
|
|
218
|
+
#: most useful thing a speed benchmark of this library can say. Hooked vLLM capture rules CUDA graphs
|
|
219
|
+
#: out, because graph replay does not re-execute the Python forward and so never fires a
|
|
220
|
+
#: ``register_forward_hook``. That is what separates the three vLLM backends, and it is why
|
|
221
|
+
#: ``vllm-cudagraph`` -- ``backend="vllm-generate"``, vLLM left at its own defaults, hence the
|
|
222
|
+
#: "vanilla" label -- is here at all: it measures what ours costs. The capture workloads run there too
|
|
223
|
+
#: rather than being skipped, so the report can show *what* a capture returns under replay instead of
|
|
224
|
+
#: asserting that it fails.
|
|
225
|
+
#:
|
|
226
|
+
#: Each variant names its backend rather than deriving it from the kwargs it passes, which is what
|
|
227
|
+
#: the engine now expects: ``vllm-static`` and ``vllm-generate`` refuse ``enforce_eager``, and a tap
|
|
228
|
+
#: set is only accepted by the backend built to bake one in.
|
|
224
229
|
VARIANTS: tuple[VariantSpec, ...] = (
|
|
225
230
|
VariantSpec(
|
|
226
231
|
"eager",
|
|
@@ -232,52 +237,52 @@ VARIANTS: tuple[VariantSpec, ...] = (
|
|
|
232
237
|
VariantSpec(
|
|
233
238
|
"vllm",
|
|
234
239
|
"vllm",
|
|
235
|
-
{
|
|
240
|
+
{},
|
|
236
241
|
"capture-capable: CUDA graphs and compile OFF",
|
|
237
242
|
"interp-engine vllm",
|
|
238
243
|
),
|
|
239
244
|
VariantSpec(
|
|
240
245
|
"vllm-cudagraph",
|
|
241
|
-
"vllm",
|
|
242
|
-
{
|
|
243
|
-
"CUDA graphs + inductor compile ON, no
|
|
246
|
+
"vllm-generate",
|
|
247
|
+
{},
|
|
248
|
+
"CUDA graphs + inductor compile ON, no static wraps; vLLM's own defaults for generate-only",
|
|
244
249
|
"vllm (vanilla)",
|
|
245
250
|
),
|
|
246
|
-
# Breakable CUDA graphs with resid_post
|
|
247
|
-
# (``
|
|
251
|
+
# Breakable CUDA graphs with resid_post static wraps at every layer. Production static
|
|
252
|
+
# (``static_points="auto"``) is this path, not Dynamo piecewise.
|
|
248
253
|
#
|
|
249
|
-
# `qwen3.8-27b` joined the row once
|
|
254
|
+
# `qwen3.8-27b` joined the row once static was correct on a hybrid trunk. Breakable graphs turn
|
|
250
255
|
# inductor off, and vLLM's `FULL_AND_PIECEWISE` capture then miscomputes prefill for a gated-delta
|
|
251
256
|
# trunk -- the engine generated fluent nonsense rather than failing, so the cell would have been a
|
|
252
|
-
# plausible number for a broken forward.
|
|
253
|
-
# runs prefill eagerly and keeps the decode graphs, and the validator's
|
|
257
|
+
# plausible number for a broken forward. Static now pins `FULL_DECODE_ONLY` on such a trunk, which
|
|
258
|
+
# runs prefill eagerly and keeps the decode graphs, and the validator's static column agrees with
|
|
254
259
|
# eager on all 28 points. Its prefill figures carry that eager prefill, which is the point of
|
|
255
260
|
# comparing it against the same model's other variants rather than against another model.
|
|
256
261
|
#
|
|
257
|
-
# `deepseek-v4-flash-0731` is in, and is the one row here whose
|
|
262
|
+
# `deepseek-v4-flash-0731` is in, and is the one row here whose static set is not one point per
|
|
258
263
|
# layer. It is a hyper-connection trunk, so `"auto"` resolves to `resid_streams` -- the whole stack
|
|
259
264
|
# of four parallel residual streams per layer, four times the width of a `resid_post` row. Its
|
|
260
265
|
# capture and transport figures therefore price four times the activations for the same question,
|
|
261
266
|
# which `cells.nonuniform` declares so the report states it and the visualizer's card drops the row
|
|
262
|
-
# rather than publishing it beside three that
|
|
267
|
+
# rather than publishing it beside three that declared a quarter as much.
|
|
263
268
|
#
|
|
264
269
|
# Its write tap is `mlp_out`, not `resid_post`: the steer workload addresses the model's
|
|
265
270
|
# `capture_point`, and a hyper-connection trunk refuses the default name (`run_bench._steer_site`).
|
|
266
271
|
#
|
|
267
|
-
# `
|
|
272
|
+
# `static_writes` was once what made the `steer` cell a number instead of `n/a`: `"auto"` installed
|
|
268
273
|
# reads, and a steering op needs a write tap to land in, so this row priced capture under replay
|
|
269
274
|
# and left the other half of the feature unmeasured -- with a message that blamed graph replay for
|
|
270
275
|
# it. Auto covers writes now, so the cell stands either way, and naming them here has become a
|
|
271
276
|
# *narrowing*: one write buffer at the site the workload steers rather than one per layer, which
|
|
272
277
|
# is what keeps this row's memory comparable with the columns beside it. The value is the sentinel
|
|
273
278
|
# `run_bench.STEER_WRITES`, resolved there to the mid-stack `resid_post` the workload steers,
|
|
274
|
-
# because the layer differs per model and a
|
|
279
|
+
# because the layer differs per model and a static write has to be named before the model exists.
|
|
275
280
|
VariantSpec(
|
|
276
|
-
"vllm-
|
|
277
|
-
"vllm",
|
|
278
|
-
{"
|
|
279
|
-
"breakable CUDA graphs with resid_post
|
|
280
|
-
"interp-engine vllm
|
|
281
|
+
"vllm-static",
|
|
282
|
+
"vllm-static",
|
|
283
|
+
{"static_points": "auto", "static_writes": "steer"},
|
|
284
|
+
"breakable CUDA graphs with resid_post static wraps at every layer, and a write tap mid-stack",
|
|
285
|
+
"interp-engine vllm static",
|
|
281
286
|
models=("gemma-2-2b", "qwen3-4b", "llama-3.1-8b", "qwen3.8-27b", "deepseek-v4-flash-0731"),
|
|
282
287
|
),
|
|
283
288
|
# DSpark on, against the `vllm` column with it off -- the pair is the measurement, so read the two
|
|
@@ -303,12 +308,11 @@ VARIANTS: tuple[VariantSpec, ...] = (
|
|
|
303
308
|
"vllm-dspark",
|
|
304
309
|
"vllm",
|
|
305
310
|
{
|
|
306
|
-
"enforce_eager": True,
|
|
307
311
|
"extra_vllm_kwargs": {
|
|
308
312
|
"speculative_config": {"method": "dspark", "num_speculative_tokens": 5},
|
|
309
313
|
},
|
|
310
314
|
},
|
|
311
|
-
"DSpark speculative decoding ON;
|
|
315
|
+
"DSpark speculative decoding ON; hooked backend, so still capture-capable",
|
|
312
316
|
"interp-engine vllm +DSpark",
|
|
313
317
|
models=("deepseek-v4-flash-0731",),
|
|
314
318
|
),
|
|
@@ -325,10 +329,8 @@ VARIANTS: tuple[VariantSpec, ...] = (
|
|
|
325
329
|
# never calls the Python forward a hook is attached to.
|
|
326
330
|
VariantSpec(
|
|
327
331
|
"vllm-dspark-cudagraph",
|
|
328
|
-
"vllm",
|
|
332
|
+
"vllm-generate",
|
|
329
333
|
{
|
|
330
|
-
"enforce_eager": False,
|
|
331
|
-
"freeze_points": [],
|
|
332
334
|
"extra_vllm_kwargs": {
|
|
333
335
|
"speculative_config": {"method": "dspark", "num_speculative_tokens": 5},
|
|
334
336
|
},
|
|
@@ -77,9 +77,9 @@ def row_spec(cells: list[dict[str, Any]], model_key: str, variants: Iterable[str
|
|
|
77
77
|
"""Every override one model's row declared, merged across the row's cells.
|
|
78
78
|
|
|
79
79
|
A per-variant override is recorded by the cell that used it and by no other: on
|
|
80
|
-
`deepseek-v4-flash-0731` both `
|
|
81
|
-
|
|
82
|
-
one -- whichever happened to sort last -- silently dropped the two overrides the
|
|
80
|
+
`deepseek-v4-flash-0731` both `static_capture_point` and `per_variant_vllm_kwargs` live on the
|
|
81
|
+
static cell alone. So any single cell's ``model`` describes a column rather than a row, and picking
|
|
82
|
+
one -- whichever happened to sort last -- silently dropped the two overrides the static column is
|
|
83
83
|
the only cell to declare.
|
|
84
84
|
|
|
85
85
|
``variants`` restricts the merge to the columns a renderer shows, and each of them passes its own:
|
|
@@ -129,9 +129,9 @@ def nonuniform(model: dict[str, Any]) -> list[str]:
|
|
|
129
129
|
point = model.get("capture_point", DEFAULT_POINT)
|
|
130
130
|
if point != DEFAULT_POINT:
|
|
131
131
|
reasons.append(f"captures `{point}` rather than `{DEFAULT_POINT}`")
|
|
132
|
-
|
|
133
|
-
if
|
|
134
|
-
reasons.append(f"its
|
|
132
|
+
static_point = model.get("static_capture_point")
|
|
133
|
+
if static_point:
|
|
134
|
+
reasons.append(f"its static column captures `{static_point}`, a wider point than the others do")
|
|
135
135
|
fraction = model.get("gpu_memory_utilization")
|
|
136
136
|
if fraction:
|
|
137
137
|
reasons.append(f"vLLM reserved {fraction} of the card, not the uniform {GPU_MEMORY_UTILIZATION}")
|
|
@@ -6,7 +6,7 @@ neither: it drives :meth:`VLLMModel.lens_capture_readout_stream`, which issues o
|
|
|
6
6
|
``collective_rpc`` per yield from ``engine.generate``, and nothing so far says how many of those
|
|
7
7
|
there are per generated token or what share of the request they are.
|
|
8
8
|
|
|
9
|
-
python -m benchmarks.probe_lens_stream # qwen3.8-27b,
|
|
9
|
+
python -m benchmarks.probe_lens_stream # qwen3.8-27b, static, 512 -> 64
|
|
10
10
|
python -m benchmarks.probe_lens_stream --variant hooked # the same without CUDA graphs
|
|
11
11
|
python -m benchmarks.probe_lens_stream --no-jacobians # skip the transport (tight card)
|
|
12
12
|
|
|
@@ -16,8 +16,8 @@ means the engine already outruns the read-out -- which the method's own docstrin
|
|
|
16
16
|
which nothing had measured -- and the round trips are not where the time is.
|
|
17
17
|
|
|
18
18
|
The second figure to read is *time to first read-out* against the baseline's time to first
|
|
19
|
-
token. Everything between them is the prompt-wide capture: one
|
|
20
|
-
over every prompt position, then transported and unembedded. On a trunk where ``"auto"``
|
|
19
|
+
token. Everything between them is the prompt-wide capture: one static site per layer harvested
|
|
20
|
+
over every prompt position, then transported and unembedded. On a trunk where ``"auto"`` declares
|
|
21
21
|
a stream stack that term dominates, so the split matters before optimizing either half.
|
|
22
22
|
|
|
23
23
|
Deliberately one process and one model, like ``run_bench``: vLLM reserves its fraction of the
|
|
@@ -428,7 +428,7 @@ class Args:
|
|
|
428
428
|
top_n: int = ENDPOINT_TOP_N
|
|
429
429
|
chunk_positions: int = ENDPOINT_CHUNK_POSITIONS
|
|
430
430
|
point: str = "resid_post"
|
|
431
|
-
variant: str = "
|
|
431
|
+
variant: str = "static"
|
|
432
432
|
word_mask: bool = True
|
|
433
433
|
jacobians: bool = True
|
|
434
434
|
gpu_memory_utilization: float | None = None
|
|
@@ -448,7 +448,7 @@ def parse_args(argv: list[str]) -> Args:
|
|
|
448
448
|
p.add_argument("--top-n", type=int, default=ENDPOINT_TOP_N)
|
|
449
449
|
p.add_argument("--chunk-positions", type=int, default=ENDPOINT_CHUNK_POSITIONS)
|
|
450
450
|
p.add_argument("--point", default="resid_post", help="capture point; resid_streams on a hyper-connection trunk")
|
|
451
|
-
p.add_argument("--variant", default="
|
|
451
|
+
p.add_argument("--variant", default="static", choices=("static", "hooked"), help="static = production")
|
|
452
452
|
p.add_argument("--no-word-mask", dest="word_mask", action="store_false")
|
|
453
453
|
p.add_argument("--no-jacobians", dest="jacobians", action="store_false")
|
|
454
454
|
p.add_argument("--gpu-memory-utilization", type=float, default=None)
|
|
@@ -494,15 +494,19 @@ async def run(args: Args) -> dict[str, Any]:
|
|
|
494
494
|
"gpu_memory_utilization": utilization,
|
|
495
495
|
"extra_vllm_kwargs": extra_vllm,
|
|
496
496
|
}
|
|
497
|
-
if args.variant == "
|
|
498
|
-
|
|
499
|
-
#
|
|
500
|
-
#
|
|
501
|
-
|
|
497
|
+
if args.variant == "static":
|
|
498
|
+
backend = "vllm-static"
|
|
499
|
+
# What production takes with STATIC_POINTS=auto. Passed explicitly even though the backend
|
|
500
|
+
# would default to it, so the run records which set it measured. No `static_writes`: the
|
|
501
|
+
# request measured here carries no intervention, so a write tap would be an unused buffer
|
|
502
|
+
# competing for the same pool the graphs and the KV cache come out of.
|
|
503
|
+
kwargs["static_points"] = "auto"
|
|
504
|
+
else:
|
|
505
|
+
backend = "vllm"
|
|
502
506
|
|
|
503
507
|
print(f"loading {hf_id} ({args.variant})...", flush=True)
|
|
504
508
|
timer = Timer()
|
|
505
|
-
model = load_model(hf_id, backend=
|
|
509
|
+
model = load_model(hf_id, backend=backend, dtype=dtype, **kwargs)
|
|
506
510
|
construct_s = timer.elapsed()
|
|
507
511
|
timer.reset()
|
|
508
512
|
await model.warmup()
|
|
@@ -567,7 +571,7 @@ async def run(args: Args) -> dict[str, Any]:
|
|
|
567
571
|
|
|
568
572
|
setup = {
|
|
569
573
|
"model": f"{hf_id} ({model.n_layers} layers, d_model {model.d_model})",
|
|
570
|
-
"variant": f"{args.variant}" + (" (
|
|
574
|
+
"variant": f"{args.variant}" + (" (static_points=auto)" if args.variant == "static" else ""),
|
|
571
575
|
"request": f"{len(prompt_ids)} prompt tokens -> {args.new_tokens} new, top-{args.top_n}",
|
|
572
576
|
"read-out": f"{len(specs)} type(s) x {len(layers)} layers, chunk_positions={args.chunk_positions}",
|
|
573
577
|
"word mask": f"{word_mask.shape[0]:,} ids" if word_mask is not None else "off",
|
|
@@ -50,13 +50,13 @@ START = "<!-- THROUGHPUT:START -->"
|
|
|
50
50
|
END = "<!-- THROUGHPUT:END -->"
|
|
51
51
|
|
|
52
52
|
#: ``(variant key, README heading, field name in the visualizer's `Benchmark`)``. Three of the sweep's
|
|
53
|
-
#: six variants: the two capture-capable backends the engine offers, plus
|
|
53
|
+
#: six variants: the two capture-capable backends the engine offers, plus static taps, which is the
|
|
54
54
|
#: claim this section exists to make. `vllm-cudagraph` is vanilla vLLM and cannot capture at all, so
|
|
55
55
|
#: publishing it beside these would invite a comparison the engine is not making.
|
|
56
56
|
COLUMNS: tuple[tuple[str, str, str], ...] = (
|
|
57
57
|
("eager", "eager", "eager"),
|
|
58
58
|
("vllm", "vLLM", "vllm"),
|
|
59
|
-
("vllm-
|
|
59
|
+
("vllm-static", "vLLM + static taps", "static"),
|
|
60
60
|
)
|
|
61
61
|
|
|
62
62
|
#: ``(field name in a `Rate`, workload, metric)``.
|
|
@@ -68,7 +68,7 @@ REGIMES: tuple[tuple[str, str, str], ...] = (
|
|
|
68
68
|
#: The column every multiplier is against, and the one that carries no multiplier of its own.
|
|
69
69
|
BASELINE = "eager"
|
|
70
70
|
#: Bolded in the README, being the number the section is about.
|
|
71
|
-
HIGHLIGHT = "vllm-
|
|
71
|
+
HIGHLIGHT = "vllm-static"
|
|
72
72
|
|
|
73
73
|
MISSING_CELL = "—"
|
|
74
74
|
|
|
@@ -223,7 +223,7 @@ def card_models(cells: list[dict[str, Any]]) -> dict[str, list[str]]:
|
|
|
223
223
|
This used to drop such a row instead. Being unpublishable and being unexplained are not the same
|
|
224
224
|
problem, and conflating them cost the card its most interesting row -- `deepseek-v4-flash-0731`
|
|
225
225
|
reserves its own fraction of the GPU and serves an FP8 KV cache, and also happens to be where
|
|
226
|
-
|
|
226
|
+
static wins by the largest margin in the sweep. What made the old rule right was that the card had
|
|
227
227
|
nowhere to state an exception; the fix is the footnote, not the omission.
|
|
228
228
|
|
|
229
229
|
One reason to drop a row remains, and prints a line on the console: a model missing either baseline
|
|
@@ -44,11 +44,11 @@ REPORT_PATH = Path(__file__).resolve().parent / "results-latest.md"
|
|
|
44
44
|
#: Ordered here rather than by ``cells.variant_order``'s spec order because the spec is ordered by how
|
|
45
45
|
#: the variants relate to each other -- vanilla vLLM next to the capture-capable one it is the control
|
|
46
46
|
#: for -- while a reader of the tables wants the engine's three own configurations adjacent and the
|
|
47
|
-
#: outside reference last, so `vllm-
|
|
47
|
+
#: outside reference last, so `vllm-static` sits beside `vllm` rather than a column away from it.
|
|
48
48
|
#:
|
|
49
49
|
#: A present variant that is in neither tuple is appended rather than dropped, so adding one to
|
|
50
50
|
#: ``bench_spec`` cannot silently produce a report that does not mention it.
|
|
51
|
-
COLUMNS: tuple[str, ...] = ("eager", "vllm", "vllm-
|
|
51
|
+
COLUMNS: tuple[str, ...] = ("eager", "vllm", "vllm-static", "vllm-cudagraph")
|
|
52
52
|
|
|
53
53
|
#: Variants whose cells this report does not render, and why.
|
|
54
54
|
#:
|
|
@@ -281,11 +281,11 @@ def _exceptions_section(cells: list[dict[str, Any]]) -> list[str]:
|
|
|
281
281
|
f"the capture and steering workloads address `{point}`, not `{DEFAULT_POINT}` "
|
|
282
282
|
"(see `bench_spec.py` for why this architecture has no such point)"
|
|
283
283
|
)
|
|
284
|
-
|
|
285
|
-
if
|
|
284
|
+
static_point = m.get("static_capture_point")
|
|
285
|
+
if static_point:
|
|
286
286
|
notes.append(
|
|
287
|
-
f"under
|
|
288
|
-
f'engine serves the set it
|
|
287
|
+
f"under static those workloads address `{static_point}` instead, because a static "
|
|
288
|
+
f'engine serves the set it declared and `"auto"` on this trunk declares `{static_point}` '
|
|
289
289
|
"-- the whole stack of parallel residual streams per layer, so that one column's "
|
|
290
290
|
"capture and transport figures price the stack where every other column prices a row"
|
|
291
291
|
)
|
|
@@ -446,7 +446,7 @@ def build_report(cells: list[dict[str, Any]], sweep_command: str) -> str:
|
|
|
446
446
|
"",
|
|
447
447
|
"`steer` and `capture_gen` are the same workload apart from the steering spec, so the difference",
|
|
448
448
|
"is the cost of the mechanism: one write hook eagerly, on vLLM an extra `collective_rpc` to",
|
|
449
|
-
"install it in the worker, and under
|
|
449
|
+
"install it in the worker, and under static an add into a buffer the graph already refers to,",
|
|
450
450
|
"which is why that column lands inside the noise floor in both directions.",
|
|
451
451
|
"",
|
|
452
452
|
*_steering_overhead(cells),
|
|
@@ -22,11 +22,9 @@
|
|
|
22
22
|
},
|
|
23
23
|
"variant": {
|
|
24
24
|
"key": "vllm-cudagraph",
|
|
25
|
-
"backend": "vllm",
|
|
26
|
-
"note": "CUDA graphs + inductor compile ON, no
|
|
25
|
+
"backend": "vllm-generate",
|
|
26
|
+
"note": "CUDA graphs + inductor compile ON, no static wraps; vLLM's own defaults for generate-only",
|
|
27
27
|
"kwargs": {
|
|
28
|
-
"enforce_eager": "False",
|
|
29
|
-
"freeze_points": "[]",
|
|
30
28
|
"max_model_len": "2048",
|
|
31
29
|
"gpu_memory_utilization": "0.95",
|
|
32
30
|
"extra_vllm_kwargs": "{'kv_cache_dtype': 'fp8', 'enable_prefix_caching': False}"
|