interp-engine 1.2.8__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {interp_engine-1.2.8 → interp_engine-1.3.0}/PKG-INFO +47 -33
- {interp_engine-1.2.8 → interp_engine-1.3.0}/README.md +46 -32
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/README.md +15 -15
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/bench_spec.py +44 -42
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/cells.py +6 -6
- interp_engine-1.3.0/benchmarks/probe_lens_stream.py +626 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/publish.py +4 -4
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/report_bench.py +7 -7
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-cudagraph.json +2 -4
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark-cudagraph.json +1 -3
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark.json +0 -1
- interp_engine-1.2.8/benchmarks/results/deepseek-v4-flash-0731__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/deepseek-v4-flash-0731__vllm-static.json +8 -8
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm.json +0 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.8/benchmarks/results/gemma-2-2b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/gemma-2-2b__vllm-static.json +6 -6
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__vllm.json +0 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.8/benchmarks/results/llama-3.1-8b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/llama-3.1-8b__vllm-static.json +6 -6
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__vllm.json +0 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.8/benchmarks/results/qwen3-4b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/qwen3-4b__vllm-static.json +6 -6
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__vllm.json +0 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__vllm-cudagraph.json +2 -4
- interp_engine-1.2.8/benchmarks/results/qwen3.8-27b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/qwen3.8-27b__vllm-static.json +6 -6
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__vllm.json +0 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results-latest.md +14 -14
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/run_bench.py +15 -10
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/workloads.py +13 -13
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/AGENT_INTEGRATION.md +46 -4
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/PERFORMANCE.md +54 -47
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/USAGE.md +11 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/__init__.py +2 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/_loop.py +47 -4
- interp_engine-1.3.0/interp_engine/load.py +216 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/model.py +2 -2
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/protocol.py +13 -7
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/sync.py +4 -4
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_backend.py +369 -244
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/__init__.py +25 -25
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_hooks.py +2 -2
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/graphs.py +5 -4
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/readout.py +22 -22
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/mhc.py +1 -1
- interp_engine-1.2.8/interp_engine/vllm_capture/freeze.py → interp_engine-1.3.0/interp_engine/vllm_capture/static.py +217 -214
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_plugin.py +25 -25
- {interp_engine-1.2.8 → interp_engine-1.3.0}/pyproject.toml +2 -2
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_bench_workloads.py +14 -14
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_layer_kinds.py +1 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_load.py +46 -10
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_published_benchmarks.py +8 -8
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_qk_norm.py +1 -1
- interp_engine-1.2.8/tests/test_freeze_dsv4_gpu.py → interp_engine-1.3.0/tests/test_static_dsv4_gpu.py +17 -16
- interp_engine-1.2.8/tests/test_freeze_parity_gpu.py → interp_engine-1.3.0/tests/test_static_parity_gpu.py +51 -48
- interp_engine-1.2.8/tests/test_freeze_set.py → interp_engine-1.3.0/tests/test_static_set.py +113 -113
- interp_engine-1.2.8/tests/test_freeze_warmup.py → interp_engine-1.3.0/tests/test_static_warmup.py +17 -17
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sync_loop.py +46 -1
- interp_engine-1.3.0/tests/test_vllm_engine_loop.py +101 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_graph_path.py +2 -1
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_graphs_on_gpu.py +26 -31
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_hook_availability.py +57 -32
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_hyper_connections.py +97 -97
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_worker_lens_capture_readout.py +11 -11
- interp_engine-1.2.8/interp_engine/load.py +0 -152
- {interp_engine-1.2.8 → interp_engine-1.3.0}/.gitignore +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/LICENSE +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/__init__.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/probe.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__eager.json +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__eager.json +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__eager.json +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__eager.json +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__eager.json +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/run_all.sh +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/ARCHITECTURE_QUIRKS.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/COMPATIBILITY.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/ENGINE_HOOK_MAPPINGS.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/GRADIENTS.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/INTERNALS.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/PORTING.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/README.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/SUPPORTED_POINTS.md +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/address.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/arch.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/attn_config.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/attn_scores.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/autograd_support.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/capture.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/chat_compose.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/chat_conventions.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/chat_formatters.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/cuda_preflight.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/dispatch.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/facts.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/hooks.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/lens.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/mappers.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/moe_routing.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/points.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/residual_basis.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/select.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/steer.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/steer_specs.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/tokenize.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_demux.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_payload.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_tree.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/attn.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/capture.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/__init__.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/intervene.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/unembed.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/native.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/requests.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/steering.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/conftest.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/harness.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/model_expectations.yaml +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/synthetic_families.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_address.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_config_tripwire.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_probs_indexing.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_scores.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_z_gqa.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_autograd_support.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_capability_refusals.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_capture_addressing.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_chat_compose.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_chat_formatters.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_chat_templates.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_core.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_cuda_preflight.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_doc_code_fences.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_eager_autograd.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_facts.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_family_points.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_gated_attn_out.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_head_contributions.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_hook_call_conventions.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_logit_transform.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_mappers.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_mlp_internals.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_model_expectations.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_moe.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_multimodal_arch.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_new_models_gpu.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_no_chat_template.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_normalized_hook.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_packaging.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_parity_gpt2.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_per_layer_attn_dims.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_points_registry.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_protocol.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_qkv_layout.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_reasoning_spans.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_release.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_resid_mid.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_residual_basis.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sandwich_norms.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_select.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sliding_window_attn.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_small_models_gpu.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_steer_context.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_steer_math_parity.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sync_parity.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_unified_free_functions.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_unresolved_families.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_capture_gpu.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_capture_scales.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_kv_isolation.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_new_points.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_only_families.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_plugin.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_wire_grammar.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vocabulary_boundary.py +0 -0
- {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_worker_lens_readout.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: interp-engine
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: A fast, standardized interpretability engine that supports most modern models and architectures. Powers Neuronpedia.
|
|
5
5
|
Project-URL: Homepage, https://github.com/decoderesearch/interp-engine
|
|
6
6
|
Project-URL: Repository, https://github.com/decoderesearch/interp-engine
|
|
@@ -31,22 +31,33 @@ Description-Content-Type: text/markdown
|
|
|
31
31
|
|
|
32
32
|
# interp-engine
|
|
33
33
|
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
34
|
+
<p align="center">
|
|
35
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
|
|
36
|
+
</p>
|
|
37
|
+
<p align="center">
|
|
38
|
+
🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
|
|
39
|
+
</p>
|
|
40
|
+
<p align="center">
|
|
41
|
+
<a href="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml"><img src="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml/badge.svg?branch=main" alt="CI status"></a>
|
|
42
|
+
<a href="https://pypi.org/project/interp-engine/"><img src="https://img.shields.io/pypi/v/interp-engine.svg" alt="PyPI version"></a>
|
|
43
|
+
<a href="LICENSE"><img src="https://img.shields.io/pypi/l/interp-engine.svg" alt="Apache-2.0 license"></a>
|
|
44
|
+
<a href="https://join.slack.com/t/opensourcemechanistic/shared_invite/zt-3z9o0hxjl-MDX9pbATO2qESOazNDLpdQ"><img src="https://img.shields.io/badge/Slack-Open%20Source%20Mechanistic%20Interpretability-4A154B?logo=slack&logoColor=white" alt="Join the Slack"></a>
|
|
45
|
+
</p>
|
|
38
46
|
|
|
39
47
|
|
|
40
48
|
`interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference and is checked for accuracy against HF Transformers and other engines.
|
|
41
49
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
50
|
+
<p align="center">
|
|
51
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-static on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="100%">
|
|
52
|
+
</p>
|
|
53
|
+
<p align="center">
|
|
54
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="100%">
|
|
55
|
+
</p>
|
|
45
56
|
|
|
46
57
|
This repo contains:
|
|
47
58
|
|
|
48
|
-
1. `
|
|
49
|
-
2. `
|
|
59
|
+
1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
|
|
60
|
+
2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
|
|
50
61
|
|
|
51
62
|
## Installation
|
|
52
63
|
|
|
@@ -60,13 +71,16 @@ pip install interp-engine # eager backend only
|
|
|
60
71
|
```python
|
|
61
72
|
from interp_engine import Address, load_model, run_with_cache
|
|
62
73
|
|
|
63
|
-
# VLLM
|
|
64
|
-
model = load_model("Qwen/Qwen3-8B")
|
|
74
|
+
# VLLM (default): low VRAM, medium speed, every point, chosen per request
|
|
75
|
+
model = load_model("Qwen/Qwen3-8B")
|
|
76
|
+
|
|
77
|
+
# VLLM-STATIC: high VRAM, high speed, only the points you declare (default resid_post)
|
|
78
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-static")
|
|
65
79
|
|
|
66
|
-
# VLLM-
|
|
67
|
-
# model = load_model("Qwen/Qwen3-8B",
|
|
80
|
+
# VLLM-GENERATE: fastest, generation only -- no capture, no steering
|
|
81
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-generate")
|
|
68
82
|
|
|
69
|
-
# EAGER
|
|
83
|
+
# EAGER: low VRAM, low speed
|
|
70
84
|
# model = load_model("Qwen/Qwen3-8B", backend="eager")
|
|
71
85
|
|
|
72
86
|
point = Address("resid_post", 10) # or string: "resid_post.10"
|
|
@@ -84,7 +98,7 @@ Add "use interp-engine" to your prompt and let your agent figure it out - everyt
|
|
|
84
98
|
|
|
85
99
|
## Performance / Speed
|
|
86
100
|
|
|
87
|
-
vLLM gives `interp-engine` high throughput via concurrency, and
|
|
101
|
+
vLLM gives `interp-engine` high throughput via concurrency, and `backend="vllm-static"` adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
|
|
88
102
|
|
|
89
103
|
<!-- THROUGHPUT:START -->
|
|
90
104
|
|
|
@@ -94,34 +108,34 @@ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
|
|
|
94
108
|
|
|
95
109
|
One stream (tok/s):
|
|
96
110
|
|
|
97
|
-
| model | eager | vLLM | vLLM +
|
|
98
|
-
| ------------------------ | ----- | ---------- |
|
|
99
|
-
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)**
|
|
100
|
-
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)**
|
|
101
|
-
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)**
|
|
102
|
-
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)**
|
|
103
|
-
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)**
|
|
111
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
112
|
+
| ------------------------ | ----- | ---------- | ------------------ |
|
|
113
|
+
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
|
|
114
|
+
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
|
|
115
|
+
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
|
|
116
|
+
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
|
|
117
|
+
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
|
|
104
118
|
|
|
105
119
|
8 concurrent requests (aggregate tok/s):
|
|
106
120
|
|
|
107
|
-
| model | eager | vLLM | vLLM +
|
|
108
|
-
| ------------------------ | ----- | ----------- |
|
|
109
|
-
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)**
|
|
110
|
-
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)**
|
|
111
|
-
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)**
|
|
112
|
-
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)**
|
|
113
|
-
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)**
|
|
121
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
122
|
+
| ------------------------ | ----- | ----------- | ------------------ |
|
|
123
|
+
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
|
|
124
|
+
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
|
|
125
|
+
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
|
|
126
|
+
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
|
|
127
|
+
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
|
|
114
128
|
|
|
115
129
|
<!-- THROUGHPUT:END -->
|
|
116
130
|
|
|
117
|
-
|
|
131
|
+
`backend="vllm-static"` is opt-in, and serves only the tap set it declared — `static_points="auto"` by default, or a list you name. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; [benchmarks/results-latest.md](benchmarks/results-latest.md) has every figure at full precision, including capture, steering and lens latencies; [benchmarks/README.md](benchmarks/README.md) has how the tables above are rounded.
|
|
118
132
|
|
|
119
133
|
## Correctness
|
|
120
134
|
|
|
121
135
|
We verify correctness in two main ways:
|
|
122
136
|
|
|
123
137
|
1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
|
|
124
|
-
2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at `
|
|
138
|
+
2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
|
|
125
139
|
|
|
126
140
|
## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
|
|
127
141
|
|
|
@@ -145,4 +159,4 @@ Bugs and feature requests belong in [issues](https://github.com/decoderesearch/i
|
|
|
145
159
|
|
|
146
160
|
## License
|
|
147
161
|
|
|
148
|
-
Apache 2.0
|
|
162
|
+
Apache 2.0
|
|
@@ -1,21 +1,32 @@
|
|
|
1
1
|
# interp-engine
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
3
|
+
<p align="center">
|
|
4
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
|
|
5
|
+
</p>
|
|
6
|
+
<p align="center">
|
|
7
|
+
🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
|
|
8
|
+
</p>
|
|
9
|
+
<p align="center">
|
|
10
|
+
<a href="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml"><img src="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml/badge.svg?branch=main" alt="CI status"></a>
|
|
11
|
+
<a href="https://pypi.org/project/interp-engine/"><img src="https://img.shields.io/pypi/v/interp-engine.svg" alt="PyPI version"></a>
|
|
12
|
+
<a href="LICENSE"><img src="https://img.shields.io/pypi/l/interp-engine.svg" alt="Apache-2.0 license"></a>
|
|
13
|
+
<a href="https://join.slack.com/t/opensourcemechanistic/shared_invite/zt-3z9o0hxjl-MDX9pbATO2qESOazNDLpdQ"><img src="https://img.shields.io/badge/Slack-Open%20Source%20Mechanistic%20Interpretability-4A154B?logo=slack&logoColor=white" alt="Join the Slack"></a>
|
|
14
|
+
</p>
|
|
7
15
|
|
|
8
16
|
|
|
9
17
|
`interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference and is checked for accuracy against HF Transformers and other engines.
|
|
10
18
|
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
19
|
+
<p align="center">
|
|
20
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-static on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="100%">
|
|
21
|
+
</p>
|
|
22
|
+
<p align="center">
|
|
23
|
+
<img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="100%">
|
|
24
|
+
</p>
|
|
14
25
|
|
|
15
26
|
This repo contains:
|
|
16
27
|
|
|
17
|
-
1. `
|
|
18
|
-
2. `
|
|
28
|
+
1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
|
|
29
|
+
2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
|
|
19
30
|
|
|
20
31
|
## Installation
|
|
21
32
|
|
|
@@ -29,13 +40,16 @@ pip install interp-engine # eager backend only
|
|
|
29
40
|
```python
|
|
30
41
|
from interp_engine import Address, load_model, run_with_cache
|
|
31
42
|
|
|
32
|
-
# VLLM
|
|
33
|
-
model = load_model("Qwen/Qwen3-8B")
|
|
43
|
+
# VLLM (default): low VRAM, medium speed, every point, chosen per request
|
|
44
|
+
model = load_model("Qwen/Qwen3-8B")
|
|
45
|
+
|
|
46
|
+
# VLLM-STATIC: high VRAM, high speed, only the points you declare (default resid_post)
|
|
47
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-static")
|
|
34
48
|
|
|
35
|
-
# VLLM-
|
|
36
|
-
# model = load_model("Qwen/Qwen3-8B",
|
|
49
|
+
# VLLM-GENERATE: fastest, generation only -- no capture, no steering
|
|
50
|
+
# model = load_model("Qwen/Qwen3-8B", backend="vllm-generate")
|
|
37
51
|
|
|
38
|
-
# EAGER
|
|
52
|
+
# EAGER: low VRAM, low speed
|
|
39
53
|
# model = load_model("Qwen/Qwen3-8B", backend="eager")
|
|
40
54
|
|
|
41
55
|
point = Address("resid_post", 10) # or string: "resid_post.10"
|
|
@@ -53,7 +67,7 @@ Add "use interp-engine" to your prompt and let your agent figure it out - everyt
|
|
|
53
67
|
|
|
54
68
|
## Performance / Speed
|
|
55
69
|
|
|
56
|
-
vLLM gives `interp-engine` high throughput via concurrency, and
|
|
70
|
+
vLLM gives `interp-engine` high throughput via concurrency, and `backend="vllm-static"` adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
|
|
57
71
|
|
|
58
72
|
<!-- THROUGHPUT:START -->
|
|
59
73
|
|
|
@@ -63,34 +77,34 @@ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
|
|
|
63
77
|
|
|
64
78
|
One stream (tok/s):
|
|
65
79
|
|
|
66
|
-
| model | eager | vLLM | vLLM +
|
|
67
|
-
| ------------------------ | ----- | ---------- |
|
|
68
|
-
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)**
|
|
69
|
-
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)**
|
|
70
|
-
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)**
|
|
71
|
-
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)**
|
|
72
|
-
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)**
|
|
80
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
81
|
+
| ------------------------ | ----- | ---------- | ------------------ |
|
|
82
|
+
| `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
|
|
83
|
+
| `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
|
|
84
|
+
| `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
|
|
85
|
+
| `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
|
|
86
|
+
| `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
|
|
73
87
|
|
|
74
88
|
8 concurrent requests (aggregate tok/s):
|
|
75
89
|
|
|
76
|
-
| model | eager | vLLM | vLLM +
|
|
77
|
-
| ------------------------ | ----- | ----------- |
|
|
78
|
-
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)**
|
|
79
|
-
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)**
|
|
80
|
-
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)**
|
|
81
|
-
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)**
|
|
82
|
-
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)**
|
|
90
|
+
| model | eager | vLLM | vLLM + static taps |
|
|
91
|
+
| ------------------------ | ----- | ----------- | ------------------ |
|
|
92
|
+
| `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
|
|
93
|
+
| `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
|
|
94
|
+
| `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
|
|
95
|
+
| `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
|
|
96
|
+
| `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
|
|
83
97
|
|
|
84
98
|
<!-- THROUGHPUT:END -->
|
|
85
99
|
|
|
86
|
-
|
|
100
|
+
`backend="vllm-static"` is opt-in, and serves only the tap set it declared — `static_points="auto"` by default, or a list you name. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; [benchmarks/results-latest.md](benchmarks/results-latest.md) has every figure at full precision, including capture, steering and lens latencies; [benchmarks/README.md](benchmarks/README.md) has how the tables above are rounded.
|
|
87
101
|
|
|
88
102
|
## Correctness
|
|
89
103
|
|
|
90
104
|
We verify correctness in two main ways:
|
|
91
105
|
|
|
92
106
|
1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
|
|
93
|
-
2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at `
|
|
107
|
+
2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
|
|
94
108
|
|
|
95
109
|
## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
|
|
96
110
|
|
|
@@ -114,4 +128,4 @@ Bugs and feature requests belong in [issues](https://github.com/decoderesearch/i
|
|
|
114
128
|
|
|
115
129
|
## License
|
|
116
130
|
|
|
117
|
-
Apache 2.0
|
|
131
|
+
Apache 2.0
|
|
@@ -92,7 +92,7 @@ full record into `results-latest.md` and then calls `publish`, which rewrites:
|
|
|
92
92
|
|
|
93
93
|
| target | what it gets |
|
|
94
94
|
| --- | --- |
|
|
95
|
-
| `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and
|
|
95
|
+
| `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and static taps, every model |
|
|
96
96
|
| `visualizer-web/data/benchmarks.generated.ts` | the same figures for the card behind the site's **Fast** claim, each row that ran differently carrying a footnote saying how |
|
|
97
97
|
|
|
98
98
|
Both were transcribed by hand until this existed, and both had drifted -- one carried percentages the
|
|
@@ -129,7 +129,7 @@ write at all when the cells disagree about the GPU or the dtype, because those t
|
|
|
129
129
|
conditions line. And it footnotes a card row when the sweep gave that model anything of its own -- its
|
|
130
130
|
own memory fraction, its own engine arguments, its own capture point -- so that the conditions line
|
|
131
131
|
keeps covering the rows it claims to. `deepseek-v4-flash-0731` earns a footnote for all four reasons.
|
|
132
|
-
The card dropped such a row until the footnote existed, which is why the largest
|
|
132
|
+
The card dropped such a row until the footnote existed, which is why the largest static win in the
|
|
133
133
|
sweep was for a while the one figure the site did not show; a row is still dropped, but only when it is
|
|
134
134
|
missing a baseline figure and so has no multiplier to print. A scratch sweep of ad-hoc models should
|
|
135
135
|
pass `--no-publish` to `report_bench` rather than publish rows nobody deployed.
|
|
@@ -217,7 +217,7 @@ The report names these by what they are; `--variant` and the result filenames us
|
|
|
217
217
|
| interp-engine eager | `eager` | raw HF forward; `attn_implementation="eager"`, which is what the engine sets |
|
|
218
218
|
| interp-engine vllm | `vllm` | vLLM with `enforce_eager=True` — CUDA graphs and inductor compile off |
|
|
219
219
|
| vllm (vanilla) | `vllm-cudagraph` | vLLM left at its own defaults, graphs and compile on |
|
|
220
|
-
| interp-engine vllm
|
|
220
|
+
| interp-engine vllm static | `vllm-static` | breakable graphs with `resid_post` static wraps at every layer, plus one write tap mid-stack |
|
|
221
221
|
|
|
222
222
|
`bench_spec.VARIANTS` also carries two speculative-decoding variants that exist on one checkpoint
|
|
223
223
|
only, and are not part of these tables: `report_bench.EXCLUDED` says why, and `--variant` still
|
|
@@ -233,31 +233,31 @@ capture actually returns under replay instead of asserting the outcome. A captur
|
|
|
233
233
|
with no points, or with fewer rows than the prompt had tokens, is recorded as `unsupported` with the
|
|
234
234
|
shape it got, and the report renders that cell as `n/a`.
|
|
235
235
|
|
|
236
|
-
### `vllm-
|
|
236
|
+
### `vllm-static`, and what its `steer` cell needs
|
|
237
237
|
|
|
238
|
-
The fourth is the answer to the third:
|
|
238
|
+
The fourth is the answer to the third: static copies activations in and out of buffers the graph
|
|
239
239
|
already refers to, so a replay serves capture and steering without a Python forward. It is the path
|
|
240
|
-
`
|
|
240
|
+
`static_points="auto"` takes in production, and the row exists to price it against the
|
|
241
241
|
`enforce_eager=True` column capture would otherwise have to use.
|
|
242
242
|
|
|
243
243
|
`"auto"` once installed **reads** only, and a steering op needs a write tap to land in, so this row
|
|
244
244
|
priced half the feature and reported the other half as `n/a` -- with a message that blamed graph
|
|
245
|
-
replay for it, which is the thing
|
|
246
|
-
cell would be a number either way, and this row still passes `
|
|
245
|
+
replay for it, which is the thing static exists to work around. Auto now covers both halves, so the
|
|
246
|
+
cell would be a number either way, and this row still passes `static_writes` on purpose: an explicit
|
|
247
247
|
list *narrows* what auto would install, to the one mid-stack site the `steer` workload actually
|
|
248
248
|
writes. A row that priced a write buffer at every layer would not be comparable with the ones beside
|
|
249
249
|
it, which is the whole job of the column. Its value is the sentinel `run_bench.STEER_WRITES` rather
|
|
250
|
-
than a site, because that layer differs per model and a
|
|
250
|
+
than a site, because that layer differs per model and a static write is a `load_model` argument, so
|
|
251
251
|
it has to be resolved from the config before a model exists to ask.
|
|
252
252
|
|
|
253
|
-
`VariantSpec.models` restricts the row to the checkpoints
|
|
253
|
+
`VariantSpec.models` restricts the row to the checkpoints static has been shown correct on, so a model
|
|
254
254
|
missing from it renders `--` rather than a number nobody checked.
|
|
255
255
|
|
|
256
|
-
One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk
|
|
257
|
-
`resid_streams` — the whole stack of four parallel streams per layer — and a
|
|
258
|
-
set it
|
|
259
|
-
That row's
|
|
260
|
-
`ModelSpec.
|
|
256
|
+
One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk declares
|
|
257
|
+
`resid_streams` — the whole stack of four parallel streams per layer — and a static engine serves the
|
|
258
|
+
set it declared, so `deepseek-v4-flash-0731` cannot be asked for the `mlp_out` its other columns capture.
|
|
259
|
+
That row's static cell therefore prices the stack where every other cell prices one row, declared in
|
|
260
|
+
`ModelSpec.static_capture_point`, stated by the report under *Where a row differs*, and carried onto
|
|
261
261
|
the visualizer's card as one line of that row's footnote.
|
|
262
262
|
|
|
263
263
|
## Reading the numbers honestly
|
|
@@ -50,21 +50,21 @@ class ModelSpec:
|
|
|
50
50
|
sweep's shared budget, which is all but the largest.
|
|
51
51
|
|
|
52
52
|
Separate from :attr:`extra_vllm_kwargs` because that is a fact about the weights and applies to
|
|
53
|
-
every vLLM column, so putting a
|
|
53
|
+
every vLLM column, so putting a static-only budget there would change three cells that have
|
|
54
54
|
already been measured and are comparable as they stand. The overrides here are recorded in the cell
|
|
55
55
|
and stated by the report, so the one cell that ran on a different budget says so."""
|
|
56
|
-
|
|
57
|
-
"""The point those workloads address instead when the variant under test
|
|
56
|
+
static_capture_point: str | None = None
|
|
57
|
+
"""The point those workloads address instead when the variant under test declared a set, for a
|
|
58
58
|
checkpoint where the two cannot be the same point. ``None`` means :attr:`capture_point` serves both.
|
|
59
59
|
|
|
60
|
-
A
|
|
61
|
-
name a point the
|
|
62
|
-
``
|
|
60
|
+
A static engine serves *the set it declared* and refuses anything outside it, so the workload has to
|
|
61
|
+
name a point the static set covers. On a hyper-connection trunk it cannot be the same one:
|
|
62
|
+
``static_points="auto"`` resolves to ``resid_streams`` -- the whole stack -- while the other columns
|
|
63
63
|
need one ``d_model`` row to stay comparable with the rest of the table. Declared here rather than
|
|
64
64
|
inferred at run time so the cell records it, ``report_bench`` states it under *Where a row differs*,
|
|
65
65
|
and ``cells.nonuniform`` carries it onto the visualizer's card as one line of that row's footnote.
|
|
66
66
|
|
|
67
|
-
Only the
|
|
67
|
+
Only the static cell records it, so read it through ``cells.row_spec`` rather than off a cell.
|
|
68
68
|
"""
|
|
69
69
|
extra_eager_kwargs: dict[str, object] = field(default_factory=dict)
|
|
70
70
|
"""``load_model`` arguments the eager variants need for this checkpoint, merged under the
|
|
@@ -138,15 +138,15 @@ MODELS: tuple[ModelSpec, ...] = (
|
|
|
138
138
|
# bytes, which would price this row's transport as though DeepSeek moved four times the
|
|
139
139
|
# activations to answer the same question.
|
|
140
140
|
capture_point="mlp_out",
|
|
141
|
-
# ...except under
|
|
141
|
+
# ...except under static, which serves the set it declared: `"auto"` on this trunk declares
|
|
142
142
|
# `resid_streams`, so `mlp_out` is outside the set and would be refused. That column therefore
|
|
143
143
|
# prices the stack rather than a row, which is declared as a row exception instead of hidden.
|
|
144
|
-
|
|
145
|
-
# The
|
|
144
|
+
static_capture_point="resid_streams",
|
|
145
|
+
# The static column is the one configuration here that does not fit the shared budget, and both
|
|
146
146
|
# of these are why. 149 GiB of weights at 0.95 of a 178 GiB card leave about 15 GiB for the
|
|
147
|
-
# activation peak, the
|
|
147
|
+
# activation peak, the static buffers, the KV pool and the graph pool together.
|
|
148
148
|
#
|
|
149
|
-
# A
|
|
149
|
+
# A static buffer is allocated per batched row, so at the 4096 `max_num_batched_tokens` static
|
|
150
150
|
# lowered vLLM's default to, `resid_streams` costs 4096 rows x 4 streams x 4096 wide x 2 bytes
|
|
151
151
|
# x 43 layers = 5.8 GiB -- and vLLM then sized the KV pool from what was left and refused to
|
|
152
152
|
# start: "No available memory for the cache blocks". 1024 rows costs a quarter of that and is
|
|
@@ -157,7 +157,7 @@ MODELS: tuple[ModelSpec, ...] = (
|
|
|
157
157
|
# sizes it will actually replay keeps the rest of that memory, and the startup time, unspent.
|
|
158
158
|
# Neither changes what the workloads measure: both sizes they use are still captured.
|
|
159
159
|
per_variant_vllm_kwargs={
|
|
160
|
-
"vllm-
|
|
160
|
+
"vllm-static": {
|
|
161
161
|
"max_num_batched_tokens": 1024,
|
|
162
162
|
"compilation_config": {"cudagraph_capture_sizes": [1, 2, 4, 8, 16, 32]},
|
|
163
163
|
}
|
|
@@ -215,12 +215,17 @@ class VariantSpec:
|
|
|
215
215
|
|
|
216
216
|
|
|
217
217
|
#: One of these three exists to price a default that the engine chose for capture's sake, which is the
|
|
218
|
-
#: most useful thing a speed benchmark of this library can say. vLLM capture
|
|
219
|
-
#:
|
|
220
|
-
#:
|
|
221
|
-
#: ``vllm-cudagraph`` -- vLLM left at its own defaults, hence the
|
|
222
|
-
#: it measures what ours costs. The capture workloads run there too
|
|
223
|
-
#: report can show *what* a capture returns under replay instead of
|
|
218
|
+
#: most useful thing a speed benchmark of this library can say. Hooked vLLM capture rules CUDA graphs
|
|
219
|
+
#: out, because graph replay does not re-execute the Python forward and so never fires a
|
|
220
|
+
#: ``register_forward_hook``. That is what separates the three vLLM backends, and it is why
|
|
221
|
+
#: ``vllm-cudagraph`` -- ``backend="vllm-generate"``, vLLM left at its own defaults, hence the
|
|
222
|
+
#: "vanilla" label -- is here at all: it measures what ours costs. The capture workloads run there too
|
|
223
|
+
#: rather than being skipped, so the report can show *what* a capture returns under replay instead of
|
|
224
|
+
#: asserting that it fails.
|
|
225
|
+
#:
|
|
226
|
+
#: Each variant names its backend rather than deriving it from the kwargs it passes, which is what
|
|
227
|
+
#: the engine now expects: ``vllm-static`` and ``vllm-generate`` refuse ``enforce_eager``, and a tap
|
|
228
|
+
#: set is only accepted by the backend built to bake one in.
|
|
224
229
|
VARIANTS: tuple[VariantSpec, ...] = (
|
|
225
230
|
VariantSpec(
|
|
226
231
|
"eager",
|
|
@@ -232,52 +237,52 @@ VARIANTS: tuple[VariantSpec, ...] = (
|
|
|
232
237
|
VariantSpec(
|
|
233
238
|
"vllm",
|
|
234
239
|
"vllm",
|
|
235
|
-
{
|
|
240
|
+
{},
|
|
236
241
|
"capture-capable: CUDA graphs and compile OFF",
|
|
237
242
|
"interp-engine vllm",
|
|
238
243
|
),
|
|
239
244
|
VariantSpec(
|
|
240
245
|
"vllm-cudagraph",
|
|
241
|
-
"vllm",
|
|
242
|
-
{
|
|
243
|
-
"CUDA graphs + inductor compile ON, no
|
|
246
|
+
"vllm-generate",
|
|
247
|
+
{},
|
|
248
|
+
"CUDA graphs + inductor compile ON, no static wraps; vLLM's own defaults for generate-only",
|
|
244
249
|
"vllm (vanilla)",
|
|
245
250
|
),
|
|
246
|
-
# Breakable CUDA graphs with resid_post
|
|
247
|
-
# (``
|
|
251
|
+
# Breakable CUDA graphs with resid_post static wraps at every layer. Production static
|
|
252
|
+
# (``static_points="auto"``) is this path, not Dynamo piecewise.
|
|
248
253
|
#
|
|
249
|
-
# `qwen3.8-27b` joined the row once
|
|
254
|
+
# `qwen3.8-27b` joined the row once static was correct on a hybrid trunk. Breakable graphs turn
|
|
250
255
|
# inductor off, and vLLM's `FULL_AND_PIECEWISE` capture then miscomputes prefill for a gated-delta
|
|
251
256
|
# trunk -- the engine generated fluent nonsense rather than failing, so the cell would have been a
|
|
252
|
-
# plausible number for a broken forward.
|
|
253
|
-
# runs prefill eagerly and keeps the decode graphs, and the validator's
|
|
257
|
+
# plausible number for a broken forward. Static now pins `FULL_DECODE_ONLY` on such a trunk, which
|
|
258
|
+
# runs prefill eagerly and keeps the decode graphs, and the validator's static column agrees with
|
|
254
259
|
# eager on all 28 points. Its prefill figures carry that eager prefill, which is the point of
|
|
255
260
|
# comparing it against the same model's other variants rather than against another model.
|
|
256
261
|
#
|
|
257
|
-
# `deepseek-v4-flash-0731` is in, and is the one row here whose
|
|
262
|
+
# `deepseek-v4-flash-0731` is in, and is the one row here whose static set is not one point per
|
|
258
263
|
# layer. It is a hyper-connection trunk, so `"auto"` resolves to `resid_streams` -- the whole stack
|
|
259
264
|
# of four parallel residual streams per layer, four times the width of a `resid_post` row. Its
|
|
260
265
|
# capture and transport figures therefore price four times the activations for the same question,
|
|
261
266
|
# which `cells.nonuniform` declares so the report states it and the visualizer's card drops the row
|
|
262
|
-
# rather than publishing it beside three that
|
|
267
|
+
# rather than publishing it beside three that declared a quarter as much.
|
|
263
268
|
#
|
|
264
269
|
# Its write tap is `mlp_out`, not `resid_post`: the steer workload addresses the model's
|
|
265
270
|
# `capture_point`, and a hyper-connection trunk refuses the default name (`run_bench._steer_site`).
|
|
266
271
|
#
|
|
267
|
-
# `
|
|
272
|
+
# `static_writes` was once what made the `steer` cell a number instead of `n/a`: `"auto"` installed
|
|
268
273
|
# reads, and a steering op needs a write tap to land in, so this row priced capture under replay
|
|
269
274
|
# and left the other half of the feature unmeasured -- with a message that blamed graph replay for
|
|
270
275
|
# it. Auto covers writes now, so the cell stands either way, and naming them here has become a
|
|
271
276
|
# *narrowing*: one write buffer at the site the workload steers rather than one per layer, which
|
|
272
277
|
# is what keeps this row's memory comparable with the columns beside it. The value is the sentinel
|
|
273
278
|
# `run_bench.STEER_WRITES`, resolved there to the mid-stack `resid_post` the workload steers,
|
|
274
|
-
# because the layer differs per model and a
|
|
279
|
+
# because the layer differs per model and a static write has to be named before the model exists.
|
|
275
280
|
VariantSpec(
|
|
276
|
-
"vllm-
|
|
277
|
-
"vllm",
|
|
278
|
-
{"
|
|
279
|
-
"breakable CUDA graphs with resid_post
|
|
280
|
-
"interp-engine vllm
|
|
281
|
+
"vllm-static",
|
|
282
|
+
"vllm-static",
|
|
283
|
+
{"static_points": "auto", "static_writes": "steer"},
|
|
284
|
+
"breakable CUDA graphs with resid_post static wraps at every layer, and a write tap mid-stack",
|
|
285
|
+
"interp-engine vllm static",
|
|
281
286
|
models=("gemma-2-2b", "qwen3-4b", "llama-3.1-8b", "qwen3.8-27b", "deepseek-v4-flash-0731"),
|
|
282
287
|
),
|
|
283
288
|
# DSpark on, against the `vllm` column with it off -- the pair is the measurement, so read the two
|
|
@@ -303,12 +308,11 @@ VARIANTS: tuple[VariantSpec, ...] = (
|
|
|
303
308
|
"vllm-dspark",
|
|
304
309
|
"vllm",
|
|
305
310
|
{
|
|
306
|
-
"enforce_eager": True,
|
|
307
311
|
"extra_vllm_kwargs": {
|
|
308
312
|
"speculative_config": {"method": "dspark", "num_speculative_tokens": 5},
|
|
309
313
|
},
|
|
310
314
|
},
|
|
311
|
-
"DSpark speculative decoding ON;
|
|
315
|
+
"DSpark speculative decoding ON; hooked backend, so still capture-capable",
|
|
312
316
|
"interp-engine vllm +DSpark",
|
|
313
317
|
models=("deepseek-v4-flash-0731",),
|
|
314
318
|
),
|
|
@@ -325,10 +329,8 @@ VARIANTS: tuple[VariantSpec, ...] = (
|
|
|
325
329
|
# never calls the Python forward a hook is attached to.
|
|
326
330
|
VariantSpec(
|
|
327
331
|
"vllm-dspark-cudagraph",
|
|
328
|
-
"vllm",
|
|
332
|
+
"vllm-generate",
|
|
329
333
|
{
|
|
330
|
-
"enforce_eager": False,
|
|
331
|
-
"freeze_points": [],
|
|
332
334
|
"extra_vllm_kwargs": {
|
|
333
335
|
"speculative_config": {"method": "dspark", "num_speculative_tokens": 5},
|
|
334
336
|
},
|
|
@@ -77,9 +77,9 @@ def row_spec(cells: list[dict[str, Any]], model_key: str, variants: Iterable[str
|
|
|
77
77
|
"""Every override one model's row declared, merged across the row's cells.
|
|
78
78
|
|
|
79
79
|
A per-variant override is recorded by the cell that used it and by no other: on
|
|
80
|
-
`deepseek-v4-flash-0731` both `
|
|
81
|
-
|
|
82
|
-
one -- whichever happened to sort last -- silently dropped the two overrides the
|
|
80
|
+
`deepseek-v4-flash-0731` both `static_capture_point` and `per_variant_vllm_kwargs` live on the
|
|
81
|
+
static cell alone. So any single cell's ``model`` describes a column rather than a row, and picking
|
|
82
|
+
one -- whichever happened to sort last -- silently dropped the two overrides the static column is
|
|
83
83
|
the only cell to declare.
|
|
84
84
|
|
|
85
85
|
``variants`` restricts the merge to the columns a renderer shows, and each of them passes its own:
|
|
@@ -129,9 +129,9 @@ def nonuniform(model: dict[str, Any]) -> list[str]:
|
|
|
129
129
|
point = model.get("capture_point", DEFAULT_POINT)
|
|
130
130
|
if point != DEFAULT_POINT:
|
|
131
131
|
reasons.append(f"captures `{point}` rather than `{DEFAULT_POINT}`")
|
|
132
|
-
|
|
133
|
-
if
|
|
134
|
-
reasons.append(f"its
|
|
132
|
+
static_point = model.get("static_capture_point")
|
|
133
|
+
if static_point:
|
|
134
|
+
reasons.append(f"its static column captures `{static_point}`, a wider point than the others do")
|
|
135
135
|
fraction = model.get("gpu_memory_utilization")
|
|
136
136
|
if fraction:
|
|
137
137
|
reasons.append(f"vLLM reserved {fraction} of the card, not the uniform {GPU_MEMORY_UTILIZATION}")
|