interp-engine 1.2.8__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (175) hide show
  1. {interp_engine-1.2.8 → interp_engine-1.3.0}/PKG-INFO +47 -33
  2. {interp_engine-1.2.8 → interp_engine-1.3.0}/README.md +46 -32
  3. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/README.md +15 -15
  4. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/bench_spec.py +44 -42
  5. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/cells.py +6 -6
  6. interp_engine-1.3.0/benchmarks/probe_lens_stream.py +626 -0
  7. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/publish.py +4 -4
  8. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/report_bench.py +7 -7
  9. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-cudagraph.json +2 -4
  10. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark-cudagraph.json +1 -3
  11. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark.json +0 -1
  12. interp_engine-1.2.8/benchmarks/results/deepseek-v4-flash-0731__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/deepseek-v4-flash-0731__vllm-static.json +8 -8
  13. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__vllm.json +0 -1
  14. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__vllm-cudagraph.json +2 -4
  15. interp_engine-1.2.8/benchmarks/results/gemma-2-2b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/gemma-2-2b__vllm-static.json +6 -6
  16. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__vllm.json +0 -1
  17. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__vllm-cudagraph.json +2 -4
  18. interp_engine-1.2.8/benchmarks/results/llama-3.1-8b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/llama-3.1-8b__vllm-static.json +6 -6
  19. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__vllm.json +0 -1
  20. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__vllm-cudagraph.json +2 -4
  21. interp_engine-1.2.8/benchmarks/results/qwen3-4b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/qwen3-4b__vllm-static.json +6 -6
  22. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__vllm.json +0 -1
  23. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__vllm-cudagraph.json +2 -4
  24. interp_engine-1.2.8/benchmarks/results/qwen3.8-27b__vllm-freeze.json → interp_engine-1.3.0/benchmarks/results/qwen3.8-27b__vllm-static.json +6 -6
  25. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__vllm.json +0 -1
  26. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results-latest.md +14 -14
  27. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/run_bench.py +15 -10
  28. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/workloads.py +13 -13
  29. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/AGENT_INTEGRATION.md +46 -4
  30. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/PERFORMANCE.md +54 -47
  31. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/USAGE.md +11 -0
  32. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/__init__.py +2 -1
  33. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/_loop.py +47 -4
  34. interp_engine-1.3.0/interp_engine/load.py +216 -0
  35. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/model.py +2 -2
  36. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/protocol.py +13 -7
  37. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/sync.py +4 -4
  38. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_backend.py +369 -244
  39. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/__init__.py +25 -25
  40. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_hooks.py +2 -2
  41. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/graphs.py +5 -4
  42. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/readout.py +22 -22
  43. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/mhc.py +1 -1
  44. interp_engine-1.2.8/interp_engine/vllm_capture/freeze.py → interp_engine-1.3.0/interp_engine/vllm_capture/static.py +217 -214
  45. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_plugin.py +25 -25
  46. {interp_engine-1.2.8 → interp_engine-1.3.0}/pyproject.toml +2 -2
  47. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_bench_workloads.py +14 -14
  48. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_layer_kinds.py +1 -1
  49. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_load.py +46 -10
  50. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_published_benchmarks.py +8 -8
  51. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_qk_norm.py +1 -1
  52. interp_engine-1.2.8/tests/test_freeze_dsv4_gpu.py → interp_engine-1.3.0/tests/test_static_dsv4_gpu.py +17 -16
  53. interp_engine-1.2.8/tests/test_freeze_parity_gpu.py → interp_engine-1.3.0/tests/test_static_parity_gpu.py +51 -48
  54. interp_engine-1.2.8/tests/test_freeze_set.py → interp_engine-1.3.0/tests/test_static_set.py +113 -113
  55. interp_engine-1.2.8/tests/test_freeze_warmup.py → interp_engine-1.3.0/tests/test_static_warmup.py +17 -17
  56. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sync_loop.py +46 -1
  57. interp_engine-1.3.0/tests/test_vllm_engine_loop.py +101 -0
  58. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_graph_path.py +2 -1
  59. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_graphs_on_gpu.py +26 -31
  60. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_hook_availability.py +57 -32
  61. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_hyper_connections.py +97 -97
  62. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_worker_lens_capture_readout.py +11 -11
  63. interp_engine-1.2.8/interp_engine/load.py +0 -152
  64. {interp_engine-1.2.8 → interp_engine-1.3.0}/.gitignore +0 -0
  65. {interp_engine-1.2.8 → interp_engine-1.3.0}/LICENSE +0 -0
  66. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/__init__.py +0 -0
  67. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/probe.py +0 -0
  68. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/deepseek-v4-flash-0731__eager.json +0 -0
  69. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/gemma-2-2b__eager.json +0 -0
  70. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/llama-3.1-8b__eager.json +0 -0
  71. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3-4b__eager.json +0 -0
  72. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/results/qwen3.8-27b__eager.json +0 -0
  73. {interp_engine-1.2.8 → interp_engine-1.3.0}/benchmarks/run_all.sh +0 -0
  74. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/ARCHITECTURE_QUIRKS.md +0 -0
  75. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/COMPATIBILITY.md +0 -0
  76. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/ENGINE_HOOK_MAPPINGS.md +0 -0
  77. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/GRADIENTS.md +0 -0
  78. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/INTERNALS.md +0 -0
  79. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/PORTING.md +0 -0
  80. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/README.md +0 -0
  81. {interp_engine-1.2.8 → interp_engine-1.3.0}/docs/SUPPORTED_POINTS.md +0 -0
  82. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/address.py +0 -0
  83. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/arch.py +0 -0
  84. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/attn_config.py +0 -0
  85. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/attn_scores.py +0 -0
  86. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/autograd_support.py +0 -0
  87. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/capture.py +0 -0
  88. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/chat_compose.py +0 -0
  89. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/chat_conventions.py +0 -0
  90. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/chat_formatters.py +0 -0
  91. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/cuda_preflight.py +0 -0
  92. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/dispatch.py +0 -0
  93. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/facts.py +0 -0
  94. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/hooks.py +0 -0
  95. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/lens.py +0 -0
  96. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/mappers.py +0 -0
  97. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/moe_routing.py +0 -0
  98. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/points.py +0 -0
  99. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/residual_basis.py +0 -0
  100. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/select.py +0 -0
  101. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/steer.py +0 -0
  102. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/steer_specs.py +0 -0
  103. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/tokenize.py +0 -0
  104. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_demux.py +0 -0
  105. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_payload.py +0 -0
  106. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/_tree.py +0 -0
  107. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/attn.py +0 -0
  108. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/capture.py +0 -0
  109. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/__init__.py +0 -0
  110. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/intervene.py +0 -0
  111. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/lens/unembed.py +0 -0
  112. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/native.py +0 -0
  113. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/requests.py +0 -0
  114. {interp_engine-1.2.8 → interp_engine-1.3.0}/interp_engine/vllm_capture/steering.py +0 -0
  115. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/conftest.py +0 -0
  116. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/harness.py +0 -0
  117. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/model_expectations.yaml +0 -0
  118. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/synthetic_families.py +0 -0
  119. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_address.py +0 -0
  120. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_config_tripwire.py +0 -0
  121. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_probs_indexing.py +0 -0
  122. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_scores.py +0 -0
  123. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_attn_z_gqa.py +0 -0
  124. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_autograd_support.py +0 -0
  125. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_capability_refusals.py +0 -0
  126. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_capture_addressing.py +0 -0
  127. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_chat_compose.py +0 -0
  128. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_chat_formatters.py +0 -0
  129. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_chat_templates.py +0 -0
  130. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_core.py +0 -0
  131. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_cuda_preflight.py +0 -0
  132. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_doc_code_fences.py +0 -0
  133. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_eager_autograd.py +0 -0
  134. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_facts.py +0 -0
  135. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_family_points.py +0 -0
  136. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_gated_attn_out.py +0 -0
  137. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_head_contributions.py +0 -0
  138. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_hook_call_conventions.py +0 -0
  139. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_logit_transform.py +0 -0
  140. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_mappers.py +0 -0
  141. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_mlp_internals.py +0 -0
  142. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_model_expectations.py +0 -0
  143. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_moe.py +0 -0
  144. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_multimodal_arch.py +0 -0
  145. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_new_models_gpu.py +0 -0
  146. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_no_chat_template.py +0 -0
  147. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_normalized_hook.py +0 -0
  148. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_packaging.py +0 -0
  149. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_parity_gpt2.py +0 -0
  150. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_per_layer_attn_dims.py +0 -0
  151. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_points_registry.py +0 -0
  152. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_protocol.py +0 -0
  153. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_qkv_layout.py +0 -0
  154. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_reasoning_spans.py +0 -0
  155. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_release.py +0 -0
  156. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_resid_mid.py +0 -0
  157. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_residual_basis.py +0 -0
  158. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sandwich_norms.py +0 -0
  159. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_select.py +0 -0
  160. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sliding_window_attn.py +0 -0
  161. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_small_models_gpu.py +0 -0
  162. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_steer_context.py +0 -0
  163. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_steer_math_parity.py +0 -0
  164. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_sync_parity.py +0 -0
  165. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_unified_free_functions.py +0 -0
  166. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_unresolved_families.py +0 -0
  167. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_capture_gpu.py +0 -0
  168. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_capture_scales.py +0 -0
  169. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_kv_isolation.py +0 -0
  170. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_new_points.py +0 -0
  171. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_only_families.py +0 -0
  172. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_plugin.py +0 -0
  173. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vllm_wire_grammar.py +0 -0
  174. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_vocabulary_boundary.py +0 -0
  175. {interp_engine-1.2.8 → interp_engine-1.3.0}/tests/test_worker_lens_readout.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: interp-engine
3
- Version: 1.2.8
3
+ Version: 1.3.0
4
4
  Summary: A fast, standardized interpretability engine that supports most modern models and architectures. Powers Neuronpedia.
5
5
  Project-URL: Homepage, https://github.com/decoderesearch/interp-engine
6
6
  Project-URL: Repository, https://github.com/decoderesearch/interp-engine
@@ -31,22 +31,33 @@ Description-Content-Type: text/markdown
31
31
 
32
32
  # interp-engine
33
33
 
34
-
35
-
36
- 🔗 **[interp-engine.org](https://interp-engine.org)**
37
-
34
+ <p align="center">
35
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
36
+ </p>
37
+ <p align="center">
38
+ 🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
39
+ </p>
40
+ <p align="center">
41
+ <a href="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml"><img src="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml/badge.svg?branch=main" alt="CI status"></a>
42
+ <a href="https://pypi.org/project/interp-engine/"><img src="https://img.shields.io/pypi/v/interp-engine.svg" alt="PyPI version"></a>
43
+ <a href="LICENSE"><img src="https://img.shields.io/pypi/l/interp-engine.svg" alt="Apache-2.0 license"></a>
44
+ <a href="https://join.slack.com/t/opensourcemechanistic/shared_invite/zt-3z9o0hxjl-MDX9pbATO2qESOazNDLpdQ"><img src="https://img.shields.io/badge/Slack-Open%20Source%20Mechanistic%20Interpretability-4A154B?logo=slack&logoColor=white" alt="Join the Slack"></a>
45
+ </p>
38
46
 
39
47
 
40
48
  `interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference and is checked for accuracy against HF Transformers and other engines.
41
49
 
42
-
43
-
44
-
50
+ <p align="center">
51
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-static on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="100%">
52
+ </p>
53
+ <p align="center">
54
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="100%">
55
+ </p>
45
56
 
46
57
  This repo contains:
47
58
 
48
- 1. `[validator/](validator/)`, which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
49
- 2. `[visualizer-web/](visualizer-web/)`, a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
59
+ 1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
60
+ 2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
50
61
 
51
62
  ## Installation
52
63
 
@@ -60,13 +71,16 @@ pip install interp-engine # eager backend only
60
71
  ```python
61
72
  from interp_engine import Address, load_model, run_with_cache
62
73
 
63
- # VLLM MODE (default): low VRAM, medium speed
64
- model = load_model("Qwen/Qwen3-8B")
74
+ # VLLM (default): low VRAM, medium speed, every point, chosen per request
75
+ model = load_model("Qwen/Qwen3-8B")
76
+
77
+ # VLLM-STATIC: high VRAM, high speed, only the points you declare (default resid_post)
78
+ # model = load_model("Qwen/Qwen3-8B", backend="vllm-static")
65
79
 
66
- # VLLM-FREEZE MODE: high VRAM, high speed, only frozen points (default resid_post)
67
- # model = load_model("Qwen/Qwen3-8B", freeze_points="auto")
80
+ # VLLM-GENERATE: fastest, generation only -- no capture, no steering
81
+ # model = load_model("Qwen/Qwen3-8B", backend="vllm-generate")
68
82
 
69
- # EAGER MODE: low VRAM, low speed
83
+ # EAGER: low VRAM, low speed
70
84
  # model = load_model("Qwen/Qwen3-8B", backend="eager")
71
85
 
72
86
  point = Address("resid_post", 10) # or string: "resid_post.10"
@@ -84,7 +98,7 @@ Add "use interp-engine" to your prompt and let your agent figure it out - everyt
84
98
 
85
99
  ## Performance / Speed
86
100
 
87
- vLLM gives `interp-engine` high throughput via concurrency, and **graph freeze** adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
101
+ vLLM gives `interp-engine` high throughput via concurrency, and `backend="vllm-static"` adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
88
102
 
89
103
  <!-- THROUGHPUT:START -->
90
104
 
@@ -94,34 +108,34 @@ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
94
108
 
95
109
  One stream (tok/s):
96
110
 
97
- | model | eager | vLLM | vLLM + graph freeze |
98
- | ------------------------ | ----- | ---------- | ------------------- |
99
- | `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
100
- | `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
101
- | `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
102
- | `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
103
- | `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
111
+ | model | eager | vLLM | vLLM + static taps |
112
+ | ------------------------ | ----- | ---------- | ------------------ |
113
+ | `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
114
+ | `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
115
+ | `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
116
+ | `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
117
+ | `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
104
118
 
105
119
  8 concurrent requests (aggregate tok/s):
106
120
 
107
- | model | eager | vLLM | vLLM + graph freeze |
108
- | ------------------------ | ----- | ----------- | ------------------- |
109
- | `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
110
- | `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
111
- | `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
112
- | `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
113
- | `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
121
+ | model | eager | vLLM | vLLM + static taps |
122
+ | ------------------------ | ----- | ----------- | ------------------ |
123
+ | `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
124
+ | `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
125
+ | `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
126
+ | `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
127
+ | `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
114
128
 
115
129
  <!-- THROUGHPUT:END -->
116
130
 
117
- **Graph freeze** is opt-in via `freeze_points`, and a frozen engine serves only the set it froze. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; `[benchmarks/results-latest.md](benchmarks/results-latest.md)` has every figure at full precision, including capture, steering and lens latencies; `[benchmarks/README.md](benchmarks/README.md)` has how the tables above are rounded.
131
+ `backend="vllm-static"` is opt-in, and serves only the tap set it declared — `static_points="auto"` by default, or a list you name. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; [benchmarks/results-latest.md](benchmarks/results-latest.md) has every figure at full precision, including capture, steering and lens latencies; [benchmarks/README.md](benchmarks/README.md) has how the tables above are rounded.
118
132
 
119
133
  ## Correctness
120
134
 
121
135
  We verify correctness in two main ways:
122
136
 
123
137
  1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
124
- 2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at `[validator/](validator/)`.
138
+ 2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
125
139
 
126
140
  ## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
127
141
 
@@ -145,4 +159,4 @@ Bugs and feature requests belong in [issues](https://github.com/decoderesearch/i
145
159
 
146
160
  ## License
147
161
 
148
- Apache 2.0
162
+ Apache 2.0
@@ -1,21 +1,32 @@
1
1
  # interp-engine
2
2
 
3
-
4
-
5
- 🔗 **[interp-engine.org](https://interp-engine.org)**
6
-
3
+ <p align="center">
4
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
5
+ </p>
6
+ <p align="center">
7
+ 🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
8
+ </p>
9
+ <p align="center">
10
+ <a href="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml"><img src="https://github.com/decoderesearch/interp-engine/actions/workflows/engine-tests.yml/badge.svg?branch=main" alt="CI status"></a>
11
+ <a href="https://pypi.org/project/interp-engine/"><img src="https://img.shields.io/pypi/v/interp-engine.svg" alt="PyPI version"></a>
12
+ <a href="LICENSE"><img src="https://img.shields.io/pypi/l/interp-engine.svg" alt="Apache-2.0 license"></a>
13
+ <a href="https://join.slack.com/t/opensourcemechanistic/shared_invite/zt-3z9o0hxjl-MDX9pbATO2qESOazNDLpdQ"><img src="https://img.shields.io/badge/Slack-Open%20Source%20Mechanistic%20Interpretability-4A154B?logo=slack&logoColor=white" alt="Join the Slack"></a>
14
+ </p>
7
15
 
8
16
 
9
17
  `interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference and is checked for accuracy against HF Transformers and other engines.
10
18
 
11
-
12
-
13
-
19
+ <p align="center">
20
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-static on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="100%">
21
+ </p>
22
+ <p align="center">
23
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="100%">
24
+ </p>
14
25
 
15
26
  This repo contains:
16
27
 
17
- 1. `[validator/](validator/)`, which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
18
- 2. `[visualizer-web/](visualizer-web/)`, a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
28
+ 1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
29
+ 2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
19
30
 
20
31
  ## Installation
21
32
 
@@ -29,13 +40,16 @@ pip install interp-engine # eager backend only
29
40
  ```python
30
41
  from interp_engine import Address, load_model, run_with_cache
31
42
 
32
- # VLLM MODE (default): low VRAM, medium speed
33
- model = load_model("Qwen/Qwen3-8B")
43
+ # VLLM (default): low VRAM, medium speed, every point, chosen per request
44
+ model = load_model("Qwen/Qwen3-8B")
45
+
46
+ # VLLM-STATIC: high VRAM, high speed, only the points you declare (default resid_post)
47
+ # model = load_model("Qwen/Qwen3-8B", backend="vllm-static")
34
48
 
35
- # VLLM-FREEZE MODE: high VRAM, high speed, only frozen points (default resid_post)
36
- # model = load_model("Qwen/Qwen3-8B", freeze_points="auto")
49
+ # VLLM-GENERATE: fastest, generation only -- no capture, no steering
50
+ # model = load_model("Qwen/Qwen3-8B", backend="vllm-generate")
37
51
 
38
- # EAGER MODE: low VRAM, low speed
52
+ # EAGER: low VRAM, low speed
39
53
  # model = load_model("Qwen/Qwen3-8B", backend="eager")
40
54
 
41
55
  point = Address("resid_post", 10) # or string: "resid_post.10"
@@ -53,7 +67,7 @@ Add "use interp-engine" to your prompt and let your agent figure it out - everyt
53
67
 
54
68
  ## Performance / Speed
55
69
 
56
- vLLM gives `interp-engine` high throughput via concurrency, and **graph freeze** adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
70
+ vLLM gives `interp-engine` high throughput via concurrency, and `backend="vllm-static"` adds CUDA-graph replay on top of that *without giving up capture or steering*. Every column below is capture-capable.
57
71
 
58
72
  <!-- THROUGHPUT:START -->
59
73
 
@@ -63,34 +77,34 @@ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
63
77
 
64
78
  One stream (tok/s):
65
79
 
66
- | model | eager | vLLM | vLLM + graph freeze |
67
- | ------------------------ | ----- | ---------- | ------------------- |
68
- | `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
69
- | `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
70
- | `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
71
- | `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
72
- | `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
80
+ | model | eager | vLLM | vLLM + static taps |
81
+ | ------------------------ | ----- | ---------- | ------------------ |
82
+ | `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
83
+ | `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
84
+ | `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
85
+ | `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
86
+ | `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
73
87
 
74
88
  8 concurrent requests (aggregate tok/s):
75
89
 
76
- | model | eager | vLLM | vLLM + graph freeze |
77
- | ------------------------ | ----- | ----------- | ------------------- |
78
- | `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
79
- | `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
80
- | `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
81
- | `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
82
- | `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
90
+ | model | eager | vLLM | vLLM + static taps |
91
+ | ------------------------ | ----- | ----------- | ------------------ |
92
+ | `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
93
+ | `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
94
+ | `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
95
+ | `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
96
+ | `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
83
97
 
84
98
  <!-- THROUGHPUT:END -->
85
99
 
86
- **Graph freeze** is opt-in via `freeze_points`, and a frozen engine serves only the set it froze. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; `[benchmarks/results-latest.md](benchmarks/results-latest.md)` has every figure at full precision, including capture, steering and lens latencies; `[benchmarks/README.md](benchmarks/README.md)` has how the tables above are rounded.
100
+ `backend="vllm-static"` is opt-in, and serves only the tap set it declared — `static_points="auto"` by default, or a list you name. [PERFORMANCE.md](docs/PERFORMANCE.md) has how it works and what it trades; [benchmarks/results-latest.md](benchmarks/results-latest.md) has every figure at full precision, including capture, steering and lens latencies; [benchmarks/README.md](benchmarks/README.md) has how the tables above are rounded.
87
101
 
88
102
  ## Correctness
89
103
 
90
104
  We verify correctness in two main ways:
91
105
 
92
106
  1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
93
- 2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at `[validator/](validator/)`.
107
+ 2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
94
108
 
95
109
  ## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
96
110
 
@@ -114,4 +128,4 @@ Bugs and feature requests belong in [issues](https://github.com/decoderesearch/i
114
128
 
115
129
  ## License
116
130
 
117
- Apache 2.0
131
+ Apache 2.0
@@ -92,7 +92,7 @@ full record into `results-latest.md` and then calls `publish`, which rewrites:
92
92
 
93
93
  | target | what it gets |
94
94
  | --- | --- |
95
- | `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and graph freeze, every model |
95
+ | `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and static taps, every model |
96
96
  | `visualizer-web/data/benchmarks.generated.ts` | the same figures for the card behind the site's **Fast** claim, each row that ran differently carrying a footnote saying how |
97
97
 
98
98
  Both were transcribed by hand until this existed, and both had drifted -- one carried percentages the
@@ -129,7 +129,7 @@ write at all when the cells disagree about the GPU or the dtype, because those t
129
129
  conditions line. And it footnotes a card row when the sweep gave that model anything of its own -- its
130
130
  own memory fraction, its own engine arguments, its own capture point -- so that the conditions line
131
131
  keeps covering the rows it claims to. `deepseek-v4-flash-0731` earns a footnote for all four reasons.
132
- The card dropped such a row until the footnote existed, which is why the largest freeze win in the
132
+ The card dropped such a row until the footnote existed, which is why the largest static win in the
133
133
  sweep was for a while the one figure the site did not show; a row is still dropped, but only when it is
134
134
  missing a baseline figure and so has no multiplier to print. A scratch sweep of ad-hoc models should
135
135
  pass `--no-publish` to `report_bench` rather than publish rows nobody deployed.
@@ -217,7 +217,7 @@ The report names these by what they are; `--variant` and the result filenames us
217
217
  | interp-engine eager | `eager` | raw HF forward; `attn_implementation="eager"`, which is what the engine sets |
218
218
  | interp-engine vllm | `vllm` | vLLM with `enforce_eager=True` — CUDA graphs and inductor compile off |
219
219
  | vllm (vanilla) | `vllm-cudagraph` | vLLM left at its own defaults, graphs and compile on |
220
- | interp-engine vllm freeze | `vllm-freeze` | breakable graphs with `resid_post` freeze wraps at every layer, plus one write tap mid-stack |
220
+ | interp-engine vllm static | `vllm-static` | breakable graphs with `resid_post` static wraps at every layer, plus one write tap mid-stack |
221
221
 
222
222
  `bench_spec.VARIANTS` also carries two speculative-decoding variants that exist on one checkpoint
223
223
  only, and are not part of these tables: `report_bench.EXCLUDED` says why, and `--variant` still
@@ -233,31 +233,31 @@ capture actually returns under replay instead of asserting the outcome. A captur
233
233
  with no points, or with fewer rows than the prompt had tokens, is recorded as `unsupported` with the
234
234
  shape it got, and the report renders that cell as `n/a`.
235
235
 
236
- ### `vllm-freeze`, and what its `steer` cell needs
236
+ ### `vllm-static`, and what its `steer` cell needs
237
237
 
238
- The fourth is the answer to the third: freeze copies activations in and out of buffers the graph
238
+ The fourth is the answer to the third: static copies activations in and out of buffers the graph
239
239
  already refers to, so a replay serves capture and steering without a Python forward. It is the path
240
- `freeze_points="auto"` takes in production, and the row exists to price it against the
240
+ `static_points="auto"` takes in production, and the row exists to price it against the
241
241
  `enforce_eager=True` column capture would otherwise have to use.
242
242
 
243
243
  `"auto"` once installed **reads** only, and a steering op needs a write tap to land in, so this row
244
244
  priced half the feature and reported the other half as `n/a` -- with a message that blamed graph
245
- replay for it, which is the thing freeze exists to work around. Auto now covers both halves, so the
246
- cell would be a number either way, and this row still passes `freeze_writes` on purpose: an explicit
245
+ replay for it, which is the thing static exists to work around. Auto now covers both halves, so the
246
+ cell would be a number either way, and this row still passes `static_writes` on purpose: an explicit
247
247
  list *narrows* what auto would install, to the one mid-stack site the `steer` workload actually
248
248
  writes. A row that priced a write buffer at every layer would not be comparable with the ones beside
249
249
  it, which is the whole job of the column. Its value is the sentinel `run_bench.STEER_WRITES` rather
250
- than a site, because that layer differs per model and a freeze write is a `load_model` argument, so
250
+ than a site, because that layer differs per model and a static write is a `load_model` argument, so
251
251
  it has to be resolved from the config before a model exists to ask.
252
252
 
253
- `VariantSpec.models` restricts the row to the checkpoints freeze has been shown correct on, so a model
253
+ `VariantSpec.models` restricts the row to the checkpoints static has been shown correct on, so a model
254
254
  missing from it renders `--` rather than a number nobody checked.
255
255
 
256
- One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk freezes
257
- `resid_streams` — the whole stack of four parallel streams per layer — and a frozen engine serves the
258
- set it froze, so `deepseek-v4-flash-0731` cannot be asked for the `mlp_out` its other columns capture.
259
- That row's freeze cell therefore prices the stack where every other cell prices one row, declared in
260
- `ModelSpec.freeze_capture_point`, stated by the report under *Where a row differs*, and carried onto
256
+ One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk declares
257
+ `resid_streams` — the whole stack of four parallel streams per layer — and a static engine serves the
258
+ set it declared, so `deepseek-v4-flash-0731` cannot be asked for the `mlp_out` its other columns capture.
259
+ That row's static cell therefore prices the stack where every other cell prices one row, declared in
260
+ `ModelSpec.static_capture_point`, stated by the report under *Where a row differs*, and carried onto
261
261
  the visualizer's card as one line of that row's footnote.
262
262
 
263
263
  ## Reading the numbers honestly
@@ -50,21 +50,21 @@ class ModelSpec:
50
50
  sweep's shared budget, which is all but the largest.
51
51
 
52
52
  Separate from :attr:`extra_vllm_kwargs` because that is a fact about the weights and applies to
53
- every vLLM column, so putting a freeze-only budget there would change three cells that have
53
+ every vLLM column, so putting a static-only budget there would change three cells that have
54
54
  already been measured and are comparable as they stand. The overrides here are recorded in the cell
55
55
  and stated by the report, so the one cell that ran on a different budget says so."""
56
- freeze_capture_point: str | None = None
57
- """The point those workloads address instead when the variant under test froze a set, for a
56
+ static_capture_point: str | None = None
57
+ """The point those workloads address instead when the variant under test declared a set, for a
58
58
  checkpoint where the two cannot be the same point. ``None`` means :attr:`capture_point` serves both.
59
59
 
60
- A frozen engine serves *the set it froze* and refuses anything outside it, so the workload has to
61
- name a point the freeze set covers. On a hyper-connection trunk it cannot be the same one:
62
- ``freeze_points="auto"`` resolves to ``resid_streams`` -- the whole stack -- while the other columns
60
+ A static engine serves *the set it declared* and refuses anything outside it, so the workload has to
61
+ name a point the static set covers. On a hyper-connection trunk it cannot be the same one:
62
+ ``static_points="auto"`` resolves to ``resid_streams`` -- the whole stack -- while the other columns
63
63
  need one ``d_model`` row to stay comparable with the rest of the table. Declared here rather than
64
64
  inferred at run time so the cell records it, ``report_bench`` states it under *Where a row differs*,
65
65
  and ``cells.nonuniform`` carries it onto the visualizer's card as one line of that row's footnote.
66
66
 
67
- Only the freeze cell records it, so read it through ``cells.row_spec`` rather than off a cell.
67
+ Only the static cell records it, so read it through ``cells.row_spec`` rather than off a cell.
68
68
  """
69
69
  extra_eager_kwargs: dict[str, object] = field(default_factory=dict)
70
70
  """``load_model`` arguments the eager variants need for this checkpoint, merged under the
@@ -138,15 +138,15 @@ MODELS: tuple[ModelSpec, ...] = (
138
138
  # bytes, which would price this row's transport as though DeepSeek moved four times the
139
139
  # activations to answer the same question.
140
140
  capture_point="mlp_out",
141
- # ...except under freeze, which serves the set it froze: `"auto"` on this trunk freezes
141
+ # ...except under static, which serves the set it declared: `"auto"` on this trunk declares
142
142
  # `resid_streams`, so `mlp_out` is outside the set and would be refused. That column therefore
143
143
  # prices the stack rather than a row, which is declared as a row exception instead of hidden.
144
- freeze_capture_point="resid_streams",
145
- # The freeze column is the one configuration here that does not fit the shared budget, and both
144
+ static_capture_point="resid_streams",
145
+ # The static column is the one configuration here that does not fit the shared budget, and both
146
146
  # of these are why. 149 GiB of weights at 0.95 of a 178 GiB card leave about 15 GiB for the
147
- # activation peak, the freeze buffers, the KV pool and the graph pool together.
147
+ # activation peak, the static buffers, the KV pool and the graph pool together.
148
148
  #
149
- # A freeze buffer is allocated per batched row, so at the 4096 `max_num_batched_tokens` freeze
149
+ # A static buffer is allocated per batched row, so at the 4096 `max_num_batched_tokens` static
150
150
  # lowered vLLM's default to, `resid_streams` costs 4096 rows x 4 streams x 4096 wide x 2 bytes
151
151
  # x 43 layers = 5.8 GiB -- and vLLM then sized the KV pool from what was left and refused to
152
152
  # start: "No available memory for the cache blocks". 1024 rows costs a quarter of that and is
@@ -157,7 +157,7 @@ MODELS: tuple[ModelSpec, ...] = (
157
157
  # sizes it will actually replay keeps the rest of that memory, and the startup time, unspent.
158
158
  # Neither changes what the workloads measure: both sizes they use are still captured.
159
159
  per_variant_vllm_kwargs={
160
- "vllm-freeze": {
160
+ "vllm-static": {
161
161
  "max_num_batched_tokens": 1024,
162
162
  "compilation_config": {"cudagraph_capture_sizes": [1, 2, 4, 8, 16, 32]},
163
163
  }
@@ -215,12 +215,17 @@ class VariantSpec:
215
215
 
216
216
 
217
217
  #: One of these three exists to price a default that the engine chose for capture's sake, which is the
218
- #: most useful thing a speed benchmark of this library can say. vLLM capture needs
219
- #: ``enforce_eager=True``, because CUDA-graph replay does not re-execute the Python forward and so
220
- #: never fires a ``register_forward_hook``. That is why ``VLLMModel`` defaults it to True, and why
221
- #: ``vllm-cudagraph`` -- vLLM left at its own defaults, hence the "vanilla" label -- is here at all:
222
- #: it measures what ours costs. The capture workloads run there too rather than being skipped, so the
223
- #: report can show *what* a capture returns under replay instead of asserting that it fails.
218
+ #: most useful thing a speed benchmark of this library can say. Hooked vLLM capture rules CUDA graphs
219
+ #: out, because graph replay does not re-execute the Python forward and so never fires a
220
+ #: ``register_forward_hook``. That is what separates the three vLLM backends, and it is why
221
+ #: ``vllm-cudagraph`` -- ``backend="vllm-generate"``, vLLM left at its own defaults, hence the
222
+ #: "vanilla" label -- is here at all: it measures what ours costs. The capture workloads run there too
223
+ #: rather than being skipped, so the report can show *what* a capture returns under replay instead of
224
+ #: asserting that it fails.
225
+ #:
226
+ #: Each variant names its backend rather than deriving it from the kwargs it passes, which is what
227
+ #: the engine now expects: ``vllm-static`` and ``vllm-generate`` refuse ``enforce_eager``, and a tap
228
+ #: set is only accepted by the backend built to bake one in.
224
229
  VARIANTS: tuple[VariantSpec, ...] = (
225
230
  VariantSpec(
226
231
  "eager",
@@ -232,52 +237,52 @@ VARIANTS: tuple[VariantSpec, ...] = (
232
237
  VariantSpec(
233
238
  "vllm",
234
239
  "vllm",
235
- {"enforce_eager": True},
240
+ {},
236
241
  "capture-capable: CUDA graphs and compile OFF",
237
242
  "interp-engine vllm",
238
243
  ),
239
244
  VariantSpec(
240
245
  "vllm-cudagraph",
241
- "vllm",
242
- {"enforce_eager": False, "freeze_points": []},
243
- "CUDA graphs + inductor compile ON, no freeze wraps; vLLM's own defaults for generate-only",
246
+ "vllm-generate",
247
+ {},
248
+ "CUDA graphs + inductor compile ON, no static wraps; vLLM's own defaults for generate-only",
244
249
  "vllm (vanilla)",
245
250
  ),
246
- # Breakable CUDA graphs with resid_post freeze wraps at every layer. Production freeze
247
- # (``freeze_points="auto"``) is this path, not Dynamo piecewise.
251
+ # Breakable CUDA graphs with resid_post static wraps at every layer. Production static
252
+ # (``static_points="auto"``) is this path, not Dynamo piecewise.
248
253
  #
249
- # `qwen3.8-27b` joined the row once freeze was correct on a hybrid trunk. Breakable graphs turn
254
+ # `qwen3.8-27b` joined the row once static was correct on a hybrid trunk. Breakable graphs turn
250
255
  # inductor off, and vLLM's `FULL_AND_PIECEWISE` capture then miscomputes prefill for a gated-delta
251
256
  # trunk -- the engine generated fluent nonsense rather than failing, so the cell would have been a
252
- # plausible number for a broken forward. Freeze now pins `FULL_DECODE_ONLY` on such a trunk, which
253
- # runs prefill eagerly and keeps the decode graphs, and the validator's freeze column agrees with
257
+ # plausible number for a broken forward. Static now pins `FULL_DECODE_ONLY` on such a trunk, which
258
+ # runs prefill eagerly and keeps the decode graphs, and the validator's static column agrees with
254
259
  # eager on all 28 points. Its prefill figures carry that eager prefill, which is the point of
255
260
  # comparing it against the same model's other variants rather than against another model.
256
261
  #
257
- # `deepseek-v4-flash-0731` is in, and is the one row here whose freeze set is not one point per
262
+ # `deepseek-v4-flash-0731` is in, and is the one row here whose static set is not one point per
258
263
  # layer. It is a hyper-connection trunk, so `"auto"` resolves to `resid_streams` -- the whole stack
259
264
  # of four parallel residual streams per layer, four times the width of a `resid_post` row. Its
260
265
  # capture and transport figures therefore price four times the activations for the same question,
261
266
  # which `cells.nonuniform` declares so the report states it and the visualizer's card drops the row
262
- # rather than publishing it beside three that froze a quarter as much.
267
+ # rather than publishing it beside three that declared a quarter as much.
263
268
  #
264
269
  # Its write tap is `mlp_out`, not `resid_post`: the steer workload addresses the model's
265
270
  # `capture_point`, and a hyper-connection trunk refuses the default name (`run_bench._steer_site`).
266
271
  #
267
- # `freeze_writes` was once what made the `steer` cell a number instead of `n/a`: `"auto"` installed
272
+ # `static_writes` was once what made the `steer` cell a number instead of `n/a`: `"auto"` installed
268
273
  # reads, and a steering op needs a write tap to land in, so this row priced capture under replay
269
274
  # and left the other half of the feature unmeasured -- with a message that blamed graph replay for
270
275
  # it. Auto covers writes now, so the cell stands either way, and naming them here has become a
271
276
  # *narrowing*: one write buffer at the site the workload steers rather than one per layer, which
272
277
  # is what keeps this row's memory comparable with the columns beside it. The value is the sentinel
273
278
  # `run_bench.STEER_WRITES`, resolved there to the mid-stack `resid_post` the workload steers,
274
- # because the layer differs per model and a freeze write has to be named before the model exists.
279
+ # because the layer differs per model and a static write has to be named before the model exists.
275
280
  VariantSpec(
276
- "vllm-freeze",
277
- "vllm",
278
- {"freeze_points": "auto", "freeze_writes": "steer"},
279
- "breakable CUDA graphs with resid_post freeze wraps at every layer, and a write tap mid-stack",
280
- "interp-engine vllm freeze",
281
+ "vllm-static",
282
+ "vllm-static",
283
+ {"static_points": "auto", "static_writes": "steer"},
284
+ "breakable CUDA graphs with resid_post static wraps at every layer, and a write tap mid-stack",
285
+ "interp-engine vllm static",
281
286
  models=("gemma-2-2b", "qwen3-4b", "llama-3.1-8b", "qwen3.8-27b", "deepseek-v4-flash-0731"),
282
287
  ),
283
288
  # DSpark on, against the `vllm` column with it off -- the pair is the measurement, so read the two
@@ -303,12 +308,11 @@ VARIANTS: tuple[VariantSpec, ...] = (
303
308
  "vllm-dspark",
304
309
  "vllm",
305
310
  {
306
- "enforce_eager": True,
307
311
  "extra_vllm_kwargs": {
308
312
  "speculative_config": {"method": "dspark", "num_speculative_tokens": 5},
309
313
  },
310
314
  },
311
- "DSpark speculative decoding ON; enforce_eager, so still capture-capable",
315
+ "DSpark speculative decoding ON; hooked backend, so still capture-capable",
312
316
  "interp-engine vllm +DSpark",
313
317
  models=("deepseek-v4-flash-0731",),
314
318
  ),
@@ -325,10 +329,8 @@ VARIANTS: tuple[VariantSpec, ...] = (
325
329
  # never calls the Python forward a hook is attached to.
326
330
  VariantSpec(
327
331
  "vllm-dspark-cudagraph",
328
- "vllm",
332
+ "vllm-generate",
329
333
  {
330
- "enforce_eager": False,
331
- "freeze_points": [],
332
334
  "extra_vllm_kwargs": {
333
335
  "speculative_config": {"method": "dspark", "num_speculative_tokens": 5},
334
336
  },
@@ -77,9 +77,9 @@ def row_spec(cells: list[dict[str, Any]], model_key: str, variants: Iterable[str
77
77
  """Every override one model's row declared, merged across the row's cells.
78
78
 
79
79
  A per-variant override is recorded by the cell that used it and by no other: on
80
- `deepseek-v4-flash-0731` both `freeze_capture_point` and `per_variant_vllm_kwargs` live on the
81
- freeze cell alone. So any single cell's ``model`` describes a column rather than a row, and picking
82
- one -- whichever happened to sort last -- silently dropped the two overrides the freeze column is
80
+ `deepseek-v4-flash-0731` both `static_capture_point` and `per_variant_vllm_kwargs` live on the
81
+ static cell alone. So any single cell's ``model`` describes a column rather than a row, and picking
82
+ one -- whichever happened to sort last -- silently dropped the two overrides the static column is
83
83
  the only cell to declare.
84
84
 
85
85
  ``variants`` restricts the merge to the columns a renderer shows, and each of them passes its own:
@@ -129,9 +129,9 @@ def nonuniform(model: dict[str, Any]) -> list[str]:
129
129
  point = model.get("capture_point", DEFAULT_POINT)
130
130
  if point != DEFAULT_POINT:
131
131
  reasons.append(f"captures `{point}` rather than `{DEFAULT_POINT}`")
132
- frozen_point = model.get("freeze_capture_point")
133
- if frozen_point:
134
- reasons.append(f"its freeze column captures `{frozen_point}`, a wider point than the others do")
132
+ static_point = model.get("static_capture_point")
133
+ if static_point:
134
+ reasons.append(f"its static column captures `{static_point}`, a wider point than the others do")
135
135
  fraction = model.get("gpu_memory_utilization")
136
136
  if fraction:
137
137
  reasons.append(f"vLLM reserved {fraction} of the card, not the uniform {GPU_MEMORY_UTILIZATION}")