interp-engine 1.0.1__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (208) hide show
  1. interp_engine-1.2.0/.gitignore +36 -0
  2. interp_engine-1.2.0/PKG-INFO +140 -0
  3. interp_engine-1.2.0/README.md +109 -0
  4. interp_engine-1.2.0/benchmarks/README.md +349 -0
  5. interp_engine-1.2.0/benchmarks/bench_spec.py +516 -0
  6. interp_engine-1.2.0/benchmarks/cells.py +163 -0
  7. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/probe.py +19 -62
  8. interp_engine-1.2.0/benchmarks/publish.py +391 -0
  9. interp_engine-1.2.0/benchmarks/report_bench.py +509 -0
  10. interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__eager.json +175 -0
  11. interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-cudagraph.json +135 -0
  12. interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark-cudagraph.json +106 -0
  13. interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-dspark.json +180 -0
  14. interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm-freeze.json +197 -0
  15. interp_engine-1.2.0/benchmarks/results/deepseek-v4-flash-0731__vllm.json +180 -0
  16. interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__eager.json → interp_engine-1.2.0/benchmarks/results/gemma-2-2b__eager.json +48 -63
  17. interp_engine-1.2.0/benchmarks/results/gemma-2-2b__vllm-cudagraph.json +131 -0
  18. interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__vllm.json → interp_engine-1.2.0/benchmarks/results/gemma-2-2b__vllm-freeze.json +56 -69
  19. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/gemma-2-2b__vllm.json +53 -67
  20. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/llama-3.1-8b__eager.json +48 -63
  21. interp_engine-1.2.0/benchmarks/results/llama-3.1-8b__vllm-cudagraph.json +131 -0
  22. interp_engine-1.0.1/benchmarks/results/llama-3.1-8b__eager-sdpa.json → interp_engine-1.2.0/benchmarks/results/llama-3.1-8b__vllm-freeze.json +63 -73
  23. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/llama-3.1-8b__vllm.json +53 -67
  24. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/qwen3-4b__eager.json +49 -64
  25. interp_engine-1.2.0/benchmarks/results/qwen3-4b__vllm-cudagraph.json +131 -0
  26. interp_engine-1.2.0/benchmarks/results/qwen3-4b__vllm-freeze.json +177 -0
  27. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/results/qwen3-4b__vllm.json +53 -67
  28. interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__eager.json +171 -0
  29. interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__vllm-cudagraph.json +131 -0
  30. interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__vllm-freeze.json +177 -0
  31. interp_engine-1.2.0/benchmarks/results/qwen3.8-27b__vllm.json +176 -0
  32. interp_engine-1.2.0/benchmarks/results-latest.md +254 -0
  33. interp_engine-1.2.0/benchmarks/run_all.sh +209 -0
  34. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/run_bench.py +113 -16
  35. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/workloads.py +112 -61
  36. interp_engine-1.2.0/docs/AGENT_INTEGRATION.md +274 -0
  37. {interp_engine-1.0.1 → interp_engine-1.2.0}/docs/ARCHITECTURE_QUIRKS.md +294 -170
  38. interp_engine-1.2.0/docs/COMPATIBILITY.md +70 -0
  39. interp_engine-1.2.0/docs/ENGINE_HOOK_MAPPINGS.md +596 -0
  40. {interp_engine-1.0.1 → interp_engine-1.2.0}/docs/GRADIENTS.md +28 -3
  41. interp_engine-1.2.0/docs/INTERNALS.md +121 -0
  42. interp_engine-1.2.0/docs/PERFORMANCE.md +247 -0
  43. {interp_engine-1.0.1 → interp_engine-1.2.0}/docs/PORTING.md +19 -6
  44. interp_engine-1.2.0/docs/README.md +21 -0
  45. interp_engine-1.2.0/docs/SUPPORTED_POINTS.md +119 -0
  46. interp_engine-1.2.0/docs/USAGE.md +398 -0
  47. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/__init__.py +44 -2
  48. interp_engine-1.2.0/interp_engine/_loop.py +174 -0
  49. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/address.py +2 -2
  50. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/arch.py +181 -36
  51. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/attn_config.py +1 -1
  52. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/autograd_support.py +50 -5
  53. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/capture.py +326 -22
  54. interp_engine-1.2.0/interp_engine/dispatch.py +204 -0
  55. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/facts.py +855 -33
  56. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/hooks.py +75 -10
  57. interp_engine-1.2.0/interp_engine/lens.py +224 -0
  58. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/load.py +29 -19
  59. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/mappers.py +185 -8
  60. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/model.py +188 -55
  61. interp_engine-1.2.0/interp_engine/moe_routing.py +81 -0
  62. interp_engine-1.2.0/interp_engine/points.py +580 -0
  63. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/protocol.py +55 -2
  64. interp_engine-1.2.0/interp_engine/residual_basis.py +534 -0
  65. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/select.py +14 -0
  66. interp_engine-1.2.0/interp_engine/steer.py +608 -0
  67. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/steer_specs.py +31 -7
  68. interp_engine-1.2.0/interp_engine/sync.py +235 -0
  69. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/tokenize.py +19 -3
  70. interp_engine-1.2.0/interp_engine/vllm_backend.py +2562 -0
  71. interp_engine-1.2.0/interp_engine/vllm_capture/__init__.py +262 -0
  72. interp_engine-1.2.0/interp_engine/vllm_capture/_demux.py +265 -0
  73. interp_engine-1.2.0/interp_engine/vllm_capture/_hooks.py +215 -0
  74. interp_engine-1.2.0/interp_engine/vllm_capture/_payload.py +168 -0
  75. interp_engine-1.2.0/interp_engine/vllm_capture/_tree.py +787 -0
  76. interp_engine-1.2.0/interp_engine/vllm_capture/attn.py +291 -0
  77. interp_engine-1.2.0/interp_engine/vllm_capture/capture.py +178 -0
  78. interp_engine-1.2.0/interp_engine/vllm_capture/freeze.py +1785 -0
  79. interp_engine-1.2.0/interp_engine/vllm_capture/graphs.py +114 -0
  80. interp_engine-1.2.0/interp_engine/vllm_capture/lens/__init__.py +56 -0
  81. interp_engine-1.2.0/interp_engine/vllm_capture/lens/intervene.py +142 -0
  82. interp_engine-1.2.0/interp_engine/vllm_capture/lens/readout.py +443 -0
  83. interp_engine-1.2.0/interp_engine/vllm_capture/lens/unembed.py +300 -0
  84. interp_engine-1.2.0/interp_engine/vllm_capture/mhc.py +715 -0
  85. interp_engine-1.2.0/interp_engine/vllm_capture/native.py +95 -0
  86. interp_engine-1.2.0/interp_engine/vllm_capture/requests.py +702 -0
  87. interp_engine-1.2.0/interp_engine/vllm_capture/steering.py +173 -0
  88. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/vllm_plugin.py +65 -0
  89. interp_engine-1.2.0/pyproject.toml +245 -0
  90. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/harness.py +22 -0
  91. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/model_expectations.yaml +3 -5
  92. interp_engine-1.2.0/tests/synthetic_families.py +408 -0
  93. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_autograd_support.py +63 -0
  94. interp_engine-1.2.0/tests/test_bench_workloads.py +44 -0
  95. interp_engine-1.2.0/tests/test_capability_refusals.py +186 -0
  96. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_chat_templates.py +33 -0
  97. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_core.py +8 -8
  98. interp_engine-1.2.0/tests/test_doc_code_fences.py +135 -0
  99. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_eager_autograd.py +54 -0
  100. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_facts.py +122 -1
  101. interp_engine-1.2.0/tests/test_family_points.py +430 -0
  102. interp_engine-1.2.0/tests/test_freeze_dsv4_gpu.py +124 -0
  103. interp_engine-1.2.0/tests/test_freeze_parity_gpu.py +253 -0
  104. interp_engine-1.2.0/tests/test_freeze_set.py +1009 -0
  105. interp_engine-1.2.0/tests/test_freeze_warmup.py +138 -0
  106. interp_engine-1.2.0/tests/test_hook_call_conventions.py +163 -0
  107. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_layer_kinds.py +92 -2
  108. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_load.py +124 -4
  109. interp_engine-1.2.0/tests/test_logit_transform.py +201 -0
  110. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_mappers.py +143 -6
  111. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_mlp_internals.py +61 -3
  112. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_model_expectations.py +2 -4
  113. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_moe.py +258 -3
  114. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_new_models_gpu.py +97 -5
  115. interp_engine-1.2.0/tests/test_normalized_hook.py +303 -0
  116. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_packaging.py +30 -1
  117. interp_engine-1.2.0/tests/test_per_layer_attn_dims.py +434 -0
  118. interp_engine-1.2.0/tests/test_points_registry.py +330 -0
  119. interp_engine-1.2.0/tests/test_published_benchmarks.py +166 -0
  120. interp_engine-1.2.0/tests/test_qk_norm.py +382 -0
  121. interp_engine-1.2.0/tests/test_release.py +211 -0
  122. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_resid_mid.py +113 -0
  123. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_residual_basis.py +70 -3
  124. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_select.py +33 -0
  125. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_small_models_gpu.py +7 -1
  126. interp_engine-1.2.0/tests/test_steer_context.py +173 -0
  127. interp_engine-1.2.0/tests/test_steer_math_parity.py +278 -0
  128. interp_engine-1.2.0/tests/test_sync_loop.py +145 -0
  129. interp_engine-1.2.0/tests/test_sync_parity.py +202 -0
  130. interp_engine-1.2.0/tests/test_unified_free_functions.py +114 -0
  131. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_capture_gpu.py +94 -4
  132. interp_engine-1.2.0/tests/test_vllm_capture_scales.py +168 -0
  133. interp_engine-1.2.0/tests/test_vllm_graph_path.py +135 -0
  134. interp_engine-1.2.0/tests/test_vllm_graphs_on_gpu.py +212 -0
  135. interp_engine-1.2.0/tests/test_vllm_hook_availability.py +259 -0
  136. interp_engine-1.2.0/tests/test_vllm_hyper_connections.py +1374 -0
  137. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_kv_isolation.py +20 -10
  138. interp_engine-1.2.0/tests/test_vllm_new_points.py +746 -0
  139. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_wire_grammar.py +48 -0
  140. interp_engine-1.2.0/tests/test_worker_lens_capture_readout.py +392 -0
  141. interp_engine-1.2.0/tests/test_worker_lens_readout.py +200 -0
  142. interp_engine-1.0.1/.gitignore +0 -33
  143. interp_engine-1.0.1/AGENTS.md +0 -54
  144. interp_engine-1.0.1/CLAUDE.md +0 -1
  145. interp_engine-1.0.1/Makefile +0 -12
  146. interp_engine-1.0.1/PKG-INFO +0 -210
  147. interp_engine-1.0.1/README.md +0 -179
  148. interp_engine-1.0.1/benchmarks/README.md +0 -257
  149. interp_engine-1.0.1/benchmarks/bench_spec.py +0 -213
  150. interp_engine-1.0.1/benchmarks/report_bench.py +0 -547
  151. interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__eager-sdpa.json +0 -187
  152. interp_engine-1.0.1/benchmarks/results/gemma-2-2b-bf16__vllm-cudagraph.json +0 -190
  153. interp_engine-1.0.1/benchmarks/results/gemma-2-2b__eager-sdpa.json +0 -187
  154. interp_engine-1.0.1/benchmarks/results/gemma-2-2b__eager.json +0 -186
  155. interp_engine-1.0.1/benchmarks/results/gemma-2-2b__vllm-cudagraph.json +0 -190
  156. interp_engine-1.0.1/benchmarks/results/gemma-3-1b__eager-sdpa.json +0 -187
  157. interp_engine-1.0.1/benchmarks/results/gemma-3-1b__eager.json +0 -186
  158. interp_engine-1.0.1/benchmarks/results/gemma-3-1b__vllm-cudagraph.json +0 -190
  159. interp_engine-1.0.1/benchmarks/results/gemma-3-1b__vllm.json +0 -190
  160. interp_engine-1.0.1/benchmarks/results/llama-3.1-8b__vllm-cudagraph.json +0 -190
  161. interp_engine-1.0.1/benchmarks/results/qwen3-4b__eager-sdpa.json +0 -187
  162. interp_engine-1.0.1/benchmarks/results/qwen3-4b__vllm-cudagraph.json +0 -190
  163. interp_engine-1.0.1/benchmarks/results-latest.md +0 -329
  164. interp_engine-1.0.1/benchmarks/run_all.sh +0 -149
  165. interp_engine-1.0.1/docs/ENGINE_HOOK_MAPPINGS.md +0 -212
  166. interp_engine-1.0.1/docs/PERFORMANCE.md +0 -108
  167. interp_engine-1.0.1/interp_engine/lens.py +0 -124
  168. interp_engine-1.0.1/interp_engine/points.py +0 -390
  169. interp_engine-1.0.1/interp_engine/residual_basis.py +0 -320
  170. interp_engine-1.0.1/interp_engine/steer.py +0 -359
  171. interp_engine-1.0.1/interp_engine/vllm_backend.py +0 -1223
  172. interp_engine-1.0.1/interp_engine/vllm_capture.py +0 -2038
  173. interp_engine-1.0.1/pyproject.toml +0 -171
  174. interp_engine-1.0.1/tests/synthetic_families.py +0 -153
  175. interp_engine-1.0.1/tests/test_family_points.py +0 -249
  176. interp_engine-1.0.1/tests/test_per_layer_attn_dims.py +0 -214
  177. interp_engine-1.0.1/tests/test_points_registry.py +0 -165
  178. interp_engine-1.0.1/tests/test_qk_norm.py +0 -199
  179. interp_engine-1.0.1/uv.lock +0 -4516
  180. {interp_engine-1.0.1 → interp_engine-1.2.0}/LICENSE +0 -0
  181. {interp_engine-1.0.1 → interp_engine-1.2.0}/benchmarks/__init__.py +0 -0
  182. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/attn_scores.py +0 -0
  183. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/chat_compose.py +0 -0
  184. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/chat_conventions.py +0 -0
  185. {interp_engine-1.0.1 → interp_engine-1.2.0}/interp_engine/cuda_preflight.py +0 -0
  186. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/conftest.py +0 -0
  187. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_address.py +0 -0
  188. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_config_tripwire.py +0 -0
  189. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_probs_indexing.py +0 -0
  190. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_scores.py +0 -0
  191. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_attn_z_gqa.py +0 -0
  192. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_capture_addressing.py +0 -0
  193. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_chat_compose.py +0 -0
  194. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_cuda_preflight.py +0 -0
  195. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_gated_attn_out.py +0 -0
  196. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_head_contributions.py +0 -0
  197. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_multimodal_arch.py +0 -0
  198. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_no_chat_template.py +0 -0
  199. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_parity_gpt2.py +0 -0
  200. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_protocol.py +0 -0
  201. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_qkv_layout.py +0 -0
  202. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_reasoning_spans.py +0 -0
  203. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_sandwich_norms.py +0 -0
  204. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_sliding_window_attn.py +0 -0
  205. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_unresolved_families.py +0 -0
  206. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_only_families.py +0 -0
  207. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vllm_plugin.py +0 -0
  208. {interp_engine-1.0.1 → interp_engine-1.2.0}/tests/test_vocabulary_boundary.py +0 -0
@@ -0,0 +1,36 @@
1
+ # Secrets. Gated-model tests and the comparison sweep both read HF_TOKEN, so a real token lives
2
+ # here on any box that runs them.
3
+ .env
4
+ .env.*
5
+
6
+ # Python caches
7
+ __pycache__/
8
+ *__pycache__
9
+ *.pyc
10
+ *.pyo
11
+
12
+ # Virtual envs. Each project in this repo has its own: interp-engine at the root, the validator in
13
+ # validator/, and the per-engine venvs the sweep builds (.venv-tlens, .venv-vllm, ...).
14
+ .venv
15
+ .venv/
16
+ .venv-*/
17
+ venv/
18
+ venv.bak/
19
+
20
+ # Build / packaging artifacts
21
+ build/
22
+ dist/
23
+ *.egg-info/
24
+ .eggs/
25
+
26
+ # Tooling caches
27
+ .pytest_cache/
28
+ .ruff_cache/
29
+ .coverage
30
+ .mypy_cache/
31
+
32
+ # Local CI venv (see .github/workflows/engine-tests.yml)
33
+ .venv-ci/
34
+
35
+ plans/
36
+ *.plan.md
@@ -0,0 +1,140 @@
1
+ Metadata-Version: 2.5
2
+ Name: interp-engine
3
+ Version: 1.2.0
4
+ Summary: A fast, standardized interpretability engine that supports most modern models and architectures. Powers Neuronpedia.
5
+ Project-URL: Homepage, https://github.com/decoderesearch/interp-engine
6
+ Project-URL: Repository, https://github.com/decoderesearch/interp-engine
7
+ Project-URL: Issues, https://github.com/decoderesearch/interp-engine/issues
8
+ License-Expression: Apache-2.0
9
+ License-File: LICENSE
10
+ Requires-Python: <3.14,>=3.11
11
+ Requires-Dist: einops
12
+ Requires-Dist: numpy>=1.24
13
+ Requires-Dist: torch>=1.10
14
+ Requires-Dist: transformers>=4.57.1
15
+ Provides-Extra: awq
16
+ Requires-Dist: accelerate>=1.0; extra == 'awq'
17
+ Requires-Dist: gptqmodel>=5.0; extra == 'awq'
18
+ Provides-Extra: dev
19
+ Requires-Dist: pyright<1.2,>=1.1.411; extra == 'dev'
20
+ Requires-Dist: pytest<9,>=8.3.1; extra == 'dev'
21
+ Requires-Dist: pyyaml>=6; extra == 'dev'
22
+ Requires-Dist: ruff<0.17,>=0.16.2; extra == 'dev'
23
+ Provides-Extra: parity
24
+ Requires-Dist: transformer-lens>=3.0; extra == 'parity'
25
+ Provides-Extra: quant
26
+ Requires-Dist: accelerate>=1.0; extra == 'quant'
27
+ Requires-Dist: kernels<0.17.0,>=0.15.2; extra == 'quant'
28
+ Provides-Extra: vllm
29
+ Requires-Dist: vllm>=0.25.1; (sys_platform == 'linux') and extra == 'vllm'
30
+ Description-Content-Type: text/markdown
31
+
32
+ # interp-engine
33
+
34
+ <p align="center">
35
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
36
+ </p>
37
+ <p align="center">
38
+ 🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
39
+ </p>
40
+
41
+ `interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference (circuit tracing, j-lens, steering, NLAs, etc) and is checked for accuracy against HF Transformers and other engines.
42
+
43
+ <p align="center">
44
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-freeze on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="620">
45
+ </p>
46
+ <p align="center">
47
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="480">
48
+ </p>
49
+
50
+ This repo contains:
51
+
52
+ 1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
53
+ 2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
54
+
55
+ ## Installation
56
+
57
+ ```bash
58
+ pip install 'interp-engine[vllm]' # preferred install: includes vLLM support (CUDA required)
59
+ pip install interp-engine # eager backend only
60
+ ```
61
+
62
+ ## Simple Usage
63
+
64
+ ```python
65
+ from interp_engine import Address, load_model, run_with_cache
66
+
67
+ model = load_model("Qwen/Qwen3-8B") # vLLM default, use backend='eager' to override
68
+ point = Address("resid_post", 10) # or string: "resid_post.10"
69
+ cache = run_with_cache(model, model.to_tokens("Hello, world"), [point])
70
+ cache[point] # [batch, pos, ...]
71
+ ```
72
+
73
+ ### AI Agents
74
+
75
+ Add "use interp-engine" to your prompt and let your agent figure it out - everything is fully documented in this repo and open source.
76
+
77
+ ## Supported Points ("Addresses")
78
+
79
+ `interp-engine` supports 34 standardized points ("Addresses") across architectures: every one of them on the eager backend, 28 of them on vLLM. Check [interp-engine.org](https://interp-engine.org) for the "cheat sheet", or [SUPPORTED_POINTS.md](docs/SUPPORTED_POINTS.md) for a markdown version with the per-backend detail.
80
+
81
+ ## Performance / Speed
82
+
83
+ vLLM gives `interp-engine` high throughput via concurrency, and **graph freeze** adds CUDA-graph replay on top of that _without giving up capture or steering_. Every column below is capture-capable.
84
+
85
+ <!-- THROUGHPUT:START -->
86
+
87
+ <!-- Generated by `python -m benchmarks.report_bench`. Do not edit: rerun the sweep. -->
88
+
89
+ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
90
+
91
+ One stream (tok/s):
92
+
93
+ | model | eager | vLLM | vLLM + graph freeze |
94
+ | ------------------------ | ----- | ---------- | ------------------- |
95
+ | `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
96
+ | `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
97
+ | `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
98
+ | `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
99
+ | `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
100
+
101
+ 8 concurrent requests (aggregate tok/s):
102
+
103
+ | model | eager | vLLM | vLLM + graph freeze |
104
+ | ------------------------ | ----- | ----------- | ------------------- |
105
+ | `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
106
+ | `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
107
+ | `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
108
+ | `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
109
+ | `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
110
+
111
+ <!-- THROUGHPUT:END -->
112
+
113
+ Every multiplier is against eager on the same workload, and is the ratio of the _unrounded_ figures — tok/s is printed whole at 10 and above, so dividing two cells by hand can differ in the last place. Eager's generation loop is synchronous underneath, so eight requests serialize rather than batch — which is why it earns almost nothing from the second table and why the multipliers there are so much larger. `deepseek-v4-flash-0731` is block-quantized FP8 on both backends rather than bf16, so read its row as two backends serving the same quantized weights; it also keeps a tenth, because at 3 tok/s that digit is worth several percent. [`benchmarks/results-latest.md`](benchmarks/results-latest.md) has every figure at full precision.
114
+
115
+ **Graph freeze** is opt-in, via `freeze_points`. Ordinary hooks cannot survive CUDA graphs — replay never calls the Python forward a hook is attached to — so the tap becomes a preallocated buffer plus a `copy_` that graph capture records and replay re-executes. Steering rides the same wrap, so additive, orthogonal, projection-cap and j-lens steer/ablate/swap all keep working under graphs, position masks included.
116
+
117
+ ```python
118
+ model = load_model("Qwen/Qwen3-8B", freeze_points="auto") # resid_post at every layer, under graphs
119
+ ```
120
+
121
+ The trade is that a frozen engine serves _the set it froze_ rather than any point on request: `"auto"` covers `resid_post` at every layer on a conventional trunk, an explicit list covers whatever you name, and anything outside the set is refused rather than quietly returned empty. Omit `freeze_points` for today's hooked vLLM, which still serves every point. `qwen3.8-27b` is a hybrid trunk, so its freeze row runs prefill eagerly and keeps the decode graphs — freeze pins that mode, because breakable graphs turn `torch.compile` off and vLLM's mixed prefill-decode capture then miscomputes prefill on a gated-delta trunk. On a hyper-connection trunk such as `deepseek-v4-flash-0731`, whose block carries four parallel residual streams, `"auto"` freezes `resid_streams` instead — the whole stack per layer, which is four times the width and so four times the buffer, and is what that row's freeze cell prices. See [PERFORMANCE.md](docs/PERFORMANCE.md) for the full trade-off, and [`benchmarks/results-latest.md`](benchmarks/results-latest.md) for the sweep, including capture, steering and lens latencies.
122
+
123
+ ## Correctness
124
+
125
+ We verify correctness in two main ways:
126
+
127
+ 1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
128
+ 2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
129
+
130
+ ## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
131
+
132
+ Of course, software is easy to make these days. But as smart as as AIs are these days, there are benefits to leveraging a pre-written engine:
133
+
134
+ 1. **Speed**: Get performance without sacrificing correctness.
135
+ 1. **Standardization + Verification**: Eliminate ambiguity when referring to points, plus a full test suite included.
136
+ 1. **Faster Dev / Fewer Tokens Used**: You could spend ten million tokens and have your AI write, test, and make production-ready an interpretability engine. Or you could just `pip install interp-engine[vllm]`.
137
+
138
+ ## License
139
+
140
+ Apache 2.0
@@ -0,0 +1,109 @@
1
+ # interp-engine
2
+
3
+ <p align="center">
4
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ielogo.png" alt="interp-engine logo, a magnifying glass where the handle is a rocket" width="160">
5
+ </p>
6
+ <p align="center">
7
+ 🔗 <a href="https://interp-engine.org"><strong>interp-engine.org</strong></a>
8
+ </p>
9
+
10
+ `interp-engine` is an interpretability engine that is fast (>40x tok/s vs HF eager), standardized (34 'points'/addresses across architectures), and easy to use and debug. It powers all of [Neuronpedia](https://neuronpedia.org)'s inference (circuit tracing, j-lens, steering, NLAs, etc) and is checked for accuracy against HF Transformers and other engines.
11
+
12
+ <p align="center">
13
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/ie-perf.png" alt="Tokens per second while capturing and generating: eager against IE-vLLM and IE-vLLM-freeze on qwen3.8-27b and deepseek-v4-flash-0731, 8 requests in flight" width="620">
14
+ </p>
15
+ <p align="center">
16
+ <img src="https://neuronpedia.s3.amazonaws.com/site-assets/interp-engine-demo.gif" alt="interp-engine demo gif" width="480">
17
+ </p>
18
+
19
+ This repo contains:
20
+
21
+ 1. [`validator/`](validator/), which compares/validates it against TransformerLens, and nnsight/nnterp on real architectures.
22
+ 2. [`visualizer-web/`](visualizer-web/), a "cheat sheet" hosted at [interp-engine.org](https://interp-engine.org) of each 'point' (eg `resid_post.16`), standardized across architectures.
23
+
24
+ ## Installation
25
+
26
+ ```bash
27
+ pip install 'interp-engine[vllm]' # preferred install: includes vLLM support (CUDA required)
28
+ pip install interp-engine # eager backend only
29
+ ```
30
+
31
+ ## Simple Usage
32
+
33
+ ```python
34
+ from interp_engine import Address, load_model, run_with_cache
35
+
36
+ model = load_model("Qwen/Qwen3-8B") # vLLM default, use backend='eager' to override
37
+ point = Address("resid_post", 10) # or string: "resid_post.10"
38
+ cache = run_with_cache(model, model.to_tokens("Hello, world"), [point])
39
+ cache[point] # [batch, pos, ...]
40
+ ```
41
+
42
+ ### AI Agents
43
+
44
+ Add "use interp-engine" to your prompt and let your agent figure it out - everything is fully documented in this repo and open source.
45
+
46
+ ## Supported Points ("Addresses")
47
+
48
+ `interp-engine` supports 34 standardized points ("Addresses") across architectures: every one of them on the eager backend, 28 of them on vLLM. Check [interp-engine.org](https://interp-engine.org) for the "cheat sheet", or [SUPPORTED_POINTS.md](docs/SUPPORTED_POINTS.md) for a markdown version with the per-backend detail.
49
+
50
+ ## Performance / Speed
51
+
52
+ vLLM gives `interp-engine` high throughput via concurrency, and **graph freeze** adds CUDA-graph replay on top of that _without giving up capture or steering_. Every column below is capture-capable.
53
+
54
+ <!-- THROUGHPUT:START -->
55
+
56
+ <!-- Generated by `python -m benchmarks.report_bench`. Do not edit: rerun the sweep. -->
57
+
58
+ Measured on NVIDIA B200, bf16, 512-token prompt, 128 new tokens.
59
+
60
+ One stream (tok/s):
61
+
62
+ | model | eager | vLLM | vLLM + graph freeze |
63
+ | ------------------------ | ----- | ---------- | ------------------- |
64
+ | `gemma-2-2b` | 31 | 31 (1.0x) | **214 (6.9x)** |
65
+ | `qwen3-4b` | 24 | 47 (2.0x) | **296 (12.3x)** |
66
+ | `llama-3.1-8b` | 33 | 57 (1.7x) | **256 (7.9x)** |
67
+ | `qwen3.8-27b` | 9.9 | 12 (1.2x) | **63 (6.4x)** |
68
+ | `deepseek-v4-flash-0731` | 3.3 | 2.9 (0.9x) | **119 (36x)** |
69
+
70
+ 8 concurrent requests (aggregate tok/s):
71
+
72
+ | model | eager | vLLM | vLLM + graph freeze |
73
+ | ------------------------ | ----- | ----------- | ------------------- |
74
+ | `gemma-2-2b` | 30 | 226 (7.5x) | **1,238 (41x)** |
75
+ | `qwen3-4b` | 24 | 333 (14.0x) | **1,018 (43x)** |
76
+ | `llama-3.1-8b` | 32 | 419 (13.0x) | **1,536 (48x)** |
77
+ | `qwen3.8-27b` | 9.5 | 87 (9.2x) | **386 (41x)** |
78
+ | `deepseek-v4-flash-0731` | 3.2 | 23 (7.2x) | **402 (127x)** |
79
+
80
+ <!-- THROUGHPUT:END -->
81
+
82
+ Every multiplier is against eager on the same workload, and is the ratio of the _unrounded_ figures — tok/s is printed whole at 10 and above, so dividing two cells by hand can differ in the last place. Eager's generation loop is synchronous underneath, so eight requests serialize rather than batch — which is why it earns almost nothing from the second table and why the multipliers there are so much larger. `deepseek-v4-flash-0731` is block-quantized FP8 on both backends rather than bf16, so read its row as two backends serving the same quantized weights; it also keeps a tenth, because at 3 tok/s that digit is worth several percent. [`benchmarks/results-latest.md`](benchmarks/results-latest.md) has every figure at full precision.
83
+
84
+ **Graph freeze** is opt-in, via `freeze_points`. Ordinary hooks cannot survive CUDA graphs — replay never calls the Python forward a hook is attached to — so the tap becomes a preallocated buffer plus a `copy_` that graph capture records and replay re-executes. Steering rides the same wrap, so additive, orthogonal, projection-cap and j-lens steer/ablate/swap all keep working under graphs, position masks included.
85
+
86
+ ```python
87
+ model = load_model("Qwen/Qwen3-8B", freeze_points="auto") # resid_post at every layer, under graphs
88
+ ```
89
+
90
+ The trade is that a frozen engine serves _the set it froze_ rather than any point on request: `"auto"` covers `resid_post` at every layer on a conventional trunk, an explicit list covers whatever you name, and anything outside the set is refused rather than quietly returned empty. Omit `freeze_points` for today's hooked vLLM, which still serves every point. `qwen3.8-27b` is a hybrid trunk, so its freeze row runs prefill eagerly and keeps the decode graphs — freeze pins that mode, because breakable graphs turn `torch.compile` off and vLLM's mixed prefill-decode capture then miscomputes prefill on a gated-delta trunk. On a hyper-connection trunk such as `deepseek-v4-flash-0731`, whose block carries four parallel residual streams, `"auto"` freezes `resid_streams` instead — the whole stack per layer, which is four times the width and so four times the buffer, and is what that row's freeze cell prices. See [PERFORMANCE.md](docs/PERFORMANCE.md) for the full trade-off, and [`benchmarks/results-latest.md`](benchmarks/results-latest.md) for the sweep, including capture, steering and lens latencies.
91
+
92
+ ## Correctness
93
+
94
+ We verify correctness in two main ways:
95
+
96
+ 1. A test suite that checks results across several models - what each check is designed to catch is in [INTERNALS.md](docs/INTERNALS.md#correctness).
97
+ 2. A full `validator` comparison engine that checks most hook points across 50+ models, at early, middle and late layers - fully reproducible, with detailed results saved in the git repo at [`validator/`](validator/).
98
+
99
+ ## Why use an Interpretability Engine, instead of just having my AI code whatever it needs on the fly?
100
+
101
+ Of course, software is easy to make these days. But as smart as as AIs are these days, there are benefits to leveraging a pre-written engine:
102
+
103
+ 1. **Speed**: Get performance without sacrificing correctness.
104
+ 1. **Standardization + Verification**: Eliminate ambiguity when referring to points, plus a full test suite included.
105
+ 1. **Faster Dev / Fewer Tokens Used**: You could spend ten million tokens and have your AI write, test, and make production-ready an interpretability engine. Or you could just `pip install interp-engine[vllm]`.
106
+
107
+ ## License
108
+
109
+ Apache 2.0
@@ -0,0 +1,349 @@
1
+ # Speed benchmarks
2
+
3
+ How fast the two backends are at the things the engine does: generating, capturing activations,
4
+ steering, and the lens read-out. Latest numbers are in [`results-latest.md`](results-latest.md).
5
+
6
+ Every workload is written against the shared `InterpModel` protocol, so the same harness code drives
7
+ both backends and a difference in the numbers is a difference in the backend rather than in the
8
+ measurement.
9
+
10
+ `lens_topk` is the single deliberate exception, and it selects on `hasattr` rather than on a backend
11
+ name: it benchmarks the route each backend's *serving* code actually takes to the same answer, because
12
+ for that one read-out the two are genuinely different code (see below).
13
+
14
+ ## Requirements
15
+
16
+ - An interpreter with this checkout installed and, for the vLLM variants, the `vllm` extra:
17
+ `uv sync --extra vllm` in the repo root, or any venv that already has `interp-engine[vllm]`.
18
+ Without vLLM you can still run `--variants eager`.
19
+ - The `quant` extra (`uv sync --extra quant`) for the quantized rows, which today means
20
+ `deepseek-v4-flash-0731`. It brings `accelerate` and `kernels`, both of which transformers *requires* to
21
+ load and run a block-quantized FP8 checkpoint rather than merely preferring: without them the eager
22
+ cell fails at load with `Loading an FP8 quantized model requires accelerate`, or at the first
23
+ forward with `finegrained-fp8 kernel requires the kernels package`. The vLLM variants of the same
24
+ model are unaffected — vLLM has its own kernels — so it is easy to read the missing extra as the
25
+ eager backend not supporting the model.
26
+ - A CUDA GPU. The workload sizes below assume something in the 24-32 GiB class, which covers the
27
+ spec's first three models. `qwen3.8-27b` and `deepseek-v4-flash-0731` do not fit that class — 52 GiB and
28
+ 149 GiB of weights — so the full sweep needs a 180 GB card. Nothing has to be passed for that: a
29
+ sweep with no `--models` runs what the card can hold and names what it dropped, and each model
30
+ carries the memory fraction and engine arguments its own weights need
31
+ (`ModelSpec.min_gpu_gib`, `.gpu_memory_utilization`, `.extra_vllm_kwargs`).
32
+ - `HF_TOKEN` for gated repos (Gemma, Llama). Put it in a gitignored `.env` at the repo root:
33
+
34
+ ```
35
+ HF_TOKEN=hf_...
36
+ ```
37
+
38
+ The runner reads that file when `HF_TOKEN` is not already in the environment. This matters more
39
+ than it looks: without a token the eager backend quietly succeeds from your local HF cache while
40
+ vLLM dies with a 401 several minutes into engine bring-up.
41
+ - Nothing else on the GPU. vLLM reserves `gpu_memory_utilization` of the *whole card* up front, so
42
+ another process holding a few GiB can starve the largest model, and anything sharing the card moves
43
+ the timings.
44
+
45
+ ## Running it
46
+
47
+ The whole sweep — every model in the spec, every backend variant, every workload:
48
+
49
+ ```bash
50
+ bash benchmarks/run_all.sh
51
+ ```
52
+
53
+ A subset, which is what you want while iterating:
54
+
55
+ ```bash
56
+ bash benchmarks/run_all.sh --models gemma-2-2b --variants eager,vllm
57
+ bash benchmarks/run_all.sh --workloads generate,capture_mid --no-report
58
+ ```
59
+
60
+ One cell directly, which is what the sweep loops over:
61
+
62
+ ```bash
63
+ python -m benchmarks.run_bench --model gemma-2-2b --variant vllm
64
+ ```
65
+
66
+ See what is defined without loading anything:
67
+
68
+ ```bash
69
+ python -m benchmarks.run_bench --list
70
+ ```
71
+
72
+ Point it at a different interpreter with `BENCH_PYTHON` or `--python`:
73
+
74
+ ```bash
75
+ BENCH_PYTHON=../apps/inference/.venv/bin/python bash benchmarks/run_all.sh
76
+ ```
77
+
78
+ Each cell writes `benchmarks/results/<model>__<variant>.json`; `report_bench.py` renders those into
79
+ `results-latest.md`. Regenerate the report without re-measuring:
80
+
81
+ ```bash
82
+ python -m benchmarks.report_bench
83
+ ```
84
+
85
+ `run_all.sh` re-runs cells it already has, so a plain rerun replaces stale numbers rather than
86
+ mixing them with fresh ones. Pass `--skip-existing` to resume an interrupted sweep instead.
87
+
88
+ ## Where the numbers get published
89
+
90
+ Three files carry these measurements, and one command writes all three. `report_bench` renders the
91
+ full record into `results-latest.md` and then calls `publish`, which rewrites:
92
+
93
+ | target | what it gets |
94
+ | --- | --- |
95
+ | `README.md`, between the `THROUGHPUT` markers | decode and concurrency-8 tok/s for eager, vLLM and graph freeze, every model |
96
+ | `visualizer-web/data/benchmarks.generated.ts` | the same figures for the card behind the site's **Fast** claim, each row that ran differently carrying a footnote saying how |
97
+
98
+ Both were transcribed by hand until this existed, and both had drifted -- one carried percentages the
99
+ other had dropped, and two of the card's multipliers were ratios of already-rounded figures. So:
100
+ **never edit a number in either.** Re-render from the cells on disk, which needs no GPU:
101
+
102
+ ```bash
103
+ python -m benchmarks.publish # rewrite both
104
+ python -m benchmarks.publish --check # exit 1 if either has drifted, and name it
105
+ ```
106
+
107
+ `tests/test_published_benchmarks.py` runs that check over the committed cells, so a stale copy is a
108
+ red suite rather than a claim nobody re-read. The visualizer's chatbot answers out of a bundle holding
109
+ the README verbatim, so a publish that changed the README also wants `make viz-knowledge` -- the
110
+ command says so when it happens.
111
+
112
+ Two things `publish` decides for itself, rather than from a list someone maintains. It refuses to
113
+ write at all when the cells disagree about the GPU or the dtype, because those tables print one shared
114
+ conditions line. And it footnotes a card row when the sweep gave that model anything of its own -- its
115
+ own memory fraction, its own engine arguments, its own capture point -- so that the conditions line
116
+ keeps covering the rows it claims to. `deepseek-v4-flash-0731` earns a footnote for all four reasons.
117
+ The card dropped such a row until the footnote existed, which is why the largest freeze win in the
118
+ sweep was for a while the one figure the site did not show; a row is still dropped, but only when it is
119
+ missing a baseline figure and so has no multiplier to print. A scratch sweep of ad-hoc models should
120
+ pass `--no-publish` to `report_bench` rather than publish rows nobody deployed.
121
+
122
+ ## Benchmarking your own model
123
+
124
+ Any model `interp_engine.load_model` can load works, with no edit to the spec:
125
+
126
+ ```bash
127
+ python -m benchmarks.run_bench --hf-id mistralai/Mistral-7B-v0.1 --variant eager
128
+ python -m benchmarks.run_bench --hf-id mistralai/Mistral-7B-v0.1 --variant vllm
129
+ python -m benchmarks.report_bench
130
+ ```
131
+
132
+ `--family` and `--params` are optional labels for the report's model table. The row key is derived
133
+ from the repo id, so `mistralai/Mistral-7B-v0.1` becomes `mistral-7b-v0.1` and sorts after the
134
+ spec's own models.
135
+
136
+ Three things to check for a model that is much larger than the ones in the spec:
137
+
138
+ - **vLLM memory.** `GPU_MEMORY_UTILIZATION` in `bench_spec.py` is `0.8`, below vLLM's own `0.9`
139
+ default, because `worker_lens_readout` needs roughly twice the vocab-logits in worker scratch (it
140
+ takes a `logsumexp` over them) and at `0.9` there is none left — `lens_topk` dies with a worker-side
141
+ CUDA OOM. It is uniform across models on purpose, since a per-model fraction measures each model
142
+ under a different reservation, and the report says so where one is used. A model whose weights do
143
+ not fit inside that fraction can declare its own `ModelSpec.gpu_memory_utilization` — as
144
+ `deepseek-v4-flash-0731` does, at 0.95, because 149 GiB does not fit in what 0.8 of a 180 GB card reserves
145
+ — and `--gpu-memory-utilization` on either script still overrides both.
146
+ - **Context length.** `MAX_MODEL_LEN` is capped at 2048, because vLLM refuses to boot unless its KV
147
+ pool can hold one request at the model's advertised context and these checkpoints advertise
148
+ 32k-131k. It must stay above the longest `prompt_tokens + max_new_tokens` in `WORKLOADS`.
149
+ - **Prefix caching** is forced **off** for the vLLM variants, though the engine defaults it on. Every
150
+ workload issues the same prompt for each repeat, so with caching on the second and third would be
151
+ served from the KV cache and the reported median would be a cache hit rather than the work — and
152
+ the eager column, which has no such cache, would stop measuring the same thing. What caching is
153
+ worth is priced separately in `docs/PERFORMANCE.md`.
154
+
155
+ To add a model permanently, append a `ModelSpec` to `MODELS` in `bench_spec.py`. `run_all.sh` reads
156
+ that list, so nothing else needs changing.
157
+
158
+ ## What the workloads are
159
+
160
+ | workload | what it does | what it isolates |
161
+ | --- | --- | --- |
162
+ | `generate` | 512-token prompt, 128 new tokens, greedy, one request | single-stream decode rate and time to first token |
163
+ | `generate_x8` | the same request 8x concurrently | batching: vLLM batches, eager serves one at a time |
164
+ | `capture_mid` | `resid_post` at the middle layer over a 512-token prompt | prefill plus a single point's transport |
165
+ | `capture_all` | `resid_post` at every layer, same prompt | transport cost, since the forward is identical to `capture_mid` |
166
+ | `capture_gen` | generate 32 tokens capturing `resid_post` | decode-time capture |
167
+ | `steer` | `capture_gen` again with an add-steering vector | steering, since nothing else differs |
168
+ | `lens_topk` | 512 rows read out to top-10 ids, the way lens serving does it | the read-out alone, with no forward attached |
169
+
170
+ The four capture and steering workloads address `resid_post` on every model that has one. A model
171
+ that does not can name a substitute in `ModelSpec.capture_point`, and the report lists the rows that
172
+ did. `deepseek-v4-flash-0731` is the case: its blocks carry four parallel residual streams
173
+ (hyper-connections), so `resid_post` names four tensors rather than one and the engine refuses it
174
+ instead of silently picking the first. That row uses `mlp_out`, which keeps both properties the
175
+ tables depend on — one `d_model`-wide row per position per layer, so the transport figures still
176
+ compare, and a plain module output, so steering has something to write to.
177
+
178
+ Pairs are deliberate. `capture_mid` and `capture_all` share a prompt length and differ only in how
179
+ many points come back, so the difference between them is transport rather than compute. `steer` and
180
+ `capture_gen` are identical apart from the spec, so the difference is what steering costs. `generate`
181
+ and `generate_x8` differ only in concurrency.
182
+
183
+ `lens_topk` is the one place the two backends run different code. `VLLMModel` has
184
+ `decode_residuals_topk`, which does the norm, unembed and `topk` in the worker and returns
185
+ `[rows, 10]`; that is what `apps/inference` calls, and it is not part of the `InterpModel` protocol
186
+ because the eager backend needs no such thing — it has no boundary to keep a large tensor away from,
187
+ so it decodes and reduces in process. The unreduced `decode_residuals` is deliberately **not**
188
+ benchmarked: it returns `[rows, vocab]`, which on vLLM is hundreds of MiB crossing a process boundary
189
+ per call, and nothing serves a lens that way. Each `lens_topk` cell checks its ids against the full
190
+ read-out, so a speedup that returned different tokens would be recorded as an error rather than a win.
191
+
192
+ Prompts are normalized to a **token count**, not a character count: the same passage is 20% more
193
+ tokens under one tokenizer than another, and prefill cost scales with tokens. Every model therefore
194
+ does the same amount of work.
195
+
196
+ ## The backend variants
197
+
198
+ The report names these by what they are; `--variant` and the result filenames use the short key.
199
+
200
+ | variant | `--variant` | what it is |
201
+ | --- | --- | --- |
202
+ | interp-engine eager | `eager` | raw HF forward; `attn_implementation="eager"`, which is what the engine sets |
203
+ | interp-engine vllm | `vllm` | vLLM with `enforce_eager=True` — CUDA graphs and inductor compile off |
204
+ | vllm (vanilla) | `vllm-cudagraph` | vLLM left at its own defaults, graphs and compile on |
205
+ | interp-engine vllm freeze | `vllm-freeze` | breakable graphs with `resid_post` freeze wraps at every layer, plus one write tap mid-stack |
206
+
207
+ `bench_spec.VARIANTS` also carries two speculative-decoding variants that exist on one checkpoint
208
+ only, and are not part of these tables: `report_bench.EXCLUDED` says why, and `--variant` still
209
+ measures them.
210
+
211
+ The third exists to price a default the engine chose for capture's sake, which is the most useful
212
+ thing a speed benchmark of this library can say: `VLLMModel` defaults `enforce_eager=True`, because
213
+ CUDA-graph replay does not re-execute the Python forward and so never fires a
214
+ `register_forward_hook`.
215
+
216
+ The capture workloads are **run** on `vllm-cudagraph` rather than skipped, so the report shows what a
217
+ capture actually returns under replay instead of asserting the outcome. A capture that comes back
218
+ with no points, or with fewer rows than the prompt had tokens, is recorded as `unsupported` with the
219
+ shape it got, and the report renders that cell as `n/a`.
220
+
221
+ ### `vllm-freeze`, and what its `steer` cell needs
222
+
223
+ The fourth is the answer to the third: freeze copies activations in and out of buffers the graph
224
+ already refers to, so a replay serves capture and steering without a Python forward. It is the path
225
+ `freeze_points="auto"` takes in production, and the row exists to price it against the
226
+ `enforce_eager=True` column capture would otherwise have to use.
227
+
228
+ `"auto"` installs **reads** only, and a steering op needs a write tap to land in, so this row priced
229
+ half the feature and reported the other half as `n/a` -- with a message that blamed graph replay for
230
+ it, which is the thing freeze exists to work around. It now also passes `freeze_writes`, whose value
231
+ is the sentinel `run_bench.STEER_WRITES` rather than a site: the `steer` workload writes mid-stack,
232
+ that layer differs per model, and a freeze write is a `load_model` argument, so it has to be resolved
233
+ from the config before a model exists to ask.
234
+
235
+ `VariantSpec.models` restricts the row to the checkpoints freeze has been shown correct on, so a model
236
+ missing from it renders `--` rather than a number nobody checked.
237
+
238
+ One of them measures a wider point than its neighbours. `"auto"` on a hyper-connection trunk freezes
239
+ `resid_streams` — the whole stack of four parallel streams per layer — and a frozen engine serves the
240
+ set it froze, so `deepseek-v4-flash-0731` cannot be asked for the `mlp_out` its other columns capture.
241
+ That row's freeze cell therefore prices the stack where every other cell prices one row, declared in
242
+ `ModelSpec.freeze_capture_point`, stated by the report under *Where a row differs*, and carried onto
243
+ the visualizer's card as one line of that row's footnote.
244
+
245
+ ## Reading the numbers honestly
246
+
247
+ - Each figure is the **median of the measured repeats** (`repeats` per workload in `bench_spec.py`),
248
+ after one unmeasured warmup run. The warmup is load-bearing: the first call of any workload pays a
249
+ lazy import, the allocator's first growth to working size, and on vLLM the first `collective_rpc`
250
+ round trip.
251
+ - **Every model asks for bfloat16**, pinned in `ModelSpec.dtype` rather than left at `"auto"`, which
252
+ is the checkpoint's own precision and does not resolve equally on both backends: eager honors a
253
+ float32 checkpoint while vLLM downcasts it. Each cell records what its backend resolved to, so an
254
+ added model can be checked. On a quantized checkpoint this is the **compute** dtype and not a
255
+ request to expand the weights: `deepseek-v4-flash-0731` stays block-quantized FP8 under both backends
256
+ (transformers dequantizes only when asked with `dequantize=True`, which nothing here passes), so
257
+ those rows compare two backends serving the same quantized weights.
258
+ - **Decode throughput** excludes the first token, whose cost is the prefill already reported as time
259
+ to first token. Dividing all tokens by the total would blend prefill into the decode figure.
260
+ - **`capture_gen` is not the same algorithm on both backends.** vLLM captures during decode; eager
261
+ generates and then re-runs one forward over prompt plus generated tokens (documented at
262
+ `EagerModel.capture_generation`), so eager pays an extra prefill that vLLM does not. The numbers
263
+ are still the right comparison — this is what each backend does when you call the method — but the
264
+ gap is not purely kernel speed.
265
+ - **`generate_x8` on eager is expected to look flat.** Its generation loop is synchronous
266
+ underneath, so awaiting it does not yield to the event loop and the eight requests serialize. That
267
+ is a correct result for that backend, not a harness artifact.
268
+
269
+ ## Files
270
+
271
+ | file | what it is |
272
+ | --- | --- |
273
+ | `bench_spec.py` | models, variants, workloads, prompt normalization — plain data, imports no torch |
274
+ | `probe.py` | the environment stamp and the timing primitives |
275
+ | `workloads.py` | the timed operations, written against the protocol only |
276
+ | `run_bench.py` | runs one `(model, variant)` cell, writes JSON |
277
+ | `cells.py` | what a directory of JSON cells says — row order, and "not measured" against "measured as zero" |
278
+ | `report_bench.py` | JSON cells to `results-latest.md`, then `publish` |
279
+ | `publish.py` | the three published columns to the root README's tables and the visualizer's card |
280
+ | `run_all.sh` | the sweep loop, one process per cell |
281
+
282
+ One process per cell is a requirement, not tidiness: vLLM reserves its memory fraction of the whole
283
+ card during bring-up and keeps its KV cache in a worker subprocess that a dropped Python reference
284
+ does not reap, so two cells in one interpreter would have the second fighting the first for free
285
+ memory.
286
+
287
+ This directory is not part of the installed package — `pyproject.toml` ships only `interp_engine`.
288
+ It is dev tooling, run from a checkout.
289
+
290
+ ## Troubleshooting
291
+
292
+ **`FileNotFoundError: 'ninja'` inside `EngineCore`, minutes into a vLLM load.** vLLM's flashinfer
293
+ sampler JIT-compiles a CUDA extension on first use and shells out to `ninja` and `nvcc`. `run_bench`
294
+ puts the venv's script directory and `$CUDA_HOME/bin` on `PATH` for you; if your CUDA toolkit is
295
+ somewhere unusual, set `CUDA_HOME`.
296
+
297
+ **`CUDA_ERROR_LAUNCH_FAILED` during `DeepGEMM warmup`, after the weights have loaded.** Preceded by a
298
+ wall of `Assertion failed: ... smxx_layout.cuh:131, condition: (values[j] & 0x807fffffu) == 0`. The
299
+ warmup precompiles DeepGEMM's FP8 kernels by calling them on synthetic scales, and on a UE8M0-scaled
300
+ checkpoint those carry mantissa bits, which is precisely what that assertion rejects. A device-side
301
+ assertion takes the CUDA context with it, so the engine dies at startup having spent ten minutes
302
+ loading weights. `run_bench` sets `VLLM_DEEP_GEMM_WARMUP=skip` for you and records it in the cell's
303
+ environment stamp; the real forwards pass correctly formed scales and keep DeepGEMM. The cost is a
304
+ slower `warmup_s` on the FP8 rows, since that first compile moves into the first forward — which is
305
+ outside every measured median.
306
+
307
+ **An eager cell on `deepseek-v4-flash-0731` fails at load with `Loading an FP8 quantized model requires
308
+ accelerate`, or at the first forward with `finegrained-fp8 kernel requires the kernels package`.**
309
+ The `quant` extra is missing from that interpreter — see Requirements. Both are hard requirements of
310
+ transformers' FP8 path, and only the eager variants use it, so the vLLM cells of the same model pass
311
+ and the row reads as an eager-backend limitation rather than a missing wheel.
312
+
313
+ **`GatedRepoError: 401` for a model that loads fine on eager.** No `HF_TOKEN` — see Requirements.
314
+ Eager found the weights in your local HF cache; vLLM resolves the safetensors index through the hub.
315
+
316
+ **A cell reports `unsupported` for every capture workload.** Expected on `vllm-cudagraph`, and the
317
+ point of that variant. On `vllm` it means something turned CUDA graphs back on.
318
+
319
+ **vLLM OOMs or sizes a tiny KV cache.** Something else is on the GPU, or the previous cell's worker
320
+ has not exited. `run_all.sh` waits for free VRAM between cells and kills stragglers after a failure;
321
+ if you are running cells by hand, check `nvidia-smi` first.
322
+
323
+ **`lens_topk` fails with a CUDA OOM inside the worker while every other workload passes.** Not a
324
+ sweep problem: the read-out allocates the full `[rows, vocab]` logits and then a `logsumexp`
325
+ intermediate of the same size, so it needs roughly twice the vocab-logits free *inside* vLLM's
326
+ reservation — about 0.5 GiB for a 256k-vocab model at 512 rows. Lower
327
+ `--gpu-memory-utilization`. Worth knowing outside the benchmark too, since this is the recommended
328
+ lens path on vLLM.
329
+
330
+ **A cell hangs.** `run_all.sh` caps each cell at `BENCH_TIMEOUT_S` (default 1800) and kills it,
331
+ recording the failure and moving on, so a wedged engine cannot stall the sweep.
332
+
333
+ **An eager cell warns `You have loaded an FP8 model on CPU`, then takes an hour to load.** The eager
334
+ backend loads through transformers and then calls `.to(device)`, which stages the whole checkpoint in
335
+ host RAM first. That is unremarkable at 5 GiB and pathological at 149. A model this size should
336
+ declare `extra_eager_kwargs={"device_map": "cuda"}`, as `deepseek-v4-flash-0731` does, so transformers places
337
+ each shard on the card as it reads it; `load_model` drops `device` whenever a `device_map` is given,
338
+ so the two do not fight. It stays per-model rather than becoming the default because it changes what
339
+ `construct_s` measures.
340
+
341
+ **An eager cell reports single-digit tok/s.** Usually the model ran on the CPU — though not always:
342
+ `deepseek-v4-flash-0731` is genuinely single-digit on eager, because a 291B MoE dispatches 256 experts per
343
+ layer through Python. Check `load.device` in the cell's JSON before assuming. For the CPU case,
344
+ `load_model` only runs its device-selection ladder for `backend="auto"`; with an explicit
345
+ `backend="eager"` the `device` argument is passed straight through, and `EagerModel` skips its
346
+ `.to(device)` when that is None, so the model stays where transformers loaded it — on the CPU, with
347
+ no error and no warning. The harness passes `device="cuda"` for exactly this reason and now refuses to
348
+ report a cell that landed on the CPU. If you call `load_model(..., backend="eager")` yourself, pass a
349
+ device.