actlens 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- actlens-0.1.0/.gitignore +11 -0
- actlens-0.1.0/LICENSE +21 -0
- actlens-0.1.0/PKG-INFO +169 -0
- actlens-0.1.0/README.md +139 -0
- actlens-0.1.0/backend/actlens/__init__.py +3 -0
- actlens-0.1.0/backend/actlens/app.py +318 -0
- actlens-0.1.0/backend/actlens/archs/__init__.py +7 -0
- actlens-0.1.0/backend/actlens/archs/base.py +153 -0
- actlens-0.1.0/backend/actlens/archs/gpt2.py +42 -0
- actlens-0.1.0/backend/actlens/archs/llama.py +79 -0
- actlens-0.1.0/backend/actlens/archs/registry.py +78 -0
- actlens-0.1.0/backend/actlens/archs/rope.py +18 -0
- actlens-0.1.0/backend/actlens/cache.py +81 -0
- actlens-0.1.0/backend/actlens/capture.py +119 -0
- actlens-0.1.0/backend/actlens/cli.py +75 -0
- actlens-0.1.0/backend/actlens/colab.py +129 -0
- actlens-0.1.0/backend/actlens/corpus.json +86 -0
- actlens-0.1.0/backend/actlens/providers.py +125 -0
- actlens-0.1.0/backend/actlens/slicing.py +340 -0
- actlens-0.1.0/backend/actlens/wire.py +20 -0
- actlens-0.1.0/backend/pytest.ini +3 -0
- actlens-0.1.0/backend/tests/__init__.py +0 -0
- actlens-0.1.0/backend/tests/test_api.py +439 -0
- actlens-0.1.0/backend/tests/test_archs.py +128 -0
- actlens-0.1.0/backend/tests/test_cli.py +33 -0
- actlens-0.1.0/backend/tests/test_colab.py +7 -0
- actlens-0.1.0/backend/tests/test_gpt2.py +94 -0
- actlens-0.1.0/backend/tests/test_provider.py +86 -0
- actlens-0.1.0/backend/tests/test_real_gpt2.py +86 -0
- actlens-0.1.0/backend/tests/test_real_model.py +252 -0
- actlens-0.1.0/backend/tests/test_slicing.py +216 -0
- actlens-0.1.0/backend/tests/tiny_models.py +30 -0
- actlens-0.1.0/frontend/dist/assets/html2canvas-uQpzslGr.js +5 -0
- actlens-0.1.0/frontend/dist/assets/index-DR5UFnGu.js +9 -0
- actlens-0.1.0/frontend/dist/assets/index-TyrpFNMz.css +1 -0
- actlens-0.1.0/frontend/dist/assets/index.es-E5o7ij1y.js +5 -0
- actlens-0.1.0/frontend/dist/assets/jspdf.es.min-Bhouoq4q.js +79 -0
- actlens-0.1.0/frontend/dist/assets/purify.es-Bvo9QlJ8.js +3 -0
- actlens-0.1.0/frontend/dist/favicon.svg +9 -0
- actlens-0.1.0/frontend/dist/index.html +14 -0
- actlens-0.1.0/frontend/index.html +13 -0
- actlens-0.1.0/frontend/package-lock.json +1997 -0
- actlens-0.1.0/frontend/package.json +32 -0
- actlens-0.1.0/frontend/src/App.tsx +54 -0
- actlens-0.1.0/frontend/src/Heatmap.tsx +339 -0
- actlens-0.1.0/frontend/src/api.ts +214 -0
- actlens-0.1.0/frontend/src/axisStats.ts +307 -0
- actlens-0.1.0/frontend/src/channels.test.ts +84 -0
- actlens-0.1.0/frontend/src/channels.ts +64 -0
- actlens-0.1.0/frontend/src/colormaps.ts +166 -0
- actlens-0.1.0/frontend/src/components/DistributionPanel.tsx +425 -0
- actlens-0.1.0/frontend/src/components/Histogram.tsx +256 -0
- actlens-0.1.0/frontend/src/components/Logo.tsx +14 -0
- actlens-0.1.0/frontend/src/components/Minimap.tsx +99 -0
- actlens-0.1.0/frontend/src/components/ModelBar.tsx +50 -0
- actlens-0.1.0/frontend/src/components/PromptPanel.tsx +82 -0
- actlens-0.1.0/frontend/src/components/SelectionBar.tsx +57 -0
- actlens-0.1.0/frontend/src/components/StatsCard.tsx +102 -0
- actlens-0.1.0/frontend/src/components/Trajectory.tsx +31 -0
- actlens-0.1.0/frontend/src/components/WindowBar.tsx +49 -0
- actlens-0.1.0/frontend/src/components/controls.tsx +129 -0
- actlens-0.1.0/frontend/src/components/distribution.css +47 -0
- actlens-0.1.0/frontend/src/components/histogramGeometry.ts +76 -0
- actlens-0.1.0/frontend/src/distribution.test.ts +218 -0
- actlens-0.1.0/frontend/src/export.ts +31 -0
- actlens-0.1.0/frontend/src/figure.ts +379 -0
- actlens-0.1.0/frontend/src/logic.test.ts +148 -0
- actlens-0.1.0/frontend/src/main.tsx +17 -0
- actlens-0.1.0/frontend/src/store.ts +137 -0
- actlens-0.1.0/frontend/src/styles.css +101 -0
- actlens-0.1.0/frontend/src/tokens.ts +13 -0
- actlens-0.1.0/frontend/src/viewState.test.ts +163 -0
- actlens-0.1.0/frontend/src/viewState.ts +213 -0
- actlens-0.1.0/frontend/src/viewport.ts +82 -0
- actlens-0.1.0/frontend/src/views/AcrossLayersView.tsx +155 -0
- actlens-0.1.0/frontend/src/views/ActivationView.tsx +27 -0
- actlens-0.1.0/frontend/src/views/AttentionPanel.tsx +322 -0
- actlens-0.1.0/frontend/src/views/TokenChannelView.tsx +219 -0
- actlens-0.1.0/frontend/src/views/attnGridExport.ts +46 -0
- actlens-0.1.0/frontend/tsconfig.json +17 -0
- actlens-0.1.0/frontend/vite.config.ts +12 -0
- actlens-0.1.0/hatch_build.py +18 -0
- actlens-0.1.0/pyproject.toml +55 -0
actlens-0.1.0/.gitignore
ADDED
actlens-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Evan704
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
actlens-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: actlens
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Interactive viewer for the activations of a Hugging Face model during a forward pass
|
|
5
|
+
Project-URL: Homepage, https://github.com/Evan704/ActLens
|
|
6
|
+
Project-URL: Repository, https://github.com/Evan704/ActLens
|
|
7
|
+
Project-URL: Issues, https://github.com/Evan704/ActLens/issues
|
|
8
|
+
Author: Evan704
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: activations,huggingface,interpretability,transformers,visualization
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Visualization
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Requires-Dist: fastapi>=0.110
|
|
20
|
+
Requires-Dist: nnsight>=0.7
|
|
21
|
+
Requires-Dist: numpy>=1.26
|
|
22
|
+
Requires-Dist: torch>=2
|
|
23
|
+
Requires-Dist: transformers>=5
|
|
24
|
+
Requires-Dist: uvicorn[standard]>=0.30
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: build; extra == 'dev'
|
|
27
|
+
Requires-Dist: httpx; extra == 'dev'
|
|
28
|
+
Requires-Dist: pytest; extra == 'dev'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# ActLens
|
|
32
|
+
|
|
33
|
+
**See inside a language model while it reads your prompt.**
|
|
34
|
+
|
|
35
|
+
ActLens is an interactive viewer for the activations of any supported Hugging Face model. Type a prompt, pick an
|
|
36
|
+
activation and a layer, and browse the result as a token × channel heatmap: residual stream, attention patterns, Q/K/V,
|
|
37
|
+
MLP internals and more. It runs locally, in one command.
|
|
38
|
+
|
|
39
|
+
[](https://pypi.org/project/actlens/)
|
|
40
|
+
[](https://pypi.org/project/actlens/)
|
|
41
|
+
[](https://github.com/Evan704/ActLens/blob/main/LICENSE)
|
|
42
|
+
|
|
43
|
+
## Features
|
|
44
|
+
|
|
45
|
+
- **Every activation, every layer.** Residual stream, attention (Q, K, V, RoPE, patterns, context, output) and MLP
|
|
46
|
+
(gate, up, SwiGLU, down), captured lazily and cached.
|
|
47
|
+
- **Two views.** *Layer* shows a token × channel heatmap for one layer; *Across layers* shows a per-token statistic
|
|
48
|
+
(L2 norm, |max|, mean, std, kurtosis, or a single channel) for every layer at once.
|
|
49
|
+
- **Attention explorer.** A grid of all heads, a full query × key map per head, and a layer × head map of entropy,
|
|
50
|
+
sink mass and attention distance.
|
|
51
|
+
- **Find outlier channels.** Rank channels by |max|, std or |mean|; jump straight to a head.
|
|
52
|
+
- **Distributions.** Histograms and summary statistics over a region, a channel, a token, or the whole layer.
|
|
53
|
+
- **Fast to navigate.** Pan, zoom, brush and inspect; large activations are pooled server-side so zooming out stays cheap.
|
|
54
|
+
- **Publication-ready export.** PNG and PDF with title, prompt, axes and colorbar at up to 4×.
|
|
55
|
+
- **Extensible.** Add support for a new architecture with a small adapter.
|
|
56
|
+
|
|
57
|
+
## Quick start
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
pip install actlens # Python >= 3.10; use a fresh virtual environment
|
|
61
|
+
actlens # loads Qwen/Qwen3-0.6B and serves the UI at http://127.0.0.1:8000
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
The first run downloads the model weights from the Hugging Face Hub. ActLens is developed and tested against
|
|
65
|
+
torch 2, transformers 5 and nnsight 0.7.
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
actlens -m Qwen/Qwen2.5-0.5B # another Hugging Face model id or a local path
|
|
69
|
+
actlens --device cuda --dtype bfloat16 # device: auto | cpu | cuda | mps; dtype: float32 | float16 | bfloat16
|
|
70
|
+
actlens --port 9000 --open # custom port, open the browser when ready
|
|
71
|
+
actlens --cache-mb 4096 # activation cache budget (or $ACTLENS_CACHE_MB)
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
You can load more models from the UI at any time.
|
|
75
|
+
|
|
76
|
+
> **Security note.** The server binds to `127.0.0.1` by default. The API can load any model and has no
|
|
77
|
+
> authentication, so only use `--host 0.0.0.0` on a network you trust.
|
|
78
|
+
|
|
79
|
+
### Google Colab
|
|
80
|
+
|
|
81
|
+
No local GPU? Run the model on a free Colab GPU and view the visualization in your own browser.
|
|
82
|
+
|
|
83
|
+
[](https://colab.research.google.com/github/Evan704/ActLens/blob/main/colab/ActLens.ipynb)
|
|
84
|
+
|
|
85
|
+
1. Open the notebook and pick a GPU runtime (Runtime → Change runtime type → T4 GPU).
|
|
86
|
+
2. Run the first cell. It installs ActLens, starts the server and opens a Cloudflare tunnel.
|
|
87
|
+
3. Click the **Open ActLens** link it prints. The UI opens in your browser; keep the Colab tab open while you use it.
|
|
88
|
+
|
|
89
|
+
Or in any notebook:
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
!pip install -q actlens
|
|
93
|
+
from actlens.colab import launch
|
|
94
|
+
launch(model="Qwen/Qwen3-0.6B", dtype="float16")
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
The link contains a random access token, and the server rejects requests without it. Anyone who has the full link can use
|
|
98
|
+
your session, so do not share it. Use `actlens.colab.stop()` to shut everything down. If the page does not load, check
|
|
99
|
+
`actlens.log`.
|
|
100
|
+
|
|
101
|
+
You can also protect a server of your own with `actlens --token` (generates a token and prints the URL) or `--token VALUE`.
|
|
102
|
+
|
|
103
|
+
## Using the viewer
|
|
104
|
+
|
|
105
|
+
Pick an **activation** and a **layer** in the selection bar (`[` and `]` step through layers), then choose a view.
|
|
106
|
+
|
|
107
|
+
| Interaction | Action |
|
|
108
|
+
|---|---|
|
|
109
|
+
| Drag | Pan |
|
|
110
|
+
| Pinch, or ⌘/Ctrl + scroll | Zoom |
|
|
111
|
+
| Shift + drag | Select a region |
|
|
112
|
+
| Click | Place the cursor |
|
|
113
|
+
| Double-click | Reset the view |
|
|
114
|
+
| `[` / `]` | Previous / next layer |
|
|
115
|
+
| Esc | Clear the selection |
|
|
116
|
+
|
|
117
|
+
For head-structured activations (`q k v q_norm k_norm q_rope k_rope attn_ctx`) the channel axis is `head × head_dim`,
|
|
118
|
+
with ticks like `h3·17` and faint separators between heads; the **Head** control jumps the window to one head.
|
|
119
|
+
|
|
120
|
+
### Available activations
|
|
121
|
+
|
|
122
|
+
The picker lists only what the loaded model provides.
|
|
123
|
+
|
|
124
|
+
| Group | Activations |
|
|
125
|
+
|---|---|
|
|
126
|
+
| Residual | `resid_pre`, `resid_mid`, `resid_post` |
|
|
127
|
+
| Attention | `attn_norm`, `q`, `k`, `v`, `q_norm`, `k_norm`, `q_rope`, `k_rope`, `attn_pattern`, `attn_ctx`, `o` |
|
|
128
|
+
| MLP | `mlp_norm`, `gate`, `up`, `silu`, `swiglu`, `mlp_act`, `down` |
|
|
129
|
+
|
|
130
|
+
### Distribution panel
|
|
131
|
+
|
|
132
|
+
- **Values**: histogram and summary (mean, std, percentiles, kurtosis, skew) of the raw values for the visible window,
|
|
133
|
+
a brushed selection, the cursor's channel or token, or the whole layer.
|
|
134
|
+
- **Per channel / Per token**: one statistic per channel or token, shown as a histogram with a clickable list of the
|
|
135
|
+
top outliers.
|
|
136
|
+
|
|
137
|
+
### Export
|
|
138
|
+
|
|
139
|
+
The PNG and PDF buttons render the current view at 1–4×. The histogram and the attention head grid have their own PNG
|
|
140
|
+
export. PDFs embed the figure as a high-resolution image so CJK tokens render correctly.
|
|
141
|
+
|
|
142
|
+
## Supported models
|
|
143
|
+
|
|
144
|
+
| Adapter | `model_type` | Models |
|
|
145
|
+
|---|---|---|
|
|
146
|
+
| `llama` | `llama`, `qwen2`, `qwen3`, `mistral`, and models with the same module layout | Llama, Qwen2/2.5/3, Mistral, SmolLM |
|
|
147
|
+
| `gpt2` | `gpt2` | GPT-2, DistilGPT-2 |
|
|
148
|
+
|
|
149
|
+
Qwen3-0.6B and GPT-2 are tested on real checkpoints; every adapter is also tested on a tiny random model. An
|
|
150
|
+
unsupported model is rejected at load time with a message naming the missing modules.
|
|
151
|
+
|
|
152
|
+
Want another architecture? See [Adding an architecture](https://github.com/Evan704/ActLens/blob/main/CONTRIBUTING.md#adding-an-architecture).
|
|
153
|
+
|
|
154
|
+
## Troubleshooting
|
|
155
|
+
|
|
156
|
+
- **Attention patterns need eager attention.** ActLens loads models with `attn_implementation="eager"` because SDPA does
|
|
157
|
+
not return attention probabilities.
|
|
158
|
+
- **Slow first open of an activation.** The first time you open an activation, one forward pass captures it for every
|
|
159
|
+
layer (about 0.1–0.4 s for Qwen3-0.6B); after that it comes from the cache.
|
|
160
|
+
- **Running out of memory.** Use a smaller model, `--dtype float16` or `bfloat16`, or lower `--cache-mb`.
|
|
161
|
+
|
|
162
|
+
## Contributing
|
|
163
|
+
|
|
164
|
+
Development setup, tests, architecture notes and the release process are in
|
|
165
|
+
[CONTRIBUTING.md](https://github.com/Evan704/ActLens/blob/main/CONTRIBUTING.md).
|
|
166
|
+
|
|
167
|
+
## License
|
|
168
|
+
|
|
169
|
+
[MIT](https://github.com/Evan704/ActLens/blob/main/LICENSE)
|
actlens-0.1.0/README.md
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
# ActLens
|
|
2
|
+
|
|
3
|
+
**See inside a language model while it reads your prompt.**
|
|
4
|
+
|
|
5
|
+
ActLens is an interactive viewer for the activations of any supported Hugging Face model. Type a prompt, pick an
|
|
6
|
+
activation and a layer, and browse the result as a token × channel heatmap: residual stream, attention patterns, Q/K/V,
|
|
7
|
+
MLP internals and more. It runs locally, in one command.
|
|
8
|
+
|
|
9
|
+
[](https://pypi.org/project/actlens/)
|
|
10
|
+
[](https://pypi.org/project/actlens/)
|
|
11
|
+
[](https://github.com/Evan704/ActLens/blob/main/LICENSE)
|
|
12
|
+
|
|
13
|
+
## Features
|
|
14
|
+
|
|
15
|
+
- **Every activation, every layer.** Residual stream, attention (Q, K, V, RoPE, patterns, context, output) and MLP
|
|
16
|
+
(gate, up, SwiGLU, down), captured lazily and cached.
|
|
17
|
+
- **Two views.** *Layer* shows a token × channel heatmap for one layer; *Across layers* shows a per-token statistic
|
|
18
|
+
(L2 norm, |max|, mean, std, kurtosis, or a single channel) for every layer at once.
|
|
19
|
+
- **Attention explorer.** A grid of all heads, a full query × key map per head, and a layer × head map of entropy,
|
|
20
|
+
sink mass and attention distance.
|
|
21
|
+
- **Find outlier channels.** Rank channels by |max|, std or |mean|; jump straight to a head.
|
|
22
|
+
- **Distributions.** Histograms and summary statistics over a region, a channel, a token, or the whole layer.
|
|
23
|
+
- **Fast to navigate.** Pan, zoom, brush and inspect; large activations are pooled server-side so zooming out stays cheap.
|
|
24
|
+
- **Publication-ready export.** PNG and PDF with title, prompt, axes and colorbar at up to 4×.
|
|
25
|
+
- **Extensible.** Add support for a new architecture with a small adapter.
|
|
26
|
+
|
|
27
|
+
## Quick start
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install actlens # Python >= 3.10; use a fresh virtual environment
|
|
31
|
+
actlens # loads Qwen/Qwen3-0.6B and serves the UI at http://127.0.0.1:8000
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
The first run downloads the model weights from the Hugging Face Hub. ActLens is developed and tested against
|
|
35
|
+
torch 2, transformers 5 and nnsight 0.7.
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
actlens -m Qwen/Qwen2.5-0.5B # another Hugging Face model id or a local path
|
|
39
|
+
actlens --device cuda --dtype bfloat16 # device: auto | cpu | cuda | mps; dtype: float32 | float16 | bfloat16
|
|
40
|
+
actlens --port 9000 --open # custom port, open the browser when ready
|
|
41
|
+
actlens --cache-mb 4096 # activation cache budget (or $ACTLENS_CACHE_MB)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
You can load more models from the UI at any time.
|
|
45
|
+
|
|
46
|
+
> **Security note.** The server binds to `127.0.0.1` by default. The API can load any model and has no
|
|
47
|
+
> authentication, so only use `--host 0.0.0.0` on a network you trust.
|
|
48
|
+
|
|
49
|
+
### Google Colab
|
|
50
|
+
|
|
51
|
+
No local GPU? Run the model on a free Colab GPU and view the visualization in your own browser.
|
|
52
|
+
|
|
53
|
+
[](https://colab.research.google.com/github/Evan704/ActLens/blob/main/colab/ActLens.ipynb)
|
|
54
|
+
|
|
55
|
+
1. Open the notebook and pick a GPU runtime (Runtime → Change runtime type → T4 GPU).
|
|
56
|
+
2. Run the first cell. It installs ActLens, starts the server and opens a Cloudflare tunnel.
|
|
57
|
+
3. Click the **Open ActLens** link it prints. The UI opens in your browser; keep the Colab tab open while you use it.
|
|
58
|
+
|
|
59
|
+
Or in any notebook:
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
!pip install -q actlens
|
|
63
|
+
from actlens.colab import launch
|
|
64
|
+
launch(model="Qwen/Qwen3-0.6B", dtype="float16")
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
The link contains a random access token, and the server rejects requests without it. Anyone who has the full link can use
|
|
68
|
+
your session, so do not share it. Use `actlens.colab.stop()` to shut everything down. If the page does not load, check
|
|
69
|
+
`actlens.log`.
|
|
70
|
+
|
|
71
|
+
You can also protect a server of your own with `actlens --token` (generates a token and prints the URL) or `--token VALUE`.
|
|
72
|
+
|
|
73
|
+
## Using the viewer
|
|
74
|
+
|
|
75
|
+
Pick an **activation** and a **layer** in the selection bar (`[` and `]` step through layers), then choose a view.
|
|
76
|
+
|
|
77
|
+
| Interaction | Action |
|
|
78
|
+
|---|---|
|
|
79
|
+
| Drag | Pan |
|
|
80
|
+
| Pinch, or ⌘/Ctrl + scroll | Zoom |
|
|
81
|
+
| Shift + drag | Select a region |
|
|
82
|
+
| Click | Place the cursor |
|
|
83
|
+
| Double-click | Reset the view |
|
|
84
|
+
| `[` / `]` | Previous / next layer |
|
|
85
|
+
| Esc | Clear the selection |
|
|
86
|
+
|
|
87
|
+
For head-structured activations (`q k v q_norm k_norm q_rope k_rope attn_ctx`) the channel axis is `head × head_dim`,
|
|
88
|
+
with ticks like `h3·17` and faint separators between heads; the **Head** control jumps the window to one head.
|
|
89
|
+
|
|
90
|
+
### Available activations
|
|
91
|
+
|
|
92
|
+
The picker lists only what the loaded model provides.
|
|
93
|
+
|
|
94
|
+
| Group | Activations |
|
|
95
|
+
|---|---|
|
|
96
|
+
| Residual | `resid_pre`, `resid_mid`, `resid_post` |
|
|
97
|
+
| Attention | `attn_norm`, `q`, `k`, `v`, `q_norm`, `k_norm`, `q_rope`, `k_rope`, `attn_pattern`, `attn_ctx`, `o` |
|
|
98
|
+
| MLP | `mlp_norm`, `gate`, `up`, `silu`, `swiglu`, `mlp_act`, `down` |
|
|
99
|
+
|
|
100
|
+
### Distribution panel
|
|
101
|
+
|
|
102
|
+
- **Values**: histogram and summary (mean, std, percentiles, kurtosis, skew) of the raw values for the visible window,
|
|
103
|
+
a brushed selection, the cursor's channel or token, or the whole layer.
|
|
104
|
+
- **Per channel / Per token**: one statistic per channel or token, shown as a histogram with a clickable list of the
|
|
105
|
+
top outliers.
|
|
106
|
+
|
|
107
|
+
### Export
|
|
108
|
+
|
|
109
|
+
The PNG and PDF buttons render the current view at 1–4×. The histogram and the attention head grid have their own PNG
|
|
110
|
+
export. PDFs embed the figure as a high-resolution image so CJK tokens render correctly.
|
|
111
|
+
|
|
112
|
+
## Supported models
|
|
113
|
+
|
|
114
|
+
| Adapter | `model_type` | Models |
|
|
115
|
+
|---|---|---|
|
|
116
|
+
| `llama` | `llama`, `qwen2`, `qwen3`, `mistral`, and models with the same module layout | Llama, Qwen2/2.5/3, Mistral, SmolLM |
|
|
117
|
+
| `gpt2` | `gpt2` | GPT-2, DistilGPT-2 |
|
|
118
|
+
|
|
119
|
+
Qwen3-0.6B and GPT-2 are tested on real checkpoints; every adapter is also tested on a tiny random model. An
|
|
120
|
+
unsupported model is rejected at load time with a message naming the missing modules.
|
|
121
|
+
|
|
122
|
+
Want another architecture? See [Adding an architecture](https://github.com/Evan704/ActLens/blob/main/CONTRIBUTING.md#adding-an-architecture).
|
|
123
|
+
|
|
124
|
+
## Troubleshooting
|
|
125
|
+
|
|
126
|
+
- **Attention patterns need eager attention.** ActLens loads models with `attn_implementation="eager"` because SDPA does
|
|
127
|
+
not return attention probabilities.
|
|
128
|
+
- **Slow first open of an activation.** The first time you open an activation, one forward pass captures it for every
|
|
129
|
+
layer (about 0.1–0.4 s for Qwen3-0.6B); after that it comes from the cache.
|
|
130
|
+
- **Running out of memory.** Use a smaller model, `--dtype float16` or `bfloat16`, or lower `--cache-mb`.
|
|
131
|
+
|
|
132
|
+
## Contributing
|
|
133
|
+
|
|
134
|
+
Development setup, tests, architecture notes and the release process are in
|
|
135
|
+
[CONTRIBUTING.md](https://github.com/Evan704/ActLens/blob/main/CONTRIBUTING.md).
|
|
136
|
+
|
|
137
|
+
## License
|
|
138
|
+
|
|
139
|
+
[MIT](https://github.com/Evan704/ActLens/blob/main/LICENSE)
|
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
"""FastAPI app: model management, runs, and windowed activation slices."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import asyncio
|
|
5
|
+
import gc
|
|
6
|
+
import hmac
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import threading
|
|
10
|
+
import time
|
|
11
|
+
import uuid
|
|
12
|
+
from contextlib import asynccontextmanager
|
|
13
|
+
from collections import OrderedDict
|
|
14
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Callable
|
|
17
|
+
|
|
18
|
+
from fastapi import FastAPI, HTTPException, Query, Request
|
|
19
|
+
from fastapi.responses import FileResponse, PlainTextResponse, RedirectResponse
|
|
20
|
+
from fastapi.staticfiles import StaticFiles
|
|
21
|
+
from pydantic import BaseModel
|
|
22
|
+
|
|
23
|
+
from . import slicing
|
|
24
|
+
from .cache import ActivationCache
|
|
25
|
+
from .capture import ATTN_ACT, Capture, Run
|
|
26
|
+
from .providers import ActivationProvider, NNsightProvider
|
|
27
|
+
from .wire import frame_response
|
|
28
|
+
|
|
29
|
+
DEFAULT_MODEL = "Qwen/Qwen3-0.6B"
|
|
30
|
+
PRESET_MODELS = [
|
|
31
|
+
{"id": "Qwen/Qwen3-0.6B", "label": "Qwen3-0.6B"},
|
|
32
|
+
{"id": "Qwen/Qwen2.5-0.5B", "label": "Qwen2.5-0.5B"},
|
|
33
|
+
{"id": "HuggingFaceTB/SmolLM2-360M", "label": "SmolLM2-360M"},
|
|
34
|
+
{"id": "openai-community/gpt2", "label": "GPT-2"},
|
|
35
|
+
]
|
|
36
|
+
MAX_RUNS = 3
|
|
37
|
+
TOKEN_COOKIE = "actlens_token"
|
|
38
|
+
DEFAULT_CACHE_MB = 2048
|
|
39
|
+
CORPUS = json.loads((Path(__file__).parent / "corpus.json").read_text())
|
|
40
|
+
_PACKAGED_DIST = Path(__file__).parent / "static" # bundled into the wheel
|
|
41
|
+
_DEV_DIST = Path(__file__).resolve().parents[2] / "frontend" / "dist" # source checkout
|
|
42
|
+
FRONTEND_DIST = _PACKAGED_DIST if (_PACKAGED_DIST / "index.html").is_file() else _DEV_DIST
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def cache_budget_bytes() -> int:
|
|
46
|
+
try:
|
|
47
|
+
mb = float(os.environ.get("ACTLENS_CACHE_MB", DEFAULT_CACHE_MB))
|
|
48
|
+
except ValueError:
|
|
49
|
+
mb = DEFAULT_CACHE_MB
|
|
50
|
+
return int(mb * 2**20)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class ModelManager:
|
|
54
|
+
"""Owns the single loaded model. All model work runs on one dedicated thread."""
|
|
55
|
+
|
|
56
|
+
def __init__(self, factory: Callable[[str, str, str], ActivationProvider] = NNsightProvider,
|
|
57
|
+
cache_bytes: int | None = None):
|
|
58
|
+
self.factory = factory
|
|
59
|
+
self.executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="model")
|
|
60
|
+
self.provider: ActivationProvider | None = None
|
|
61
|
+
self.state = "idle" # idle | loading | ready | error
|
|
62
|
+
self.target: str | None = None
|
|
63
|
+
self.error: str | None = None
|
|
64
|
+
self.runs: OrderedDict[str, Run] = OrderedDict()
|
|
65
|
+
self.cache = ActivationCache(cache_budget_bytes() if cache_bytes is None else cache_bytes)
|
|
66
|
+
self._lock = threading.Lock()
|
|
67
|
+
|
|
68
|
+
def status(self) -> dict:
|
|
69
|
+
return {
|
|
70
|
+
"state": self.state,
|
|
71
|
+
"model_id": self.provider.model_id if self.provider else self.target,
|
|
72
|
+
"target": self.target,
|
|
73
|
+
"error": self.error,
|
|
74
|
+
"info": self.provider.info if self.state == "ready" and self.provider else None,
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
def load(self, model_id: str, device: str = "auto", dtype: str = "float32") -> None:
|
|
78
|
+
with self._lock:
|
|
79
|
+
if self.state == "loading":
|
|
80
|
+
raise HTTPException(409, "A model is already loading")
|
|
81
|
+
self.state, self.target, self.error = "loading", model_id, None
|
|
82
|
+
self.executor.submit(self._load, model_id, device, dtype)
|
|
83
|
+
|
|
84
|
+
def _load(self, model_id: str, device: str, dtype: str) -> None:
|
|
85
|
+
try:
|
|
86
|
+
if self.provider is not None:
|
|
87
|
+
self.provider.close()
|
|
88
|
+
self.provider = None
|
|
89
|
+
self.runs.clear()
|
|
90
|
+
self.cache.clear()
|
|
91
|
+
gc.collect()
|
|
92
|
+
self.provider = self.factory(model_id, device, dtype)
|
|
93
|
+
self.state = "ready"
|
|
94
|
+
except Exception as e: # surfaced to the UI via /api/status
|
|
95
|
+
self.error = f"{type(e).__name__}: {e}"
|
|
96
|
+
self.state = "error"
|
|
97
|
+
|
|
98
|
+
def _register(self, provider: ActivationProvider, text: str, max_tokens: int) -> Run:
|
|
99
|
+
t0 = time.perf_counter()
|
|
100
|
+
ids, tokens, truncated = provider.tokenize(text, max_tokens)
|
|
101
|
+
specs = {s.id: s for s in provider.activations()}
|
|
102
|
+
keys = ("n_layers", "hidden_size", "intermediate_size", "n_heads", "n_kv_heads", "head_dim")
|
|
103
|
+
return Run(run_id=uuid.uuid4().hex[:12], model_id=provider.model_id, text=text, token_ids=list(ids),
|
|
104
|
+
tokens=list(tokens), truncated=truncated, specs=specs,
|
|
105
|
+
model={k: provider.info.get(k) for k in keys}, elapsed_ms=(time.perf_counter() - t0) * 1000)
|
|
106
|
+
|
|
107
|
+
async def run(self, text: str, max_tokens: int) -> Run:
|
|
108
|
+
"""Tokenize and register a run. No activations are captured yet."""
|
|
109
|
+
if self.state != "ready" or self.provider is None:
|
|
110
|
+
raise HTTPException(409, f"Model not ready (state: {self.state})")
|
|
111
|
+
loop = asyncio.get_running_loop()
|
|
112
|
+
try:
|
|
113
|
+
run = await loop.run_in_executor(self.executor, self._register, self.provider, text, max_tokens)
|
|
114
|
+
except ValueError as e:
|
|
115
|
+
raise HTTPException(400, str(e))
|
|
116
|
+
self.runs[run.run_id] = run
|
|
117
|
+
while len(self.runs) > MAX_RUNS:
|
|
118
|
+
old, _ = self.runs.popitem(last=False)
|
|
119
|
+
self.cache.drop_run(old)
|
|
120
|
+
return run
|
|
121
|
+
|
|
122
|
+
def get(self, run_id: str) -> Run:
|
|
123
|
+
run = self.runs.get(run_id)
|
|
124
|
+
if run is None:
|
|
125
|
+
raise HTTPException(404, "Unknown run (it may have been evicted); run the prompt again")
|
|
126
|
+
self.runs.move_to_end(run_id)
|
|
127
|
+
return run
|
|
128
|
+
|
|
129
|
+
def capture(self, run: Run, act: str) -> Capture:
|
|
130
|
+
"""Blocking (call from a worker thread): return the cached capture, building it on the model thread if needed.
|
|
131
|
+
Concurrent callers for the same (run, act) share one forward pass."""
|
|
132
|
+
provider = self.provider
|
|
133
|
+
if provider is None or run.run_id not in self.runs or provider.model_id != run.model_id:
|
|
134
|
+
raise HTTPException(404, "Unknown run (it may have been evicted); run the prompt again")
|
|
135
|
+
|
|
136
|
+
def build() -> Capture:
|
|
137
|
+
arr = self.executor.submit(provider.capture, run.token_ids, act).result()
|
|
138
|
+
return Capture(act=act, arr=arr)
|
|
139
|
+
|
|
140
|
+
try:
|
|
141
|
+
cap = self.cache.get((run.run_id, act), build)
|
|
142
|
+
except HTTPException:
|
|
143
|
+
raise
|
|
144
|
+
except ValueError as e:
|
|
145
|
+
raise HTTPException(400, str(e))
|
|
146
|
+
except Exception as e:
|
|
147
|
+
raise HTTPException(500, f"Capture of {act!r} failed: {type(e).__name__}: {e}")
|
|
148
|
+
if run.run_id not in self.runs: # run was evicted while we were capturing
|
|
149
|
+
self.cache.drop_run(run.run_id)
|
|
150
|
+
return cap
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
class LoadRequest(BaseModel):
|
|
154
|
+
model_id: str
|
|
155
|
+
device: str = "auto"
|
|
156
|
+
dtype: str = "float32"
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class RunRequest(BaseModel):
|
|
160
|
+
text: str
|
|
161
|
+
max_tokens: int = 512
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def create_app(manager: ModelManager | None = None, autoload: bool = True, model_id: str = DEFAULT_MODEL,
|
|
165
|
+
device: str = "auto", dtype: str = "float32", token: str | None = None) -> FastAPI:
|
|
166
|
+
mgr = manager or ModelManager()
|
|
167
|
+
|
|
168
|
+
@asynccontextmanager
|
|
169
|
+
async def lifespan(_: FastAPI):
|
|
170
|
+
if autoload and mgr.state == "idle":
|
|
171
|
+
mgr.load(model_id, device, dtype)
|
|
172
|
+
yield
|
|
173
|
+
|
|
174
|
+
app = FastAPI(title="ActLens", lifespan=lifespan)
|
|
175
|
+
app.state.manager = mgr
|
|
176
|
+
|
|
177
|
+
if token:
|
|
178
|
+
# Access token for a server that is reachable from outside (e.g. a tunnel from Colab). Open
|
|
179
|
+
# `/?token=...` once: the token is stored in an HttpOnly cookie and the URL is cleaned up.
|
|
180
|
+
# Scripts can send `Authorization: Bearer <token>` instead.
|
|
181
|
+
def valid(candidate: str | None) -> bool:
|
|
182
|
+
return bool(candidate) and hmac.compare_digest(candidate.encode(), token.encode())
|
|
183
|
+
|
|
184
|
+
@app.middleware("http")
|
|
185
|
+
async def require_token(request: Request, call_next):
|
|
186
|
+
bearer = request.headers.get("authorization", "")
|
|
187
|
+
if valid(request.cookies.get(TOKEN_COOKIE)) or valid(bearer.removeprefix("Bearer ").strip()):
|
|
188
|
+
return await call_next(request)
|
|
189
|
+
if request.method == "GET" and valid(request.query_params.get("token")):
|
|
190
|
+
url = request.url.remove_query_params("token")
|
|
191
|
+
resp = RedirectResponse(url.path + (f"?{url.query}" if url.query else ""), status_code=303)
|
|
192
|
+
# Lax, not Strict: the link is usually opened from another site (Colab), and a Strict cookie is
|
|
193
|
+
# not sent on the redirect that follows a cross-site navigation.
|
|
194
|
+
https = "https" in (request.url.scheme, request.headers.get("x-forwarded-proto", ""))
|
|
195
|
+
resp.set_cookie(TOKEN_COOKIE, token, httponly=True, samesite="lax", secure=https)
|
|
196
|
+
return resp
|
|
197
|
+
return PlainTextResponse("ActLens: missing or invalid access token. Open the full link printed "
|
|
198
|
+
"by the server (it ends in ?token=...).", status_code=401)
|
|
199
|
+
|
|
200
|
+
def guarded(fn, *a, **kw):
|
|
201
|
+
try:
|
|
202
|
+
return fn(*a, **kw)
|
|
203
|
+
except ValueError as e:
|
|
204
|
+
raise HTTPException(400, str(e))
|
|
205
|
+
|
|
206
|
+
def token_capture(run_id: str, act: str, layer: int | None = None) -> Capture:
|
|
207
|
+
"""Validate `act` (a token-kind activation of this run's model) and capture it lazily."""
|
|
208
|
+
run = mgr.get(run_id)
|
|
209
|
+
spec = run.specs.get(act)
|
|
210
|
+
if spec is None:
|
|
211
|
+
raise HTTPException(400, f"unknown act {act!r}; expected one of {list(run.specs)}")
|
|
212
|
+
if spec.kind != "token":
|
|
213
|
+
raise HTTPException(400, f"act {act!r} is an attention pattern; use the /attn endpoints")
|
|
214
|
+
if layer is not None and not 0 <= layer < spec.n_layers:
|
|
215
|
+
raise HTTPException(400, f"layer {layer} out of range for {act} (0..{spec.n_layers - 1})")
|
|
216
|
+
return mgr.capture(run, act)
|
|
217
|
+
|
|
218
|
+
def attn_capture(run_id: str) -> Capture:
|
|
219
|
+
run = mgr.get(run_id)
|
|
220
|
+
if ATTN_ACT not in run.specs:
|
|
221
|
+
raise HTTPException(400, "this model does not expose attention patterns")
|
|
222
|
+
return mgr.capture(run, ATTN_ACT)
|
|
223
|
+
|
|
224
|
+
# ----- models / corpus -----
|
|
225
|
+
@app.get("/api/status")
|
|
226
|
+
def status():
|
|
227
|
+
return {**mgr.status(), "presets": PRESET_MODELS}
|
|
228
|
+
|
|
229
|
+
@app.post("/api/models/load", status_code=202)
|
|
230
|
+
def load_model(req: LoadRequest):
|
|
231
|
+
mgr.load(req.model_id, req.device, req.dtype)
|
|
232
|
+
return mgr.status()
|
|
233
|
+
|
|
234
|
+
@app.get("/api/corpus")
|
|
235
|
+
def corpus():
|
|
236
|
+
return CORPUS
|
|
237
|
+
|
|
238
|
+
# ----- runs -----
|
|
239
|
+
@app.post("/api/run")
|
|
240
|
+
async def run(req: RunRequest):
|
|
241
|
+
r = await mgr.run(req.text, max(1, min(req.max_tokens, 1024)))
|
|
242
|
+
return {
|
|
243
|
+
"run_id": r.run_id, "model_id": r.model_id, "tokens": r.tokens, "token_ids": r.token_ids,
|
|
244
|
+
"truncated": r.truncated, "elapsed_ms": round(r.elapsed_ms, 1), "model": r.model,
|
|
245
|
+
"activations": [sp.info() for sp in r.specs.values()],
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
@app.get("/api/run/{run_id}/overview")
|
|
249
|
+
def overview(run_id: str, act: str = "resid_post", stat: str = "norm", dim: int = 0):
|
|
250
|
+
cap = token_capture(run_id, act)
|
|
251
|
+
arr, meta = guarded(slicing.overview, cap, stat, dim)
|
|
252
|
+
return frame_response(meta, arr)
|
|
253
|
+
|
|
254
|
+
@app.get("/api/run/{run_id}/slice")
|
|
255
|
+
def slice_(run_id: str, act: str, layer: int, t0: int = 0, t1: int = 64, d0: int = 0, d1: int = 128,
|
|
256
|
+
max_h: int = Query(512, ge=1, le=2048), max_w: int = Query(1024, ge=1, le=4096),
|
|
257
|
+
agg: str = "absmax", order: str = "natural"):
|
|
258
|
+
cap = token_capture(run_id, act, layer)
|
|
259
|
+
arr, meta = guarded(slicing.token_dim_slice, cap, layer, t0, t1, d0, d1, max_h, max_w, agg, order)
|
|
260
|
+
return frame_response(meta, arr)
|
|
261
|
+
|
|
262
|
+
@app.get("/api/run/{run_id}/profile")
|
|
263
|
+
def profile(run_id: str, act: str, layer: int, order: str = "natural", bins: int = Query(512, ge=8, le=4096)):
|
|
264
|
+
cap = token_capture(run_id, act, layer)
|
|
265
|
+
arr, meta = guarded(slicing.dim_profile, cap, layer, order, bins)
|
|
266
|
+
return frame_response(meta, arr)
|
|
267
|
+
|
|
268
|
+
@app.get("/api/run/{run_id}/stats")
|
|
269
|
+
def stats(run_id: str, act: str, layer: int, t0: int = 0, t1: int = 64, d0: int = 0, d1: int = 128,
|
|
270
|
+
order: str = "natural", clip: bool = False):
|
|
271
|
+
cap = token_capture(run_id, act, layer)
|
|
272
|
+
return guarded(slicing.token_region_stats, cap, layer, t0, t1, d0, d1, order, clip)
|
|
273
|
+
|
|
274
|
+
@app.get("/api/run/{run_id}/axis_stats")
|
|
275
|
+
def axis_stats(run_id: str, act: str, layer: int, axis: str = "channel", stat: str = "norm",
|
|
276
|
+
t0: int = 0, t1: int = 64, d0: int = 0, d1: int = 128, order: str = "natural",
|
|
277
|
+
clip: bool = False, top: int = 10, bins: int = 64):
|
|
278
|
+
cap = token_capture(run_id, act, layer)
|
|
279
|
+
return guarded(slicing.axis_stats, cap, layer, axis, stat, t0, t1, d0, d1, order, clip, top, bins)
|
|
280
|
+
|
|
281
|
+
@app.get("/api/run/{run_id}/attn")
|
|
282
|
+
def attn(run_id: str, layer: int, head: int = -1, q0: int = 0, q1: int = 100000, k0: int = 0, k1: int = 100000,
|
|
283
|
+
max_q: int = Query(512, ge=1, le=1024), max_k: int = Query(512, ge=1, le=1024), agg: str = "max"):
|
|
284
|
+
cap = attn_capture(run_id)
|
|
285
|
+
if not 0 <= layer < cap.n_layers:
|
|
286
|
+
raise HTTPException(400, f"layer {layer} out of range (0..{cap.n_layers - 1})")
|
|
287
|
+
arr, meta = guarded(slicing.attn_slice, cap, layer, head, q0, q1, k0, k1, max_q, max_k, agg)
|
|
288
|
+
return frame_response(meta, arr)
|
|
289
|
+
|
|
290
|
+
@app.get("/api/run/{run_id}/attn_overview")
|
|
291
|
+
def attn_overview(run_id: str, stat: str = "entropy"):
|
|
292
|
+
cap = attn_capture(run_id)
|
|
293
|
+
arr, meta = guarded(slicing.attn_overview, cap, stat)
|
|
294
|
+
return frame_response(meta, arr)
|
|
295
|
+
|
|
296
|
+
@app.get("/api/run/{run_id}/attn_stats")
|
|
297
|
+
def attn_stats(run_id: str, layer: int, head: int, q0: int = 0, q1: int = 100000, k0: int = 0, k1: int = 100000,
|
|
298
|
+
clip: bool = False):
|
|
299
|
+
cap = attn_capture(run_id)
|
|
300
|
+
if not (0 <= layer < cap.n_layers and 0 <= head < cap.n_heads):
|
|
301
|
+
raise HTTPException(400, "layer/head out of range")
|
|
302
|
+
return guarded(slicing.attn_region_stats, cap, layer, head, q0, q1, k0, k1, clip)
|
|
303
|
+
|
|
304
|
+
# ----- static frontend (production build) -----
|
|
305
|
+
if (FRONTEND_DIST / "index.html").is_file():
|
|
306
|
+
app.mount("/assets", StaticFiles(directory=FRONTEND_DIST / "assets"), name="assets")
|
|
307
|
+
|
|
308
|
+
@app.get("/{path:path}")
|
|
309
|
+
def spa(path: str):
|
|
310
|
+
if path.startswith("api/"):
|
|
311
|
+
raise HTTPException(404)
|
|
312
|
+
f = FRONTEND_DIST / path
|
|
313
|
+
return FileResponse(f if path and f.is_file() else FRONTEND_DIST / "index.html")
|
|
314
|
+
|
|
315
|
+
return app
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
app = create_app()
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Architecture adapters: how each model family exposes its activations. See `base.ArchAdapter`."""
|
|
2
|
+
from . import gpt2, llama # noqa: F401 (importing registers the built-in adapters)
|
|
3
|
+
from .base import ActDef, ArchAdapter, Dims, PreNormBlockAdapter, TraceCtx, get, has
|
|
4
|
+
from .registry import ADAPTERS, load_plugins, register, resolve_adapter, supported_model_types
|
|
5
|
+
|
|
6
|
+
__all__ = ["ActDef", "ArchAdapter", "Dims", "PreNormBlockAdapter", "TraceCtx", "ADAPTERS", "get", "has",
|
|
7
|
+
"load_plugins", "register", "resolve_adapter", "supported_model_types"]
|