icalens 0.3.0.dev1__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {icalens-0.3.0.dev1 → icalens-0.3.2}/.gitignore +5 -0
- icalens-0.3.2/PKG-INFO +214 -0
- icalens-0.3.2/README.md +186 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/demo/apply_chat.py +4 -13
- icalens-0.3.2/demo/commands-2.sh +91 -0
- icalens-0.3.2/demo/demo-1.ipynb +499 -0
- icalens-0.3.2/demo/demo-2.ipynb +408 -0
- icalens-0.3.2/demo/demo-3.ipynb +474 -0
- icalens-0.3.2/demo/demo-4.ipynb +100 -0
- icalens-0.3.2/demo/demo-5.ipynb +137 -0
- icalens-0.3.2/demo/demo-6.ipynb +1229 -0
- icalens-0.3.2/demo/demo-7.ipynb +805 -0
- icalens-0.3.2/demo/fit.py +6 -0
- icalens-0.3.2/demo/fit_chat.py +6 -0
- icalens-0.3.2/demo/publish.py +6 -0
- icalens-0.3.2/docs/api.md +602 -0
- icalens-0.3.2/docs/assets/conversation-analysis-notebook.png +0 -0
- icalens-0.3.2/docs/assets/fit.png +0 -0
- icalens-0.3.2/docs/assets/text-analysis-notebook.png +0 -0
- icalens-0.3.2/docs/assets/text-analysis-profile.png +0 -0
- icalens-0.3.2/docs/component-profiles.md +80 -0
- icalens-0.3.2/docs/fit-and-publish.md +364 -0
- icalens-0.3.2/docs/getting-started.md +128 -0
- icalens-0.3.2/docs/index.md +55 -0
- icalens-0.3.2/docs/reconstruction.md +234 -0
- icalens-0.3.2/docs/scores-and-energy.md +174 -0
- icalens-0.3.2/docs/steering.md +195 -0
- icalens-0.3.2/docs/text-and-chat.md +166 -0
- icalens-0.3.2/docs-zh/.readthedocs.yaml +13 -0
- icalens-0.3.2/docs-zh/api.md +435 -0
- icalens-0.3.2/docs-zh/assets/text-analysis-profile.png +0 -0
- icalens-0.3.2/docs-zh/component-profiles.md +66 -0
- icalens-0.3.2/docs-zh/fit-and-publish.md +340 -0
- icalens-0.3.2/docs-zh/getting-started.md +115 -0
- icalens-0.3.2/docs-zh/index.md +53 -0
- icalens-0.3.2/docs-zh/reconstruction.md +209 -0
- icalens-0.3.2/docs-zh/scores-and-energy.md +154 -0
- icalens-0.3.2/docs-zh/steering.md +178 -0
- icalens-0.3.2/docs-zh/text-and-chat.md +155 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/mkdocs.yml +13 -0
- icalens-0.3.2/mkdocs.zh.yml +71 -0
- icalens-0.3.2/model_framing.json +34 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/pyproject.toml +9 -2
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/__init__.py +1 -1
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/_artifact.py +97 -6
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/_capture.py +31 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/_fastica.py +3 -9
- icalens-0.3.2/src/icalens/_model_framing.py +179 -0
- icalens-0.3.2/src/icalens/analysis.py +627 -0
- icalens-0.3.2/src/icalens/cli/__init__.py +73 -0
- icalens-0.3.2/src/icalens/cli/__main__.py +8 -0
- {icalens-0.3.0.dev1/demo → icalens-0.3.2/src/icalens/cli}/fit_chat.py +12 -12
- icalens-0.3.0.dev1/demo/fit.py → icalens-0.3.2/src/icalens/cli/fit_text.py +214 -54
- icalens-0.3.2/src/icalens/cli/profile.py +161 -0
- {icalens-0.3.0.dev1/demo → icalens-0.3.2/src/icalens/cli}/publish.py +9 -9
- icalens-0.3.2/src/icalens/html.py +570 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/lens.py +214 -1
- icalens-0.3.2/src/icalens/profiling.py +268 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/smoke_test.py +6 -5
- icalens-0.3.2/tests/test_analysis.py +260 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/tests/test_artifacts.py +33 -6
- icalens-0.3.2/tests/test_cli.py +72 -0
- icalens-0.3.2/tests/test_demo_fit.py +117 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/tests/test_demo_fit_chat.py +1 -1
- icalens-0.3.2/tests/test_docs_i18n.py +47 -0
- icalens-0.3.2/tests/test_html_explorer.py +235 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/tests/test_lens.py +66 -0
- icalens-0.3.2/tests/test_model_framing_registry.py +88 -0
- icalens-0.3.2/tests/test_profiling.py +66 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/tests/test_public_api.py +1 -1
- {icalens-0.3.0.dev1 → icalens-0.3.2}/uv.lock +1477 -1
- icalens-0.3.0.dev1/PKG-INFO +0 -153
- icalens-0.3.0.dev1/README.md +0 -127
- icalens-0.3.0.dev1/docs/api.md +0 -87
- icalens-0.3.0.dev1/docs/fit-and-publish.md +0 -67
- icalens-0.3.0.dev1/docs/getting-started.md +0 -61
- icalens-0.3.0.dev1/docs/index.md +0 -44
- icalens-0.3.0.dev1/docs/scores-and-energy.md +0 -47
- icalens-0.3.0.dev1/docs/text-and-chat.md +0 -63
- icalens-0.3.0.dev1/src/icalens/analysis.py +0 -281
- icalens-0.3.0.dev1/src/icalens/html.py +0 -299
- icalens-0.3.0.dev1/tests/test_analysis.py +0 -72
- icalens-0.3.0.dev1/tests/test_demo_fit.py +0 -17
- icalens-0.3.0.dev1/tests/test_html_explorer.py +0 -108
- {icalens-0.3.0.dev1 → icalens-0.3.2}/.env.template +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/.github/workflows/release.yml +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/.readthedocs.yaml +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/LICENSE +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/THIRD_PARTY_NOTICES.md +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/demo/README.md +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/demo/__init__.py +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/demo/apply.py +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/demo/get_started.py +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/demo/plot_objective.py +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/docs/requirements.txt +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/notes/api.md +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/notes/artifact-format.md +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/notes/naming.md +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/_arrays.py +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/exceptions.py +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/src/icalens/py.typed +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/tests/conftest.py +0 -0
- {icalens-0.3.0.dev1 → icalens-0.3.2}/tests/test_capture.py +0 -0
icalens-0.3.2/PKG-INFO
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: icalens
|
|
3
|
+
Version: 0.3.2
|
|
4
|
+
Summary: Fit, share, and apply ICA lenses for language-model activations.
|
|
5
|
+
Project-URL: Homepage, https://liusida.github.io/ica-lens-paper/
|
|
6
|
+
Project-URL: Documentation, https://icalens.readthedocs.io/
|
|
7
|
+
Project-URL: Repository, https://github.com/liusida/icalens
|
|
8
|
+
Author: Sida Liu, Feijiang Han
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
License-File: THIRD_PARTY_NOTICES.md
|
|
12
|
+
Keywords: ICA,activations,interpretability,language-models
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Requires-Dist: datasets>=4.0
|
|
19
|
+
Requires-Dist: gb10-load-llm>=0.1.2
|
|
20
|
+
Requires-Dist: huggingface-hub>=0.25
|
|
21
|
+
Requires-Dist: numpy>=1.24
|
|
22
|
+
Requires-Dist: python-dotenv>=1.0
|
|
23
|
+
Requires-Dist: safetensors>=0.4
|
|
24
|
+
Requires-Dist: torch>=2.1
|
|
25
|
+
Requires-Dist: tqdm>=4.66
|
|
26
|
+
Requires-Dist: transformers>=5.0
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# ICA Lens
|
|
30
|
+
|
|
31
|
+
ICA Lens interprets language-model activations with Independent Component
|
|
32
|
+
Analysis. It is substantially more compute-efficient to fit than an SAE
|
|
33
|
+
dictionary and supports base and instruction-tuned language models.
|
|
34
|
+
|
|
35
|
+
**[Documentation](https://icalens.readthedocs.io/en/latest/)** ·
|
|
36
|
+
**[中文文档](https://icalens.readthedocs.io/zh_CN/latest/)** ·
|
|
37
|
+
**[Paper](https://arxiv.org/abs/2606.11722)** ·
|
|
38
|
+
**[Model collection](https://huggingface.co/sida)**
|
|
39
|
+
|
|
40
|
+
## Get started
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install icalens
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Load a published Lens and analyze text:
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from icalens import ICALens
|
|
50
|
+
|
|
51
|
+
lens = ICALens.from_pretrained("sida/icalens-gpt2-small-pile10k")
|
|
52
|
+
result = lens.analyze("She deposited the check at the bank.", layer=6)
|
|
53
|
+
result
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
In Jupyter or Colab, the final `result` expression displays an interactive
|
|
57
|
+
token-level analysis:
|
|
58
|
+
|
|
59
|
+

|
|
60
|
+
|
|
61
|
+
Use signed ICA scores or switch the explorer to per-token component energy.
|
|
62
|
+
Save the same view as a standalone HTML file with:
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
result.to_html("analysis.html")
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
The first analysis loads the language model and requested Lens layer. Later
|
|
69
|
+
calls on the same `lens` reuse the model in memory. `device="auto"` uses CUDA
|
|
70
|
+
when available and otherwise uses the CPU.
|
|
71
|
+
|
|
72
|
+
## Analyze conversations
|
|
73
|
+
|
|
74
|
+
Instruction-tuned models accept completed conversations using the standard
|
|
75
|
+
`{role, content}` format:
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
lens = ICALens.from_pretrained("sida/icalens-qwen3.5-2b-ultrachat-1m")
|
|
79
|
+
result = lens.analyze(
|
|
80
|
+
[
|
|
81
|
+
{"role": "user", "content": "What is the most interesting science?"},
|
|
82
|
+
{"role": "assistant", "content": "Physics."},
|
|
83
|
+
],
|
|
84
|
+
layer=16,
|
|
85
|
+
)
|
|
86
|
+
result
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Chat templates are applied automatically, and template tokens and message
|
|
90
|
+
turns are grouped in the interactive result.
|
|
91
|
+
|
|
92
|
+
## Steering
|
|
93
|
+
|
|
94
|
+
Generate normally or clamp a signed ICA coordinate during generation:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
messages = [{
|
|
98
|
+
"role": "user",
|
|
99
|
+
"content": "If you had to pick one, what is the most interesting science? Be brief.",
|
|
100
|
+
}]
|
|
101
|
+
|
|
102
|
+
baseline = lens.generate(messages, max_new_tokens=16)
|
|
103
|
+
steered = lens.generate(
|
|
104
|
+
messages,
|
|
105
|
+
layer=5,
|
|
106
|
+
clamp=(188, -20.0),
|
|
107
|
+
max_new_tokens=16,
|
|
108
|
+
)
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Component labels, signs, and suitable targets must be established empirically
|
|
112
|
+
for the exact Lens and layer. See the
|
|
113
|
+
**[steering tutorial](https://icalens.readthedocs.io/en/latest/steering/)** for
|
|
114
|
+
the reproducible inspection and calibration workflow.
|
|
115
|
+
|
|
116
|
+
## Fit a Lens
|
|
117
|
+
|
|
118
|
+
Run a small GPT-2/Pile-10k example with the installed CLI:
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
icalens fit text \
|
|
122
|
+
--model openai-community/gpt2 \
|
|
123
|
+
--dataset NeelNanda/pile-10k \
|
|
124
|
+
--layers 6 \
|
|
125
|
+
--token-budget 1000 \
|
|
126
|
+
--max-iter 20 \
|
|
127
|
+
--output icalens-output/gpt2-demo
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Fit an instruction-tuned model from UltraChat conversations:
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
icalens fit chat \
|
|
134
|
+
--model Qwen/Qwen3.5-2B \
|
|
135
|
+
--dataset HuggingFaceH4/ultrachat_200k \
|
|
136
|
+
--layers 12 \
|
|
137
|
+
--token-budget 100000 \
|
|
138
|
+
--output icalens-output/qwen-demo
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
ICA Lens includes a PyTorch FastICA implementation and does not depend on
|
|
142
|
+
SciPy or scikit-learn. Blockwise fitting and layer-at-a-time capture support
|
|
143
|
+
larger token collections while bounding memory use.
|
|
144
|
+
|
|
145
|
+
## Profile every fitted layer
|
|
146
|
+
|
|
147
|
+
After fitting, profile the components against a representative corpus:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
icalens profile \
|
|
151
|
+
--lens icalens-output/gpt2-demo \
|
|
152
|
+
--layers all \
|
|
153
|
+
--dataset NeelNanda/pile-10k \
|
|
154
|
+
--split train \
|
|
155
|
+
--max-tokens 10000
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Profiles add sign statistics, high-energy examples, and logit-lens tokens to
|
|
159
|
+
the existing Lens directory. They help label and inspect components without
|
|
160
|
+
changing the fitted directions.
|
|
161
|
+
|
|
162
|
+
## Publish to Hugging Face
|
|
163
|
+
|
|
164
|
+
Authenticate with `hf auth login`, set `HF_TOKEN`, or add a `.env` file in the
|
|
165
|
+
current directory containing a write-enabled token:
|
|
166
|
+
|
|
167
|
+
```dotenv
|
|
168
|
+
HF_TOKEN=hf_...
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
Then publish the saved Lens as a Hugging Face model repository:
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
icalens publish \
|
|
175
|
+
--lens icalens-output/gpt2-demo \
|
|
176
|
+
username/icalens-gpt2-demo
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
The artifact records the analyzed model, activation site, fitted layers,
|
|
180
|
+
preprocessing, component profiles, and fitting and profiling provenance.
|
|
181
|
+
Individual layer and profile files are downloaded lazily when a published Lens
|
|
182
|
+
is used.
|
|
183
|
+
|
|
184
|
+
## Learn more
|
|
185
|
+
|
|
186
|
+
The documentation covers:
|
|
187
|
+
|
|
188
|
+
- [Getting started](https://icalens.readthedocs.io/en/latest/getting-started/)
|
|
189
|
+
- [Text and chat](https://icalens.readthedocs.io/en/latest/text-and-chat/)
|
|
190
|
+
- [Component profiles](https://icalens.readthedocs.io/en/latest/component-profiles/)
|
|
191
|
+
- [Scores and energy](https://icalens.readthedocs.io/en/latest/scores-and-energy/)
|
|
192
|
+
- [Steering](https://icalens.readthedocs.io/en/latest/steering/)
|
|
193
|
+
- [Reconstruction](https://icalens.readthedocs.io/en/latest/reconstruction/)
|
|
194
|
+
- [Fitting and publishing](https://icalens.readthedocs.io/en/latest/fit-and-publish/)
|
|
195
|
+
- [Python API](https://icalens.readthedocs.io/en/latest/api/)
|
|
196
|
+
|
|
197
|
+
The repository also contains compact notebooks in [`demo/`](demo/) covering
|
|
198
|
+
text analysis, conversations, reconstruction, fitting, and steering.
|
|
199
|
+
|
|
200
|
+
## Authors
|
|
201
|
+
|
|
202
|
+
- [Sida Liu](https://liusida.com/)
|
|
203
|
+
- [Feijiang Han](https://feijianghan.com/)
|
|
204
|
+
|
|
205
|
+
## Citation
|
|
206
|
+
|
|
207
|
+
```bibtex
|
|
208
|
+
@article{liu2026icalens,
|
|
209
|
+
title={ICA Lens: Interpreting Language Models Without Training Another Dictionary},
|
|
210
|
+
author={Liu, Sida and Han, Feijiang},
|
|
211
|
+
journal={arXiv preprint arXiv:2606.11722},
|
|
212
|
+
year={2026}
|
|
213
|
+
}
|
|
214
|
+
```
|
icalens-0.3.2/README.md
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
# ICA Lens
|
|
2
|
+
|
|
3
|
+
ICA Lens interprets language-model activations with Independent Component
|
|
4
|
+
Analysis. It is substantially more compute-efficient to fit than an SAE
|
|
5
|
+
dictionary and supports base and instruction-tuned language models.
|
|
6
|
+
|
|
7
|
+
**[Documentation](https://icalens.readthedocs.io/en/latest/)** ·
|
|
8
|
+
**[中文文档](https://icalens.readthedocs.io/zh_CN/latest/)** ·
|
|
9
|
+
**[Paper](https://arxiv.org/abs/2606.11722)** ·
|
|
10
|
+
**[Model collection](https://huggingface.co/sida)**
|
|
11
|
+
|
|
12
|
+
## Get started
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
pip install icalens
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Load a published Lens and analyze text:
|
|
19
|
+
|
|
20
|
+
```python
|
|
21
|
+
from icalens import ICALens
|
|
22
|
+
|
|
23
|
+
lens = ICALens.from_pretrained("sida/icalens-gpt2-small-pile10k")
|
|
24
|
+
result = lens.analyze("She deposited the check at the bank.", layer=6)
|
|
25
|
+
result
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
In Jupyter or Colab, the final `result` expression displays an interactive
|
|
29
|
+
token-level analysis:
|
|
30
|
+
|
|
31
|
+

|
|
32
|
+
|
|
33
|
+
Use signed ICA scores or switch the explorer to per-token component energy.
|
|
34
|
+
Save the same view as a standalone HTML file with:
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
result.to_html("analysis.html")
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
The first analysis loads the language model and requested Lens layer. Later
|
|
41
|
+
calls on the same `lens` reuse the model in memory. `device="auto"` uses CUDA
|
|
42
|
+
when available and otherwise uses the CPU.
|
|
43
|
+
|
|
44
|
+
## Analyze conversations
|
|
45
|
+
|
|
46
|
+
Instruction-tuned models accept completed conversations using the standard
|
|
47
|
+
`{role, content}` format:
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
lens = ICALens.from_pretrained("sida/icalens-qwen3.5-2b-ultrachat-1m")
|
|
51
|
+
result = lens.analyze(
|
|
52
|
+
[
|
|
53
|
+
{"role": "user", "content": "What is the most interesting science?"},
|
|
54
|
+
{"role": "assistant", "content": "Physics."},
|
|
55
|
+
],
|
|
56
|
+
layer=16,
|
|
57
|
+
)
|
|
58
|
+
result
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Chat templates are applied automatically, and template tokens and message
|
|
62
|
+
turns are grouped in the interactive result.
|
|
63
|
+
|
|
64
|
+
## Steering
|
|
65
|
+
|
|
66
|
+
Generate normally or clamp a signed ICA coordinate during generation:
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
messages = [{
|
|
70
|
+
"role": "user",
|
|
71
|
+
"content": "If you had to pick one, what is the most interesting science? Be brief.",
|
|
72
|
+
}]
|
|
73
|
+
|
|
74
|
+
baseline = lens.generate(messages, max_new_tokens=16)
|
|
75
|
+
steered = lens.generate(
|
|
76
|
+
messages,
|
|
77
|
+
layer=5,
|
|
78
|
+
clamp=(188, -20.0),
|
|
79
|
+
max_new_tokens=16,
|
|
80
|
+
)
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Component labels, signs, and suitable targets must be established empirically
|
|
84
|
+
for the exact Lens and layer. See the
|
|
85
|
+
**[steering tutorial](https://icalens.readthedocs.io/en/latest/steering/)** for
|
|
86
|
+
the reproducible inspection and calibration workflow.
|
|
87
|
+
|
|
88
|
+
## Fit a Lens
|
|
89
|
+
|
|
90
|
+
Run a small GPT-2/Pile-10k example with the installed CLI:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
icalens fit text \
|
|
94
|
+
--model openai-community/gpt2 \
|
|
95
|
+
--dataset NeelNanda/pile-10k \
|
|
96
|
+
--layers 6 \
|
|
97
|
+
--token-budget 1000 \
|
|
98
|
+
--max-iter 20 \
|
|
99
|
+
--output icalens-output/gpt2-demo
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Fit an instruction-tuned model from UltraChat conversations:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
icalens fit chat \
|
|
106
|
+
--model Qwen/Qwen3.5-2B \
|
|
107
|
+
--dataset HuggingFaceH4/ultrachat_200k \
|
|
108
|
+
--layers 12 \
|
|
109
|
+
--token-budget 100000 \
|
|
110
|
+
--output icalens-output/qwen-demo
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
ICA Lens includes a PyTorch FastICA implementation and does not depend on
|
|
114
|
+
SciPy or scikit-learn. Blockwise fitting and layer-at-a-time capture support
|
|
115
|
+
larger token collections while bounding memory use.
|
|
116
|
+
|
|
117
|
+
## Profile every fitted layer
|
|
118
|
+
|
|
119
|
+
After fitting, profile the components against a representative corpus:
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
icalens profile \
|
|
123
|
+
--lens icalens-output/gpt2-demo \
|
|
124
|
+
--layers all \
|
|
125
|
+
--dataset NeelNanda/pile-10k \
|
|
126
|
+
--split train \
|
|
127
|
+
--max-tokens 10000
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Profiles add sign statistics, high-energy examples, and logit-lens tokens to
|
|
131
|
+
the existing Lens directory. They help label and inspect components without
|
|
132
|
+
changing the fitted directions.
|
|
133
|
+
|
|
134
|
+
## Publish to Hugging Face
|
|
135
|
+
|
|
136
|
+
Authenticate with `hf auth login`, set `HF_TOKEN`, or add a `.env` file in the
|
|
137
|
+
current directory containing a write-enabled token:
|
|
138
|
+
|
|
139
|
+
```dotenv
|
|
140
|
+
HF_TOKEN=hf_...
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Then publish the saved Lens as a Hugging Face model repository:
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
icalens publish \
|
|
147
|
+
--lens icalens-output/gpt2-demo \
|
|
148
|
+
username/icalens-gpt2-demo
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
The artifact records the analyzed model, activation site, fitted layers,
|
|
152
|
+
preprocessing, component profiles, and fitting and profiling provenance.
|
|
153
|
+
Individual layer and profile files are downloaded lazily when a published Lens
|
|
154
|
+
is used.
|
|
155
|
+
|
|
156
|
+
## Learn more
|
|
157
|
+
|
|
158
|
+
The documentation covers:
|
|
159
|
+
|
|
160
|
+
- [Getting started](https://icalens.readthedocs.io/en/latest/getting-started/)
|
|
161
|
+
- [Text and chat](https://icalens.readthedocs.io/en/latest/text-and-chat/)
|
|
162
|
+
- [Component profiles](https://icalens.readthedocs.io/en/latest/component-profiles/)
|
|
163
|
+
- [Scores and energy](https://icalens.readthedocs.io/en/latest/scores-and-energy/)
|
|
164
|
+
- [Steering](https://icalens.readthedocs.io/en/latest/steering/)
|
|
165
|
+
- [Reconstruction](https://icalens.readthedocs.io/en/latest/reconstruction/)
|
|
166
|
+
- [Fitting and publishing](https://icalens.readthedocs.io/en/latest/fit-and-publish/)
|
|
167
|
+
- [Python API](https://icalens.readthedocs.io/en/latest/api/)
|
|
168
|
+
|
|
169
|
+
The repository also contains compact notebooks in [`demo/`](demo/) covering
|
|
170
|
+
text analysis, conversations, reconstruction, fitting, and steering.
|
|
171
|
+
|
|
172
|
+
## Authors
|
|
173
|
+
|
|
174
|
+
- [Sida Liu](https://liusida.com/)
|
|
175
|
+
- [Feijiang Han](https://feijianghan.com/)
|
|
176
|
+
|
|
177
|
+
## Citation
|
|
178
|
+
|
|
179
|
+
```bibtex
|
|
180
|
+
@article{liu2026icalens,
|
|
181
|
+
title={ICA Lens: Interpreting Language Models Without Training Another Dictionary},
|
|
182
|
+
author={Liu, Sida and Han, Feijiang},
|
|
183
|
+
journal={arXiv preprint arXiv:2606.11722},
|
|
184
|
+
year={2026}
|
|
185
|
+
}
|
|
186
|
+
```
|
|
@@ -109,9 +109,7 @@ def main() -> None:
|
|
|
109
109
|
if tokenizer.chat_template is None:
|
|
110
110
|
raise RuntimeError(f"{lens.model_id} tokenizer does not define a chat template.")
|
|
111
111
|
|
|
112
|
-
user_turns = tuple(
|
|
113
|
-
turn for argument in (args.user or [(DEFAULT_USER,)]) for turn in argument
|
|
114
|
-
)
|
|
112
|
+
user_turns = tuple(turn for argument in (args.user or [(DEFAULT_USER,)]) for turn in argument)
|
|
115
113
|
messages = []
|
|
116
114
|
if args.system is not None:
|
|
117
115
|
messages.append({"role": "system", "content": args.system})
|
|
@@ -247,9 +245,7 @@ def group_tokens_by_message(
|
|
|
247
245
|
tokens: list[dict[str, object]],
|
|
248
246
|
) -> list[dict[str, object]]:
|
|
249
247
|
"""Group formatted token cards by the message that introduced them."""
|
|
250
|
-
rendered = tokenizer.apply_chat_template(
|
|
251
|
-
messages, tokenize=False, add_generation_prompt=False
|
|
252
|
-
)
|
|
248
|
+
rendered = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=False)
|
|
253
249
|
encoded = tokenizer(
|
|
254
250
|
rendered,
|
|
255
251
|
add_special_tokens=False,
|
|
@@ -279,10 +275,7 @@ def group_tokens_by_message(
|
|
|
279
275
|
opening_candidates = [
|
|
280
276
|
index
|
|
281
277
|
for index in range(content_token + 1)
|
|
282
|
-
if (
|
|
283
|
-
"start" in input_tokens[index].lower()
|
|
284
|
-
or "begin" in input_tokens[index].lower()
|
|
285
|
-
)
|
|
278
|
+
if ("start" in input_tokens[index].lower() or "begin" in input_tokens[index].lower())
|
|
286
279
|
and (not message_starts or index > message_starts[-1])
|
|
287
280
|
]
|
|
288
281
|
message_starts.append(opening_candidates[-1] if opening_candidates else content_token)
|
|
@@ -292,9 +285,7 @@ def group_tokens_by_message(
|
|
|
292
285
|
|
|
293
286
|
role_counts = {"system": 0, "user": 0, "assistant": 0}
|
|
294
287
|
groups = []
|
|
295
|
-
for message, start, end in zip(
|
|
296
|
-
visible_messages, message_starts, end_positions, strict=True
|
|
297
|
-
):
|
|
288
|
+
for message, start, end in zip(visible_messages, message_starts, end_positions, strict=True):
|
|
298
289
|
role = message["role"]
|
|
299
290
|
role_counts[role] += 1
|
|
300
291
|
title = role.title() if role == "system" else f"{role.title()} {role_counts[role]}"
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
set -euo pipefail
|
|
3
|
+
|
|
4
|
+
# Refit the three base-model ICA lenses with model-aware document framing.
|
|
5
|
+
# Each fit samples 1M rows from all usable Pile-10k tokens. New output paths
|
|
6
|
+
# preserve the existing artifacts until these corrected lenses are validated.
|
|
7
|
+
#
|
|
8
|
+
# Gemma prerequisite: accept the license for google/gemma-2-2b on Hugging Face.
|
|
9
|
+
|
|
10
|
+
echo "[1/6] Fitting GPT-2 Small"
|
|
11
|
+
uv run icalens fit text \
|
|
12
|
+
--model openai-community/gpt2 \
|
|
13
|
+
--dataset NeelNanda/pile-10k \
|
|
14
|
+
--split train \
|
|
15
|
+
--text-field text \
|
|
16
|
+
--layers all \
|
|
17
|
+
--capture-layers-at-once 5 \
|
|
18
|
+
--candidate-tokens all \
|
|
19
|
+
--token-budget 1000000 \
|
|
20
|
+
--fit-batch-size 32768 \
|
|
21
|
+
--max-iter 50 \
|
|
22
|
+
--refresh-model-registry \
|
|
23
|
+
--output three-icalens-fit/icalens-gpt2-small-pile10k-1m
|
|
24
|
+
|
|
25
|
+
echo "[2/6] Profiling GPT-2 Small"
|
|
26
|
+
uv run icalens profile \
|
|
27
|
+
--lens three-icalens-fit/icalens-gpt2-small-pile10k-1m \
|
|
28
|
+
--layers all \
|
|
29
|
+
--dataset NeelNanda/pile-10k \
|
|
30
|
+
--split train \
|
|
31
|
+
--input-type text \
|
|
32
|
+
--text-field text \
|
|
33
|
+
--max-tokens 1000000 \
|
|
34
|
+
--top-k-examples 20 \
|
|
35
|
+
--min-energy 0.05
|
|
36
|
+
|
|
37
|
+
echo "[3/6] Fitting Gemma 2 2B Base"
|
|
38
|
+
uv run icalens fit text \
|
|
39
|
+
--model google/gemma-2-2b \
|
|
40
|
+
--dataset NeelNanda/pile-10k \
|
|
41
|
+
--split train \
|
|
42
|
+
--text-field text \
|
|
43
|
+
--layers all \
|
|
44
|
+
--capture-layers-at-once 5 \
|
|
45
|
+
--candidate-tokens all \
|
|
46
|
+
--token-budget 1000000 \
|
|
47
|
+
--fit-batch-size 32768 \
|
|
48
|
+
--max-iter 50 \
|
|
49
|
+
--refresh-model-registry \
|
|
50
|
+
--output three-icalens-fit/icalens-gemma-2-2b-pile10k-1m
|
|
51
|
+
|
|
52
|
+
echo "[4/6] Profiling Gemma 2 2B Base"
|
|
53
|
+
uv run icalens profile \
|
|
54
|
+
--lens three-icalens-fit/icalens-gemma-2-2b-pile10k-1m \
|
|
55
|
+
--layers all \
|
|
56
|
+
--dataset NeelNanda/pile-10k \
|
|
57
|
+
--split train \
|
|
58
|
+
--input-type text \
|
|
59
|
+
--text-field text \
|
|
60
|
+
--max-tokens 1000000 \
|
|
61
|
+
--top-k-examples 20 \
|
|
62
|
+
--min-energy 0.05
|
|
63
|
+
|
|
64
|
+
echo "[5/6] Fitting Qwen3.5 2B Base"
|
|
65
|
+
uv run icalens fit text \
|
|
66
|
+
--model Qwen/Qwen3.5-2B-Base \
|
|
67
|
+
--dataset NeelNanda/pile-10k \
|
|
68
|
+
--split train \
|
|
69
|
+
--text-field text \
|
|
70
|
+
--layers all \
|
|
71
|
+
--capture-layers-at-once 5 \
|
|
72
|
+
--candidate-tokens all \
|
|
73
|
+
--token-budget 1000000 \
|
|
74
|
+
--fit-batch-size 32768 \
|
|
75
|
+
--max-iter 50 \
|
|
76
|
+
--refresh-model-registry \
|
|
77
|
+
--output three-icalens-fit/icalens-qwen3.5-2b-base-pile10k-1m
|
|
78
|
+
|
|
79
|
+
echo "[6/6] Profiling Qwen3.5 2B Base"
|
|
80
|
+
uv run icalens profile \
|
|
81
|
+
--lens three-icalens-fit/icalens-qwen3.5-2b-base-pile10k-1m \
|
|
82
|
+
--layers all \
|
|
83
|
+
--dataset NeelNanda/pile-10k \
|
|
84
|
+
--split train \
|
|
85
|
+
--input-type text \
|
|
86
|
+
--text-field text \
|
|
87
|
+
--max-tokens 1000000 \
|
|
88
|
+
--top-k-examples 20 \
|
|
89
|
+
--min-energy 0.05
|
|
90
|
+
|
|
91
|
+
echo "Finished fitting and profiling all three corrected base-model lenses."
|