interp-engine 0.0.24__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interp_engine-0.0.24.dist-info/METADATA +21 -0
- interp_engine-0.0.24.dist-info/RECORD +28 -0
- interp_engine-0.0.24.dist-info/WHEEL +4 -0
- interp_engine-0.0.24.dist-info/licenses/LICENSE +19 -0
- neuron_explainer/__init__.py +0 -0
- neuron_explainer/activations/__init__.py +0 -0
- neuron_explainer/activations/activation_records.py +130 -0
- neuron_explainer/activations/activations.py +311 -0
- neuron_explainer/activations/attention_utils.py +121 -0
- neuron_explainer/activations/token_connections.py +59 -0
- neuron_explainer/api_client.py +190 -0
- neuron_explainer/azure.py +5 -0
- neuron_explainer/explanations/__init__.py +0 -0
- neuron_explainer/explanations/calibrated_simulator.py +194 -0
- neuron_explainer/explanations/explainer.py +2585 -0
- neuron_explainer/explanations/explanations.py +230 -0
- neuron_explainer/explanations/few_shot_examples.py +3125 -0
- neuron_explainer/explanations/prompt_builder.py +118 -0
- neuron_explainer/explanations/puzzles.json +399 -0
- neuron_explainer/explanations/puzzles.py +50 -0
- neuron_explainer/explanations/scoring.py +155 -0
- neuron_explainer/explanations/simulator.py +1121 -0
- neuron_explainer/explanations/test_explainer.py +227 -0
- neuron_explainer/explanations/test_simulator.py +269 -0
- neuron_explainer/explanations/token_space_few_shot_examples.py +212 -0
- neuron_explainer/fast_dataclasses/__init__.py +3 -0
- neuron_explainer/fast_dataclasses/fast_dataclasses.py +85 -0
- neuron_explainer/fast_dataclasses/test_fast_dataclasses.py +83 -0
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
# Dataclasses and enums for storing neuron explanations, their scores, and related data. Also,
|
|
2
|
+
# related helper functions.
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from enum import Enum
|
|
9
|
+
from typing import List, Optional, Union
|
|
10
|
+
|
|
11
|
+
import blobfile as bf
|
|
12
|
+
import boostedblob as bbb
|
|
13
|
+
from neuron_explainer.activations.activations import NeuronId
|
|
14
|
+
from neuron_explainer.fast_dataclasses import FastDataclass, loads, register_dataclass
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ActivationScale(str, Enum):
|
|
18
|
+
"""Which "units" are stored in the expected_activations/distribution_values fields of a
|
|
19
|
+
SequenceSimulation.
|
|
20
|
+
|
|
21
|
+
This enum identifies whether the values represent real activations of the neuron or something
|
|
22
|
+
else. Different scales are not necessarily related by a linear transformation.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
NEURON_ACTIVATIONS = "neuron_activations"
|
|
26
|
+
"""Values represent real activations of the neuron."""
|
|
27
|
+
SIMULATED_NORMALIZED_ACTIVATIONS = "simulated_normalized_activations"
|
|
28
|
+
"""
|
|
29
|
+
Values represent simulated activations of the neuron, normalized to the range [0, 10]. This
|
|
30
|
+
scale is arbitrary and should not be interpreted as a neuron activation.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@register_dataclass
|
|
35
|
+
@dataclass
|
|
36
|
+
class SequenceSimulation(FastDataclass):
|
|
37
|
+
"""The result of a simulation of neuron activations on one text sequence."""
|
|
38
|
+
|
|
39
|
+
tokens: list[str]
|
|
40
|
+
"""The sequence of tokens that was simulated."""
|
|
41
|
+
expected_activations: list[float]
|
|
42
|
+
"""Expected value of the possibly-normalized activation for each token in the sequence."""
|
|
43
|
+
activation_scale: ActivationScale
|
|
44
|
+
"""What scale is used for values in the expected_activations field."""
|
|
45
|
+
distribution_values: list[list[float]]
|
|
46
|
+
"""
|
|
47
|
+
For each token in the sequence, a list of values from the discrete distribution of activations
|
|
48
|
+
produced from simulation. Tokens will be included here if and only if they are in the top K=15
|
|
49
|
+
tokens predicted by the simulator, and excluded otherwise.
|
|
50
|
+
|
|
51
|
+
May be transformed to another unit by calibration. When we simulate a neuron, we produce a
|
|
52
|
+
discrete distribution with values in the arbitrary discretized space of the neuron, e.g. 10%
|
|
53
|
+
chance of 0, 70% chance of 1, 20% chance of 2. Which we store as distribution_values =
|
|
54
|
+
[0, 1, 2], distribution_probabilities = [0.1, 0.7, 0.2]. When we transform the distribution to
|
|
55
|
+
the real activation units, we can correspondingly transform the values of this distribution
|
|
56
|
+
to get a distribution in the units of the neuron. e.g. if the mapping from the discretized space
|
|
57
|
+
to the real activation unit of the neuron is f(x) = x/2, then the distribution becomes 10%
|
|
58
|
+
chance of 0, 70% chance of 0.5, 20% chance of 1. Which we store as distribution_values =
|
|
59
|
+
[0, 0.5, 1], distribution_probabilities = [0.1, 0.7, 0.2].
|
|
60
|
+
"""
|
|
61
|
+
distribution_probabilities: list[list[float]]
|
|
62
|
+
"""
|
|
63
|
+
For each token in the sequence, the probability of the corresponding value in
|
|
64
|
+
distribution_values.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
uncalibrated_simulation: Optional["SequenceSimulation"] = None
|
|
68
|
+
"""The result of the simulation before calibration."""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@register_dataclass
|
|
72
|
+
@dataclass
|
|
73
|
+
class ScoredSequenceSimulation(FastDataclass):
|
|
74
|
+
"""
|
|
75
|
+
SequenceSimulation result with a score (for that sequence only) and ground truth activations.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
simulation: SequenceSimulation
|
|
79
|
+
"""The result of a simulation of neuron activations."""
|
|
80
|
+
true_activations: List[float]
|
|
81
|
+
"""Ground truth activations on the sequence (not normalized)"""
|
|
82
|
+
ev_correlation_score: float
|
|
83
|
+
"""
|
|
84
|
+
Correlation coefficient between the expected values of the normalized activations from the
|
|
85
|
+
simulation and the unnormalized true activations of the neuron on the text sequence.
|
|
86
|
+
"""
|
|
87
|
+
rsquared_score: Optional[float] = None
|
|
88
|
+
"""R^2 of the simulated activations."""
|
|
89
|
+
absolute_dev_explained_score: Optional[float] = None
|
|
90
|
+
"""
|
|
91
|
+
Score based on absolute difference between real and simulated activations.
|
|
92
|
+
absolute_dev_explained_score = 1 - mean(abs(real-predicted))/ mean(abs(real))
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@register_dataclass
|
|
97
|
+
@dataclass
|
|
98
|
+
class ScoredSimulation(FastDataclass):
|
|
99
|
+
"""Result of scoring a neuron simulation on multiple sequences."""
|
|
100
|
+
|
|
101
|
+
scored_sequence_simulations: List[ScoredSequenceSimulation]
|
|
102
|
+
"""ScoredSequenceSimulation for each sequence"""
|
|
103
|
+
ev_correlation_score: Optional[float] = None
|
|
104
|
+
"""
|
|
105
|
+
Correlation coefficient between the expected values of the normalized activations from the
|
|
106
|
+
simulation and the unnormalized true activations on a dataset created from all score_results.
|
|
107
|
+
(Note that this is not equivalent to averaging across sequences.)
|
|
108
|
+
"""
|
|
109
|
+
rsquared_score: Optional[float] = None
|
|
110
|
+
"""R^2 of the simulated activations."""
|
|
111
|
+
absolute_dev_explained_score: Optional[float] = None
|
|
112
|
+
"""
|
|
113
|
+
Score based on absolute difference between real and simulated activations.
|
|
114
|
+
absolute_dev_explained_score = 1 - mean(abs(real-predicted))/ mean(abs(real)).
|
|
115
|
+
"""
|
|
116
|
+
|
|
117
|
+
def get_preferred_score(self) -> Optional[float]:
|
|
118
|
+
"""
|
|
119
|
+
This method may return None in cases where the score is undefined, for example if the
|
|
120
|
+
normalized activations were all zero, yielding a correlation coefficient of NaN.
|
|
121
|
+
"""
|
|
122
|
+
return self.ev_correlation_score
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@register_dataclass
|
|
126
|
+
@dataclass
|
|
127
|
+
class ScoredExplanation(FastDataclass):
|
|
128
|
+
"""Simulator parameters and the results of scoring it on multiple sequences"""
|
|
129
|
+
|
|
130
|
+
explanation: str
|
|
131
|
+
"""The explanation used for simulation."""
|
|
132
|
+
|
|
133
|
+
scored_simulation: ScoredSimulation
|
|
134
|
+
"""Result of scoring the neuron simulator on multiple sequences."""
|
|
135
|
+
|
|
136
|
+
def get_preferred_score(self) -> Optional[float]:
|
|
137
|
+
"""
|
|
138
|
+
This method may return None in cases where the score is undefined, for example if the
|
|
139
|
+
normalized activations were all zero, yielding a correlation coefficient of NaN.
|
|
140
|
+
"""
|
|
141
|
+
return self.scored_simulation.get_preferred_score()
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@register_dataclass
|
|
145
|
+
@dataclass
|
|
146
|
+
class NeuronSimulationResults(FastDataclass):
|
|
147
|
+
"""Simulation results and scores for a neuron."""
|
|
148
|
+
|
|
149
|
+
neuron_id: NeuronId
|
|
150
|
+
scored_explanations: list[ScoredExplanation]
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def load_neuron_explanations(
|
|
154
|
+
explanations_path: str, layer_index: Union[str, int], neuron_index: Union[str, int]
|
|
155
|
+
) -> Optional[NeuronSimulationResults]:
|
|
156
|
+
"""Load scored explanations for the specified neuron."""
|
|
157
|
+
file = bf.join(explanations_path, str(layer_index), f"{neuron_index}.jsonl")
|
|
158
|
+
if not bf.exists(file):
|
|
159
|
+
return None
|
|
160
|
+
with bf.BlobFile(file) as f:
|
|
161
|
+
for line in f:
|
|
162
|
+
return loads(line)
|
|
163
|
+
return None
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
@bbb.ensure_session
|
|
167
|
+
async def load_neuron_explanations_async(
|
|
168
|
+
explanations_path: str, layer_index: Union[str, int], neuron_index: Union[str, int]
|
|
169
|
+
) -> Optional[NeuronSimulationResults]:
|
|
170
|
+
"""Load scored explanations for the specified neuron, asynchronously."""
|
|
171
|
+
return await read_explanation_file(
|
|
172
|
+
bf.join(explanations_path, str(layer_index), f"{neuron_index}.jsonl")
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@bbb.ensure_session
|
|
177
|
+
async def read_file(filename: str) -> Optional[str]:
|
|
178
|
+
"""Read the contents of the given file as a string, asynchronously."""
|
|
179
|
+
try:
|
|
180
|
+
raw_contents = await bbb.read.read_single(filename)
|
|
181
|
+
except FileNotFoundError:
|
|
182
|
+
print(f"Could not read {filename}")
|
|
183
|
+
return None
|
|
184
|
+
lines = []
|
|
185
|
+
for line in raw_contents.decode("utf-8").split("\n"):
|
|
186
|
+
if len(line) > 0:
|
|
187
|
+
lines.append(line)
|
|
188
|
+
assert len(lines) == 1, filename
|
|
189
|
+
return lines[0]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
@bbb.ensure_session
|
|
193
|
+
async def read_explanation_file(explanation_filename: str) -> Optional[NeuronSimulationResults]:
|
|
194
|
+
"""Load scored explanations from the given filename, asynchronously."""
|
|
195
|
+
line = await read_file(explanation_filename)
|
|
196
|
+
return loads(line) if line is not None else None
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
@bbb.ensure_session
|
|
200
|
+
async def read_json_file(filename: str) -> Optional[dict]:
|
|
201
|
+
"""Read the contents of the given file as a JSON object, asynchronously."""
|
|
202
|
+
line = await read_file(filename)
|
|
203
|
+
return json.loads(line) if line is not None else None
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def get_numerical_subdirs(dataset_path: str) -> list[str]:
|
|
207
|
+
"""Return the names of all numbered subdirectories in the specified directory.
|
|
208
|
+
|
|
209
|
+
Used to get all layer directories in an explanation directory.
|
|
210
|
+
"""
|
|
211
|
+
return [
|
|
212
|
+
str(x)
|
|
213
|
+
for x in sorted(
|
|
214
|
+
[
|
|
215
|
+
int(x)
|
|
216
|
+
for x in bf.listdir(dataset_path)
|
|
217
|
+
if bf.isdir(bf.join(dataset_path, x)) and x.isnumeric()
|
|
218
|
+
]
|
|
219
|
+
)
|
|
220
|
+
]
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def get_sorted_neuron_indices_from_explanations(
|
|
224
|
+
explanations_path: str, layer: Union[str, int]
|
|
225
|
+
) -> list[int]:
|
|
226
|
+
"""Return the indices of all neurons in this layer, in ascending order."""
|
|
227
|
+
layer_dir = bf.join(explanations_path, str(layer))
|
|
228
|
+
return sorted(
|
|
229
|
+
[int(f.split(".")[0]) for f in bf.listdir(layer_dir) if f.split(".")[0].isnumeric()]
|
|
230
|
+
)
|