interp-engine 0.0.24__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,230 @@
1
+ # Dataclasses and enums for storing neuron explanations, their scores, and related data. Also,
2
+ # related helper functions.
3
+
4
+ from __future__ import annotations
5
+
6
+ import json
7
+ from dataclasses import dataclass
8
+ from enum import Enum
9
+ from typing import List, Optional, Union
10
+
11
+ import blobfile as bf
12
+ import boostedblob as bbb
13
+ from neuron_explainer.activations.activations import NeuronId
14
+ from neuron_explainer.fast_dataclasses import FastDataclass, loads, register_dataclass
15
+
16
+
17
+ class ActivationScale(str, Enum):
18
+ """Which "units" are stored in the expected_activations/distribution_values fields of a
19
+ SequenceSimulation.
20
+
21
+ This enum identifies whether the values represent real activations of the neuron or something
22
+ else. Different scales are not necessarily related by a linear transformation.
23
+ """
24
+
25
+ NEURON_ACTIVATIONS = "neuron_activations"
26
+ """Values represent real activations of the neuron."""
27
+ SIMULATED_NORMALIZED_ACTIVATIONS = "simulated_normalized_activations"
28
+ """
29
+ Values represent simulated activations of the neuron, normalized to the range [0, 10]. This
30
+ scale is arbitrary and should not be interpreted as a neuron activation.
31
+ """
32
+
33
+
34
+ @register_dataclass
35
+ @dataclass
36
+ class SequenceSimulation(FastDataclass):
37
+ """The result of a simulation of neuron activations on one text sequence."""
38
+
39
+ tokens: list[str]
40
+ """The sequence of tokens that was simulated."""
41
+ expected_activations: list[float]
42
+ """Expected value of the possibly-normalized activation for each token in the sequence."""
43
+ activation_scale: ActivationScale
44
+ """What scale is used for values in the expected_activations field."""
45
+ distribution_values: list[list[float]]
46
+ """
47
+ For each token in the sequence, a list of values from the discrete distribution of activations
48
+ produced from simulation. Tokens will be included here if and only if they are in the top K=15
49
+ tokens predicted by the simulator, and excluded otherwise.
50
+
51
+ May be transformed to another unit by calibration. When we simulate a neuron, we produce a
52
+ discrete distribution with values in the arbitrary discretized space of the neuron, e.g. 10%
53
+ chance of 0, 70% chance of 1, 20% chance of 2. Which we store as distribution_values =
54
+ [0, 1, 2], distribution_probabilities = [0.1, 0.7, 0.2]. When we transform the distribution to
55
+ the real activation units, we can correspondingly transform the values of this distribution
56
+ to get a distribution in the units of the neuron. e.g. if the mapping from the discretized space
57
+ to the real activation unit of the neuron is f(x) = x/2, then the distribution becomes 10%
58
+ chance of 0, 70% chance of 0.5, 20% chance of 1. Which we store as distribution_values =
59
+ [0, 0.5, 1], distribution_probabilities = [0.1, 0.7, 0.2].
60
+ """
61
+ distribution_probabilities: list[list[float]]
62
+ """
63
+ For each token in the sequence, the probability of the corresponding value in
64
+ distribution_values.
65
+ """
66
+
67
+ uncalibrated_simulation: Optional["SequenceSimulation"] = None
68
+ """The result of the simulation before calibration."""
69
+
70
+
71
+ @register_dataclass
72
+ @dataclass
73
+ class ScoredSequenceSimulation(FastDataclass):
74
+ """
75
+ SequenceSimulation result with a score (for that sequence only) and ground truth activations.
76
+ """
77
+
78
+ simulation: SequenceSimulation
79
+ """The result of a simulation of neuron activations."""
80
+ true_activations: List[float]
81
+ """Ground truth activations on the sequence (not normalized)"""
82
+ ev_correlation_score: float
83
+ """
84
+ Correlation coefficient between the expected values of the normalized activations from the
85
+ simulation and the unnormalized true activations of the neuron on the text sequence.
86
+ """
87
+ rsquared_score: Optional[float] = None
88
+ """R^2 of the simulated activations."""
89
+ absolute_dev_explained_score: Optional[float] = None
90
+ """
91
+ Score based on absolute difference between real and simulated activations.
92
+ absolute_dev_explained_score = 1 - mean(abs(real-predicted))/ mean(abs(real))
93
+ """
94
+
95
+
96
+ @register_dataclass
97
+ @dataclass
98
+ class ScoredSimulation(FastDataclass):
99
+ """Result of scoring a neuron simulation on multiple sequences."""
100
+
101
+ scored_sequence_simulations: List[ScoredSequenceSimulation]
102
+ """ScoredSequenceSimulation for each sequence"""
103
+ ev_correlation_score: Optional[float] = None
104
+ """
105
+ Correlation coefficient between the expected values of the normalized activations from the
106
+ simulation and the unnormalized true activations on a dataset created from all score_results.
107
+ (Note that this is not equivalent to averaging across sequences.)
108
+ """
109
+ rsquared_score: Optional[float] = None
110
+ """R^2 of the simulated activations."""
111
+ absolute_dev_explained_score: Optional[float] = None
112
+ """
113
+ Score based on absolute difference between real and simulated activations.
114
+ absolute_dev_explained_score = 1 - mean(abs(real-predicted))/ mean(abs(real)).
115
+ """
116
+
117
+ def get_preferred_score(self) -> Optional[float]:
118
+ """
119
+ This method may return None in cases where the score is undefined, for example if the
120
+ normalized activations were all zero, yielding a correlation coefficient of NaN.
121
+ """
122
+ return self.ev_correlation_score
123
+
124
+
125
+ @register_dataclass
126
+ @dataclass
127
+ class ScoredExplanation(FastDataclass):
128
+ """Simulator parameters and the results of scoring it on multiple sequences"""
129
+
130
+ explanation: str
131
+ """The explanation used for simulation."""
132
+
133
+ scored_simulation: ScoredSimulation
134
+ """Result of scoring the neuron simulator on multiple sequences."""
135
+
136
+ def get_preferred_score(self) -> Optional[float]:
137
+ """
138
+ This method may return None in cases where the score is undefined, for example if the
139
+ normalized activations were all zero, yielding a correlation coefficient of NaN.
140
+ """
141
+ return self.scored_simulation.get_preferred_score()
142
+
143
+
144
+ @register_dataclass
145
+ @dataclass
146
+ class NeuronSimulationResults(FastDataclass):
147
+ """Simulation results and scores for a neuron."""
148
+
149
+ neuron_id: NeuronId
150
+ scored_explanations: list[ScoredExplanation]
151
+
152
+
153
+ def load_neuron_explanations(
154
+ explanations_path: str, layer_index: Union[str, int], neuron_index: Union[str, int]
155
+ ) -> Optional[NeuronSimulationResults]:
156
+ """Load scored explanations for the specified neuron."""
157
+ file = bf.join(explanations_path, str(layer_index), f"{neuron_index}.jsonl")
158
+ if not bf.exists(file):
159
+ return None
160
+ with bf.BlobFile(file) as f:
161
+ for line in f:
162
+ return loads(line)
163
+ return None
164
+
165
+
166
+ @bbb.ensure_session
167
+ async def load_neuron_explanations_async(
168
+ explanations_path: str, layer_index: Union[str, int], neuron_index: Union[str, int]
169
+ ) -> Optional[NeuronSimulationResults]:
170
+ """Load scored explanations for the specified neuron, asynchronously."""
171
+ return await read_explanation_file(
172
+ bf.join(explanations_path, str(layer_index), f"{neuron_index}.jsonl")
173
+ )
174
+
175
+
176
+ @bbb.ensure_session
177
+ async def read_file(filename: str) -> Optional[str]:
178
+ """Read the contents of the given file as a string, asynchronously."""
179
+ try:
180
+ raw_contents = await bbb.read.read_single(filename)
181
+ except FileNotFoundError:
182
+ print(f"Could not read {filename}")
183
+ return None
184
+ lines = []
185
+ for line in raw_contents.decode("utf-8").split("\n"):
186
+ if len(line) > 0:
187
+ lines.append(line)
188
+ assert len(lines) == 1, filename
189
+ return lines[0]
190
+
191
+
192
+ @bbb.ensure_session
193
+ async def read_explanation_file(explanation_filename: str) -> Optional[NeuronSimulationResults]:
194
+ """Load scored explanations from the given filename, asynchronously."""
195
+ line = await read_file(explanation_filename)
196
+ return loads(line) if line is not None else None
197
+
198
+
199
+ @bbb.ensure_session
200
+ async def read_json_file(filename: str) -> Optional[dict]:
201
+ """Read the contents of the given file as a JSON object, asynchronously."""
202
+ line = await read_file(filename)
203
+ return json.loads(line) if line is not None else None
204
+
205
+
206
+ def get_numerical_subdirs(dataset_path: str) -> list[str]:
207
+ """Return the names of all numbered subdirectories in the specified directory.
208
+
209
+ Used to get all layer directories in an explanation directory.
210
+ """
211
+ return [
212
+ str(x)
213
+ for x in sorted(
214
+ [
215
+ int(x)
216
+ for x in bf.listdir(dataset_path)
217
+ if bf.isdir(bf.join(dataset_path, x)) and x.isnumeric()
218
+ ]
219
+ )
220
+ ]
221
+
222
+
223
+ def get_sorted_neuron_indices_from_explanations(
224
+ explanations_path: str, layer: Union[str, int]
225
+ ) -> list[int]:
226
+ """Return the indices of all neurons in this layer, in ascending order."""
227
+ layer_dir = bf.join(explanations_path, str(layer))
228
+ return sorted(
229
+ [int(f.split(".")[0]) for f in bf.listdir(layer_dir) if f.split(".")[0].isnumeric()]
230
+ )