logogram 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. logogram/__init__.py +6 -0
  2. logogram/__main__.py +5 -0
  3. logogram/analysis.py +419 -0
  4. logogram/atp.py +120 -0
  5. logogram/backends/__init__.py +5 -0
  6. logogram/backends/base.py +202 -0
  7. logogram/backends/hub.py +375 -0
  8. logogram/backends/saes.py +277 -0
  9. logogram/backends/transformer_lens.py +872 -0
  10. logogram/cli.py +496 -0
  11. logogram/compare.py +177 -0
  12. logogram/datasets.py +159 -0
  13. logogram/direct.py +193 -0
  14. logogram/engine.py +550 -0
  15. logogram/examples/ioi-gpt2/.gitignore +3 -0
  16. logogram/examples/ioi-gpt2/datasets/ioi.jsonl +32 -0
  17. logogram/examples/ioi-gpt2/experiments/ioi-head-patching/spec.json +42 -0
  18. logogram/examples/ioi-gpt2/project.json +6 -0
  19. logogram/exports.py +33 -0
  20. logogram/features.py +368 -0
  21. logogram/fileio.py +63 -0
  22. logogram/ioi.py +220 -0
  23. logogram/paths.py +204 -0
  24. logogram/project.py +444 -0
  25. logogram/prompts.py +204 -0
  26. logogram/research.py +84 -0
  27. logogram/results.py +240 -0
  28. logogram/runner.py +396 -0
  29. logogram/runs.py +98 -0
  30. logogram/sae.py +161 -0
  31. logogram/schema.py +302 -0
  32. logogram/server/__init__.py +1 -0
  33. logogram/server/app.py +1083 -0
  34. logogram/server/models.py +426 -0
  35. logogram/server/security.py +212 -0
  36. logogram/server/state.py +585 -0
  37. logogram/sites.py +249 -0
  38. logogram/spec.py +518 -0
  39. logogram/stats.py +171 -0
  40. logogram/steering.py +258 -0
  41. logogram/system.py +379 -0
  42. logogram/updates.py +194 -0
  43. logogram/verify.py +39 -0
  44. logogram/web_dist/assets/index-BvCU-2uy.js +54 -0
  45. logogram/web_dist/assets/index-DTr8_ucV.css +1 -0
  46. logogram/web_dist/assets/instrument-sans-latin-ext-standard-normal-C5E2Gvlv.woff2 +0 -0
  47. logogram/web_dist/assets/instrument-sans-latin-standard-normal-BVScPF0l.woff2 +0 -0
  48. logogram/web_dist/favicon.svg +1 -0
  49. logogram/web_dist/index.html +15 -0
  50. logogram-0.1.0.dist-info/METADATA +550 -0
  51. logogram-0.1.0.dist-info/RECORD +54 -0
  52. logogram-0.1.0.dist-info/WHEEL +4 -0
  53. logogram-0.1.0.dist-info/entry_points.txt +2 -0
  54. logogram-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,550 @@
1
+ Metadata-Version: 2.5
2
+ Name: logogram
3
+ Version: 0.1.0
4
+ Summary: A local-first workbench for causal experiments inside language models.
5
+ Author: The Logogram contributors
6
+ License-Expression: MIT
7
+ License-File: LICENSE
8
+ Keywords: activation-patching,interpretability,mechanistic-interpretability,transformers
9
+ Classifier: Development Status :: 4 - Beta
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Requires-Python: >=3.11
18
+ Requires-Dist: anyio>=4
19
+ Requires-Dist: fastapi>=0.115
20
+ Requires-Dist: huggingface-hub>=0.23
21
+ Requires-Dist: numpy>=1.26
22
+ Requires-Dist: platformdirs>=4
23
+ Requires-Dist: psutil>=5.9
24
+ Requires-Dist: pyarrow>=15
25
+ Requires-Dist: pydantic>=2.7
26
+ Requires-Dist: torch>=2.4
27
+ Requires-Dist: transformer-lens<5,>=4.0
28
+ Requires-Dist: transformers>=4.45
29
+ Requires-Dist: typer>=0.12
30
+ Requires-Dist: uvicorn>=0.30
31
+ Requires-Dist: websockets>=12
32
+ Description-Content-Type: text/markdown
33
+
34
+ # Logogram
35
+
36
+ A local workbench for causal experiments inside language models.
37
+
38
+ In *Arrival*, an unfamiliar language is decoded by running small, controlled experiments and
39
+ watching what changes. Activation patching works the same way: replace one activation with its
40
+ value from a different input, and measure what happens to the output. Logogram makes those
41
+ experiments visual, fast and rigorous. Load GPT-2, set up a clean/corrupted prompt pair, sweep
42
+ every attention head, watch the results paint onto a map of the model, then click a head and see
43
+ the evidence behind its number.
44
+
45
+ * **A research instrument, not a demo.** Every number is traceable to the method that produced
46
+ it: the intervention, baseline, direction, position, metric, normalization, n and confidence
47
+ interval are always shown.
48
+ * **Local-first.** No account, no cloud, no telemetry. Everything runs on your machine.
49
+ * **Plain files.** Experiments are folders you can commit, share and rerun.
50
+ * **One spec, two front ends.** The app and `logogram run` execute the same `spec.json`, and
51
+ produce identical results on the same machine.
52
+
53
+ ## Install
54
+
55
+ Logogram needs Python 3.11 or newer and [uv](https://docs.astral.sh/uv/).
56
+
57
+ ```bash
58
+ uv tool install logogram
59
+ ```
60
+
61
+ Until Logogram is published on PyPI, install it from a clone of this repository with
62
+ `uv tool install .` instead. The web app is prebuilt inside the package, so you never need Node.
63
+ The commands in the table below work the same way with `.` in place of `logogram`.
64
+
65
+ PyTorch is chosen per machine:
66
+
67
+ | Machine | Install | Notes |
68
+ |---|---|---|
69
+ | Linux with an NVIDIA GPU | `uv tool install logogram` | The default Linux wheels include CUDA. |
70
+ | Windows with an NVIDIA GPU | `uv tool install --torch-backend=auto logogram` | Picks the CUDA build that matches your driver. |
71
+ | Apple Silicon | `uv tool install logogram` | Uses Metal (MPS). Use a native arm64 Python, not one under Rosetta. |
72
+ | CPU only | `uv tool install --torch-backend=cpu logogram` | Smaller download. GPT-2 small runs well on a CPU. |
73
+
74
+ Then check the setup:
75
+
76
+ ```bash
77
+ logogram doctor
78
+ ```
79
+
80
+ It reports the operating system, CPU, memory, GPU, video memory, compute backend and recommended
81
+ precision, and gives the exact command to fix common problems, such as an NVIDIA GPU with a
82
+ CPU-only PyTorch build.
83
+
84
+ ## Quickstart
85
+
86
+ ```bash
87
+ logogram
88
+ ```
89
+
90
+ This starts a local server on 127.0.0.1, prints its address and opens your browser.
91
+
92
+ 1. The first time, a system check shows what Logogram found. Choose **Open the example project**.
93
+ Logogram copies it to `Documents/Logogram/ioi-example`, an ordinary project folder.
94
+ 2. The example contains 32 indirect object identification (IOI) prompts and a ready-made
95
+ experiment: patch each head's output from the clean prompt into the corrupt prompt. Press
96
+ **Run**. GPT-2 small (about 0.5 GB) downloads once from Hugging Face; the sweep then takes
97
+ about a minute on a CPU, a few seconds on a GPU.
98
+ 3. Results paint onto the model map layer by layer. Click the strongest heads (L9 H9, L10 H7,
99
+ L8 H6 …) to see, in the inspector, the exact intervention, the metric, the confidence
100
+ interval, how many prompts flipped, the distribution of per-prompt effects and the prompts
101
+ with the strongest and weakest effects.
102
+
103
+ Each run also gets a **logogram**, a circular ink glyph written by its results: layers run
104
+ clockwise from the top, and the ink swells outward (cobalt) where a layer's strongest effect
105
+ is positive and inward (ochre) where it is negative. A run with a few strong layers reads as
106
+ a heavy, lopsided ring; a null result as a thin, even one. Logograms appear in the history,
107
+ on the results page, and beside the model atlas, where each layer number opens that layer's
108
+ strongest component.
109
+ 4. Press **A** on a head to see its attention pattern, or right-click any cell for **Patch here**,
110
+ **Ablate here** and **Compare across runs**.
111
+ 5. Press **Check robustness** to rerun the sweep with a different baseline, direction or donor
112
+ count. Logogram reports the rank correlation, the overlap of the top components and the
113
+ components whose conclusion changed, and flags them on the map.
114
+
115
+ Keyboard: arrow keys move across the map, **Ctrl/⌘ K** opens the command palette (type `L9H9` to
116
+ jump to a head), **1–7** switch views, **[** and **]** step through prompts, **P**, **B**, **A**
117
+ and **C** patch, ablate, show attention or compare the selected component, and **Ctrl/⌘ Enter**
118
+ runs the experiment form.
119
+
120
+ Use **History** to reopen experiments or compare two runs. On a narrow window, **Inspector**
121
+ opens a panel that closes with Escape.
122
+ Attention heatmaps support arrow keys, Home and End, with a spoken value readout and an optional
123
+ values table. Baseline and attention show whether they use the selected run or the experiment
124
+ form, including BOS, prompt limit and batch size. Inspecting a saved run requires its model,
125
+ revision, dtype, device and weight processing; **Load a model → Use saved experiment model
126
+ settings** fills those choices in.
127
+
128
+ Finished results offer a per-prompt CSV, a PNG heatmap with a symmetric legend, and a copyable or
129
+ downloadable methods description. The CSV retains every site/prompt row; text fields that could
130
+ be interpreted as spreadsheet formulas are prefixed with an apostrophe.
131
+
132
+ ## Explore, experiment, evidence
133
+
134
+ The workbench has three workspaces:
135
+
136
+ * **Explore** opens the model atlas, a layer cutaway, attention patterns, head comparison,
137
+ and layer predictions. The atlas uses the model's real dimensions. Double-click a component
138
+ to open its layer, or choose **Layer explorer** and use the layer rail. The cutaway exposes
139
+ residual inputs and outputs, individual head outputs, the combined attention output, and the
140
+ MLP, connected the way the model was measured to add them when it loaded (one after the
141
+ other, or both from the same input); GPT-2 also shows its layer norms.
142
+ * **Experiment** contains prompts, baseline checks, the intervention form, and the complete
143
+ spec. Select a token, choose components, and press **Add to experiment**. The staging tray
144
+ keeps exact sites and positions. **Configure experiment** preserves the source run's
145
+ methodological settings for review. Each staged site is a separate intervention in a sweep;
146
+ the tray does not perform a simultaneous circuit intervention.
147
+ * **Evidence** contains the full result heatmap, run comparisons, and research notes. Selecting
148
+ an effect opens its per-prompt evidence and recorded method in the inspector.
149
+
150
+ Pin up to two heads for **Head comparison**. Both attention maps use the same prompt variant,
151
+ analysis context, and 0–1 ink scale. Selecting a query token in either map or the token ribbon
152
+ links both readouts. Attention weights describe where a head reads; causal effect and its
153
+ confidence interval come from a separate intervention and remain labelled as such.
154
+
155
+ **Save selection** keeps named sites, exact positions, model settings, source run, and written
156
+ observations in `research.json`. Notes can be reopened, edited, or deleted from Evidence.
157
+ Revision checks reject conflicting edits from another tab, keeping the unsaved text available
158
+ so it can be saved as a separate note. New run summaries retain the model's dimensions and
159
+ validated block layout, so saved runs can be explored without loading weights.
160
+
161
+ ### Layer predictions
162
+
163
+ **Layer predictions** implements a final-norm logit lens: project each layer's `resid_post` at
164
+ the chosen token through the loaded model's final normalization and vocabulary projection,
165
+ including biases and the logit soft-capping some models apply, then apply softmax. The normalization is recomputed at every
166
+ layer, following the distinction explained in the [TransformerLens lens documentation](https://transformerlensorg.github.io/TransformerLens/content/backward_lens.html).
167
+ Intermediate projections are descriptive diagnostics, not causal measurements or calibrated
168
+ early predictions. No trained lens is fitted. The diagnostic is offered only for models whose
169
+ final normalization and projection reproduce their own output when they load (see Models).
170
+
171
+ Choose clean or corrupt input, a prompt, a last-token or indexed position, and 1–20 top tokens,
172
+ then press **Measure predictions**. BOS handling, dataset limit and hash, model revision,
173
+ weight processing, dtype, and original length-group batch composition follow the active
174
+ analysis context. The trajectory and per-layer table show answer and distractor probabilities,
175
+ logit difference, and top projected tokens. At an earlier position, the answer and distractor
176
+ are still the dataset's reference tokens, not necessarily the expected next token there.
177
+
178
+ Measuring adds these settings to the experiment form. The optional spec field is:
179
+
180
+ ```json
181
+ "predictions": {
182
+ "method": "final_norm_logit_lens",
183
+ "prompt_index": 0,
184
+ "which": "clean",
185
+ "position": {"kind": "last"},
186
+ "top_k": 5
187
+ }
188
+ ```
189
+
190
+ Save and run that spec from the app or CLI to write `predictions.json` beside the intervention
191
+ results. The shared runner calls the same diagnostic implementation as the preview. Saved
192
+ reports can be reopened without model weights. Removing the optional diagnostic in Experiment
193
+ leaves the intervention settings intact.
194
+
195
+ ## Command line
196
+
197
+ ```text
198
+ logogram start the app and open the browser
199
+ logogram open PATH start with a project open
200
+ logogram run SPEC.json run an experiment headlessly and print a summary
201
+ logogram doctor environment and hardware report, with fixes
202
+ logogram check-updates ask PyPI whether a newer Logogram is out
203
+ logogram serve --dev API only, for working on the web app
204
+ logogram --version
205
+ ```
206
+
207
+ The app listens on port 8765, or the next free one; `--port N` picks another (`--port 0` takes
208
+ any free port) and `--no-browser` skips opening the browser.
209
+
210
+ `logogram run` writes a new experiment folder next to the original. When the spec came from a
211
+ finished run, it also checks the new results against the old ones:
212
+
213
+ ```text
214
+ Identical to 20261007-203446-which-heads-restore-the-answer: all 4,608 per-prompt values match exactly.
215
+ ```
216
+
217
+ ## Prompts
218
+
219
+ A dataset is a JSONL file in the project's `datasets/` folder, one prompt pair per line:
220
+
221
+ ```json
222
+ {"clean": "When Mary and John went to the store, John gave a drink to",
223
+ "corrupt": "When Mary and John went to the store, Mary gave a drink to",
224
+ "answer": " Mary", "distractor": " John",
225
+ "positions": {"IO": [5, 9], "S1": [14, 18], "S2": [38, 42], "end": [56, 58]}}
226
+ ```
227
+
228
+ * `clean` and `corrupt` must tokenize to the same length, so positions line up.
229
+ * `answer` and `distractor` must each be a single token (usually with a leading space).
230
+ * `positions` is optional: named character spans in the clean prompt. A label refers to the last
231
+ token that overlaps its span. Named positions let you patch at, say, `S2` across templates of
232
+ different lengths.
233
+
234
+ The app shows both prompts token by token, highlights where they differ, and refuses to run on
235
+ prompts that can't be used, saying which and why. The built-in IOI generator writes ABBA and BABA
236
+ prompts from several templates with generic names, with a chosen size and seed, and two
237
+ corruptions: `flip` (the repeated name becomes the other name) and `abc` (three unrelated names).
238
+
239
+ ## Experiments
240
+
241
+ **Activation patching** copies an activation from one prompt of each pair into the run on the
242
+ other. *Clean → corrupt* runs the corrupt prompt and patches in clean activations: does this
243
+ restore the behavior? *Corrupt → clean* runs the clean prompt and patches in corrupt
244
+ activations: does this break it?
245
+
246
+ **Ablation** runs the clean prompt and replaces the activation with a baseline, which you must
247
+ choose:
248
+
249
+ * *zero*: zeros;
250
+ * *mean*: the mean over a stated reference set (the dataset's clean or corrupt prompts), per
251
+ position when every position is replaced;
252
+ * *resample*: the same activation in other prompts. Each prompt draws `donors` donors from the
253
+ pool without replacement, never itself, with a seed; the result is averaged over donors, and
254
+ the same donors are used at every site.
255
+
256
+ **Attribution patching** estimates activation patching at every site at once. The change patching
257
+ would cause is estimated to first order, as (source activation − receiver activation) · the
258
+ gradient of the logit difference at the receiver run: one forward pass on the source prompts and
259
+ one forward and backward pass on the receivers cover a whole sweep, so every head at every
260
+ position takes seconds even on larger models. Values are labelled as estimates. A first-order
261
+ estimate misses saturation (in attention, normalization and the final softmax) and can miss or
262
+ even invert an effect, so **Verify top 10 by patching** on the results page patches the sites
263
+ with the largest estimated effects for real and opens the comparison: the rank correlation, and
264
+ any estimate whose sign patching confidently reverses. **Check robustness** can also patch the
265
+ whole sweep.
266
+
267
+ **Direct logit attribution** splits the logit difference of the clean or the corrupt prompts
268
+ (your choice; there is no default) into what each head, attention output and MLP output writes
269
+ directly into the residual stream at the last token. The final normalization is read with its
270
+ scale held at its value in the run, so the logit difference is exactly the sum of one term per
271
+ component plus the embeddings and biases; the results page shows that split. A component's direct
272
+ effect leaves out what it does through later components, which patching measures: **Check
273
+ robustness** offers the same components patched, so direct and total effects can be compared
274
+ side by side. Nothing is replaced, so there are no sign flips or patched probabilities.
275
+ Logogram refuses it where it isn't defined: for residual stream states (no component writes
276
+ them), at positions other than the last token, for single heads in models that normalize the
277
+ attention output after combining heads (Gemma 2, OLMo 2), and for models that soft-cap their
278
+ logits.
279
+
280
+ **Path patching** measures a component's effect through chosen receivers only: a later head's
281
+ query, key or value, or the logits read directly from the final residual stream. For each sender
282
+ (a head, attention output or MLP output of the sweep), the receiver prompt runs with the sender's
283
+ activation from the source prompt while every other attention head is held at its own value, so the
284
+ change travels only through the residual stream (and the MLPs, unless you hold them too). What the
285
+ receivers read in that run is recorded and patched into an unchanged receiver run, where the metric
286
+ is read. Senders in or after the last receiver's layer have no path, so a sweep keeps only the
287
+ layers before it. In models that share keys and values across heads (grouped-query attention),
288
+ only queries can be receivers. Tests check that from the last layer to the logits a path is the
289
+ whole effect, exactly as patching measures it. **Check robustness** offers holding or releasing
290
+ the MLPs, the opposite direction, and the senders' whole effect by patching.
291
+
292
+ **Steering** adds a direction to the residual stream and measures what it does. At each steered
293
+ site (one residual stream state per layer, or chosen ones, at one token of each prompt) the
294
+ direction is the mean difference between the two prompts of each pair, from the prompts you steer
295
+ toward the other ones. It is computed on a seeded training split of the pairs and measured only on
296
+ the held-out rest, so no prompt receives its own difference. Each strength multiplies the whole
297
+ mean difference: 1 adds all of it, negative values push the other way. The effect is normalized
298
+ like patching, so 1 means the steered prompts moved as far as switching to the other prompt. A
299
+ random direction of the same length, at the same strengths, runs alongside as a control, and the
300
+ results show both: a layer × strength map with the control columns muted, and, for a selected site,
301
+ its effect at every strength with intervals. **Check robustness** offers another split of the pairs
302
+ or the other direction.
303
+
304
+ **SAE features.** A sparse autoencoder (SAE) rewrites one of the model's activations as a few active
305
+ features out of thousands, each a direction in the model, plus an error it misses. In
306
+ **Explore → Features**, load a published SAE from Hugging Face (Logogram suggests ones made for the
307
+ suggested models, or give any repository), then **Measure the fit** on your prompts: the fraction
308
+ of the activations' variance the SAE explains, how many features fire per token, and the logit
309
+ difference with the SAE's reconstruction spliced in. SAEs only fit activations like those they were
310
+ trained on, so a poor fit (for example with TransformerLens's weight processing switched the other
311
+ way) is shown before anything else. The view then lists the features that fire on each token of a
312
+ prompt and, for one feature, its activation along the prompt and the prompts where it fires most,
313
+ all computed from your own prompts. Feature descriptions live on sites like Neuronpedia, which
314
+ Logogram doesn't contact.
315
+
316
+ Features are sites like any other (`sae_feature`, with a feature index). **Patching a feature**
317
+ encodes the receiver's activation and changes only that feature, to its value in the source prompt
318
+ (or to zero, for zero ablation): the activation moves along the feature's decoder direction, and the
319
+ SAE's error is kept. **Attribution patching** can sweep **every SAE feature** at once and keep the
320
+ strongest, then verify them by patching; the results also show how much of the whole site's
321
+ estimated effect the features account for (the rest is the SAE's error). Logogram reads the
322
+ SAELens and EleutherAI formats from safetensors files only; Gemma Scope's NumPy archives are
323
+ refused, as are SAEs whose input scaling or architecture it can't reproduce exactly.
324
+
325
+ **Sites** are abstract: `resid_pre`, `resid_mid`, `resid_post`, `attn_out`, `mlp_out` and `head`
326
+ (an attention head's output `z`), each at a layer, head and position. Positions are `all`,
327
+ `last`, a token `index` or a named `label`. **Sweeps** cover every head at one position
328
+ (layer × head), one stream site at every layer and position (layer × position), or attention
329
+ and MLP outputs per layer.
330
+
331
+ **The metric** is the logit difference at the last token, answer minus distractor. The
332
+ **normalized effect** of each prompt is (patched − receiver) ÷ gap, where the gap is
333
+ (source − receiver) for patching and (corrupt − clean) for ablation. By default the gap is the
334
+ dataset's mean, so the mean effect is the usual normalized metric; you can choose each prompt's
335
+ own gap instead. 0 is no change; 1 is a change as large as switching to the other prompt.
336
+
337
+ **Statistics** for every site: n, mean, standard deviation, a percentile bootstrap confidence
338
+ interval over prompts (seeded; the same resamples for every site), the number of prompts whose
339
+ logit difference changed sign, and the number whose effect has the opposite sign to the mean.
340
+ Per-prompt values are always kept.
341
+
342
+ ## The spec
343
+
344
+ Every experiment is a `spec.json`. Nothing that can change a number is left implicit.
345
+
346
+ ```json
347
+ {
348
+ "logogram_spec": 1,
349
+ "name": "Which heads restore the answer?",
350
+ "notes": "",
351
+ "model": {
352
+ "id": "openai-community/gpt2",
353
+ "revision": "607a30d783dfa663caf39e06633721c8d4cfcd7e",
354
+ "dtype": "float32",
355
+ "device": "auto",
356
+ "process_weights": true
357
+ },
358
+ "dataset": {"path": "datasets/ioi.jsonl", "sha256": "2bd046d5…", "limit": null},
359
+ "tokenization": {"prepend_bos": true},
360
+ "experiment": {"kind": "activation_patching", "direction": "clean_to_corrupt"},
361
+ "scope": {"kind": "heads", "position": {"kind": "all"}},
362
+ "metric": {"kind": "logit_diff", "normalization": "dataset_gap"},
363
+ "statistics": {"bootstrap": 1000, "ci": 0.95, "seed": 0},
364
+ "execution": {"batch_size": 64}
365
+ }
366
+ ```
367
+
368
+ | Field | Values |
369
+ |---|---|
370
+ | `model.revision` | A commit. `null` resolves the current main branch at run time; the saved spec pins what ran. |
371
+ | `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change; zero ablation of head outputs does, because value biases are folded. |
372
+ | `dataset.sha256` | If set, the run refuses a dataset file that has changed. |
373
+ | `experiment` | `{"kind": "activation_patching", "direction": "clean_to_corrupt" \| "corrupt_to_clean"}`, `{"kind": "ablation", "baseline": …}` with `{"kind": "zero"}`, `{"kind": "mean", "reference": "clean" \| "corrupt"}` or `{"kind": "resample", "pool": "clean" \| "corrupt", "donors": 10, "seed": 0}`, `{"kind": "attribution_patching", "direction": …}` (same directions as patching), `{"kind": "direct_logit_attribution", "prompts": "clean" \| "corrupt"}`, `{"kind": "path_patching", "direction": …, "receivers": [{"kind": "head", "layer": 9, "head": 9, "input": "q" \| "k" \| "v"}, {"kind": "logits"}], "freeze_mlps": false}`, or `{"kind": "steering", "apply_to": "clean" \| "corrupt", "coefficients": [-1, 1, 2], "train_fraction": 0.5, "seed": 0, "control": true}` with a scope of one residual component per layer (`layer_components`) or residual `sites`, at one token |
374
+ | `scope` | `{"kind": "heads", "position": …}`, `{"kind": "layer_position", "site": "resid_pre", "positions": "each" \| "labels"}`, `{"kind": "layer_components", "components": ["attn_out", "mlp_out"], "position": …}` `{"kind": "sites", "sites": [{"kind": "head", "layer": 9, "head": 9, "position": {"kind": "label", "label": "end"}}]}` (feature sites are `{"kind": "sae_feature", "layer": 8, "feature": 1234, "position": …}`), or, for attribution patching, `{"kind": "features", "position": …, "top": 50}` |
375
+ | `sae` | The SAE that feature sites belong to: `{"repo": "…", "path": "blocks.8.hook_resid_pre", "revision": "…"}`. The revision is pinned when a run starts. |
376
+ | `metric.normalization` | `dataset_gap` or `prompt_gap` |
377
+ | `execution.batch_size` | Recorded because batch shape can change floating-point results in the last digits. |
378
+
379
+ ## Projects
380
+
381
+ ```text
382
+ my-project/
383
+ project.json
384
+ research.json saved selections and research notes (created on first save)
385
+ datasets/*.jsonl
386
+ datasets/snapshots/<sha256>.jsonl immutable input bytes, shared by runs with identical data
387
+ experiments/<id>/spec.json the experiment, with revision and dataset hash pinned
388
+ experiments/<id>/results.parquet one row per site and prompt
389
+ experiments/<id>/summary.json per-site statistics and the layout of the results
390
+ experiments/<id>/predictions.json optional per-layer prediction diagnostic
391
+ experiments/<id>/manifest.json versions, device, dtype, model revision, dataset hash, timing
392
+ .gitignore
393
+ ```
394
+
395
+ `results.parquet` has everything needed to recompute the statistics: site, layer, head,
396
+ position, prompt, the patched logit difference and answer probability, and the receiver and
397
+ reference logit differences. Paths inside a project are relative, so folders can be moved,
398
+ committed and shared.
399
+
400
+ Each new run snapshots its input dataset before loading the model and pins that snapshot in the
401
+ executed spec. Editing the original dataset leaves existing runs reproducible. A modified
402
+ snapshot is refused by its hash; it is never silently overwritten. Keep snapshots alongside
403
+ experiments when sharing a project. Older runs still use the dataset path in their original spec.
404
+
405
+ Browser tabs follow the server's open project. Switching projects clears the other tabs' forms,
406
+ prompts, cached analyses and results; stale requests are rejected. A project cannot be switched
407
+ while a job runs. The status line shows lost connections, and interrupted runs are marked failed
408
+ when their project is reopened, with their saved spec available to rerun.
409
+
410
+ **Reproducibility.** Runs seed Python, NumPy and PyTorch from `statistics.seed`, require deterministic
411
+ PyTorch algorithms (unsupported operations fail rather than merely warn), and disable
412
+ TF32, and record Logogram, Python, PyTorch, TransformerLens and transformers versions, the device
413
+ and GPU model, the dtype, the model id and revision, whether weights were processed, the dataset
414
+ hash and the wall time. The same spec on the same machine gives bit-identical results. Results
415
+ on different devices or library versions can differ; compare their per-prompt values before
416
+ treating them as equivalent.
417
+
418
+ ## Privacy
419
+
420
+ * Everything runs locally. There is no account, no telemetry and no analytics. The app loads no
421
+ fonts or scripts from the internet; its fonts are bundled.
422
+ * Logogram goes online in two cases only. When you load a model, it asks Hugging Face to resolve
423
+ its revision, read its size for the memory estimate and download it once (your own Hugging
424
+ Face login is used for gated models; Hugging Face telemetry is turned off); loading an SAE
425
+ downloads it the same way. And if you allow
426
+ it, it asks pypi.org for the newest Logogram version number once a day (see Updating). Nothing
427
+ about you, your machine or your work is sent.
428
+ * The server listens on 127.0.0.1 only. Every request needs the random session token from the
429
+ address printed in the terminal (exchanged for a same-site cookie when the page opens). The
430
+ browser is opened with a single-use code instead, because other users of a shared machine can
431
+ read a program's command line. The server checks the Host and Origin headers, sends no CORS
432
+ headers, and refuses cross-site requests.
433
+ * Projects are often shared, so Logogram treats their contents with care: it writes only into
434
+ folders that are really inside the project (never through a symlink that leads out of it), and
435
+ a damaged or unexpected file shows up as an error on that file rather than breaking the project.
436
+ * Weights are loaded only from safetensors files, which cannot run code. Logogram never unpickles
437
+ files.
438
+ * Settings and the list of recent projects are kept in your user configuration folder, not in
439
+ projects.
440
+
441
+ ## Updating
442
+
443
+ Logogram tells you when a new version is out, in the way you choose:
444
+
445
+ * **Never online by itself.** Each version knows its release date. A few months after it, the
446
+ app and the terminal suggest checking for a newer one.
447
+ * **Check now.** Press it in the update panel (or the System check), or run
448
+ `logogram check-updates`. Logogram asks pypi.org for the newest version number.
449
+ * **Daily, if you allow it.** The System check asks once whether Logogram may look once a day.
450
+ You can change your mind in the update panel at any time.
451
+
452
+ When a newer version is known, a small **Logogram 0.x.y** button appears in the top bar with what's
453
+ new and the exact update command for how you installed it, and starting `logogram` prints it in
454
+ the terminal. Updating never touches your projects, results or settings:
455
+
456
+ | Installed with | Update with |
457
+ |---|---|
458
+ | `uv tool install logogram` | `uv tool upgrade logogram` |
459
+ | `uv tool install .` from a copy of this repository | `git pull`, then `uv tool install --reinstall .` |
460
+ | `uv sync` (development) | `git pull`, then `uv sync` |
461
+
462
+ A check is one HTTPS request for Logogram's public package information on pypi.org. It carries
463
+ Logogram's version number and nothing else.
464
+
465
+ ## Models
466
+
467
+ TransformerLens 4 loads well over a hundred architectures, and Logogram works with the
468
+ decoder-only language models among them: GPT-2, Pythia, Llama, Mistral, SmolLM, Qwen, Gemma,
469
+ OLMo, Phi and more. The model dialog lists starting points, and any other Hugging Face id can be
470
+ typed in. Before downloading, Logogram checks that TransformerLens supports the architecture and
471
+ estimates the memory needed (weights, activations and a margin) against what is free.
472
+
473
+ Rather than trusting a list, Logogram checks every model when it loads, on a short fixed input:
474
+
475
+ * TransformerLens's version of the model must predict what the original model predicts. If weight
476
+ processing changes the predictions, loading with processed weights is refused.
477
+ * It measures how each layer adds attention and the MLP to the residual stream: one after the
478
+ other (sequential), or both from the same input (parallel, as in Pythia, GPT-J and Phi). Only
479
+ sequential layers have a residual stream between attention and MLP (`resid_mid`), and the layer
480
+ explorer draws what was measured.
481
+ * Layer predictions are offered when the final normalization and unembedding, with any logit
482
+ soft-capping, reproduce the model's output.
483
+
484
+ GPT-2 small is the model the bundled example was made for, and the one run end to end with real
485
+ weights. The test suite runs the sanity checks on tiny random models of the Llama, Pythia, Qwen 2,
486
+ Gemma 2 and OLMo 2 families without downloading anything. Pythia publishes checkpoints from
487
+ throughout training as revisions (`step1000` to `step143000`), so an experiment can be rerun at
488
+ several points of training.
489
+
490
+ Some tokenizers, such as Qwen's, have no beginning-of-sequence token. Logogram then runs prompts
491
+ without one, and the spec records it. Answers and distractors must still be single tokens.
492
+
493
+ ## Development
494
+
495
+ ```bash
496
+ uv sync # Python environment with dev tools
497
+ uv run pytest # tests use a tiny random model and never use the network
498
+ uv run ruff check src tests scripts
499
+ python3 scripts/check_privacy.py
500
+ ```
501
+
502
+ The web app lives in `web/` (React, TypeScript, Vite). For live reloading, run the API and the
503
+ dev server side by side:
504
+
505
+ ```bash
506
+ uv run logogram serve --dev
507
+ ```
508
+
509
+ ```bash
510
+ cd web && npm install && npm run dev
511
+ ```
512
+
513
+ Open the development address printed by `logogram serve --dev`. Before committing changes to the
514
+ web app, build it into the package with `npm run build`; CI checks that the committed build
515
+ matches the sources.
516
+
517
+ Web regression checks:
518
+
519
+ ```bash
520
+ cd web
521
+ npm test # context, cache and keyboard navigation checks
522
+ npx playwright install chromium
523
+ npm run test:browser # temporary local project and tiny model; no Hub access
524
+ ```
525
+
526
+ The browser fixture never opens the user's projects or cached models. CI runs these workflows
527
+ on Linux, and the Python suite plus an installed-wheel smoke check on Linux, macOS and Windows.
528
+ The wheel check covers the CLI, bundled example and fonts, authenticated API and served web
529
+ assets. Source distributions include the web sources and lockfiles so the bundle can be rebuilt.
530
+
531
+ Releasing: raise `__version__` and set `__released__` to the release date in
532
+ `src/logogram/__init__.py` (the date drives the "this version is getting old" hint), commit, and
533
+ push a tag with the same version, such as `v0.1.0`. The Release workflow runs every CI check on
534
+ that commit, builds the wheel and source distribution, and uploads them to PyPI through Trusted
535
+ Publishing, so no token is stored anywhere. PyPI never accepts the same version twice.
536
+
537
+ Before tagging, manually check a first GPT-2 download, cancel and retry it, then run the example
538
+ on each supported compute backend. CPU and CUDA are covered by local development checks; MPS
539
+ still needs a check on Apple Silicon. Other model families are checked when they load and in the
540
+ tiny-model tests, but not yet run end to end with their real weights.
541
+
542
+ Model access goes through `logogram.backends.base.ModelBackend`. TransformerLens is the only
543
+ backend today; remote execution and other libraries can be added behind the same interface.
544
+ Not built yet, with room left for them: attribution graphs with transcoders, Gemma Scope's NumPy
545
+ SAE files, steering with SAE features, remote compute, a native desktop wrapper, plugins and
546
+ assistants.
547
+
548
+ ## License
549
+
550
+ MIT. See [LICENSE](LICENSE).
@@ -0,0 +1,54 @@
1
+ logogram/__init__.py,sha256=BCnqUALarkYy5TAObS3rEKQplA-ZYtQqE8-OOQRxcZA,284
2
+ logogram/__main__.py,sha256=VDqoNcO4KO8OasdCjwdewUn7zbKHJSV3qDsOIEp1BNM,64
3
+ logogram/analysis.py,sha256=kaD1hGr7sWwiZ_TCKRQHPHqy8Zueq4ud1QjA0bvoVwg,16928
4
+ logogram/atp.py,sha256=yn4Qdr2oHZ5c9D-LkHJhd4-6OpKq9uTjlG9kg3aXWjE,4869
5
+ logogram/cli.py,sha256=eFqJFE5QJ4pjvmioxGaOQAwXMpD2PWChvvme4sUyTBg,17978
6
+ logogram/compare.py,sha256=jctO-MUA8MMoM-iu7w5OHDQMMBXWwOyoJgiaTt-Fi18,6222
7
+ logogram/datasets.py,sha256=LAHrHBFJkQP-uSCVMF3HVdRLeauHvdomscWQaYOzU-E,6169
8
+ logogram/direct.py,sha256=-KiZmTPdd8NDO3dLBbrOWWNHH3Dv7EQ0zYhsBwRUQ8o,7952
9
+ logogram/engine.py,sha256=y6LNoWoGUZkfiqxq39ky6MCIMyFP-egiZ46OkmgyIUw,21859
10
+ logogram/exports.py,sha256=wt_5lpZiyGlbax62PAZUB0-E73DEF8OHySHirdtxSeo,1033
11
+ logogram/features.py,sha256=cWDMSBcAVv7ZwMutAKNznRuD-DOxWvSzvycTF5ql22M,15808
12
+ logogram/fileio.py,sha256=NX2OaN1dmmhAssLTakMsbxEpFp5HopRqTgp6fTMVuKM,2046
13
+ logogram/ioi.py,sha256=rBcc3Kxha5rqsWwPpiHQhsQp6fM65NxhWiNCsfOI4kI,8880
14
+ logogram/paths.py,sha256=_wZMPcaPk5M7IaMa1Y1oExdVTJCNIhzqdP0UWdD1jC4,8593
15
+ logogram/project.py,sha256=dbNFbML8JrmvGAQcRxFWn9xCVQQMmBcBsdVr8YxbUFQ,17843
16
+ logogram/prompts.py,sha256=o7pEYe5x0M4CtgOiSxhk9gKy24KlXS3JsTEGbDv2ZLs,6762
17
+ logogram/research.py,sha256=bw6AcGnTVNYKJlnP2a6pHKvqgcaF3jtuxc_Qmow0JZc,2847
18
+ logogram/results.py,sha256=U4p-Wo2PUJXx4B1C_LVOsCh3lffn5biiqF7VAVjMI3k,9377
19
+ logogram/runner.py,sha256=9o2WyuPnZxian-UD82FqOsxaVJcgZvEgGoRREZuFnjs,14596
20
+ logogram/runs.py,sha256=qJqqbNlqRBZfAg9EmPit_N8EViZPMiXsPxWhZloHuZc,3641
21
+ logogram/sae.py,sha256=R-G-mFrMiEx1d3ESzyddOpGH6_JroZSHGVqNuSN9IlA,5570
22
+ logogram/schema.py,sha256=srHH24qT6i_bpfHB-b28P9RlPjOkJCoZbjuZZ39prAU,7564
23
+ logogram/sites.py,sha256=P7qhrrhcAPF7pSwSHYXr1cUXt-youAdIHJDa5Fw-poA,9262
24
+ logogram/spec.py,sha256=ZuknV2GT78dtO-pUwzyDpwWM_fdgeRWn7_qDHr3NSGw,19598
25
+ logogram/stats.py,sha256=7KPjafsPrRAPWKoISuyyvCb9JQycUgNpC8VFyXN58cs,6056
26
+ logogram/steering.py,sha256=aqkcsK7J5H0LSWdscKYie-nZrB3m3dohEP9eEkjN1j4,10504
27
+ logogram/system.py,sha256=BfJOnFU5BfTaHyPCgIN8atgMFtjFY-BsmfEkmEz_nSs,11764
28
+ logogram/updates.py,sha256=JgTz85EkKETDYk_lpqZwgulkfer2auTL7o5dGgAvNq8,7205
29
+ logogram/verify.py,sha256=svglN7_N96l3UN7B9FF5VUSE8uLKFFTrid4D1dvSab8,1453
30
+ logogram/backends/__init__.py,sha256=u3BtyWSpIiyHTDeowAhg-3ubyWG1duEOedZvBQzu5Gc,247
31
+ logogram/backends/base.py,sha256=d4sDG9OR8lYzHcpz7rU6_a6kA6FX2b8OjhJpSoCcPbc,7531
32
+ logogram/backends/hub.py,sha256=06MeO6gozpfJATF4moFy3FIue7TC6oihkwUGHPX60SA,13925
33
+ logogram/backends/saes.py,sha256=mOECxINkoNbBxOx0AXpk57P2IHZtFq2xD6yyI0X6bqo,10987
34
+ logogram/backends/transformer_lens.py,sha256=xM8cyO-S1QXDPu8pxKJoQDAqo_mh_A5EMYE4dWmEadM,37059
35
+ logogram/examples/ioi-gpt2/.gitignore,sha256=QDO-Mk14u5qCKiA9RDtAq9lua9TSaiVA0Y5qtoKvKp8,42
36
+ logogram/examples/ioi-gpt2/project.json,sha256=qLqApjoB0wuZHn7OzKNdTdwL_RWCIqR2DlfMKXpefvw,197
37
+ logogram/examples/ioi-gpt2/datasets/ioi.jsonl,sha256=K9BG1dAZDUKGPfpk4YX2EijdJqWEWPe2z5MUCEdHisM,11758
38
+ logogram/examples/ioi-gpt2/experiments/ioi-head-patching/spec.json,sha256=sgjfpE0TrLc120vPGYS4N9We3B8wIwVu1oiXTJMhXh4,1170
39
+ logogram/server/__init__.py,sha256=f0Uv8RFY_Mjlkv9groRj5jw3LTLP9zjKtYQtCA7kpOY,28
40
+ logogram/server/app.py,sha256=DF4nSvc6_5xY4F6kaXEIaRbuK1hPX2hx8zqJMkPOKug,41356
41
+ logogram/server/models.py,sha256=7w_6NCnuaZVKcMH4LzjKgZLQXTI0kQkAVjPZvp_vX9c,8738
42
+ logogram/server/security.py,sha256=QTbuNDRnwzbEYd687B2P8wf9Ofp50VBv0Kz_Ajj0Kvo,8547
43
+ logogram/server/state.py,sha256=sxaSbCE0GI0yEWd27Q-J05630iBQuo2wOcBL_v3G88E,23715
44
+ logogram/web_dist/favicon.svg,sha256=3d5g0Z4krR6mq6pqVVMJEnuFP5e0argaYjXuiYStXLM,5746
45
+ logogram/web_dist/index.html,sha256=FgM_ptsvSkGBhrb0-iys4AOrulKQhPxRQyuPkmEy9ks,531
46
+ logogram/web_dist/assets/index-BvCU-2uy.js,sha256=yW0lNrfSBW11YAkxLdDyJkUflK1bft20UnpNmckKVHM,636059
47
+ logogram/web_dist/assets/index-DTr8_ucV.css,sha256=oLOIH4hSDloJ1jj8_kFjDLdxsDwQrWKN34sqnyT7m-I,77837
48
+ logogram/web_dist/assets/instrument-sans-latin-ext-standard-normal-C5E2Gvlv.woff2,sha256=8h_GSG2B1y0vjDiaxBvvzPGZucAjZ-qDdzaDIcrT53k,18856
49
+ logogram/web_dist/assets/instrument-sans-latin-standard-normal-BVScPF0l.woff2,sha256=i3TpvG5BXVZKX_0yElyS_9ULFjUEPndUPLhx4KPPyvI,57332
50
+ logogram-0.1.0.dist-info/METADATA,sha256=YXlK6QWrbTSXk7GZahaqi1BGS4BcWsxHt5RmbSIfbK0,32589
51
+ logogram-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
52
+ logogram-0.1.0.dist-info/entry_points.txt,sha256=eOec68-4Een5YDo6L34HHPIvIUmsQY6JF5lLXNDJ8ik,47
53
+ logogram-0.1.0.dist-info/licenses/LICENSE,sha256=g47OsI0PdoDX-rfZ5RIs9_eykfq8hg4dgDcu36x7TY8,1082
54
+ logogram-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ logogram = logogram.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 The Logogram contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.