logogram 0.1.0__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- logogram-0.1.2/CHANGELOG.md +94 -0
- logogram-0.1.0/README.md → logogram-0.1.2/PKG-INFO +94 -20
- logogram-0.1.0/PKG-INFO → logogram-0.1.2/README.md +56 -53
- {logogram-0.1.0 → logogram-0.1.2}/pyproject.toml +9 -2
- logogram-0.1.2/scripts/validate_real_weights.py +459 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/__init__.py +1 -1
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/analysis.py +5 -5
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/atp.py +13 -3
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/base.py +15 -2
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/hub.py +54 -7
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/saes.py +10 -10
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/transformer_lens.py +135 -31
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/cli.py +11 -14
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/direct.py +16 -6
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/engine.py +18 -3
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/features.py +8 -8
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/results.py +26 -1
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/app.py +28 -7
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/models.py +1 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/steering.py +38 -6
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/system.py +44 -11
- logogram-0.1.0/src/logogram/web_dist/assets/index-BvCU-2uy.js → logogram-0.1.2/src/logogram/web_dist/assets/index-Cw0WBZBB.js +6 -6
- logogram-0.1.2/src/logogram/web_dist/assets/index-Dt6ecB66.css +1 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/index.html +2 -2
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_atp.py +20 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_cli.py +33 -0
- logogram-0.1.2/tests/test_devices.py +64 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_direct.py +38 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_hardening.py +39 -0
- logogram-0.1.2/tests/test_loading.py +235 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_steering.py +26 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_units.py +12 -0
- logogram-0.1.2/tests/test_validation_script.py +45 -0
- {logogram-0.1.0 → logogram-0.1.2}/uv.lock +50 -2
- {logogram-0.1.0 → logogram-0.1.2}/web/src/api/client.ts +4 -6
- {logogram-0.1.0 → logogram-0.1.2}/web/src/api/types.ts +6 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelDialog.module.css +6 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelDialog.tsx +24 -3
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/RobustnessDialog.tsx +15 -3
- {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/SystemCheck.tsx +1 -1
- {logogram-0.1.0 → logogram-0.1.2}/web/src/store/app.ts +4 -4
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/BaselineView.tsx +13 -3
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/PromptsView.tsx +30 -0
- logogram-0.1.0/src/logogram/web_dist/assets/index-DTr8_ucV.css +0 -1
- {logogram-0.1.0 → logogram-0.1.2}/.gitignore +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/AGENTS.md +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/LICENSE +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/scripts/check_privacy.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/scripts/smoke_wheel.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/__main__.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/__init__.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/compare.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/datasets.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/.gitignore +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/datasets/ioi.jsonl +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/experiments/ioi-head-patching/spec.json +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/project.json +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/exports.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/fileio.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/ioi.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/paths.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/project.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/prompts.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/research.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/runner.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/runs.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/sae.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/schema.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/__init__.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/security.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/state.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/sites.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/spec.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/stats.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/updates.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/verify.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/assets/instrument-sans-latin-ext-standard-normal-C5E2Gvlv.woff2 +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/assets/instrument-sans-latin-standard-normal-BVScPF0l.woff2 +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/favicon.svg +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/browser_server.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/conftest.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_architectures.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_engine.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_explorer.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_paths.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_privacy.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_release_fixes.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_sae.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_sanity.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_server.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/tests/test_updates.py +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/index.html +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/package-lock.json +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/package.json +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/playwright.config.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/public/favicon.svg +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/App.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/App.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/api/events.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CommandPalette.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CommandPalette.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CopyCommand.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CopyCommand.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Distribution.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Distribution.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/FolderPicker.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/FolderPicker.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Header.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Header.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Heatmap/Heatmap.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Heatmap/Heatmap.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Heatmap/ScaleBar.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/History.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/History.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Inspector.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Inspector.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Logogram.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/LogogramDial.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/LogogramDial.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelMap/Legend.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelMap/ModelMap.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelMap/ModelMap.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Notices.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Notices.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Splitter.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Splitter.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/StatusLine.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/StatusLine.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TokenStrip.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TokenStrip.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TopBar.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TopBar.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/UpdateNotice.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/UpdateNotice.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ui/Icon.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ui/index.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ui/ui.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/analysis.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/canvas.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/color.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/export.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/format.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/heatmapNavigation.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/hooks.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/keys.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/logogram.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/sites.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/spec.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/main.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Projects.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Screens.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Welcome.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Workbench.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Workbench.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/styles/global.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/styles/tokens.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/AttentionView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/AttentionView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/BaselineView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/CompareView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/CompareView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExperimentView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExperimentView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExploreView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExploreView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/FeaturesView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/FeaturesView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/Forest.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/Forest.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/HeadComparisonView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/HeadComparisonView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/PredictionsView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/PredictionsView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResearchView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResearchView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResultsView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResultsView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/SpecView.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/SpecView.tsx +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/src/views/views.module.css +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/tests/context.unit.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/tests/logogram.unit.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/tests/workbench.browser.ts +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/tsconfig.json +0 -0
- {logogram-0.1.0 → logogram-0.1.2}/web/vite.config.ts +0 -0
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
What changed in each version of Logogram.
|
|
4
|
+
|
|
5
|
+
## 0.1.2 (2026-10-08)
|
|
6
|
+
|
|
7
|
+
Fixes and checks that came out of validating every method on real weights.
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- On Apple Silicon, **Automatic** runs models on the CPU. TransformerLens reports that Apple's MPS
|
|
12
|
+
can give silently wrong results, and Logogram hasn't been checked on it yet, so MPS runs only when
|
|
13
|
+
you choose it; the system check and the model dialog say why.
|
|
14
|
+
|
|
15
|
+
### Added
|
|
16
|
+
|
|
17
|
+
- **Check robustness** reruns a float16 or bfloat16 run in float32 and compares the two.
|
|
18
|
+
- When processing a model's weights is what doesn't fit, the model dialog offers to turn it off.
|
|
19
|
+
- When a generated IOI dataset has names the loaded model splits into several tokens, one click
|
|
20
|
+
makes a new one with names it doesn't.
|
|
21
|
+
- Steering results say when the direction does no more than its random control at any site and
|
|
22
|
+
strength.
|
|
23
|
+
- Attribution patching of residual stream sites warns that it can miss effects there, and that
|
|
24
|
+
**Check robustness** patches every site.
|
|
25
|
+
- `scripts/validate_real_weights.py` runs every method on a real model and checks identities,
|
|
26
|
+
agreement and bit-identical reruns, on one device or two. It is how a release, or Apple's MPS, is
|
|
27
|
+
checked.
|
|
28
|
+
|
|
29
|
+
### Fixed
|
|
30
|
+
|
|
31
|
+
- The baseline says that the clean prompts prefer the distractor, instead of preferring the answer
|
|
32
|
+
by a negative amount.
|
|
33
|
+
|
|
34
|
+
### Development
|
|
35
|
+
|
|
36
|
+
- Tests run on Python 3.14 too, and the Linux runners are pinned to Ubuntu 24.04.
|
|
37
|
+
- The test client uses httpx2, as Starlette recommends.
|
|
38
|
+
|
|
39
|
+
## 0.1.1 (2026-10-08)
|
|
40
|
+
|
|
41
|
+
Every method has now run end to end on real weights: GPT-2 small with a SAELens and an OpenAI
|
|
42
|
+
SAE, Pythia-70m with an EleutherAI SAE, and Qwen 2.5 0.5B. This release fixes what that turned up.
|
|
43
|
+
|
|
44
|
+
### Fixed
|
|
45
|
+
|
|
46
|
+
- Runs work on Apple Silicon. Every method converted tensors to float64 on the model's device,
|
|
47
|
+
which Apple's Metal (MPS) doesn't have, so no run got past its baseline on a Mac. This still
|
|
48
|
+
needs a check on a Mac.
|
|
49
|
+
- A model whose own forward pass overflows in the chosen dtype (Pythia in float16) is refused, with
|
|
50
|
+
the dtype to use. Its load checks used to read as passed.
|
|
51
|
+
- When processing a model's weights doesn't fit in memory, loading says so and frees the memory,
|
|
52
|
+
instead of failing with a raw CUDA error and keeping it.
|
|
53
|
+
- The memory estimate counts weight processing, which needs three to four times the float32 size
|
|
54
|
+
of the weights while the model loads. Qwen 2.5 0.5B on a 6 GB GPU now reads "won't fit" with
|
|
55
|
+
processing and "fits" without it.
|
|
56
|
+
- Models without a beginning-of-sequence token, such as Qwen, run without one, as documented.
|
|
57
|
+
TransformerLens 4 had given them its end-of-text token instead.
|
|
58
|
+
- Models and SAEs that Logogram downloaded open offline without a revision, and offline memory
|
|
59
|
+
estimates use the model's real size.
|
|
60
|
+
- `logogram run` reports an identical rerun of direct logit attribution or attribution patching as
|
|
61
|
+
identical.
|
|
62
|
+
- Direct logit attribution in float16 and bfloat16 no longer warns about drift that is only
|
|
63
|
+
rounding.
|
|
64
|
+
|
|
65
|
+
### Added
|
|
66
|
+
|
|
67
|
+
- Runs warn when the model prefers the distractor on the clean prompts, so it doesn't show the
|
|
68
|
+
behavior they test (Pythia-70m on IOI).
|
|
69
|
+
- The model dialog shows the memory that weight processing needs. Qwen 2.5 0.5B is marked as
|
|
70
|
+
tested.
|
|
71
|
+
- Project links on PyPI, and this changelog, which the update notice links to.
|
|
72
|
+
|
|
73
|
+
### Documentation
|
|
74
|
+
|
|
75
|
+
- What has been validated with real weights, and where attribution patching is least reliable:
|
|
76
|
+
residual stream sites at the token that differs between the prompts.
|
|
77
|
+
- Steering needs prompt pairs that differ the same way; IOI prompts that mix the ABBA and BABA
|
|
78
|
+
orders cancel out.
|
|
79
|
+
- What float16 and bfloat16 do to results, and how weight processing changes heads' direct effects.
|
|
80
|
+
|
|
81
|
+
## 0.1.0 (2026-10-08)
|
|
82
|
+
|
|
83
|
+
First public release.
|
|
84
|
+
|
|
85
|
+
- Activation patching, ablation (zero, mean or resample), direct logit attribution, attribution
|
|
86
|
+
patching verified by patching, steering with a random control, path patching, and SAE features,
|
|
87
|
+
all described by one spec that the app and `logogram run` execute the same way.
|
|
88
|
+
- Every model is checked when it loads: that TransformerLens reproduces its predictions, how its
|
|
89
|
+
layers add into the residual stream, and whether the logit lens reproduces its output.
|
|
90
|
+
- Per-prompt values with seeded bootstrap intervals, and bit-identical reruns on the same machine.
|
|
91
|
+
- The workbench: Explore, Experiment and Evidence, logograms written by results, robustness checks
|
|
92
|
+
and run comparisons.
|
|
93
|
+
- Local-first: no account and no telemetry, safetensors weights only, and an update check that
|
|
94
|
+
runs only when you allow it.
|
|
@@ -1,3 +1,41 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: logogram
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: A local-first workbench for causal experiments inside language models.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Jeevash23/logogram
|
|
6
|
+
Project-URL: Source, https://github.com/Jeevash23/logogram
|
|
7
|
+
Project-URL: Issues, https://github.com/Jeevash23/logogram/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/Jeevash23/logogram/blob/main/CHANGELOG.md
|
|
9
|
+
Author: The Logogram contributors
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: activation-patching,interpretability,mechanistic-interpretability,transformers
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Requires-Python: >=3.11
|
|
23
|
+
Requires-Dist: anyio>=4
|
|
24
|
+
Requires-Dist: fastapi>=0.115
|
|
25
|
+
Requires-Dist: huggingface-hub>=0.23
|
|
26
|
+
Requires-Dist: numpy>=1.26
|
|
27
|
+
Requires-Dist: platformdirs>=4
|
|
28
|
+
Requires-Dist: psutil>=5.9
|
|
29
|
+
Requires-Dist: pyarrow>=15
|
|
30
|
+
Requires-Dist: pydantic>=2.7
|
|
31
|
+
Requires-Dist: torch>=2.4
|
|
32
|
+
Requires-Dist: transformer-lens<5,>=4.0
|
|
33
|
+
Requires-Dist: transformers>=4.45
|
|
34
|
+
Requires-Dist: typer>=0.12
|
|
35
|
+
Requires-Dist: uvicorn>=0.30
|
|
36
|
+
Requires-Dist: websockets>=12
|
|
37
|
+
Description-Content-Type: text/markdown
|
|
38
|
+
|
|
1
39
|
# Logogram
|
|
2
40
|
|
|
3
41
|
A local workbench for causal experiments inside language models.
|
|
@@ -25,9 +63,10 @@ Logogram needs Python 3.11 or newer and [uv](https://docs.astral.sh/uv/).
|
|
|
25
63
|
uv tool install logogram
|
|
26
64
|
```
|
|
27
65
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
66
|
+
To run the code in a clone of this repository instead, install it with `uv tool install .`; the
|
|
67
|
+
commands in the table below work the same way with `.` in place of `logogram`. The web app is
|
|
68
|
+
prebuilt inside the package, so you never need Node. What changed in each version is in
|
|
69
|
+
[CHANGELOG.md](CHANGELOG.md).
|
|
31
70
|
|
|
32
71
|
PyTorch is chosen per machine:
|
|
33
72
|
|
|
@@ -35,7 +74,7 @@ PyTorch is chosen per machine:
|
|
|
35
74
|
|---|---|---|
|
|
36
75
|
| Linux with an NVIDIA GPU | `uv tool install logogram` | The default Linux wheels include CUDA. |
|
|
37
76
|
| Windows with an NVIDIA GPU | `uv tool install --torch-backend=auto logogram` | Picks the CUDA build that matches your driver. |
|
|
38
|
-
| Apple Silicon | `uv tool install logogram` |
|
|
77
|
+
| Apple Silicon | `uv tool install logogram` | Runs on the CPU unless you choose MPS (see Models). Use a native arm64 Python, not one under Rosetta. |
|
|
39
78
|
| CPU only | `uv tool install --torch-backend=cpu logogram` | Smaller download. GPT-2 small runs well on a CPU. |
|
|
40
79
|
|
|
41
80
|
Then check the setup:
|
|
@@ -76,8 +115,9 @@ This starts a local server on 127.0.0.1, prints its address and opens your brows
|
|
|
76
115
|
4. Press **A** on a head to see its attention pattern, or right-click any cell for **Patch here**,
|
|
77
116
|
**Ablate here** and **Compare across runs**.
|
|
78
117
|
5. Press **Check robustness** to rerun the sweep with a different baseline, direction or donor
|
|
79
|
-
count
|
|
80
|
-
|
|
118
|
+
count, or, for a run in float16 or bfloat16, in float32. Logogram reports the rank
|
|
119
|
+
correlation, the overlap of the top components and the components whose conclusion changed,
|
|
120
|
+
and flags them on the map.
|
|
81
121
|
|
|
82
122
|
Keyboard: arrow keys move across the map, **Ctrl/⌘ K** opens the command palette (type `L9H9` to
|
|
83
123
|
jump to a head), **1–7** switch views, **[** and **]** step through prompts, **P**, **B**, **A**
|
|
@@ -229,7 +269,11 @@ estimate misses saturation (in attention, normalization and the final softmax) a
|
|
|
229
269
|
even invert an effect, so **Verify top 10 by patching** on the results page patches the sites
|
|
230
270
|
with the largest estimated effects for real and opens the comparison: the rank correlation, and
|
|
231
271
|
any estimate whose sign patching confidently reverses. **Check robustness** can also patch the
|
|
232
|
-
whole sweep.
|
|
272
|
+
whole sweep. Do that for residual stream sweeps: an estimate is least reliable where patching
|
|
273
|
+
replaces a whole token's representation. In the IOI example on GPT-2 small, patching the residual
|
|
274
|
+
stream at the changed name in the first layer restores the whole answer while its estimate is
|
|
275
|
+
slightly negative, so verifying the top estimates would never reach it. Head outputs track
|
|
276
|
+
patching closely.
|
|
233
277
|
|
|
234
278
|
**Direct logit attribution** splits the logit difference of the clean or the corrupt prompts
|
|
235
279
|
(your choice; there is no default) into what each head, attention output and MLP output writes
|
|
@@ -266,7 +310,10 @@ like patching, so 1 means the steered prompts moved as far as switching to the o
|
|
|
266
310
|
random direction of the same length, at the same strengths, runs alongside as a control, and the
|
|
267
311
|
results show both: a layer × strength map with the control columns muted, and, for a selected site,
|
|
268
312
|
its effect at every strength with intervals. **Check robustness** offers another split of the pairs
|
|
269
|
-
or the other direction.
|
|
313
|
+
or the other direction. A mean difference steers only when the pairs differ the same way. In IOI
|
|
314
|
+
prompts that mix the ABBA and BABA orders, the difference at the last token flips with the order,
|
|
315
|
+
so the mean cancels and steering does no more than the control; generate prompts of one order to
|
|
316
|
+
steer. When no site and strength does more than the control, the results say so.
|
|
270
317
|
|
|
271
318
|
**SAE features.** A sparse autoencoder (SAE) rewrites one of the model's activations as a few active
|
|
272
319
|
features out of thousands, each a direction in the model, plus an error it misses. In
|
|
@@ -335,7 +382,7 @@ Every experiment is a `spec.json`. Nothing that can change a number is left impl
|
|
|
335
382
|
| Field | Values |
|
|
336
383
|
|---|---|
|
|
337
384
|
| `model.revision` | A commit. `null` resolves the current main branch at run time; the saved spec pins what ran. |
|
|
338
|
-
| `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change
|
|
385
|
+
| `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change. Value biases are folded into the attention output's bias, so zero ablation of head outputs and heads' direct effects do change (by up to several logits in Qwen 2.5, whose value biases are large); a layer's attention output doesn't. |
|
|
339
386
|
| `dataset.sha256` | If set, the run refuses a dataset file that has changed. |
|
|
340
387
|
| `experiment` | `{"kind": "activation_patching", "direction": "clean_to_corrupt" \| "corrupt_to_clean"}`, `{"kind": "ablation", "baseline": …}` with `{"kind": "zero"}`, `{"kind": "mean", "reference": "clean" \| "corrupt"}` or `{"kind": "resample", "pool": "clean" \| "corrupt", "donors": 10, "seed": 0}`, `{"kind": "attribution_patching", "direction": …}` (same directions as patching), `{"kind": "direct_logit_attribution", "prompts": "clean" \| "corrupt"}`, `{"kind": "path_patching", "direction": …, "receivers": [{"kind": "head", "layer": 9, "head": 9, "input": "q" \| "k" \| "v"}, {"kind": "logits"}], "freeze_mlps": false}`, or `{"kind": "steering", "apply_to": "clean" \| "corrupt", "coefficients": [-1, 1, 2], "train_fraction": 0.5, "seed": 0, "control": true}` with a scope of one residual component per layer (`layer_components`) or residual `sites`, at one token |
|
|
341
388
|
| `scope` | `{"kind": "heads", "position": …}`, `{"kind": "layer_position", "site": "resid_pre", "positions": "each" \| "labels"}`, `{"kind": "layer_components", "components": ["attn_out", "mlp_out"], "position": …}` `{"kind": "sites", "sites": [{"kind": "head", "layer": 9, "head": 9, "position": {"kind": "label", "label": "end"}}]}` (feature sites are `{"kind": "sae_feature", "layer": 8, "feature": 1234, "position": …}`), or, for attribution patching, `{"kind": "features", "position": …, "top": 50}` |
|
|
@@ -435,12 +482,27 @@ TransformerLens 4 loads well over a hundred architectures, and Logogram works wi
|
|
|
435
482
|
decoder-only language models among them: GPT-2, Pythia, Llama, Mistral, SmolLM, Qwen, Gemma,
|
|
436
483
|
OLMo, Phi and more. The model dialog lists starting points, and any other Hugging Face id can be
|
|
437
484
|
typed in. Before downloading, Logogram checks that TransformerLens supports the architecture and
|
|
438
|
-
estimates the memory needed (weights, activations and a margin) against what is free.
|
|
485
|
+
estimates the memory needed (weights, activations and a margin) against what is free. Processing
|
|
486
|
+
the weights needs more for a moment while the model loads: TransformerLens works on float32
|
|
487
|
+
copies, three to four times the float32 size of the weights. The estimate includes them, so a
|
|
488
|
+
model that fits only without processing (Qwen 2.5 0.5B on a 6 GB GPU) says so before it loads.
|
|
489
|
+
|
|
490
|
+
Use float32 whenever the model fits. float16 and bfloat16 halve the memory and run two to three
|
|
491
|
+
times faster on a GPU, but they round. In the IOI example on GPT-2 small and Qwen 2.5, float16
|
|
492
|
+
moved effects by at most 0.003 and bfloat16 by up to 0.03: the strongest sites stayed in place,
|
|
493
|
+
but effects smaller than about 0.01 changed order. On Pythia-70m, bfloat16 doubled the clean
|
|
494
|
+
logit difference and reordered the heads, and float16 overflows. Check results that matter
|
|
495
|
+
against float32: **Check robustness** reruns a 16-bit run in float32 and compares the two.
|
|
496
|
+
|
|
497
|
+
On Apple Silicon, **Automatic** runs models on the CPU. TransformerLens reports that Apple's MPS
|
|
498
|
+
can give silently wrong results, and Logogram hasn't been checked on it yet, so MPS is used only
|
|
499
|
+
when you choose it under **Device**. It is faster; check results that matter on the CPU.
|
|
439
500
|
|
|
440
501
|
Rather than trusting a list, Logogram checks every model when it loads, on a short fixed input:
|
|
441
502
|
|
|
442
503
|
* TransformerLens's version of the model must predict what the original model predicts. If weight
|
|
443
|
-
processing changes the predictions, loading with processed weights is refused.
|
|
504
|
+
processing changes the predictions, loading with processed weights is refused. So is a dtype in
|
|
505
|
+
which the model overflows (Pythia in float16).
|
|
444
506
|
* It measures how each layer adds attention and the MLP to the residual stream: one after the
|
|
445
507
|
other (sequential), or both from the same input (parallel, as in Pythia, GPT-J and Phi). Only
|
|
446
508
|
sequential layers have a residual stream between attention and MLP (`resid_mid`), and the layer
|
|
@@ -448,11 +510,15 @@ Rather than trusting a list, Logogram checks every model when it loads, on a sho
|
|
|
448
510
|
* Layer predictions are offered when the final normalization and unembedding, with any logit
|
|
449
511
|
soft-capping, reproduce the model's output.
|
|
450
512
|
|
|
451
|
-
GPT-2 small is the model the bundled example was made for
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
513
|
+
GPT-2 small is the model the bundled example was made for. Every method has been run end to end
|
|
514
|
+
with real weights on GPT-2 small (with a SAELens and an OpenAI SAE), Pythia-70m (with an
|
|
515
|
+
EleutherAI SAE) and Qwen 2.5 0.5B, and the test suite runs the sanity checks on tiny random models
|
|
516
|
+
of the Llama, Pythia, Qwen 2, Gemma 2 and OLMo 2 families without downloading anything. The
|
|
517
|
+
example's names are single tokens for GPT-2 and Qwen but not all for Pythia; for another model,
|
|
518
|
+
generate IOI prompts in the app, which keeps only names that are single tokens for it. A small
|
|
519
|
+
model may not do the task at all (Pythia-70m prefers the repeated name), and its runs then say so.
|
|
520
|
+
Pythia publishes checkpoints from throughout training as revisions (`step1000` to `step143000`),
|
|
521
|
+
so an experiment can be rerun at several points of training.
|
|
456
522
|
|
|
457
523
|
Some tokenizers, such as Qwen's, have no beginning-of-sequence token. Logogram then runs prompts
|
|
458
524
|
without one, and the spec records it. Answers and distractors must still be single tokens.
|
|
@@ -501,10 +567,18 @@ push a tag with the same version, such as `v0.1.0`. The Release workflow runs ev
|
|
|
501
567
|
that commit, builds the wheel and source distribution, and uploads them to PyPI through Trusted
|
|
502
568
|
Publishing, so no token is stored anywhere. PyPI never accepts the same version twice.
|
|
503
569
|
|
|
504
|
-
Before tagging, manually check a first GPT-2 download, cancel and retry it, then run
|
|
505
|
-
on each supported compute backend.
|
|
506
|
-
|
|
507
|
-
|
|
570
|
+
Before tagging, manually check a first GPT-2 download, cancel and retry it, then run
|
|
571
|
+
`scripts/validate_real_weights.py` on each supported compute backend. It runs every method on a
|
|
572
|
+
real model in a temporary project and checks exact identities, agreement between methods and
|
|
573
|
+
bit-identical reruns; `--compare-device` reruns the head sweep on a second device, and `--sae`
|
|
574
|
+
adds SAE features. CPU and CUDA pass it. Apple's MPS still needs it run on a Mac:
|
|
575
|
+
|
|
576
|
+
```bash
|
|
577
|
+
uv run python scripts/validate_real_weights.py --device mps --compare-device cpu
|
|
578
|
+
```
|
|
579
|
+
|
|
580
|
+
Pythia and Qwen 2.5 have also been run end to end with their real weights; other families are
|
|
581
|
+
checked when they load and in the tiny-model tests.
|
|
508
582
|
|
|
509
583
|
Model access goes through `logogram.backends.base.ModelBackend`. TransformerLens is the only
|
|
510
584
|
backend today; remote execution and other libraries can be added behind the same interface.
|
|
@@ -1,36 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.5
|
|
2
|
-
Name: logogram
|
|
3
|
-
Version: 0.1.0
|
|
4
|
-
Summary: A local-first workbench for causal experiments inside language models.
|
|
5
|
-
Author: The Logogram contributors
|
|
6
|
-
License-Expression: MIT
|
|
7
|
-
License-File: LICENSE
|
|
8
|
-
Keywords: activation-patching,interpretability,mechanistic-interpretability,transformers
|
|
9
|
-
Classifier: Development Status :: 4 - Beta
|
|
10
|
-
Classifier: Intended Audience :: Science/Research
|
|
11
|
-
Classifier: Operating System :: OS Independent
|
|
12
|
-
Classifier: Programming Language :: Python :: 3
|
|
13
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
-
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
-
Requires-Python: >=3.11
|
|
18
|
-
Requires-Dist: anyio>=4
|
|
19
|
-
Requires-Dist: fastapi>=0.115
|
|
20
|
-
Requires-Dist: huggingface-hub>=0.23
|
|
21
|
-
Requires-Dist: numpy>=1.26
|
|
22
|
-
Requires-Dist: platformdirs>=4
|
|
23
|
-
Requires-Dist: psutil>=5.9
|
|
24
|
-
Requires-Dist: pyarrow>=15
|
|
25
|
-
Requires-Dist: pydantic>=2.7
|
|
26
|
-
Requires-Dist: torch>=2.4
|
|
27
|
-
Requires-Dist: transformer-lens<5,>=4.0
|
|
28
|
-
Requires-Dist: transformers>=4.45
|
|
29
|
-
Requires-Dist: typer>=0.12
|
|
30
|
-
Requires-Dist: uvicorn>=0.30
|
|
31
|
-
Requires-Dist: websockets>=12
|
|
32
|
-
Description-Content-Type: text/markdown
|
|
33
|
-
|
|
34
1
|
# Logogram
|
|
35
2
|
|
|
36
3
|
A local workbench for causal experiments inside language models.
|
|
@@ -58,9 +25,10 @@ Logogram needs Python 3.11 or newer and [uv](https://docs.astral.sh/uv/).
|
|
|
58
25
|
uv tool install logogram
|
|
59
26
|
```
|
|
60
27
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
28
|
+
To run the code in a clone of this repository instead, install it with `uv tool install .`; the
|
|
29
|
+
commands in the table below work the same way with `.` in place of `logogram`. The web app is
|
|
30
|
+
prebuilt inside the package, so you never need Node. What changed in each version is in
|
|
31
|
+
[CHANGELOG.md](CHANGELOG.md).
|
|
64
32
|
|
|
65
33
|
PyTorch is chosen per machine:
|
|
66
34
|
|
|
@@ -68,7 +36,7 @@ PyTorch is chosen per machine:
|
|
|
68
36
|
|---|---|---|
|
|
69
37
|
| Linux with an NVIDIA GPU | `uv tool install logogram` | The default Linux wheels include CUDA. |
|
|
70
38
|
| Windows with an NVIDIA GPU | `uv tool install --torch-backend=auto logogram` | Picks the CUDA build that matches your driver. |
|
|
71
|
-
| Apple Silicon | `uv tool install logogram` |
|
|
39
|
+
| Apple Silicon | `uv tool install logogram` | Runs on the CPU unless you choose MPS (see Models). Use a native arm64 Python, not one under Rosetta. |
|
|
72
40
|
| CPU only | `uv tool install --torch-backend=cpu logogram` | Smaller download. GPT-2 small runs well on a CPU. |
|
|
73
41
|
|
|
74
42
|
Then check the setup:
|
|
@@ -109,8 +77,9 @@ This starts a local server on 127.0.0.1, prints its address and opens your brows
|
|
|
109
77
|
4. Press **A** on a head to see its attention pattern, or right-click any cell for **Patch here**,
|
|
110
78
|
**Ablate here** and **Compare across runs**.
|
|
111
79
|
5. Press **Check robustness** to rerun the sweep with a different baseline, direction or donor
|
|
112
|
-
count
|
|
113
|
-
|
|
80
|
+
count, or, for a run in float16 or bfloat16, in float32. Logogram reports the rank
|
|
81
|
+
correlation, the overlap of the top components and the components whose conclusion changed,
|
|
82
|
+
and flags them on the map.
|
|
114
83
|
|
|
115
84
|
Keyboard: arrow keys move across the map, **Ctrl/⌘ K** opens the command palette (type `L9H9` to
|
|
116
85
|
jump to a head), **1–7** switch views, **[** and **]** step through prompts, **P**, **B**, **A**
|
|
@@ -262,7 +231,11 @@ estimate misses saturation (in attention, normalization and the final softmax) a
|
|
|
262
231
|
even invert an effect, so **Verify top 10 by patching** on the results page patches the sites
|
|
263
232
|
with the largest estimated effects for real and opens the comparison: the rank correlation, and
|
|
264
233
|
any estimate whose sign patching confidently reverses. **Check robustness** can also patch the
|
|
265
|
-
whole sweep.
|
|
234
|
+
whole sweep. Do that for residual stream sweeps: an estimate is least reliable where patching
|
|
235
|
+
replaces a whole token's representation. In the IOI example on GPT-2 small, patching the residual
|
|
236
|
+
stream at the changed name in the first layer restores the whole answer while its estimate is
|
|
237
|
+
slightly negative, so verifying the top estimates would never reach it. Head outputs track
|
|
238
|
+
patching closely.
|
|
266
239
|
|
|
267
240
|
**Direct logit attribution** splits the logit difference of the clean or the corrupt prompts
|
|
268
241
|
(your choice; there is no default) into what each head, attention output and MLP output writes
|
|
@@ -299,7 +272,10 @@ like patching, so 1 means the steered prompts moved as far as switching to the o
|
|
|
299
272
|
random direction of the same length, at the same strengths, runs alongside as a control, and the
|
|
300
273
|
results show both: a layer × strength map with the control columns muted, and, for a selected site,
|
|
301
274
|
its effect at every strength with intervals. **Check robustness** offers another split of the pairs
|
|
302
|
-
or the other direction.
|
|
275
|
+
or the other direction. A mean difference steers only when the pairs differ the same way. In IOI
|
|
276
|
+
prompts that mix the ABBA and BABA orders, the difference at the last token flips with the order,
|
|
277
|
+
so the mean cancels and steering does no more than the control; generate prompts of one order to
|
|
278
|
+
steer. When no site and strength does more than the control, the results say so.
|
|
303
279
|
|
|
304
280
|
**SAE features.** A sparse autoencoder (SAE) rewrites one of the model's activations as a few active
|
|
305
281
|
features out of thousands, each a direction in the model, plus an error it misses. In
|
|
@@ -368,7 +344,7 @@ Every experiment is a `spec.json`. Nothing that can change a number is left impl
|
|
|
368
344
|
| Field | Values |
|
|
369
345
|
|---|---|
|
|
370
346
|
| `model.revision` | A commit. `null` resolves the current main branch at run time; the saved spec pins what ran. |
|
|
371
|
-
| `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change
|
|
347
|
+
| `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change. Value biases are folded into the attention output's bias, so zero ablation of head outputs and heads' direct effects do change (by up to several logits in Qwen 2.5, whose value biases are large); a layer's attention output doesn't. |
|
|
372
348
|
| `dataset.sha256` | If set, the run refuses a dataset file that has changed. |
|
|
373
349
|
| `experiment` | `{"kind": "activation_patching", "direction": "clean_to_corrupt" \| "corrupt_to_clean"}`, `{"kind": "ablation", "baseline": …}` with `{"kind": "zero"}`, `{"kind": "mean", "reference": "clean" \| "corrupt"}` or `{"kind": "resample", "pool": "clean" \| "corrupt", "donors": 10, "seed": 0}`, `{"kind": "attribution_patching", "direction": …}` (same directions as patching), `{"kind": "direct_logit_attribution", "prompts": "clean" \| "corrupt"}`, `{"kind": "path_patching", "direction": …, "receivers": [{"kind": "head", "layer": 9, "head": 9, "input": "q" \| "k" \| "v"}, {"kind": "logits"}], "freeze_mlps": false}`, or `{"kind": "steering", "apply_to": "clean" \| "corrupt", "coefficients": [-1, 1, 2], "train_fraction": 0.5, "seed": 0, "control": true}` with a scope of one residual component per layer (`layer_components`) or residual `sites`, at one token |
|
|
374
350
|
| `scope` | `{"kind": "heads", "position": …}`, `{"kind": "layer_position", "site": "resid_pre", "positions": "each" \| "labels"}`, `{"kind": "layer_components", "components": ["attn_out", "mlp_out"], "position": …}` `{"kind": "sites", "sites": [{"kind": "head", "layer": 9, "head": 9, "position": {"kind": "label", "label": "end"}}]}` (feature sites are `{"kind": "sae_feature", "layer": 8, "feature": 1234, "position": …}`), or, for attribution patching, `{"kind": "features", "position": …, "top": 50}` |
|
|
@@ -468,12 +444,27 @@ TransformerLens 4 loads well over a hundred architectures, and Logogram works wi
|
|
|
468
444
|
decoder-only language models among them: GPT-2, Pythia, Llama, Mistral, SmolLM, Qwen, Gemma,
|
|
469
445
|
OLMo, Phi and more. The model dialog lists starting points, and any other Hugging Face id can be
|
|
470
446
|
typed in. Before downloading, Logogram checks that TransformerLens supports the architecture and
|
|
471
|
-
estimates the memory needed (weights, activations and a margin) against what is free.
|
|
447
|
+
estimates the memory needed (weights, activations and a margin) against what is free. Processing
|
|
448
|
+
the weights needs more for a moment while the model loads: TransformerLens works on float32
|
|
449
|
+
copies, three to four times the float32 size of the weights. The estimate includes them, so a
|
|
450
|
+
model that fits only without processing (Qwen 2.5 0.5B on a 6 GB GPU) says so before it loads.
|
|
451
|
+
|
|
452
|
+
Use float32 whenever the model fits. float16 and bfloat16 halve the memory and run two to three
|
|
453
|
+
times faster on a GPU, but they round. In the IOI example on GPT-2 small and Qwen 2.5, float16
|
|
454
|
+
moved effects by at most 0.003 and bfloat16 by up to 0.03: the strongest sites stayed in place,
|
|
455
|
+
but effects smaller than about 0.01 changed order. On Pythia-70m, bfloat16 doubled the clean
|
|
456
|
+
logit difference and reordered the heads, and float16 overflows. Check results that matter
|
|
457
|
+
against float32: **Check robustness** reruns a 16-bit run in float32 and compares the two.
|
|
458
|
+
|
|
459
|
+
On Apple Silicon, **Automatic** runs models on the CPU. TransformerLens reports that Apple's MPS
|
|
460
|
+
can give silently wrong results, and Logogram hasn't been checked on it yet, so MPS is used only
|
|
461
|
+
when you choose it under **Device**. It is faster; check results that matter on the CPU.
|
|
472
462
|
|
|
473
463
|
Rather than trusting a list, Logogram checks every model when it loads, on a short fixed input:
|
|
474
464
|
|
|
475
465
|
* TransformerLens's version of the model must predict what the original model predicts. If weight
|
|
476
|
-
processing changes the predictions, loading with processed weights is refused.
|
|
466
|
+
processing changes the predictions, loading with processed weights is refused. So is a dtype in
|
|
467
|
+
which the model overflows (Pythia in float16).
|
|
477
468
|
* It measures how each layer adds attention and the MLP to the residual stream: one after the
|
|
478
469
|
other (sequential), or both from the same input (parallel, as in Pythia, GPT-J and Phi). Only
|
|
479
470
|
sequential layers have a residual stream between attention and MLP (`resid_mid`), and the layer
|
|
@@ -481,11 +472,15 @@ Rather than trusting a list, Logogram checks every model when it loads, on a sho
|
|
|
481
472
|
* Layer predictions are offered when the final normalization and unembedding, with any logit
|
|
482
473
|
soft-capping, reproduce the model's output.
|
|
483
474
|
|
|
484
|
-
GPT-2 small is the model the bundled example was made for
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
475
|
+
GPT-2 small is the model the bundled example was made for. Every method has been run end to end
|
|
476
|
+
with real weights on GPT-2 small (with a SAELens and an OpenAI SAE), Pythia-70m (with an
|
|
477
|
+
EleutherAI SAE) and Qwen 2.5 0.5B, and the test suite runs the sanity checks on tiny random models
|
|
478
|
+
of the Llama, Pythia, Qwen 2, Gemma 2 and OLMo 2 families without downloading anything. The
|
|
479
|
+
example's names are single tokens for GPT-2 and Qwen but not all for Pythia; for another model,
|
|
480
|
+
generate IOI prompts in the app, which keeps only names that are single tokens for it. A small
|
|
481
|
+
model may not do the task at all (Pythia-70m prefers the repeated name), and its runs then say so.
|
|
482
|
+
Pythia publishes checkpoints from throughout training as revisions (`step1000` to `step143000`),
|
|
483
|
+
so an experiment can be rerun at several points of training.
|
|
489
484
|
|
|
490
485
|
Some tokenizers, such as Qwen's, have no beginning-of-sequence token. Logogram then runs prompts
|
|
491
486
|
without one, and the spec records it. Answers and distractors must still be single tokens.
|
|
@@ -534,10 +529,18 @@ push a tag with the same version, such as `v0.1.0`. The Release workflow runs ev
|
|
|
534
529
|
that commit, builds the wheel and source distribution, and uploads them to PyPI through Trusted
|
|
535
530
|
Publishing, so no token is stored anywhere. PyPI never accepts the same version twice.
|
|
536
531
|
|
|
537
|
-
Before tagging, manually check a first GPT-2 download, cancel and retry it, then run
|
|
538
|
-
on each supported compute backend.
|
|
539
|
-
|
|
540
|
-
|
|
532
|
+
Before tagging, manually check a first GPT-2 download, cancel and retry it, then run
|
|
533
|
+
`scripts/validate_real_weights.py` on each supported compute backend. It runs every method on a
|
|
534
|
+
real model in a temporary project and checks exact identities, agreement between methods and
|
|
535
|
+
bit-identical reruns; `--compare-device` reruns the head sweep on a second device, and `--sae`
|
|
536
|
+
adds SAE features. CPU and CUDA pass it. Apple's MPS still needs it run on a Mac:
|
|
537
|
+
|
|
538
|
+
```bash
|
|
539
|
+
uv run python scripts/validate_real_weights.py --device mps --compare-device cpu
|
|
540
|
+
```
|
|
541
|
+
|
|
542
|
+
Pythia and Qwen 2.5 have also been run end to end with their real weights; other families are
|
|
543
|
+
checked when they load and in the tiny-model tests.
|
|
541
544
|
|
|
542
545
|
Model access goes through `logogram.backends.base.ModelBackend`. TransformerLens is the only
|
|
543
546
|
backend today; remote execution and other libraries can be added behind the same interface.
|
|
@@ -20,6 +20,7 @@ classifiers = [
|
|
|
20
20
|
"Programming Language :: Python :: 3.11",
|
|
21
21
|
"Programming Language :: Python :: 3.12",
|
|
22
22
|
"Programming Language :: Python :: 3.13",
|
|
23
|
+
"Programming Language :: Python :: 3.14",
|
|
23
24
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
24
25
|
]
|
|
25
26
|
dependencies = [
|
|
@@ -39,13 +40,19 @@ dependencies = [
|
|
|
39
40
|
"huggingface-hub>=0.23",
|
|
40
41
|
]
|
|
41
42
|
|
|
43
|
+
[project.urls]
|
|
44
|
+
Homepage = "https://github.com/Jeevash23/logogram"
|
|
45
|
+
Source = "https://github.com/Jeevash23/logogram"
|
|
46
|
+
Issues = "https://github.com/Jeevash23/logogram/issues"
|
|
47
|
+
Changelog = "https://github.com/Jeevash23/logogram/blob/main/CHANGELOG.md"
|
|
48
|
+
|
|
42
49
|
[project.scripts]
|
|
43
50
|
logogram = "logogram.cli:main"
|
|
44
51
|
|
|
45
52
|
[dependency-groups]
|
|
46
53
|
dev = [
|
|
47
54
|
"pytest>=8",
|
|
48
|
-
"
|
|
55
|
+
"httpx2>=2.13",
|
|
49
56
|
"ruff==0.16.10",
|
|
50
57
|
]
|
|
51
58
|
|
|
@@ -58,7 +65,7 @@ packages = ["src/logogram"]
|
|
|
58
65
|
artifacts = ["src/logogram/web_dist/**"]
|
|
59
66
|
|
|
60
67
|
[tool.hatch.build.targets.sdist]
|
|
61
|
-
include = ["src/logogram", "web/src", "web/public", "web/tests", "web/package.json", "web/package-lock.json", "web/index.html", "web/tsconfig.json", "web/vite.config.ts", "web/playwright.config.ts", "uv.lock", "tests", "scripts", "README.md", "LICENSE", "AGENTS.md"]
|
|
68
|
+
include = ["src/logogram", "web/src", "web/public", "web/tests", "web/package.json", "web/package-lock.json", "web/index.html", "web/tsconfig.json", "web/vite.config.ts", "web/playwright.config.ts", "uv.lock", "tests", "scripts", "README.md", "CHANGELOG.md", "LICENSE", "AGENTS.md"]
|
|
62
69
|
artifacts = ["src/logogram/web_dist/**"]
|
|
63
70
|
|
|
64
71
|
[tool.pytest.ini_options]
|