logogram 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (185) hide show
  1. logogram-0.1.2/CHANGELOG.md +94 -0
  2. logogram-0.1.0/README.md → logogram-0.1.2/PKG-INFO +94 -20
  3. logogram-0.1.0/PKG-INFO → logogram-0.1.2/README.md +56 -53
  4. {logogram-0.1.0 → logogram-0.1.2}/pyproject.toml +9 -2
  5. logogram-0.1.2/scripts/validate_real_weights.py +459 -0
  6. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/__init__.py +1 -1
  7. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/analysis.py +5 -5
  8. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/atp.py +13 -3
  9. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/base.py +15 -2
  10. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/hub.py +54 -7
  11. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/saes.py +10 -10
  12. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/transformer_lens.py +135 -31
  13. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/cli.py +11 -14
  14. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/direct.py +16 -6
  15. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/engine.py +18 -3
  16. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/features.py +8 -8
  17. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/results.py +26 -1
  18. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/app.py +28 -7
  19. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/models.py +1 -0
  20. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/steering.py +38 -6
  21. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/system.py +44 -11
  22. logogram-0.1.0/src/logogram/web_dist/assets/index-BvCU-2uy.js → logogram-0.1.2/src/logogram/web_dist/assets/index-Cw0WBZBB.js +6 -6
  23. logogram-0.1.2/src/logogram/web_dist/assets/index-Dt6ecB66.css +1 -0
  24. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/index.html +2 -2
  25. {logogram-0.1.0 → logogram-0.1.2}/tests/test_atp.py +20 -0
  26. {logogram-0.1.0 → logogram-0.1.2}/tests/test_cli.py +33 -0
  27. logogram-0.1.2/tests/test_devices.py +64 -0
  28. {logogram-0.1.0 → logogram-0.1.2}/tests/test_direct.py +38 -0
  29. {logogram-0.1.0 → logogram-0.1.2}/tests/test_hardening.py +39 -0
  30. logogram-0.1.2/tests/test_loading.py +235 -0
  31. {logogram-0.1.0 → logogram-0.1.2}/tests/test_steering.py +26 -0
  32. {logogram-0.1.0 → logogram-0.1.2}/tests/test_units.py +12 -0
  33. logogram-0.1.2/tests/test_validation_script.py +45 -0
  34. {logogram-0.1.0 → logogram-0.1.2}/uv.lock +50 -2
  35. {logogram-0.1.0 → logogram-0.1.2}/web/src/api/client.ts +4 -6
  36. {logogram-0.1.0 → logogram-0.1.2}/web/src/api/types.ts +6 -0
  37. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelDialog.module.css +6 -0
  38. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelDialog.tsx +24 -3
  39. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/RobustnessDialog.tsx +15 -3
  40. {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/SystemCheck.tsx +1 -1
  41. {logogram-0.1.0 → logogram-0.1.2}/web/src/store/app.ts +4 -4
  42. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/BaselineView.tsx +13 -3
  43. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/PromptsView.tsx +30 -0
  44. logogram-0.1.0/src/logogram/web_dist/assets/index-DTr8_ucV.css +0 -1
  45. {logogram-0.1.0 → logogram-0.1.2}/.gitignore +0 -0
  46. {logogram-0.1.0 → logogram-0.1.2}/AGENTS.md +0 -0
  47. {logogram-0.1.0 → logogram-0.1.2}/LICENSE +0 -0
  48. {logogram-0.1.0 → logogram-0.1.2}/scripts/check_privacy.py +0 -0
  49. {logogram-0.1.0 → logogram-0.1.2}/scripts/smoke_wheel.py +0 -0
  50. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/__main__.py +0 -0
  51. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/backends/__init__.py +0 -0
  52. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/compare.py +0 -0
  53. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/datasets.py +0 -0
  54. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/.gitignore +0 -0
  55. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/datasets/ioi.jsonl +0 -0
  56. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/experiments/ioi-head-patching/spec.json +0 -0
  57. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/examples/ioi-gpt2/project.json +0 -0
  58. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/exports.py +0 -0
  59. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/fileio.py +0 -0
  60. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/ioi.py +0 -0
  61. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/paths.py +0 -0
  62. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/project.py +0 -0
  63. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/prompts.py +0 -0
  64. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/research.py +0 -0
  65. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/runner.py +0 -0
  66. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/runs.py +0 -0
  67. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/sae.py +0 -0
  68. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/schema.py +0 -0
  69. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/__init__.py +0 -0
  70. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/security.py +0 -0
  71. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/server/state.py +0 -0
  72. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/sites.py +0 -0
  73. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/spec.py +0 -0
  74. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/stats.py +0 -0
  75. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/updates.py +0 -0
  76. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/verify.py +0 -0
  77. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/assets/instrument-sans-latin-ext-standard-normal-C5E2Gvlv.woff2 +0 -0
  78. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/assets/instrument-sans-latin-standard-normal-BVScPF0l.woff2 +0 -0
  79. {logogram-0.1.0 → logogram-0.1.2}/src/logogram/web_dist/favicon.svg +0 -0
  80. {logogram-0.1.0 → logogram-0.1.2}/tests/browser_server.py +0 -0
  81. {logogram-0.1.0 → logogram-0.1.2}/tests/conftest.py +0 -0
  82. {logogram-0.1.0 → logogram-0.1.2}/tests/test_architectures.py +0 -0
  83. {logogram-0.1.0 → logogram-0.1.2}/tests/test_engine.py +0 -0
  84. {logogram-0.1.0 → logogram-0.1.2}/tests/test_explorer.py +0 -0
  85. {logogram-0.1.0 → logogram-0.1.2}/tests/test_paths.py +0 -0
  86. {logogram-0.1.0 → logogram-0.1.2}/tests/test_privacy.py +0 -0
  87. {logogram-0.1.0 → logogram-0.1.2}/tests/test_release_fixes.py +0 -0
  88. {logogram-0.1.0 → logogram-0.1.2}/tests/test_sae.py +0 -0
  89. {logogram-0.1.0 → logogram-0.1.2}/tests/test_sanity.py +0 -0
  90. {logogram-0.1.0 → logogram-0.1.2}/tests/test_server.py +0 -0
  91. {logogram-0.1.0 → logogram-0.1.2}/tests/test_updates.py +0 -0
  92. {logogram-0.1.0 → logogram-0.1.2}/web/index.html +0 -0
  93. {logogram-0.1.0 → logogram-0.1.2}/web/package-lock.json +0 -0
  94. {logogram-0.1.0 → logogram-0.1.2}/web/package.json +0 -0
  95. {logogram-0.1.0 → logogram-0.1.2}/web/playwright.config.ts +0 -0
  96. {logogram-0.1.0 → logogram-0.1.2}/web/public/favicon.svg +0 -0
  97. {logogram-0.1.0 → logogram-0.1.2}/web/src/App.module.css +0 -0
  98. {logogram-0.1.0 → logogram-0.1.2}/web/src/App.tsx +0 -0
  99. {logogram-0.1.0 → logogram-0.1.2}/web/src/api/events.ts +0 -0
  100. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CommandPalette.module.css +0 -0
  101. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CommandPalette.tsx +0 -0
  102. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CopyCommand.module.css +0 -0
  103. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/CopyCommand.tsx +0 -0
  104. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Distribution.module.css +0 -0
  105. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Distribution.tsx +0 -0
  106. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/FolderPicker.module.css +0 -0
  107. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/FolderPicker.tsx +0 -0
  108. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Header.module.css +0 -0
  109. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Header.tsx +0 -0
  110. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Heatmap/Heatmap.module.css +0 -0
  111. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Heatmap/Heatmap.tsx +0 -0
  112. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Heatmap/ScaleBar.tsx +0 -0
  113. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/History.module.css +0 -0
  114. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/History.tsx +0 -0
  115. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Inspector.module.css +0 -0
  116. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Inspector.tsx +0 -0
  117. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Logogram.tsx +0 -0
  118. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/LogogramDial.module.css +0 -0
  119. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/LogogramDial.tsx +0 -0
  120. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelMap/Legend.tsx +0 -0
  121. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelMap/ModelMap.module.css +0 -0
  122. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ModelMap/ModelMap.tsx +0 -0
  123. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Notices.module.css +0 -0
  124. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Notices.tsx +0 -0
  125. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Splitter.module.css +0 -0
  126. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/Splitter.tsx +0 -0
  127. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/StatusLine.module.css +0 -0
  128. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/StatusLine.tsx +0 -0
  129. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TokenStrip.module.css +0 -0
  130. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TokenStrip.tsx +0 -0
  131. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TopBar.module.css +0 -0
  132. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/TopBar.tsx +0 -0
  133. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/UpdateNotice.module.css +0 -0
  134. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/UpdateNotice.tsx +0 -0
  135. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ui/Icon.tsx +0 -0
  136. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ui/index.tsx +0 -0
  137. {logogram-0.1.0 → logogram-0.1.2}/web/src/components/ui/ui.module.css +0 -0
  138. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/analysis.ts +0 -0
  139. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/canvas.ts +0 -0
  140. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/color.ts +0 -0
  141. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/export.ts +0 -0
  142. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/format.ts +0 -0
  143. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/heatmapNavigation.ts +0 -0
  144. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/hooks.ts +0 -0
  145. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/keys.ts +0 -0
  146. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/logogram.ts +0 -0
  147. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/sites.ts +0 -0
  148. {logogram-0.1.0 → logogram-0.1.2}/web/src/lib/spec.ts +0 -0
  149. {logogram-0.1.0 → logogram-0.1.2}/web/src/main.tsx +0 -0
  150. {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Projects.tsx +0 -0
  151. {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Screens.module.css +0 -0
  152. {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Welcome.tsx +0 -0
  153. {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Workbench.module.css +0 -0
  154. {logogram-0.1.0 → logogram-0.1.2}/web/src/screens/Workbench.tsx +0 -0
  155. {logogram-0.1.0 → logogram-0.1.2}/web/src/styles/global.css +0 -0
  156. {logogram-0.1.0 → logogram-0.1.2}/web/src/styles/tokens.css +0 -0
  157. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/AttentionView.module.css +0 -0
  158. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/AttentionView.tsx +0 -0
  159. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/BaselineView.module.css +0 -0
  160. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/CompareView.module.css +0 -0
  161. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/CompareView.tsx +0 -0
  162. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExperimentView.module.css +0 -0
  163. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExperimentView.tsx +0 -0
  164. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExploreView.module.css +0 -0
  165. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ExploreView.tsx +0 -0
  166. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/FeaturesView.module.css +0 -0
  167. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/FeaturesView.tsx +0 -0
  168. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/Forest.module.css +0 -0
  169. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/Forest.tsx +0 -0
  170. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/HeadComparisonView.module.css +0 -0
  171. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/HeadComparisonView.tsx +0 -0
  172. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/PredictionsView.module.css +0 -0
  173. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/PredictionsView.tsx +0 -0
  174. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResearchView.module.css +0 -0
  175. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResearchView.tsx +0 -0
  176. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResultsView.module.css +0 -0
  177. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/ResultsView.tsx +0 -0
  178. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/SpecView.module.css +0 -0
  179. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/SpecView.tsx +0 -0
  180. {logogram-0.1.0 → logogram-0.1.2}/web/src/views/views.module.css +0 -0
  181. {logogram-0.1.0 → logogram-0.1.2}/web/tests/context.unit.ts +0 -0
  182. {logogram-0.1.0 → logogram-0.1.2}/web/tests/logogram.unit.ts +0 -0
  183. {logogram-0.1.0 → logogram-0.1.2}/web/tests/workbench.browser.ts +0 -0
  184. {logogram-0.1.0 → logogram-0.1.2}/web/tsconfig.json +0 -0
  185. {logogram-0.1.0 → logogram-0.1.2}/web/vite.config.ts +0 -0
@@ -0,0 +1,94 @@
1
+ # Changelog
2
+
3
+ What changed in each version of Logogram.
4
+
5
+ ## 0.1.2 (2026-10-08)
6
+
7
+ Fixes and checks that came out of validating every method on real weights.
8
+
9
+ ### Changed
10
+
11
+ - On Apple Silicon, **Automatic** runs models on the CPU. TransformerLens reports that Apple's MPS
12
+ can give silently wrong results, and Logogram hasn't been checked on it yet, so MPS runs only when
13
+ you choose it; the system check and the model dialog say why.
14
+
15
+ ### Added
16
+
17
+ - **Check robustness** reruns a float16 or bfloat16 run in float32 and compares the two.
18
+ - When processing a model's weights is what doesn't fit, the model dialog offers to turn it off.
19
+ - When a generated IOI dataset has names the loaded model splits into several tokens, one click
20
+ makes a new one with names it doesn't.
21
+ - Steering results say when the direction does no more than its random control at any site and
22
+ strength.
23
+ - Attribution patching of residual stream sites warns that it can miss effects there, and that
24
+ **Check robustness** patches every site.
25
+ - `scripts/validate_real_weights.py` runs every method on a real model and checks identities,
26
+ agreement and bit-identical reruns, on one device or two. It is how a release, or Apple's MPS, is
27
+ checked.
28
+
29
+ ### Fixed
30
+
31
+ - The baseline says that the clean prompts prefer the distractor, instead of preferring the answer
32
+ by a negative amount.
33
+
34
+ ### Development
35
+
36
+ - Tests run on Python 3.14 too, and the Linux runners are pinned to Ubuntu 24.04.
37
+ - The test client uses httpx2, as Starlette recommends.
38
+
39
+ ## 0.1.1 (2026-10-08)
40
+
41
+ Every method has now run end to end on real weights: GPT-2 small with a SAELens and an OpenAI
42
+ SAE, Pythia-70m with an EleutherAI SAE, and Qwen 2.5 0.5B. This release fixes what that turned up.
43
+
44
+ ### Fixed
45
+
46
+ - Runs work on Apple Silicon. Every method converted tensors to float64 on the model's device,
47
+ which Apple's Metal (MPS) doesn't have, so no run got past its baseline on a Mac. This still
48
+ needs a check on a Mac.
49
+ - A model whose own forward pass overflows in the chosen dtype (Pythia in float16) is refused, with
50
+ the dtype to use. Its load checks used to read as passed.
51
+ - When processing a model's weights doesn't fit in memory, loading says so and frees the memory,
52
+ instead of failing with a raw CUDA error and keeping it.
53
+ - The memory estimate counts weight processing, which needs three to four times the float32 size
54
+ of the weights while the model loads. Qwen 2.5 0.5B on a 6 GB GPU now reads "won't fit" with
55
+ processing and "fits" without it.
56
+ - Models without a beginning-of-sequence token, such as Qwen, run without one, as documented.
57
+ TransformerLens 4 had given them its end-of-text token instead.
58
+ - Models and SAEs that Logogram downloaded open offline without a revision, and offline memory
59
+ estimates use the model's real size.
60
+ - `logogram run` reports an identical rerun of direct logit attribution or attribution patching as
61
+ identical.
62
+ - Direct logit attribution in float16 and bfloat16 no longer warns about drift that is only
63
+ rounding.
64
+
65
+ ### Added
66
+
67
+ - Runs warn when the model prefers the distractor on the clean prompts, so it doesn't show the
68
+ behavior they test (Pythia-70m on IOI).
69
+ - The model dialog shows the memory that weight processing needs. Qwen 2.5 0.5B is marked as
70
+ tested.
71
+ - Project links on PyPI, and this changelog, which the update notice links to.
72
+
73
+ ### Documentation
74
+
75
+ - What has been validated with real weights, and where attribution patching is least reliable:
76
+ residual stream sites at the token that differs between the prompts.
77
+ - Steering needs prompt pairs that differ the same way; IOI prompts that mix the ABBA and BABA
78
+ orders cancel out.
79
+ - What float16 and bfloat16 do to results, and how weight processing changes heads' direct effects.
80
+
81
+ ## 0.1.0 (2026-10-08)
82
+
83
+ First public release.
84
+
85
+ - Activation patching, ablation (zero, mean or resample), direct logit attribution, attribution
86
+ patching verified by patching, steering with a random control, path patching, and SAE features,
87
+ all described by one spec that the app and `logogram run` execute the same way.
88
+ - Every model is checked when it loads: that TransformerLens reproduces its predictions, how its
89
+ layers add into the residual stream, and whether the logit lens reproduces its output.
90
+ - Per-prompt values with seeded bootstrap intervals, and bit-identical reruns on the same machine.
91
+ - The workbench: Explore, Experiment and Evidence, logograms written by results, robustness checks
92
+ and run comparisons.
93
+ - Local-first: no account and no telemetry, safetensors weights only, and an update check that
94
+ runs only when you allow it.
@@ -1,3 +1,41 @@
1
+ Metadata-Version: 2.5
2
+ Name: logogram
3
+ Version: 0.1.2
4
+ Summary: A local-first workbench for causal experiments inside language models.
5
+ Project-URL: Homepage, https://github.com/Jeevash23/logogram
6
+ Project-URL: Source, https://github.com/Jeevash23/logogram
7
+ Project-URL: Issues, https://github.com/Jeevash23/logogram/issues
8
+ Project-URL: Changelog, https://github.com/Jeevash23/logogram/blob/main/CHANGELOG.md
9
+ Author: The Logogram contributors
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: activation-patching,interpretability,mechanistic-interpretability,transformers
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Programming Language :: Python :: 3.14
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Requires-Python: >=3.11
23
+ Requires-Dist: anyio>=4
24
+ Requires-Dist: fastapi>=0.115
25
+ Requires-Dist: huggingface-hub>=0.23
26
+ Requires-Dist: numpy>=1.26
27
+ Requires-Dist: platformdirs>=4
28
+ Requires-Dist: psutil>=5.9
29
+ Requires-Dist: pyarrow>=15
30
+ Requires-Dist: pydantic>=2.7
31
+ Requires-Dist: torch>=2.4
32
+ Requires-Dist: transformer-lens<5,>=4.0
33
+ Requires-Dist: transformers>=4.45
34
+ Requires-Dist: typer>=0.12
35
+ Requires-Dist: uvicorn>=0.30
36
+ Requires-Dist: websockets>=12
37
+ Description-Content-Type: text/markdown
38
+
1
39
  # Logogram
2
40
 
3
41
  A local workbench for causal experiments inside language models.
@@ -25,9 +63,10 @@ Logogram needs Python 3.11 or newer and [uv](https://docs.astral.sh/uv/).
25
63
  uv tool install logogram
26
64
  ```
27
65
 
28
- Until Logogram is published on PyPI, install it from a clone of this repository with
29
- `uv tool install .` instead. The web app is prebuilt inside the package, so you never need Node.
30
- The commands in the table below work the same way with `.` in place of `logogram`.
66
+ To run the code in a clone of this repository instead, install it with `uv tool install .`; the
67
+ commands in the table below work the same way with `.` in place of `logogram`. The web app is
68
+ prebuilt inside the package, so you never need Node. What changed in each version is in
69
+ [CHANGELOG.md](CHANGELOG.md).
31
70
 
32
71
  PyTorch is chosen per machine:
33
72
 
@@ -35,7 +74,7 @@ PyTorch is chosen per machine:
35
74
  |---|---|---|
36
75
  | Linux with an NVIDIA GPU | `uv tool install logogram` | The default Linux wheels include CUDA. |
37
76
  | Windows with an NVIDIA GPU | `uv tool install --torch-backend=auto logogram` | Picks the CUDA build that matches your driver. |
38
- | Apple Silicon | `uv tool install logogram` | Uses Metal (MPS). Use a native arm64 Python, not one under Rosetta. |
77
+ | Apple Silicon | `uv tool install logogram` | Runs on the CPU unless you choose MPS (see Models). Use a native arm64 Python, not one under Rosetta. |
39
78
  | CPU only | `uv tool install --torch-backend=cpu logogram` | Smaller download. GPT-2 small runs well on a CPU. |
40
79
 
41
80
  Then check the setup:
@@ -76,8 +115,9 @@ This starts a local server on 127.0.0.1, prints its address and opens your brows
76
115
  4. Press **A** on a head to see its attention pattern, or right-click any cell for **Patch here**,
77
116
  **Ablate here** and **Compare across runs**.
78
117
  5. Press **Check robustness** to rerun the sweep with a different baseline, direction or donor
79
- count. Logogram reports the rank correlation, the overlap of the top components and the
80
- components whose conclusion changed, and flags them on the map.
118
+ count, or, for a run in float16 or bfloat16, in float32. Logogram reports the rank
119
+ correlation, the overlap of the top components and the components whose conclusion changed,
120
+ and flags them on the map.
81
121
 
82
122
  Keyboard: arrow keys move across the map, **Ctrl/⌘ K** opens the command palette (type `L9H9` to
83
123
  jump to a head), **1–7** switch views, **[** and **]** step through prompts, **P**, **B**, **A**
@@ -229,7 +269,11 @@ estimate misses saturation (in attention, normalization and the final softmax) a
229
269
  even invert an effect, so **Verify top 10 by patching** on the results page patches the sites
230
270
  with the largest estimated effects for real and opens the comparison: the rank correlation, and
231
271
  any estimate whose sign patching confidently reverses. **Check robustness** can also patch the
232
- whole sweep.
272
+ whole sweep. Do that for residual stream sweeps: an estimate is least reliable where patching
273
+ replaces a whole token's representation. In the IOI example on GPT-2 small, patching the residual
274
+ stream at the changed name in the first layer restores the whole answer while its estimate is
275
+ slightly negative, so verifying the top estimates would never reach it. Head outputs track
276
+ patching closely.
233
277
 
234
278
  **Direct logit attribution** splits the logit difference of the clean or the corrupt prompts
235
279
  (your choice; there is no default) into what each head, attention output and MLP output writes
@@ -266,7 +310,10 @@ like patching, so 1 means the steered prompts moved as far as switching to the o
266
310
  random direction of the same length, at the same strengths, runs alongside as a control, and the
267
311
  results show both: a layer × strength map with the control columns muted, and, for a selected site,
268
312
  its effect at every strength with intervals. **Check robustness** offers another split of the pairs
269
- or the other direction.
313
+ or the other direction. A mean difference steers only when the pairs differ the same way. In IOI
314
+ prompts that mix the ABBA and BABA orders, the difference at the last token flips with the order,
315
+ so the mean cancels and steering does no more than the control; generate prompts of one order to
316
+ steer. When no site and strength does more than the control, the results say so.
270
317
 
271
318
  **SAE features.** A sparse autoencoder (SAE) rewrites one of the model's activations as a few active
272
319
  features out of thousands, each a direction in the model, plus an error it misses. In
@@ -335,7 +382,7 @@ Every experiment is a `spec.json`. Nothing that can change a number is left impl
335
382
  | Field | Values |
336
383
  |---|---|
337
384
  | `model.revision` | A commit. `null` resolves the current main branch at run time; the saved spec pins what ran. |
338
- | `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change; zero ablation of head outputs does, because value biases are folded. |
385
+ | `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change. Value biases are folded into the attention output's bias, so zero ablation of head outputs and heads' direct effects do change (by up to several logits in Qwen 2.5, whose value biases are large); a layer's attention output doesn't. |
339
386
  | `dataset.sha256` | If set, the run refuses a dataset file that has changed. |
340
387
  | `experiment` | `{"kind": "activation_patching", "direction": "clean_to_corrupt" \| "corrupt_to_clean"}`, `{"kind": "ablation", "baseline": …}` with `{"kind": "zero"}`, `{"kind": "mean", "reference": "clean" \| "corrupt"}` or `{"kind": "resample", "pool": "clean" \| "corrupt", "donors": 10, "seed": 0}`, `{"kind": "attribution_patching", "direction": …}` (same directions as patching), `{"kind": "direct_logit_attribution", "prompts": "clean" \| "corrupt"}`, `{"kind": "path_patching", "direction": …, "receivers": [{"kind": "head", "layer": 9, "head": 9, "input": "q" \| "k" \| "v"}, {"kind": "logits"}], "freeze_mlps": false}`, or `{"kind": "steering", "apply_to": "clean" \| "corrupt", "coefficients": [-1, 1, 2], "train_fraction": 0.5, "seed": 0, "control": true}` with a scope of one residual component per layer (`layer_components`) or residual `sites`, at one token |
341
388
  | `scope` | `{"kind": "heads", "position": …}`, `{"kind": "layer_position", "site": "resid_pre", "positions": "each" \| "labels"}`, `{"kind": "layer_components", "components": ["attn_out", "mlp_out"], "position": …}` `{"kind": "sites", "sites": [{"kind": "head", "layer": 9, "head": 9, "position": {"kind": "label", "label": "end"}}]}` (feature sites are `{"kind": "sae_feature", "layer": 8, "feature": 1234, "position": …}`), or, for attribution patching, `{"kind": "features", "position": …, "top": 50}` |
@@ -435,12 +482,27 @@ TransformerLens 4 loads well over a hundred architectures, and Logogram works wi
435
482
  decoder-only language models among them: GPT-2, Pythia, Llama, Mistral, SmolLM, Qwen, Gemma,
436
483
  OLMo, Phi and more. The model dialog lists starting points, and any other Hugging Face id can be
437
484
  typed in. Before downloading, Logogram checks that TransformerLens supports the architecture and
438
- estimates the memory needed (weights, activations and a margin) against what is free.
485
+ estimates the memory needed (weights, activations and a margin) against what is free. Processing
486
+ the weights needs more for a moment while the model loads: TransformerLens works on float32
487
+ copies, three to four times the float32 size of the weights. The estimate includes them, so a
488
+ model that fits only without processing (Qwen 2.5 0.5B on a 6 GB GPU) says so before it loads.
489
+
490
+ Use float32 whenever the model fits. float16 and bfloat16 halve the memory and run two to three
491
+ times faster on a GPU, but they round. In the IOI example on GPT-2 small and Qwen 2.5, float16
492
+ moved effects by at most 0.003 and bfloat16 by up to 0.03: the strongest sites stayed in place,
493
+ but effects smaller than about 0.01 changed order. On Pythia-70m, bfloat16 doubled the clean
494
+ logit difference and reordered the heads, and float16 overflows. Check results that matter
495
+ against float32: **Check robustness** reruns a 16-bit run in float32 and compares the two.
496
+
497
+ On Apple Silicon, **Automatic** runs models on the CPU. TransformerLens reports that Apple's MPS
498
+ can give silently wrong results, and Logogram hasn't been checked on it yet, so MPS is used only
499
+ when you choose it under **Device**. It is faster; check results that matter on the CPU.
439
500
 
440
501
  Rather than trusting a list, Logogram checks every model when it loads, on a short fixed input:
441
502
 
442
503
  * TransformerLens's version of the model must predict what the original model predicts. If weight
443
- processing changes the predictions, loading with processed weights is refused.
504
+ processing changes the predictions, loading with processed weights is refused. So is a dtype in
505
+ which the model overflows (Pythia in float16).
444
506
  * It measures how each layer adds attention and the MLP to the residual stream: one after the
445
507
  other (sequential), or both from the same input (parallel, as in Pythia, GPT-J and Phi). Only
446
508
  sequential layers have a residual stream between attention and MLP (`resid_mid`), and the layer
@@ -448,11 +510,15 @@ Rather than trusting a list, Logogram checks every model when it loads, on a sho
448
510
  * Layer predictions are offered when the final normalization and unembedding, with any logit
449
511
  soft-capping, reproduce the model's output.
450
512
 
451
- GPT-2 small is the model the bundled example was made for, and the one run end to end with real
452
- weights. The test suite runs the sanity checks on tiny random models of the Llama, Pythia, Qwen 2,
453
- Gemma 2 and OLMo 2 families without downloading anything. Pythia publishes checkpoints from
454
- throughout training as revisions (`step1000` to `step143000`), so an experiment can be rerun at
455
- several points of training.
513
+ GPT-2 small is the model the bundled example was made for. Every method has been run end to end
514
+ with real weights on GPT-2 small (with a SAELens and an OpenAI SAE), Pythia-70m (with an
515
+ EleutherAI SAE) and Qwen 2.5 0.5B, and the test suite runs the sanity checks on tiny random models
516
+ of the Llama, Pythia, Qwen 2, Gemma 2 and OLMo 2 families without downloading anything. The
517
+ example's names are single tokens for GPT-2 and Qwen but not all for Pythia; for another model,
518
+ generate IOI prompts in the app, which keeps only names that are single tokens for it. A small
519
+ model may not do the task at all (Pythia-70m prefers the repeated name), and its runs then say so.
520
+ Pythia publishes checkpoints from throughout training as revisions (`step1000` to `step143000`),
521
+ so an experiment can be rerun at several points of training.
456
522
 
457
523
  Some tokenizers, such as Qwen's, have no beginning-of-sequence token. Logogram then runs prompts
458
524
  without one, and the spec records it. Answers and distractors must still be single tokens.
@@ -501,10 +567,18 @@ push a tag with the same version, such as `v0.1.0`. The Release workflow runs ev
501
567
  that commit, builds the wheel and source distribution, and uploads them to PyPI through Trusted
502
568
  Publishing, so no token is stored anywhere. PyPI never accepts the same version twice.
503
569
 
504
- Before tagging, manually check a first GPT-2 download, cancel and retry it, then run the example
505
- on each supported compute backend. CPU and CUDA are covered by local development checks; MPS
506
- still needs a check on Apple Silicon. Other model families are checked when they load and in the
507
- tiny-model tests, but not yet run end to end with their real weights.
570
+ Before tagging, manually check a first GPT-2 download, cancel and retry it, then run
571
+ `scripts/validate_real_weights.py` on each supported compute backend. It runs every method on a
572
+ real model in a temporary project and checks exact identities, agreement between methods and
573
+ bit-identical reruns; `--compare-device` reruns the head sweep on a second device, and `--sae`
574
+ adds SAE features. CPU and CUDA pass it. Apple's MPS still needs it run on a Mac:
575
+
576
+ ```bash
577
+ uv run python scripts/validate_real_weights.py --device mps --compare-device cpu
578
+ ```
579
+
580
+ Pythia and Qwen 2.5 have also been run end to end with their real weights; other families are
581
+ checked when they load and in the tiny-model tests.
508
582
 
509
583
  Model access goes through `logogram.backends.base.ModelBackend`. TransformerLens is the only
510
584
  backend today; remote execution and other libraries can be added behind the same interface.
@@ -1,36 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: logogram
3
- Version: 0.1.0
4
- Summary: A local-first workbench for causal experiments inside language models.
5
- Author: The Logogram contributors
6
- License-Expression: MIT
7
- License-File: LICENSE
8
- Keywords: activation-patching,interpretability,mechanistic-interpretability,transformers
9
- Classifier: Development Status :: 4 - Beta
10
- Classifier: Intended Audience :: Science/Research
11
- Classifier: Operating System :: OS Independent
12
- Classifier: Programming Language :: Python :: 3
13
- Classifier: Programming Language :: Python :: 3.11
14
- Classifier: Programming Language :: Python :: 3.12
15
- Classifier: Programming Language :: Python :: 3.13
16
- Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
- Requires-Python: >=3.11
18
- Requires-Dist: anyio>=4
19
- Requires-Dist: fastapi>=0.115
20
- Requires-Dist: huggingface-hub>=0.23
21
- Requires-Dist: numpy>=1.26
22
- Requires-Dist: platformdirs>=4
23
- Requires-Dist: psutil>=5.9
24
- Requires-Dist: pyarrow>=15
25
- Requires-Dist: pydantic>=2.7
26
- Requires-Dist: torch>=2.4
27
- Requires-Dist: transformer-lens<5,>=4.0
28
- Requires-Dist: transformers>=4.45
29
- Requires-Dist: typer>=0.12
30
- Requires-Dist: uvicorn>=0.30
31
- Requires-Dist: websockets>=12
32
- Description-Content-Type: text/markdown
33
-
34
1
  # Logogram
35
2
 
36
3
  A local workbench for causal experiments inside language models.
@@ -58,9 +25,10 @@ Logogram needs Python 3.11 or newer and [uv](https://docs.astral.sh/uv/).
58
25
  uv tool install logogram
59
26
  ```
60
27
 
61
- Until Logogram is published on PyPI, install it from a clone of this repository with
62
- `uv tool install .` instead. The web app is prebuilt inside the package, so you never need Node.
63
- The commands in the table below work the same way with `.` in place of `logogram`.
28
+ To run the code in a clone of this repository instead, install it with `uv tool install .`; the
29
+ commands in the table below work the same way with `.` in place of `logogram`. The web app is
30
+ prebuilt inside the package, so you never need Node. What changed in each version is in
31
+ [CHANGELOG.md](CHANGELOG.md).
64
32
 
65
33
  PyTorch is chosen per machine:
66
34
 
@@ -68,7 +36,7 @@ PyTorch is chosen per machine:
68
36
  |---|---|---|
69
37
  | Linux with an NVIDIA GPU | `uv tool install logogram` | The default Linux wheels include CUDA. |
70
38
  | Windows with an NVIDIA GPU | `uv tool install --torch-backend=auto logogram` | Picks the CUDA build that matches your driver. |
71
- | Apple Silicon | `uv tool install logogram` | Uses Metal (MPS). Use a native arm64 Python, not one under Rosetta. |
39
+ | Apple Silicon | `uv tool install logogram` | Runs on the CPU unless you choose MPS (see Models). Use a native arm64 Python, not one under Rosetta. |
72
40
  | CPU only | `uv tool install --torch-backend=cpu logogram` | Smaller download. GPT-2 small runs well on a CPU. |
73
41
 
74
42
  Then check the setup:
@@ -109,8 +77,9 @@ This starts a local server on 127.0.0.1, prints its address and opens your brows
109
77
  4. Press **A** on a head to see its attention pattern, or right-click any cell for **Patch here**,
110
78
  **Ablate here** and **Compare across runs**.
111
79
  5. Press **Check robustness** to rerun the sweep with a different baseline, direction or donor
112
- count. Logogram reports the rank correlation, the overlap of the top components and the
113
- components whose conclusion changed, and flags them on the map.
80
+ count, or, for a run in float16 or bfloat16, in float32. Logogram reports the rank
81
+ correlation, the overlap of the top components and the components whose conclusion changed,
82
+ and flags them on the map.
114
83
 
115
84
  Keyboard: arrow keys move across the map, **Ctrl/⌘ K** opens the command palette (type `L9H9` to
116
85
  jump to a head), **1–7** switch views, **[** and **]** step through prompts, **P**, **B**, **A**
@@ -262,7 +231,11 @@ estimate misses saturation (in attention, normalization and the final softmax) a
262
231
  even invert an effect, so **Verify top 10 by patching** on the results page patches the sites
263
232
  with the largest estimated effects for real and opens the comparison: the rank correlation, and
264
233
  any estimate whose sign patching confidently reverses. **Check robustness** can also patch the
265
- whole sweep.
234
+ whole sweep. Do that for residual stream sweeps: an estimate is least reliable where patching
235
+ replaces a whole token's representation. In the IOI example on GPT-2 small, patching the residual
236
+ stream at the changed name in the first layer restores the whole answer while its estimate is
237
+ slightly negative, so verifying the top estimates would never reach it. Head outputs track
238
+ patching closely.
266
239
 
267
240
  **Direct logit attribution** splits the logit difference of the clean or the corrupt prompts
268
241
  (your choice; there is no default) into what each head, attention output and MLP output writes
@@ -299,7 +272,10 @@ like patching, so 1 means the steered prompts moved as far as switching to the o
299
272
  random direction of the same length, at the same strengths, runs alongside as a control, and the
300
273
  results show both: a layer × strength map with the control columns muted, and, for a selected site,
301
274
  its effect at every strength with intervals. **Check robustness** offers another split of the pairs
302
- or the other direction.
275
+ or the other direction. A mean difference steers only when the pairs differ the same way. In IOI
276
+ prompts that mix the ABBA and BABA orders, the difference at the last token flips with the order,
277
+ so the mean cancels and steering does no more than the control; generate prompts of one order to
278
+ steer. When no site and strength does more than the control, the results say so.
303
279
 
304
280
  **SAE features.** A sparse autoencoder (SAE) rewrites one of the model's activations as a few active
305
281
  features out of thousands, each a direction in the model, plus an error it misses. In
@@ -368,7 +344,7 @@ Every experiment is a `spec.json`. Nothing that can change a number is left impl
368
344
  | Field | Values |
369
345
  |---|---|
370
346
  | `model.revision` | A commit. `null` resolves the current main branch at run time; the saved spec pins what ran. |
371
- | `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change; zero ablation of head outputs does, because value biases are folded. |
347
+ | `model.process_weights` | Fold LayerNorm and center weights, as TransformerLens does by default. Logit differences don't change. Value biases are folded into the attention output's bias, so zero ablation of head outputs and heads' direct effects do change (by up to several logits in Qwen 2.5, whose value biases are large); a layer's attention output doesn't. |
372
348
  | `dataset.sha256` | If set, the run refuses a dataset file that has changed. |
373
349
  | `experiment` | `{"kind": "activation_patching", "direction": "clean_to_corrupt" \| "corrupt_to_clean"}`, `{"kind": "ablation", "baseline": …}` with `{"kind": "zero"}`, `{"kind": "mean", "reference": "clean" \| "corrupt"}` or `{"kind": "resample", "pool": "clean" \| "corrupt", "donors": 10, "seed": 0}`, `{"kind": "attribution_patching", "direction": …}` (same directions as patching), `{"kind": "direct_logit_attribution", "prompts": "clean" \| "corrupt"}`, `{"kind": "path_patching", "direction": …, "receivers": [{"kind": "head", "layer": 9, "head": 9, "input": "q" \| "k" \| "v"}, {"kind": "logits"}], "freeze_mlps": false}`, or `{"kind": "steering", "apply_to": "clean" \| "corrupt", "coefficients": [-1, 1, 2], "train_fraction": 0.5, "seed": 0, "control": true}` with a scope of one residual component per layer (`layer_components`) or residual `sites`, at one token |
374
350
  | `scope` | `{"kind": "heads", "position": …}`, `{"kind": "layer_position", "site": "resid_pre", "positions": "each" \| "labels"}`, `{"kind": "layer_components", "components": ["attn_out", "mlp_out"], "position": …}` `{"kind": "sites", "sites": [{"kind": "head", "layer": 9, "head": 9, "position": {"kind": "label", "label": "end"}}]}` (feature sites are `{"kind": "sae_feature", "layer": 8, "feature": 1234, "position": …}`), or, for attribution patching, `{"kind": "features", "position": …, "top": 50}` |
@@ -468,12 +444,27 @@ TransformerLens 4 loads well over a hundred architectures, and Logogram works wi
468
444
  decoder-only language models among them: GPT-2, Pythia, Llama, Mistral, SmolLM, Qwen, Gemma,
469
445
  OLMo, Phi and more. The model dialog lists starting points, and any other Hugging Face id can be
470
446
  typed in. Before downloading, Logogram checks that TransformerLens supports the architecture and
471
- estimates the memory needed (weights, activations and a margin) against what is free.
447
+ estimates the memory needed (weights, activations and a margin) against what is free. Processing
448
+ the weights needs more for a moment while the model loads: TransformerLens works on float32
449
+ copies, three to four times the float32 size of the weights. The estimate includes them, so a
450
+ model that fits only without processing (Qwen 2.5 0.5B on a 6 GB GPU) says so before it loads.
451
+
452
+ Use float32 whenever the model fits. float16 and bfloat16 halve the memory and run two to three
453
+ times faster on a GPU, but they round. In the IOI example on GPT-2 small and Qwen 2.5, float16
454
+ moved effects by at most 0.003 and bfloat16 by up to 0.03: the strongest sites stayed in place,
455
+ but effects smaller than about 0.01 changed order. On Pythia-70m, bfloat16 doubled the clean
456
+ logit difference and reordered the heads, and float16 overflows. Check results that matter
457
+ against float32: **Check robustness** reruns a 16-bit run in float32 and compares the two.
458
+
459
+ On Apple Silicon, **Automatic** runs models on the CPU. TransformerLens reports that Apple's MPS
460
+ can give silently wrong results, and Logogram hasn't been checked on it yet, so MPS is used only
461
+ when you choose it under **Device**. It is faster; check results that matter on the CPU.
472
462
 
473
463
  Rather than trusting a list, Logogram checks every model when it loads, on a short fixed input:
474
464
 
475
465
  * TransformerLens's version of the model must predict what the original model predicts. If weight
476
- processing changes the predictions, loading with processed weights is refused.
466
+ processing changes the predictions, loading with processed weights is refused. So is a dtype in
467
+ which the model overflows (Pythia in float16).
477
468
  * It measures how each layer adds attention and the MLP to the residual stream: one after the
478
469
  other (sequential), or both from the same input (parallel, as in Pythia, GPT-J and Phi). Only
479
470
  sequential layers have a residual stream between attention and MLP (`resid_mid`), and the layer
@@ -481,11 +472,15 @@ Rather than trusting a list, Logogram checks every model when it loads, on a sho
481
472
  * Layer predictions are offered when the final normalization and unembedding, with any logit
482
473
  soft-capping, reproduce the model's output.
483
474
 
484
- GPT-2 small is the model the bundled example was made for, and the one run end to end with real
485
- weights. The test suite runs the sanity checks on tiny random models of the Llama, Pythia, Qwen 2,
486
- Gemma 2 and OLMo 2 families without downloading anything. Pythia publishes checkpoints from
487
- throughout training as revisions (`step1000` to `step143000`), so an experiment can be rerun at
488
- several points of training.
475
+ GPT-2 small is the model the bundled example was made for. Every method has been run end to end
476
+ with real weights on GPT-2 small (with a SAELens and an OpenAI SAE), Pythia-70m (with an
477
+ EleutherAI SAE) and Qwen 2.5 0.5B, and the test suite runs the sanity checks on tiny random models
478
+ of the Llama, Pythia, Qwen 2, Gemma 2 and OLMo 2 families without downloading anything. The
479
+ example's names are single tokens for GPT-2 and Qwen but not all for Pythia; for another model,
480
+ generate IOI prompts in the app, which keeps only names that are single tokens for it. A small
481
+ model may not do the task at all (Pythia-70m prefers the repeated name), and its runs then say so.
482
+ Pythia publishes checkpoints from throughout training as revisions (`step1000` to `step143000`),
483
+ so an experiment can be rerun at several points of training.
489
484
 
490
485
  Some tokenizers, such as Qwen's, have no beginning-of-sequence token. Logogram then runs prompts
491
486
  without one, and the spec records it. Answers and distractors must still be single tokens.
@@ -534,10 +529,18 @@ push a tag with the same version, such as `v0.1.0`. The Release workflow runs ev
534
529
  that commit, builds the wheel and source distribution, and uploads them to PyPI through Trusted
535
530
  Publishing, so no token is stored anywhere. PyPI never accepts the same version twice.
536
531
 
537
- Before tagging, manually check a first GPT-2 download, cancel and retry it, then run the example
538
- on each supported compute backend. CPU and CUDA are covered by local development checks; MPS
539
- still needs a check on Apple Silicon. Other model families are checked when they load and in the
540
- tiny-model tests, but not yet run end to end with their real weights.
532
+ Before tagging, manually check a first GPT-2 download, cancel and retry it, then run
533
+ `scripts/validate_real_weights.py` on each supported compute backend. It runs every method on a
534
+ real model in a temporary project and checks exact identities, agreement between methods and
535
+ bit-identical reruns; `--compare-device` reruns the head sweep on a second device, and `--sae`
536
+ adds SAE features. CPU and CUDA pass it. Apple's MPS still needs it run on a Mac:
537
+
538
+ ```bash
539
+ uv run python scripts/validate_real_weights.py --device mps --compare-device cpu
540
+ ```
541
+
542
+ Pythia and Qwen 2.5 have also been run end to end with their real weights; other families are
543
+ checked when they load and in the tiny-model tests.
541
544
 
542
545
  Model access goes through `logogram.backends.base.ModelBackend`. TransformerLens is the only
543
546
  backend today; remote execution and other libraries can be added behind the same interface.
@@ -20,6 +20,7 @@ classifiers = [
20
20
  "Programming Language :: Python :: 3.11",
21
21
  "Programming Language :: Python :: 3.12",
22
22
  "Programming Language :: Python :: 3.13",
23
+ "Programming Language :: Python :: 3.14",
23
24
  "Topic :: Scientific/Engineering :: Artificial Intelligence",
24
25
  ]
25
26
  dependencies = [
@@ -39,13 +40,19 @@ dependencies = [
39
40
  "huggingface-hub>=0.23",
40
41
  ]
41
42
 
43
+ [project.urls]
44
+ Homepage = "https://github.com/Jeevash23/logogram"
45
+ Source = "https://github.com/Jeevash23/logogram"
46
+ Issues = "https://github.com/Jeevash23/logogram/issues"
47
+ Changelog = "https://github.com/Jeevash23/logogram/blob/main/CHANGELOG.md"
48
+
42
49
  [project.scripts]
43
50
  logogram = "logogram.cli:main"
44
51
 
45
52
  [dependency-groups]
46
53
  dev = [
47
54
  "pytest>=8",
48
- "httpx>=0.27",
55
+ "httpx2>=2.13",
49
56
  "ruff==0.16.10",
50
57
  ]
51
58
 
@@ -58,7 +65,7 @@ packages = ["src/logogram"]
58
65
  artifacts = ["src/logogram/web_dist/**"]
59
66
 
60
67
  [tool.hatch.build.targets.sdist]
61
- include = ["src/logogram", "web/src", "web/public", "web/tests", "web/package.json", "web/package-lock.json", "web/index.html", "web/tsconfig.json", "web/vite.config.ts", "web/playwright.config.ts", "uv.lock", "tests", "scripts", "README.md", "LICENSE", "AGENTS.md"]
68
+ include = ["src/logogram", "web/src", "web/public", "web/tests", "web/package.json", "web/package-lock.json", "web/index.html", "web/tsconfig.json", "web/vite.config.ts", "web/playwright.config.ts", "uv.lock", "tests", "scripts", "README.md", "CHANGELOG.md", "LICENSE", "AGENTS.md"]
62
69
  artifacts = ["src/logogram/web_dist/**"]
63
70
 
64
71
  [tool.pytest.ini_options]