glyph-discovery 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- glyph_discovery-0.1.0/LICENSE +21 -0
- glyph_discovery-0.1.0/PKG-INFO +250 -0
- glyph_discovery-0.1.0/README.md +199 -0
- glyph_discovery-0.1.0/glyph/__init__.py +12 -0
- glyph_discovery-0.1.0/glyph/conf/autointerp/default.yaml +75 -0
- glyph_discovery-0.1.0/glyph/conf/autointerp/llm/gemini_api.yaml +6 -0
- glyph_discovery-0.1.0/glyph/conf/autointerp/llm/vertex.yaml +7 -0
- glyph_discovery-0.1.0/glyph/conf/config.yaml +17 -0
- glyph_discovery-0.1.0/glyph/conf/data/default.yaml +33 -0
- glyph_discovery-0.1.0/glyph/conf/dev.yaml.template +49 -0
- glyph_discovery-0.1.0/glyph/conf/logging/none.yaml +6 -0
- glyph_discovery-0.1.0/glyph/conf/logging/wandb.yaml +6 -0
- glyph_discovery-0.1.0/glyph/conf/metrics/default.yaml +13 -0
- glyph_discovery-0.1.0/glyph/conf/phewas/default.yaml +6 -0
- glyph_discovery-0.1.0/glyph/conf/sae/default.yaml +12 -0
- glyph_discovery-0.1.0/glyph/conf/user.yaml.template +35 -0
- glyph_discovery-0.1.0/glyph/conf/visualization/default.yaml +5 -0
- glyph_discovery-0.1.0/glyph/config/__init__.py +15 -0
- glyph_discovery-0.1.0/glyph/config/config.py +87 -0
- glyph_discovery-0.1.0/glyph/config/default_configs.py +5 -0
- glyph_discovery-0.1.0/glyph/core/__init__.py +5 -0
- glyph_discovery-0.1.0/glyph/core/glyph.py +896 -0
- glyph_discovery-0.1.0/glyph/dataloader/__init__.py +10 -0
- glyph_discovery-0.1.0/glyph/dataloader/dataloader.py +112 -0
- glyph_discovery-0.1.0/glyph/dataloader/dataset.py +255 -0
- glyph_discovery-0.1.0/glyph/interpretation/__init__.py +5 -0
- glyph_discovery-0.1.0/glyph/interpretation/autointerp.py +600 -0
- glyph_discovery-0.1.0/glyph/mechinterp/__init__.py +6 -0
- glyph_discovery-0.1.0/glyph/mechinterp/sae.py +285 -0
- glyph_discovery-0.1.0/glyph/mechinterp/sae_trainer.py +257 -0
- glyph_discovery-0.1.0/glyph/metrics/linear_probe.py +345 -0
- glyph_discovery-0.1.0/glyph/metrics/statistics.py +54 -0
- glyph_discovery-0.1.0/glyph/models/__init__.py +15 -0
- glyph_discovery-0.1.0/glyph/models/base.py +59 -0
- glyph_discovery-0.1.0/glyph/models/embedding_extractor.py +270 -0
- glyph_discovery-0.1.0/glyph/models/image/__init__.py +5 -0
- glyph_discovery-0.1.0/glyph/models/image/resnet18.py +120 -0
- glyph_discovery-0.1.0/glyph/models/image/siglip_embed.py +24 -0
- glyph_discovery-0.1.0/glyph/models/precomputed.py +40 -0
- glyph_discovery-0.1.0/glyph/models/registry.py +112 -0
- glyph_discovery-0.1.0/glyph/phewas/__init__.py +15 -0
- glyph_discovery-0.1.0/glyph/phewas/phewas.py +305 -0
- glyph_discovery-0.1.0/glyph/phewas/statistics.py +135 -0
- glyph_discovery-0.1.0/glyph/utils/__init__.py +19 -0
- glyph_discovery-0.1.0/glyph/utils/device.py +85 -0
- glyph_discovery-0.1.0/glyph/utils/io.py +98 -0
- glyph_discovery-0.1.0/glyph/visualization/__init__.py +2 -0
- glyph_discovery-0.1.0/glyph/visualization/concept_network.py +244 -0
- glyph_discovery-0.1.0/glyph/visualization/feature_lollipop.py +114 -0
- glyph_discovery-0.1.0/glyph_discovery.egg-info/PKG-INFO +250 -0
- glyph_discovery-0.1.0/glyph_discovery.egg-info/SOURCES.txt +62 -0
- glyph_discovery-0.1.0/glyph_discovery.egg-info/dependency_links.txt +1 -0
- glyph_discovery-0.1.0/glyph_discovery.egg-info/requires.txt +31 -0
- glyph_discovery-0.1.0/glyph_discovery.egg-info/top_level.txt +1 -0
- glyph_discovery-0.1.0/pyproject.toml +76 -0
- glyph_discovery-0.1.0/setup.cfg +4 -0
- glyph_discovery-0.1.0/tests/test1_config.py +236 -0
- glyph_discovery-0.1.0/tests/test2_metrics.py +110 -0
- glyph_discovery-0.1.0/tests/test3_data.py +310 -0
- glyph_discovery-0.1.0/tests/test4_embedding_extraction.py +342 -0
- glyph_discovery-0.1.0/tests/test5_outcome_types.py +343 -0
- glyph_discovery-0.1.0/tests/test6_sae.py +188 -0
- glyph_discovery-0.1.0/tests/test7_phewas.py +205 -0
- glyph_discovery-0.1.0/tests/test_autointerp.py +391 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025-2026 Robbie Holland
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: glyph-discovery
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Domain-agnostic hypothesis generation using sparse autoencoders
|
|
5
|
+
Author-email: Robbie Holland <robbie.holland@stanford.edu>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/RobbieHolland/Glyph
|
|
8
|
+
Project-URL: Repository, https://github.com/RobbieHolland/Glyph
|
|
9
|
+
Project-URL: Paper, https://openreview.net/forum?id=rgpgukbeVf
|
|
10
|
+
Keywords: sparse autoencoders,mechanistic interpretability,hypothesis generation,phewas,medical imaging
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: torch>=2.0.0
|
|
23
|
+
Requires-Dist: torchvision>=0.15.0
|
|
24
|
+
Requires-Dist: pytorch-lightning>=2.0.0
|
|
25
|
+
Requires-Dist: numpy>=1.24.0
|
|
26
|
+
Requires-Dist: pandas>=2.0.0
|
|
27
|
+
Requires-Dist: scikit-learn>=1.3.0
|
|
28
|
+
Requires-Dist: scipy>=1.10.0
|
|
29
|
+
Requires-Dist: statsmodels>=0.14.0
|
|
30
|
+
Requires-Dist: matplotlib>=3.7.0
|
|
31
|
+
Requires-Dist: networkx>=3.0
|
|
32
|
+
Requires-Dist: umap-learn>=0.5.4
|
|
33
|
+
Requires-Dist: pillow>=9.5.0
|
|
34
|
+
Requires-Dist: pyarrow>=12.0.0
|
|
35
|
+
Requires-Dist: hydra-core>=1.3.0
|
|
36
|
+
Requires-Dist: omegaconf>=2.3.0
|
|
37
|
+
Requires-Dist: pyyaml>=6.0
|
|
38
|
+
Requires-Dist: tqdm>=4.65.0
|
|
39
|
+
Provides-Extra: autointerp
|
|
40
|
+
Requires-Dist: google-generativeai>=0.8.0; extra == "autointerp"
|
|
41
|
+
Requires-Dist: google-cloud-aiplatform>=1.60.0; extra == "autointerp"
|
|
42
|
+
Provides-Extra: tracking
|
|
43
|
+
Requires-Dist: wandb>=0.15.0; extra == "tracking"
|
|
44
|
+
Provides-Extra: dev
|
|
45
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
46
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
47
|
+
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
48
|
+
Requires-Dist: flake8>=6.0.0; extra == "dev"
|
|
49
|
+
Requires-Dist: mypy>=1.0.0; extra == "dev"
|
|
50
|
+
Dynamic: license-file
|
|
51
|
+
|
|
52
|
+
# Glyph: Mechanistic Science
|
|
53
|
+
|
|
54
|
+
**Domain-Agnostic Hypothesis Generation using Sparse Autoencoders**
|
|
55
|
+
|
|
56
|
+
Glyph is a Python toolkit for discovering interpretable features and generating scientific hypotheses from multimodal data. It uses Sparse Autoencoders (SAEs) to learn interpretable features, then conducts Phenome-Wide Association Studies (PheWAS) to identify and rank hypotheses.
|
|
57
|
+
|
|
58
|
+
📄 **Paper**: [MechSci: Scaling Clinical Science via Mechanistic Interpretability of Multimodal Medical Foundation Models](https://openreview.net/forum?id=rgpgukbeVf#discussion) — *Agents4Science 2025*
|
|
59
|
+
|
|
60
|
+
[](https://www.youtube.com/watch?v=s3t-Nu8HNFc)
|
|
61
|
+
*Click to watch the video presentation*
|
|
62
|
+
|
|
63
|
+
## Features
|
|
64
|
+
|
|
65
|
+
- **Sparse Autoencoders** with TopK activation and Matryoshka nested dictionaries
|
|
66
|
+
- **Ghost Gradient Recovery** to prevent dead features during training
|
|
67
|
+
- **PheWAS Analysis** for systematic hypothesis testing
|
|
68
|
+
- **Statistical Rigor** with odds ratios, confidence intervals, and AUC metrics
|
|
69
|
+
- **GPU Acceleration** for large-scale analysis
|
|
70
|
+
- **Domain Agnostic** - works with any structured data (images, tabular, etc.)
|
|
71
|
+
|
|
72
|
+
## Installation
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
pip install glyph-discovery
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Optional extras:
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
pip install "glyph-discovery[autointerp]" # LLM-based feature interpretation
|
|
82
|
+
pip install "glyph-discovery[tracking]" # Weights & Biases logging
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
To work on Glyph itself:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
git clone https://github.com/RobbieHolland/Glyph.git
|
|
89
|
+
cd Glyph
|
|
90
|
+
pip install -e ".[dev]"
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Quick Start
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
from glyph import Glyph
|
|
97
|
+
|
|
98
|
+
# Define your configuration overrides
|
|
99
|
+
config = [
|
|
100
|
+
"data.metadata=/path/to/metadata.csv",
|
|
101
|
+
"data.input_cols=[image_path]",
|
|
102
|
+
"data.outcome_cols=[disease]",
|
|
103
|
+
"data.embedding_map={image_path: resnet18}",
|
|
104
|
+
"data.cache_dir=./cache",
|
|
105
|
+
"data.output_dir=./outputs",
|
|
106
|
+
"seed=42",
|
|
107
|
+
]
|
|
108
|
+
|
|
109
|
+
# Initialize Glyph
|
|
110
|
+
ms = Glyph(config_path='config', config_overrides=config)
|
|
111
|
+
|
|
112
|
+
# Option 1: Run full pipeline
|
|
113
|
+
results = ms() # Equivalent to ms.fit().hypothesis_search()
|
|
114
|
+
|
|
115
|
+
# Option 2: Run in stages
|
|
116
|
+
ms.fit() # Extract embeddings, train SAE, compute linear probes
|
|
117
|
+
ms.hypothesis_search() # Run PheWAS study
|
|
118
|
+
|
|
119
|
+
# Access results directly
|
|
120
|
+
print(f"Embeddings: {list(ms.embeddings.keys())}")
|
|
121
|
+
print(f"SAE models: {list(ms.sae_models.keys())}")
|
|
122
|
+
print(f"PheWAS hypotheses: {len(ms.phewas_results)}")
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
## Pipeline Overview
|
|
126
|
+
|
|
127
|
+
```
|
|
128
|
+
Raw Data → Embeddings → SAE Training → Sparse Features → PheWAS → Ranked Hypotheses
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
### `ms.fit()` - Steps 1-4
|
|
132
|
+
|
|
133
|
+
1. **Embedding Extraction**: Extract dense representations using pretrained models (e.g., ResNet18)
|
|
134
|
+
2. **Linear Probe**: Evaluate embedding quality for outcome prediction (AUC)
|
|
135
|
+
3. **SAE Training**: Learn sparse, interpretable features with TopK activation
|
|
136
|
+
4. **SAE Linear Probe**: Evaluate sparse feature quality for outcome prediction
|
|
137
|
+
|
|
138
|
+
### `ms.hypothesis_search()` - Step 5
|
|
139
|
+
|
|
140
|
+
5. **PheWAS Study**: Test all sparse feature-outcome associations
|
|
141
|
+
- Computes odds ratios, confidence intervals, and AUC for each feature
|
|
142
|
+
- Filters by minimum activation count for statistical reliability
|
|
143
|
+
- Ranks hypotheses by odds ratio
|
|
144
|
+
|
|
145
|
+
## Example Output
|
|
146
|
+
|
|
147
|
+
```
|
|
148
|
+
[PheWAS Results]
|
|
149
|
+
Total hypotheses: 162
|
|
150
|
+
|
|
151
|
+
Outcome class balance (train):
|
|
152
|
+
outcome: 2436/3295 positive (73.9%), 859/3295 negative (26.1%)
|
|
153
|
+
|
|
154
|
+
Top 10 hypotheses (by train odds ratio):
|
|
155
|
+
──────────────────────────────────────────────────────────────────────────────────────────
|
|
156
|
+
Feature Train OR Test OR AUC % AUC Count
|
|
157
|
+
──────────────────────────────────────────────────────────────────────────────────────────
|
|
158
|
+
feature_98 3.75 3.96 0.796 60.0% 2818
|
|
159
|
+
feature_61 3.45 3.34 0.773 55.2% 2524
|
|
160
|
+
feature_140 2.86 2.64 0.614 23.2% 674
|
|
161
|
+
...
|
|
162
|
+
──────────────────────────────────────────────────────────────────────────────────────────
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
- **Train/Test OR**: Odds ratio (how much the feature increases disease odds)
|
|
166
|
+
- **AUC**: Predictive power of this single feature
|
|
167
|
+
- **% AUC**: `(feature_auc - 0.5) / (embedding_auc - 0.5)` - what % of embedding's predictive power this feature captures
|
|
168
|
+
- **Count**: Number of samples where feature activates
|
|
169
|
+
|
|
170
|
+
## Configuration
|
|
171
|
+
|
|
172
|
+
Glyph uses [Hydra](https://hydra.cc/) for configuration management.
|
|
173
|
+
|
|
174
|
+
### Key Configuration Options
|
|
175
|
+
|
|
176
|
+
```yaml
|
|
177
|
+
# Data settings
|
|
178
|
+
data:
|
|
179
|
+
metadata: /path/to/metadata.csv # CSV with sample IDs and file paths
|
|
180
|
+
input_cols: [image_path] # Columns containing input data paths
|
|
181
|
+
outcome_cols: [disease] # Columns containing outcomes to predict
|
|
182
|
+
embedding_map: {image_path: resnet18} # Map input columns to embedding models
|
|
183
|
+
cache_dir: ./cache # Cache directory for embeddings
|
|
184
|
+
output_dir: ./outputs # Output directory for results
|
|
185
|
+
|
|
186
|
+
# SAE settings
|
|
187
|
+
sae:
|
|
188
|
+
top_ks: [20] # TopK sparsity values to try
|
|
189
|
+
matryoshka: [2048] # Dictionary sizes (Matryoshka nesting)
|
|
190
|
+
max_steps: 5000 # Training steps
|
|
191
|
+
learning_rate: 0.0003 # Learning rate
|
|
192
|
+
|
|
193
|
+
# PheWAS settings
|
|
194
|
+
phewas:
|
|
195
|
+
min_activations: 25 # Minimum feature activations for hypothesis
|
|
196
|
+
top_n_print: 10 # Number of top hypotheses to display
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
## Project Status
|
|
200
|
+
|
|
201
|
+
✅ **Core Pipeline Complete**
|
|
202
|
+
|
|
203
|
+
- [x] Phase 1: Project structure
|
|
204
|
+
- [x] Phase 2: Data loading (GlyphDataset, DataLoader)
|
|
205
|
+
- [x] Phase 3: Embedding extraction (ResNet18, caching)
|
|
206
|
+
- [x] Phase 4: SAE training (TopK, Matryoshka loss, ghost gradients)
|
|
207
|
+
- [x] Phase 5: Linear probe (binary classification, regression)
|
|
208
|
+
- [x] Phase 6: PheWAS analysis (odds ratios, AUC, ranking)
|
|
209
|
+
- [x] Phase 7: Main Glyph class with fit/hypothesis_search API
|
|
210
|
+
- [x] Phase 8: Examples
|
|
211
|
+
- [x] Phase 9: Unit tests (83 tests passing)
|
|
212
|
+
- [ ] Phase 10: AutoInterp (interpretation with LLMs)
|
|
213
|
+
- [ ] Phase 11: Documentation
|
|
214
|
+
|
|
215
|
+
## Running Tests
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
# Run all tests
|
|
219
|
+
python -m pytest tests/test*.py -v
|
|
220
|
+
|
|
221
|
+
# Run specific test module
|
|
222
|
+
python -m pytest tests/test6_sae.py -v
|
|
223
|
+
python -m pytest tests/test7_phewas.py -v
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
## Inspiration
|
|
227
|
+
|
|
228
|
+
Glyph distills core algorithms from a larger medical hypothesis generation research codebase. Key innovations preserved:
|
|
229
|
+
|
|
230
|
+
- **Matryoshka SAEs** for multi-resolution feature learning
|
|
231
|
+
- **Ghost Gradients** for dead feature recovery
|
|
232
|
+
- **TopK Activation** for interpretable sparse codes
|
|
233
|
+
- **Rigorous Statistics** with odds ratios and confidence intervals
|
|
234
|
+
|
|
235
|
+
## License
|
|
236
|
+
|
|
237
|
+
MIT License - see [LICENSE](LICENSE) for details.
|
|
238
|
+
|
|
239
|
+
## Citation
|
|
240
|
+
|
|
241
|
+
If you use Glyph in your research, please cite:
|
|
242
|
+
|
|
243
|
+
```bibtex
|
|
244
|
+
@software{glyph2025,
|
|
245
|
+
title = {Glyph: Domain-Agnostic Hypothesis Generation},
|
|
246
|
+
author = {Holland, Robbie},
|
|
247
|
+
year = {2025},
|
|
248
|
+
url = {https://github.com/RobbieHolland/Glyph}
|
|
249
|
+
}
|
|
250
|
+
```
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
# Glyph: Mechanistic Science
|
|
2
|
+
|
|
3
|
+
**Domain-Agnostic Hypothesis Generation using Sparse Autoencoders**
|
|
4
|
+
|
|
5
|
+
Glyph is a Python toolkit for discovering interpretable features and generating scientific hypotheses from multimodal data. It uses Sparse Autoencoders (SAEs) to learn interpretable features, then conducts Phenome-Wide Association Studies (PheWAS) to identify and rank hypotheses.
|
|
6
|
+
|
|
7
|
+
📄 **Paper**: [MechSci: Scaling Clinical Science via Mechanistic Interpretability of Multimodal Medical Foundation Models](https://openreview.net/forum?id=rgpgukbeVf#discussion) — *Agents4Science 2025*
|
|
8
|
+
|
|
9
|
+
[](https://www.youtube.com/watch?v=s3t-Nu8HNFc)
|
|
10
|
+
*Click to watch the video presentation*
|
|
11
|
+
|
|
12
|
+
## Features
|
|
13
|
+
|
|
14
|
+
- **Sparse Autoencoders** with TopK activation and Matryoshka nested dictionaries
|
|
15
|
+
- **Ghost Gradient Recovery** to prevent dead features during training
|
|
16
|
+
- **PheWAS Analysis** for systematic hypothesis testing
|
|
17
|
+
- **Statistical Rigor** with odds ratios, confidence intervals, and AUC metrics
|
|
18
|
+
- **GPU Acceleration** for large-scale analysis
|
|
19
|
+
- **Domain Agnostic** - works with any structured data (images, tabular, etc.)
|
|
20
|
+
|
|
21
|
+
## Installation
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install glyph-discovery
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Optional extras:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install "glyph-discovery[autointerp]" # LLM-based feature interpretation
|
|
31
|
+
pip install "glyph-discovery[tracking]" # Weights & Biases logging
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
To work on Glyph itself:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
git clone https://github.com/RobbieHolland/Glyph.git
|
|
38
|
+
cd Glyph
|
|
39
|
+
pip install -e ".[dev]"
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Quick Start
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
from glyph import Glyph
|
|
46
|
+
|
|
47
|
+
# Define your configuration overrides
|
|
48
|
+
config = [
|
|
49
|
+
"data.metadata=/path/to/metadata.csv",
|
|
50
|
+
"data.input_cols=[image_path]",
|
|
51
|
+
"data.outcome_cols=[disease]",
|
|
52
|
+
"data.embedding_map={image_path: resnet18}",
|
|
53
|
+
"data.cache_dir=./cache",
|
|
54
|
+
"data.output_dir=./outputs",
|
|
55
|
+
"seed=42",
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
# Initialize Glyph
|
|
59
|
+
ms = Glyph(config_path='config', config_overrides=config)
|
|
60
|
+
|
|
61
|
+
# Option 1: Run full pipeline
|
|
62
|
+
results = ms() # Equivalent to ms.fit().hypothesis_search()
|
|
63
|
+
|
|
64
|
+
# Option 2: Run in stages
|
|
65
|
+
ms.fit() # Extract embeddings, train SAE, compute linear probes
|
|
66
|
+
ms.hypothesis_search() # Run PheWAS study
|
|
67
|
+
|
|
68
|
+
# Access results directly
|
|
69
|
+
print(f"Embeddings: {list(ms.embeddings.keys())}")
|
|
70
|
+
print(f"SAE models: {list(ms.sae_models.keys())}")
|
|
71
|
+
print(f"PheWAS hypotheses: {len(ms.phewas_results)}")
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Pipeline Overview
|
|
75
|
+
|
|
76
|
+
```
|
|
77
|
+
Raw Data → Embeddings → SAE Training → Sparse Features → PheWAS → Ranked Hypotheses
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### `ms.fit()` - Steps 1-4
|
|
81
|
+
|
|
82
|
+
1. **Embedding Extraction**: Extract dense representations using pretrained models (e.g., ResNet18)
|
|
83
|
+
2. **Linear Probe**: Evaluate embedding quality for outcome prediction (AUC)
|
|
84
|
+
3. **SAE Training**: Learn sparse, interpretable features with TopK activation
|
|
85
|
+
4. **SAE Linear Probe**: Evaluate sparse feature quality for outcome prediction
|
|
86
|
+
|
|
87
|
+
### `ms.hypothesis_search()` - Step 5
|
|
88
|
+
|
|
89
|
+
5. **PheWAS Study**: Test all sparse feature-outcome associations
|
|
90
|
+
- Computes odds ratios, confidence intervals, and AUC for each feature
|
|
91
|
+
- Filters by minimum activation count for statistical reliability
|
|
92
|
+
- Ranks hypotheses by odds ratio
|
|
93
|
+
|
|
94
|
+
## Example Output
|
|
95
|
+
|
|
96
|
+
```
|
|
97
|
+
[PheWAS Results]
|
|
98
|
+
Total hypotheses: 162
|
|
99
|
+
|
|
100
|
+
Outcome class balance (train):
|
|
101
|
+
outcome: 2436/3295 positive (73.9%), 859/3295 negative (26.1%)
|
|
102
|
+
|
|
103
|
+
Top 10 hypotheses (by train odds ratio):
|
|
104
|
+
──────────────────────────────────────────────────────────────────────────────────────────
|
|
105
|
+
Feature Train OR Test OR AUC % AUC Count
|
|
106
|
+
──────────────────────────────────────────────────────────────────────────────────────────
|
|
107
|
+
feature_98 3.75 3.96 0.796 60.0% 2818
|
|
108
|
+
feature_61 3.45 3.34 0.773 55.2% 2524
|
|
109
|
+
feature_140 2.86 2.64 0.614 23.2% 674
|
|
110
|
+
...
|
|
111
|
+
──────────────────────────────────────────────────────────────────────────────────────────
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
- **Train/Test OR**: Odds ratio (how much the feature increases disease odds)
|
|
115
|
+
- **AUC**: Predictive power of this single feature
|
|
116
|
+
- **% AUC**: `(feature_auc - 0.5) / (embedding_auc - 0.5)` - what % of embedding's predictive power this feature captures
|
|
117
|
+
- **Count**: Number of samples where feature activates
|
|
118
|
+
|
|
119
|
+
## Configuration
|
|
120
|
+
|
|
121
|
+
Glyph uses [Hydra](https://hydra.cc/) for configuration management.
|
|
122
|
+
|
|
123
|
+
### Key Configuration Options
|
|
124
|
+
|
|
125
|
+
```yaml
|
|
126
|
+
# Data settings
|
|
127
|
+
data:
|
|
128
|
+
metadata: /path/to/metadata.csv # CSV with sample IDs and file paths
|
|
129
|
+
input_cols: [image_path] # Columns containing input data paths
|
|
130
|
+
outcome_cols: [disease] # Columns containing outcomes to predict
|
|
131
|
+
embedding_map: {image_path: resnet18} # Map input columns to embedding models
|
|
132
|
+
cache_dir: ./cache # Cache directory for embeddings
|
|
133
|
+
output_dir: ./outputs # Output directory for results
|
|
134
|
+
|
|
135
|
+
# SAE settings
|
|
136
|
+
sae:
|
|
137
|
+
top_ks: [20] # TopK sparsity values to try
|
|
138
|
+
matryoshka: [2048] # Dictionary sizes (Matryoshka nesting)
|
|
139
|
+
max_steps: 5000 # Training steps
|
|
140
|
+
learning_rate: 0.0003 # Learning rate
|
|
141
|
+
|
|
142
|
+
# PheWAS settings
|
|
143
|
+
phewas:
|
|
144
|
+
min_activations: 25 # Minimum feature activations for hypothesis
|
|
145
|
+
top_n_print: 10 # Number of top hypotheses to display
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
## Project Status
|
|
149
|
+
|
|
150
|
+
✅ **Core Pipeline Complete**
|
|
151
|
+
|
|
152
|
+
- [x] Phase 1: Project structure
|
|
153
|
+
- [x] Phase 2: Data loading (GlyphDataset, DataLoader)
|
|
154
|
+
- [x] Phase 3: Embedding extraction (ResNet18, caching)
|
|
155
|
+
- [x] Phase 4: SAE training (TopK, Matryoshka loss, ghost gradients)
|
|
156
|
+
- [x] Phase 5: Linear probe (binary classification, regression)
|
|
157
|
+
- [x] Phase 6: PheWAS analysis (odds ratios, AUC, ranking)
|
|
158
|
+
- [x] Phase 7: Main Glyph class with fit/hypothesis_search API
|
|
159
|
+
- [x] Phase 8: Examples
|
|
160
|
+
- [x] Phase 9: Unit tests (83 tests passing)
|
|
161
|
+
- [ ] Phase 10: AutoInterp (interpretation with LLMs)
|
|
162
|
+
- [ ] Phase 11: Documentation
|
|
163
|
+
|
|
164
|
+
## Running Tests
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
# Run all tests
|
|
168
|
+
python -m pytest tests/test*.py -v
|
|
169
|
+
|
|
170
|
+
# Run specific test module
|
|
171
|
+
python -m pytest tests/test6_sae.py -v
|
|
172
|
+
python -m pytest tests/test7_phewas.py -v
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
## Inspiration
|
|
176
|
+
|
|
177
|
+
Glyph distills core algorithms from a larger medical hypothesis generation research codebase. Key innovations preserved:
|
|
178
|
+
|
|
179
|
+
- **Matryoshka SAEs** for multi-resolution feature learning
|
|
180
|
+
- **Ghost Gradients** for dead feature recovery
|
|
181
|
+
- **TopK Activation** for interpretable sparse codes
|
|
182
|
+
- **Rigorous Statistics** with odds ratios and confidence intervals
|
|
183
|
+
|
|
184
|
+
## License
|
|
185
|
+
|
|
186
|
+
MIT License - see [LICENSE](LICENSE) for details.
|
|
187
|
+
|
|
188
|
+
## Citation
|
|
189
|
+
|
|
190
|
+
If you use Glyph in your research, please cite:
|
|
191
|
+
|
|
192
|
+
```bibtex
|
|
193
|
+
@software{glyph2025,
|
|
194
|
+
title = {Glyph: Domain-Agnostic Hypothesis Generation},
|
|
195
|
+
author = {Holland, Robbie},
|
|
196
|
+
year = {2025},
|
|
197
|
+
url = {https://github.com/RobbieHolland/Glyph}
|
|
198
|
+
}
|
|
199
|
+
```
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Glyph: Mechanistic Science - Domain-Agnostic Hypothesis Generation
|
|
3
|
+
|
|
4
|
+
A toolkit for discovering interpretable features and generating scientific hypotheses
|
|
5
|
+
from multimodal data using Sparse Autoencoders and statistical analysis.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
__version__ = "0.1.0"
|
|
9
|
+
|
|
10
|
+
from glyph.core.glyph import Glyph
|
|
11
|
+
|
|
12
|
+
__all__ = ["Glyph"]
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# @package _global_.autointerp
|
|
2
|
+
# AutoInterp configuration for LLM-based feature interpretation
|
|
3
|
+
|
|
4
|
+
defaults:
|
|
5
|
+
- llm: gemini_api
|
|
6
|
+
|
|
7
|
+
# Selection parameters
|
|
8
|
+
top_n_hypotheses: 5 # Number of top (highest OR, risk) hypotheses per outcome to interpret
|
|
9
|
+
bottom_n_hypotheses: 0 # Number of bottom (lowest OR, protective) hypotheses per outcome to interpret
|
|
10
|
+
top_pct_activating: 0.05 # Top percentage of activating samples to consider
|
|
11
|
+
n_fit_samples: 40 # Number of train samples for interpretation
|
|
12
|
+
n_test_samples: 40 # Number of test samples for validation
|
|
13
|
+
|
|
14
|
+
# Text data column in metadata (maps modality to its text column)
|
|
15
|
+
# e.g. {image_path: report} means use the 'report' column for the 'image_path' modality
|
|
16
|
+
autointerp_text_map: {image_path: report}
|
|
17
|
+
|
|
18
|
+
# Prompt templates - use {samples} and {interpretation} as placeholders
|
|
19
|
+
full_reports_prompt: |
|
|
20
|
+
=== Task ===
|
|
21
|
+
You are an expert radiologist and diagnostic data scientist. Your task is to distill the most accurate Interpretation or Description that characterizes a specific cohort of patients compared to a reference set of controls. The primary objective is to produce a simple, high-performing explanation that maximizes discriminative accuracy on unseen samples.
|
|
22
|
+
|
|
23
|
+
=== Data provided ===
|
|
24
|
+
1. REFERENCE PATIENTS (CONTROLS): A sample of patients who DO NOT possess the feature. These represent the background baseline where the characteristic in question is generally absent.
|
|
25
|
+
2. COHORT PATIENTS: Patients who strongly activate this specific feature, listed in order of activation strength.
|
|
26
|
+
|
|
27
|
+
=== Core Directives ===
|
|
28
|
+
Your goal is to define the "boundary" or "signal" that is present in the Cohort but absent or significantly different in the Reference Patients.
|
|
29
|
+
|
|
30
|
+
1. Contrastive Filtering: Focus on findings with a clear disparity in prevalence or intensity between the two groups. If a finding appears frequently in both the cohort and the reference set, it is highly unlikely to be the characterizing feature.
|
|
31
|
+
2. Concept over Literalism: Prioritize the underlying clinical or radiological concept. While you should note the linguistic style (e.g., if the reports are consistently "vague" or "specific"), do not be misled by synonyms; "renal calculus" and "kidney stone" represent the same anchor.
|
|
32
|
+
3. Discriminative Balance: Aim for a level of abstraction that is general enough to apply to the vast majority of the cohort (not just the top few examples), but specific enough that it does not apply to the reference set.
|
|
33
|
+
4. Chain of Thought Reasoning: You must begin by verbalizing your thought process. Analyze the patterns across the *entire* cohort, cross-reference them against the reference patients to identify unique signals, and verify your interpretation's accuracy against both groups to ensure it maximizes discriminative performance. You should expect to get at least 85% accuracy on the discriminative task, and test this on your own data before providing your final interpretation.
|
|
34
|
+
|
|
35
|
+
=== Logical Guidelines ===
|
|
36
|
+
- Pure Description: Focus strictly on characterizing the feature as a finding. Do not provide clinical definitions or background explanations of the pathologies themselves (e.g., avoid "This represents X, which is defined as...").
|
|
37
|
+
- Atomic Simplicity: Simple, single atomic explanations are preferred. Aim for the most concise "unit" of clinical meaning that captures the cohort's essence.
|
|
38
|
+
- The "Or" Exception: Strictly avoid "A or B" structures in general. However, if a singular unifying concept cannot be found and an "OR" is strictly necessary to achieve high discriminative performance, you may use it sparingly.
|
|
39
|
+
- The "And" Clause: You may use "and" if the feature is a conjunctive relationship between two elements that consistently appear together.
|
|
40
|
+
- Discriminative Power: Your interpretation must act as a precise rule that allows a human (or another AI) to accurately classify a mixed/shuffled pile of reports into "Cohort" and "Reference" groups.
|
|
41
|
+
|
|
42
|
+
=== Formatting Rules ===
|
|
43
|
+
You must first provide your reasoning and analysis. Once your analysis is complete, you must provide your final interpretation as the very last line of your response, starting with an asterisk:
|
|
44
|
+
* This cohort is characterized by [your specific, synthesized, and discriminative description of the feature].
|
|
45
|
+
|
|
46
|
+
The asterisk is essential for the final line. Do not include any text after the starred sentence.
|
|
47
|
+
|
|
48
|
+
=== Data ===
|
|
49
|
+
[REFERENCE PATIENTS (CONTROLS)]
|
|
50
|
+
{reference_patients}
|
|
51
|
+
|
|
52
|
+
[COHORT PATIENTS (FEATURE ACTIVATIONS)]
|
|
53
|
+
{patient_data}
|
|
54
|
+
|
|
55
|
+
discriminative_autointerp_score_prompt: |
|
|
56
|
+
Task: Distinguish samples (randomly ordered) based on those that do and do not belong to the group characterized by the group characteristics.
|
|
57
|
+
|
|
58
|
+
Medical Findings Reports:
|
|
59
|
+
{patient_data}
|
|
60
|
+
|
|
61
|
+
Group characteristics:
|
|
62
|
+
{interpretation}
|
|
63
|
+
|
|
64
|
+
Analyze the findings reports one by one, reasoning on whether or not the characteristics apply to the report. Use the following format for each report:
|
|
65
|
+
{{i}}: 1 (if belongs to the group) or 0 (if does not belong to the group). We have designed the list so that exactly half of the reports belong to the group, and half do not. Therefore, you should use the characteristics to differentiate each of the reports into two distinct groups, rather than finding perfect matches.
|
|
66
|
+
|
|
67
|
+
You need to be as accurate as possible when assigning 1s and 0s. We will score the accuracy of your response at the end.
|
|
68
|
+
|
|
69
|
+
=== Format ===:
|
|
70
|
+
It is crucial that you present your final explanation in the following format, started with an asterisk:
|
|
71
|
+
* PREDICTIONS:
|
|
72
|
+
1. 1 (if the sample belongs to the group) or 0 (if it does not belong to the group)
|
|
73
|
+
2. 1 (if the sample belongs to the group) or 0 (if it does not belong to the group)
|
|
74
|
+
...
|
|
75
|
+
{{n}}. 1 (if the sample belongs to the group) or 0 (if it does not belong to the group)
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# Glyph Main Configuration
|
|
2
|
+
# This file defines the default configuration structure and references to config groups
|
|
3
|
+
|
|
4
|
+
defaults:
|
|
5
|
+
- data: default
|
|
6
|
+
- sae: default
|
|
7
|
+
- phewas: default
|
|
8
|
+
- autointerp: default
|
|
9
|
+
- metrics: default
|
|
10
|
+
- visualization: default
|
|
11
|
+
- logging: none
|
|
12
|
+
- _self_
|
|
13
|
+
|
|
14
|
+
# Top-level settings
|
|
15
|
+
seed: 42
|
|
16
|
+
device: auto # Options: auto, cuda, cpu
|
|
17
|
+
path_to_embeddings: null
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# @package _global_.data
|
|
2
|
+
# Default data configuration
|
|
3
|
+
|
|
4
|
+
# Data paths
|
|
5
|
+
data_path: ??? # Path to data directory (optional, if files are in one directory)
|
|
6
|
+
metadata: ??? # Path to metadata CSV file (required)
|
|
7
|
+
|
|
8
|
+
# Column names in metadata CSV
|
|
9
|
+
input_cols: ??? # List of input column names (features/modalities)
|
|
10
|
+
outcome_cols: ??? # List of outcome column names (target variables)
|
|
11
|
+
|
|
12
|
+
# Embedding models
|
|
13
|
+
# Maps metadata column names to model names
|
|
14
|
+
# Example: {'image_path': 'resnet18'}
|
|
15
|
+
embedding_map: null
|
|
16
|
+
embedding_model_kwargs: null # Optional per-modality kwargs: {col_name: {key: val}}
|
|
17
|
+
|
|
18
|
+
# Output paths
|
|
19
|
+
cache_dir: ./cache # Directory for caching intermediate results
|
|
20
|
+
output_dir: ./outputs # Directory for saving final outputs
|
|
21
|
+
|
|
22
|
+
# DataLoader settings
|
|
23
|
+
batch_size: 32 # Batch size for training
|
|
24
|
+
num_workers: 4 # Number of workers for data loading
|
|
25
|
+
shuffle_train: true # Whether to shuffle training data
|
|
26
|
+
|
|
27
|
+
# Train/val/test split ratios (must sum to 1.0)
|
|
28
|
+
split_ratios: [0.7, 0.15, 0.15] # [train, val, test]
|
|
29
|
+
|
|
30
|
+
# AutoInterp text column mapping
|
|
31
|
+
# Maps modality column names to text columns for LLM-based interpretation
|
|
32
|
+
# e.g. {image_path: report} uses the 'report' column for image modality autointerp
|
|
33
|
+
autointerp_text_map: null
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# Developer Configuration Template for MechSci
|
|
2
|
+
# Copy this file to dev.yaml and modify for experiments
|
|
3
|
+
|
|
4
|
+
# SAE Configuration
|
|
5
|
+
sae:
|
|
6
|
+
# TopK values to try (number of active features per sample)
|
|
7
|
+
top_ks:
|
|
8
|
+
- 5
|
|
9
|
+
- 10
|
|
10
|
+
- 20
|
|
11
|
+
- 40
|
|
12
|
+
|
|
13
|
+
# Matryoshka dictionary sizes (nested feature dictionaries)
|
|
14
|
+
matryoshka:
|
|
15
|
+
- 128
|
|
16
|
+
- 512
|
|
17
|
+
- 2048
|
|
18
|
+
- 8192
|
|
19
|
+
|
|
20
|
+
# Training hyperparameters
|
|
21
|
+
learning_rate: 0.0003
|
|
22
|
+
max_steps: 50000
|
|
23
|
+
sparsity_coefficient: 0.001
|
|
24
|
+
use_ghost_grads: true
|
|
25
|
+
aux_scale: 1.0
|
|
26
|
+
dead_feature_threshold: 1.0e-8
|
|
27
|
+
|
|
28
|
+
# PheWAS Configuration
|
|
29
|
+
phewas:
|
|
30
|
+
# Minimum number of feature activations required for hypothesis testing
|
|
31
|
+
min_activations: 25
|
|
32
|
+
|
|
33
|
+
# AutoInterp selection criteria
|
|
34
|
+
autointerp_selection_criteria: top_20
|
|
35
|
+
|
|
36
|
+
# Device Configuration
|
|
37
|
+
device: auto # Options: auto, cuda, cpu
|
|
38
|
+
|
|
39
|
+
# Optional: Path to precomputed embeddings (skips embedding extraction)
|
|
40
|
+
path_to_embeddings: null
|
|
41
|
+
|
|
42
|
+
# Optional: Random seed for reproducibility
|
|
43
|
+
seed: 42
|
|
44
|
+
|
|
45
|
+
# Optional: Logging
|
|
46
|
+
logging:
|
|
47
|
+
use_wandb: false
|
|
48
|
+
wandb_project: mechsci
|
|
49
|
+
log_frequency: 100
|