glyph-discovery 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. glyph_discovery-0.1.0/LICENSE +21 -0
  2. glyph_discovery-0.1.0/PKG-INFO +250 -0
  3. glyph_discovery-0.1.0/README.md +199 -0
  4. glyph_discovery-0.1.0/glyph/__init__.py +12 -0
  5. glyph_discovery-0.1.0/glyph/conf/autointerp/default.yaml +75 -0
  6. glyph_discovery-0.1.0/glyph/conf/autointerp/llm/gemini_api.yaml +6 -0
  7. glyph_discovery-0.1.0/glyph/conf/autointerp/llm/vertex.yaml +7 -0
  8. glyph_discovery-0.1.0/glyph/conf/config.yaml +17 -0
  9. glyph_discovery-0.1.0/glyph/conf/data/default.yaml +33 -0
  10. glyph_discovery-0.1.0/glyph/conf/dev.yaml.template +49 -0
  11. glyph_discovery-0.1.0/glyph/conf/logging/none.yaml +6 -0
  12. glyph_discovery-0.1.0/glyph/conf/logging/wandb.yaml +6 -0
  13. glyph_discovery-0.1.0/glyph/conf/metrics/default.yaml +13 -0
  14. glyph_discovery-0.1.0/glyph/conf/phewas/default.yaml +6 -0
  15. glyph_discovery-0.1.0/glyph/conf/sae/default.yaml +12 -0
  16. glyph_discovery-0.1.0/glyph/conf/user.yaml.template +35 -0
  17. glyph_discovery-0.1.0/glyph/conf/visualization/default.yaml +5 -0
  18. glyph_discovery-0.1.0/glyph/config/__init__.py +15 -0
  19. glyph_discovery-0.1.0/glyph/config/config.py +87 -0
  20. glyph_discovery-0.1.0/glyph/config/default_configs.py +5 -0
  21. glyph_discovery-0.1.0/glyph/core/__init__.py +5 -0
  22. glyph_discovery-0.1.0/glyph/core/glyph.py +896 -0
  23. glyph_discovery-0.1.0/glyph/dataloader/__init__.py +10 -0
  24. glyph_discovery-0.1.0/glyph/dataloader/dataloader.py +112 -0
  25. glyph_discovery-0.1.0/glyph/dataloader/dataset.py +255 -0
  26. glyph_discovery-0.1.0/glyph/interpretation/__init__.py +5 -0
  27. glyph_discovery-0.1.0/glyph/interpretation/autointerp.py +600 -0
  28. glyph_discovery-0.1.0/glyph/mechinterp/__init__.py +6 -0
  29. glyph_discovery-0.1.0/glyph/mechinterp/sae.py +285 -0
  30. glyph_discovery-0.1.0/glyph/mechinterp/sae_trainer.py +257 -0
  31. glyph_discovery-0.1.0/glyph/metrics/linear_probe.py +345 -0
  32. glyph_discovery-0.1.0/glyph/metrics/statistics.py +54 -0
  33. glyph_discovery-0.1.0/glyph/models/__init__.py +15 -0
  34. glyph_discovery-0.1.0/glyph/models/base.py +59 -0
  35. glyph_discovery-0.1.0/glyph/models/embedding_extractor.py +270 -0
  36. glyph_discovery-0.1.0/glyph/models/image/__init__.py +5 -0
  37. glyph_discovery-0.1.0/glyph/models/image/resnet18.py +120 -0
  38. glyph_discovery-0.1.0/glyph/models/image/siglip_embed.py +24 -0
  39. glyph_discovery-0.1.0/glyph/models/precomputed.py +40 -0
  40. glyph_discovery-0.1.0/glyph/models/registry.py +112 -0
  41. glyph_discovery-0.1.0/glyph/phewas/__init__.py +15 -0
  42. glyph_discovery-0.1.0/glyph/phewas/phewas.py +305 -0
  43. glyph_discovery-0.1.0/glyph/phewas/statistics.py +135 -0
  44. glyph_discovery-0.1.0/glyph/utils/__init__.py +19 -0
  45. glyph_discovery-0.1.0/glyph/utils/device.py +85 -0
  46. glyph_discovery-0.1.0/glyph/utils/io.py +98 -0
  47. glyph_discovery-0.1.0/glyph/visualization/__init__.py +2 -0
  48. glyph_discovery-0.1.0/glyph/visualization/concept_network.py +244 -0
  49. glyph_discovery-0.1.0/glyph/visualization/feature_lollipop.py +114 -0
  50. glyph_discovery-0.1.0/glyph_discovery.egg-info/PKG-INFO +250 -0
  51. glyph_discovery-0.1.0/glyph_discovery.egg-info/SOURCES.txt +62 -0
  52. glyph_discovery-0.1.0/glyph_discovery.egg-info/dependency_links.txt +1 -0
  53. glyph_discovery-0.1.0/glyph_discovery.egg-info/requires.txt +31 -0
  54. glyph_discovery-0.1.0/glyph_discovery.egg-info/top_level.txt +1 -0
  55. glyph_discovery-0.1.0/pyproject.toml +76 -0
  56. glyph_discovery-0.1.0/setup.cfg +4 -0
  57. glyph_discovery-0.1.0/tests/test1_config.py +236 -0
  58. glyph_discovery-0.1.0/tests/test2_metrics.py +110 -0
  59. glyph_discovery-0.1.0/tests/test3_data.py +310 -0
  60. glyph_discovery-0.1.0/tests/test4_embedding_extraction.py +342 -0
  61. glyph_discovery-0.1.0/tests/test5_outcome_types.py +343 -0
  62. glyph_discovery-0.1.0/tests/test6_sae.py +188 -0
  63. glyph_discovery-0.1.0/tests/test7_phewas.py +205 -0
  64. glyph_discovery-0.1.0/tests/test_autointerp.py +391 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025-2026 Robbie Holland
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,250 @@
1
+ Metadata-Version: 2.4
2
+ Name: glyph-discovery
3
+ Version: 0.1.0
4
+ Summary: Domain-agnostic hypothesis generation using sparse autoencoders
5
+ Author-email: Robbie Holland <robbie.holland@stanford.edu>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/RobbieHolland/Glyph
8
+ Project-URL: Repository, https://github.com/RobbieHolland/Glyph
9
+ Project-URL: Paper, https://openreview.net/forum?id=rgpgukbeVf
10
+ Keywords: sparse autoencoders,mechanistic interpretability,hypothesis generation,phewas,medical imaging
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Requires-Python: >=3.9
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: torch>=2.0.0
23
+ Requires-Dist: torchvision>=0.15.0
24
+ Requires-Dist: pytorch-lightning>=2.0.0
25
+ Requires-Dist: numpy>=1.24.0
26
+ Requires-Dist: pandas>=2.0.0
27
+ Requires-Dist: scikit-learn>=1.3.0
28
+ Requires-Dist: scipy>=1.10.0
29
+ Requires-Dist: statsmodels>=0.14.0
30
+ Requires-Dist: matplotlib>=3.7.0
31
+ Requires-Dist: networkx>=3.0
32
+ Requires-Dist: umap-learn>=0.5.4
33
+ Requires-Dist: pillow>=9.5.0
34
+ Requires-Dist: pyarrow>=12.0.0
35
+ Requires-Dist: hydra-core>=1.3.0
36
+ Requires-Dist: omegaconf>=2.3.0
37
+ Requires-Dist: pyyaml>=6.0
38
+ Requires-Dist: tqdm>=4.65.0
39
+ Provides-Extra: autointerp
40
+ Requires-Dist: google-generativeai>=0.8.0; extra == "autointerp"
41
+ Requires-Dist: google-cloud-aiplatform>=1.60.0; extra == "autointerp"
42
+ Provides-Extra: tracking
43
+ Requires-Dist: wandb>=0.15.0; extra == "tracking"
44
+ Provides-Extra: dev
45
+ Requires-Dist: pytest>=7.0.0; extra == "dev"
46
+ Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
47
+ Requires-Dist: black>=23.0.0; extra == "dev"
48
+ Requires-Dist: flake8>=6.0.0; extra == "dev"
49
+ Requires-Dist: mypy>=1.0.0; extra == "dev"
50
+ Dynamic: license-file
51
+
52
+ # Glyph: Mechanistic Science
53
+
54
+ **Domain-Agnostic Hypothesis Generation using Sparse Autoencoders**
55
+
56
+ Glyph is a Python toolkit for discovering interpretable features and generating scientific hypotheses from multimodal data. It uses Sparse Autoencoders (SAEs) to learn interpretable features, then conducts Phenome-Wide Association Studies (PheWAS) to identify and rank hypotheses.
57
+
58
+ 📄 **Paper**: [MechSci: Scaling Clinical Science via Mechanistic Interpretability of Multimodal Medical Foundation Models](https://openreview.net/forum?id=rgpgukbeVf#discussion) — *Agents4Science 2025*
59
+
60
+ [![Video Presentation](https://img.youtube.com/vi/s3t-Nu8HNFc/maxresdefault.jpg)](https://www.youtube.com/watch?v=s3t-Nu8HNFc)
61
+ *Click to watch the video presentation*
62
+
63
+ ## Features
64
+
65
+ - **Sparse Autoencoders** with TopK activation and Matryoshka nested dictionaries
66
+ - **Ghost Gradient Recovery** to prevent dead features during training
67
+ - **PheWAS Analysis** for systematic hypothesis testing
68
+ - **Statistical Rigor** with odds ratios, confidence intervals, and AUC metrics
69
+ - **GPU Acceleration** for large-scale analysis
70
+ - **Domain Agnostic** - works with any structured data (images, tabular, etc.)
71
+
72
+ ## Installation
73
+
74
+ ```bash
75
+ pip install glyph-discovery
76
+ ```
77
+
78
+ Optional extras:
79
+
80
+ ```bash
81
+ pip install "glyph-discovery[autointerp]" # LLM-based feature interpretation
82
+ pip install "glyph-discovery[tracking]" # Weights & Biases logging
83
+ ```
84
+
85
+ To work on Glyph itself:
86
+
87
+ ```bash
88
+ git clone https://github.com/RobbieHolland/Glyph.git
89
+ cd Glyph
90
+ pip install -e ".[dev]"
91
+ ```
92
+
93
+ ## Quick Start
94
+
95
+ ```python
96
+ from glyph import Glyph
97
+
98
+ # Define your configuration overrides
99
+ config = [
100
+ "data.metadata=/path/to/metadata.csv",
101
+ "data.input_cols=[image_path]",
102
+ "data.outcome_cols=[disease]",
103
+ "data.embedding_map={image_path: resnet18}",
104
+ "data.cache_dir=./cache",
105
+ "data.output_dir=./outputs",
106
+ "seed=42",
107
+ ]
108
+
109
+ # Initialize Glyph
110
+ ms = Glyph(config_path='config', config_overrides=config)
111
+
112
+ # Option 1: Run full pipeline
113
+ results = ms() # Equivalent to ms.fit().hypothesis_search()
114
+
115
+ # Option 2: Run in stages
116
+ ms.fit() # Extract embeddings, train SAE, compute linear probes
117
+ ms.hypothesis_search() # Run PheWAS study
118
+
119
+ # Access results directly
120
+ print(f"Embeddings: {list(ms.embeddings.keys())}")
121
+ print(f"SAE models: {list(ms.sae_models.keys())}")
122
+ print(f"PheWAS hypotheses: {len(ms.phewas_results)}")
123
+ ```
124
+
125
+ ## Pipeline Overview
126
+
127
+ ```
128
+ Raw Data → Embeddings → SAE Training → Sparse Features → PheWAS → Ranked Hypotheses
129
+ ```
130
+
131
+ ### `ms.fit()` - Steps 1-4
132
+
133
+ 1. **Embedding Extraction**: Extract dense representations using pretrained models (e.g., ResNet18)
134
+ 2. **Linear Probe**: Evaluate embedding quality for outcome prediction (AUC)
135
+ 3. **SAE Training**: Learn sparse, interpretable features with TopK activation
136
+ 4. **SAE Linear Probe**: Evaluate sparse feature quality for outcome prediction
137
+
138
+ ### `ms.hypothesis_search()` - Step 5
139
+
140
+ 5. **PheWAS Study**: Test all sparse feature-outcome associations
141
+ - Computes odds ratios, confidence intervals, and AUC for each feature
142
+ - Filters by minimum activation count for statistical reliability
143
+ - Ranks hypotheses by odds ratio
144
+
145
+ ## Example Output
146
+
147
+ ```
148
+ [PheWAS Results]
149
+ Total hypotheses: 162
150
+
151
+ Outcome class balance (train):
152
+ outcome: 2436/3295 positive (73.9%), 859/3295 negative (26.1%)
153
+
154
+ Top 10 hypotheses (by train odds ratio):
155
+ ──────────────────────────────────────────────────────────────────────────────────────────
156
+ Feature Train OR Test OR AUC % AUC Count
157
+ ──────────────────────────────────────────────────────────────────────────────────────────
158
+ feature_98 3.75 3.96 0.796 60.0% 2818
159
+ feature_61 3.45 3.34 0.773 55.2% 2524
160
+ feature_140 2.86 2.64 0.614 23.2% 674
161
+ ...
162
+ ──────────────────────────────────────────────────────────────────────────────────────────
163
+ ```
164
+
165
+ - **Train/Test OR**: Odds ratio (how much the feature increases disease odds)
166
+ - **AUC**: Predictive power of this single feature
167
+ - **% AUC**: `(feature_auc - 0.5) / (embedding_auc - 0.5)` - what % of embedding's predictive power this feature captures
168
+ - **Count**: Number of samples where feature activates
169
+
170
+ ## Configuration
171
+
172
+ Glyph uses [Hydra](https://hydra.cc/) for configuration management.
173
+
174
+ ### Key Configuration Options
175
+
176
+ ```yaml
177
+ # Data settings
178
+ data:
179
+ metadata: /path/to/metadata.csv # CSV with sample IDs and file paths
180
+ input_cols: [image_path] # Columns containing input data paths
181
+ outcome_cols: [disease] # Columns containing outcomes to predict
182
+ embedding_map: {image_path: resnet18} # Map input columns to embedding models
183
+ cache_dir: ./cache # Cache directory for embeddings
184
+ output_dir: ./outputs # Output directory for results
185
+
186
+ # SAE settings
187
+ sae:
188
+ top_ks: [20] # TopK sparsity values to try
189
+ matryoshka: [2048] # Dictionary sizes (Matryoshka nesting)
190
+ max_steps: 5000 # Training steps
191
+ learning_rate: 0.0003 # Learning rate
192
+
193
+ # PheWAS settings
194
+ phewas:
195
+ min_activations: 25 # Minimum feature activations for hypothesis
196
+ top_n_print: 10 # Number of top hypotheses to display
197
+ ```
198
+
199
+ ## Project Status
200
+
201
+ ✅ **Core Pipeline Complete**
202
+
203
+ - [x] Phase 1: Project structure
204
+ - [x] Phase 2: Data loading (GlyphDataset, DataLoader)
205
+ - [x] Phase 3: Embedding extraction (ResNet18, caching)
206
+ - [x] Phase 4: SAE training (TopK, Matryoshka loss, ghost gradients)
207
+ - [x] Phase 5: Linear probe (binary classification, regression)
208
+ - [x] Phase 6: PheWAS analysis (odds ratios, AUC, ranking)
209
+ - [x] Phase 7: Main Glyph class with fit/hypothesis_search API
210
+ - [x] Phase 8: Examples
211
+ - [x] Phase 9: Unit tests (83 tests passing)
212
+ - [ ] Phase 10: AutoInterp (interpretation with LLMs)
213
+ - [ ] Phase 11: Documentation
214
+
215
+ ## Running Tests
216
+
217
+ ```bash
218
+ # Run all tests
219
+ python -m pytest tests/test*.py -v
220
+
221
+ # Run specific test module
222
+ python -m pytest tests/test6_sae.py -v
223
+ python -m pytest tests/test7_phewas.py -v
224
+ ```
225
+
226
+ ## Inspiration
227
+
228
+ Glyph distills core algorithms from a larger medical hypothesis generation research codebase. Key innovations preserved:
229
+
230
+ - **Matryoshka SAEs** for multi-resolution feature learning
231
+ - **Ghost Gradients** for dead feature recovery
232
+ - **TopK Activation** for interpretable sparse codes
233
+ - **Rigorous Statistics** with odds ratios and confidence intervals
234
+
235
+ ## License
236
+
237
+ MIT License - see [LICENSE](LICENSE) for details.
238
+
239
+ ## Citation
240
+
241
+ If you use Glyph in your research, please cite:
242
+
243
+ ```bibtex
244
+ @software{glyph2025,
245
+ title = {Glyph: Domain-Agnostic Hypothesis Generation},
246
+ author = {Holland, Robbie},
247
+ year = {2025},
248
+ url = {https://github.com/RobbieHolland/Glyph}
249
+ }
250
+ ```
@@ -0,0 +1,199 @@
1
+ # Glyph: Mechanistic Science
2
+
3
+ **Domain-Agnostic Hypothesis Generation using Sparse Autoencoders**
4
+
5
+ Glyph is a Python toolkit for discovering interpretable features and generating scientific hypotheses from multimodal data. It uses Sparse Autoencoders (SAEs) to learn interpretable features, then conducts Phenome-Wide Association Studies (PheWAS) to identify and rank hypotheses.
6
+
7
+ 📄 **Paper**: [MechSci: Scaling Clinical Science via Mechanistic Interpretability of Multimodal Medical Foundation Models](https://openreview.net/forum?id=rgpgukbeVf#discussion) — *Agents4Science 2025*
8
+
9
+ [![Video Presentation](https://img.youtube.com/vi/s3t-Nu8HNFc/maxresdefault.jpg)](https://www.youtube.com/watch?v=s3t-Nu8HNFc)
10
+ *Click to watch the video presentation*
11
+
12
+ ## Features
13
+
14
+ - **Sparse Autoencoders** with TopK activation and Matryoshka nested dictionaries
15
+ - **Ghost Gradient Recovery** to prevent dead features during training
16
+ - **PheWAS Analysis** for systematic hypothesis testing
17
+ - **Statistical Rigor** with odds ratios, confidence intervals, and AUC metrics
18
+ - **GPU Acceleration** for large-scale analysis
19
+ - **Domain Agnostic** - works with any structured data (images, tabular, etc.)
20
+
21
+ ## Installation
22
+
23
+ ```bash
24
+ pip install glyph-discovery
25
+ ```
26
+
27
+ Optional extras:
28
+
29
+ ```bash
30
+ pip install "glyph-discovery[autointerp]" # LLM-based feature interpretation
31
+ pip install "glyph-discovery[tracking]" # Weights & Biases logging
32
+ ```
33
+
34
+ To work on Glyph itself:
35
+
36
+ ```bash
37
+ git clone https://github.com/RobbieHolland/Glyph.git
38
+ cd Glyph
39
+ pip install -e ".[dev]"
40
+ ```
41
+
42
+ ## Quick Start
43
+
44
+ ```python
45
+ from glyph import Glyph
46
+
47
+ # Define your configuration overrides
48
+ config = [
49
+ "data.metadata=/path/to/metadata.csv",
50
+ "data.input_cols=[image_path]",
51
+ "data.outcome_cols=[disease]",
52
+ "data.embedding_map={image_path: resnet18}",
53
+ "data.cache_dir=./cache",
54
+ "data.output_dir=./outputs",
55
+ "seed=42",
56
+ ]
57
+
58
+ # Initialize Glyph
59
+ ms = Glyph(config_path='config', config_overrides=config)
60
+
61
+ # Option 1: Run full pipeline
62
+ results = ms() # Equivalent to ms.fit().hypothesis_search()
63
+
64
+ # Option 2: Run in stages
65
+ ms.fit() # Extract embeddings, train SAE, compute linear probes
66
+ ms.hypothesis_search() # Run PheWAS study
67
+
68
+ # Access results directly
69
+ print(f"Embeddings: {list(ms.embeddings.keys())}")
70
+ print(f"SAE models: {list(ms.sae_models.keys())}")
71
+ print(f"PheWAS hypotheses: {len(ms.phewas_results)}")
72
+ ```
73
+
74
+ ## Pipeline Overview
75
+
76
+ ```
77
+ Raw Data → Embeddings → SAE Training → Sparse Features → PheWAS → Ranked Hypotheses
78
+ ```
79
+
80
+ ### `ms.fit()` - Steps 1-4
81
+
82
+ 1. **Embedding Extraction**: Extract dense representations using pretrained models (e.g., ResNet18)
83
+ 2. **Linear Probe**: Evaluate embedding quality for outcome prediction (AUC)
84
+ 3. **SAE Training**: Learn sparse, interpretable features with TopK activation
85
+ 4. **SAE Linear Probe**: Evaluate sparse feature quality for outcome prediction
86
+
87
+ ### `ms.hypothesis_search()` - Step 5
88
+
89
+ 5. **PheWAS Study**: Test all sparse feature-outcome associations
90
+ - Computes odds ratios, confidence intervals, and AUC for each feature
91
+ - Filters by minimum activation count for statistical reliability
92
+ - Ranks hypotheses by odds ratio
93
+
94
+ ## Example Output
95
+
96
+ ```
97
+ [PheWAS Results]
98
+ Total hypotheses: 162
99
+
100
+ Outcome class balance (train):
101
+ outcome: 2436/3295 positive (73.9%), 859/3295 negative (26.1%)
102
+
103
+ Top 10 hypotheses (by train odds ratio):
104
+ ──────────────────────────────────────────────────────────────────────────────────────────
105
+ Feature Train OR Test OR AUC % AUC Count
106
+ ──────────────────────────────────────────────────────────────────────────────────────────
107
+ feature_98 3.75 3.96 0.796 60.0% 2818
108
+ feature_61 3.45 3.34 0.773 55.2% 2524
109
+ feature_140 2.86 2.64 0.614 23.2% 674
110
+ ...
111
+ ──────────────────────────────────────────────────────────────────────────────────────────
112
+ ```
113
+
114
+ - **Train/Test OR**: Odds ratio (how much the feature increases disease odds)
115
+ - **AUC**: Predictive power of this single feature
116
+ - **% AUC**: `(feature_auc - 0.5) / (embedding_auc - 0.5)` - what % of embedding's predictive power this feature captures
117
+ - **Count**: Number of samples where feature activates
118
+
119
+ ## Configuration
120
+
121
+ Glyph uses [Hydra](https://hydra.cc/) for configuration management.
122
+
123
+ ### Key Configuration Options
124
+
125
+ ```yaml
126
+ # Data settings
127
+ data:
128
+ metadata: /path/to/metadata.csv # CSV with sample IDs and file paths
129
+ input_cols: [image_path] # Columns containing input data paths
130
+ outcome_cols: [disease] # Columns containing outcomes to predict
131
+ embedding_map: {image_path: resnet18} # Map input columns to embedding models
132
+ cache_dir: ./cache # Cache directory for embeddings
133
+ output_dir: ./outputs # Output directory for results
134
+
135
+ # SAE settings
136
+ sae:
137
+ top_ks: [20] # TopK sparsity values to try
138
+ matryoshka: [2048] # Dictionary sizes (Matryoshka nesting)
139
+ max_steps: 5000 # Training steps
140
+ learning_rate: 0.0003 # Learning rate
141
+
142
+ # PheWAS settings
143
+ phewas:
144
+ min_activations: 25 # Minimum feature activations for hypothesis
145
+ top_n_print: 10 # Number of top hypotheses to display
146
+ ```
147
+
148
+ ## Project Status
149
+
150
+ ✅ **Core Pipeline Complete**
151
+
152
+ - [x] Phase 1: Project structure
153
+ - [x] Phase 2: Data loading (GlyphDataset, DataLoader)
154
+ - [x] Phase 3: Embedding extraction (ResNet18, caching)
155
+ - [x] Phase 4: SAE training (TopK, Matryoshka loss, ghost gradients)
156
+ - [x] Phase 5: Linear probe (binary classification, regression)
157
+ - [x] Phase 6: PheWAS analysis (odds ratios, AUC, ranking)
158
+ - [x] Phase 7: Main Glyph class with fit/hypothesis_search API
159
+ - [x] Phase 8: Examples
160
+ - [x] Phase 9: Unit tests (83 tests passing)
161
+ - [ ] Phase 10: AutoInterp (interpretation with LLMs)
162
+ - [ ] Phase 11: Documentation
163
+
164
+ ## Running Tests
165
+
166
+ ```bash
167
+ # Run all tests
168
+ python -m pytest tests/test*.py -v
169
+
170
+ # Run specific test module
171
+ python -m pytest tests/test6_sae.py -v
172
+ python -m pytest tests/test7_phewas.py -v
173
+ ```
174
+
175
+ ## Inspiration
176
+
177
+ Glyph distills core algorithms from a larger medical hypothesis generation research codebase. Key innovations preserved:
178
+
179
+ - **Matryoshka SAEs** for multi-resolution feature learning
180
+ - **Ghost Gradients** for dead feature recovery
181
+ - **TopK Activation** for interpretable sparse codes
182
+ - **Rigorous Statistics** with odds ratios and confidence intervals
183
+
184
+ ## License
185
+
186
+ MIT License - see [LICENSE](LICENSE) for details.
187
+
188
+ ## Citation
189
+
190
+ If you use Glyph in your research, please cite:
191
+
192
+ ```bibtex
193
+ @software{glyph2025,
194
+ title = {Glyph: Domain-Agnostic Hypothesis Generation},
195
+ author = {Holland, Robbie},
196
+ year = {2025},
197
+ url = {https://github.com/RobbieHolland/Glyph}
198
+ }
199
+ ```
@@ -0,0 +1,12 @@
1
+ """
2
+ Glyph: Mechanistic Science - Domain-Agnostic Hypothesis Generation
3
+
4
+ A toolkit for discovering interpretable features and generating scientific hypotheses
5
+ from multimodal data using Sparse Autoencoders and statistical analysis.
6
+ """
7
+
8
+ __version__ = "0.1.0"
9
+
10
+ from glyph.core.glyph import Glyph
11
+
12
+ __all__ = ["Glyph"]
@@ -0,0 +1,75 @@
1
+ # @package _global_.autointerp
2
+ # AutoInterp configuration for LLM-based feature interpretation
3
+
4
+ defaults:
5
+ - llm: gemini_api
6
+
7
+ # Selection parameters
8
+ top_n_hypotheses: 5 # Number of top (highest OR, risk) hypotheses per outcome to interpret
9
+ bottom_n_hypotheses: 0 # Number of bottom (lowest OR, protective) hypotheses per outcome to interpret
10
+ top_pct_activating: 0.05 # Top percentage of activating samples to consider
11
+ n_fit_samples: 40 # Number of train samples for interpretation
12
+ n_test_samples: 40 # Number of test samples for validation
13
+
14
+ # Text data column in metadata (maps modality to its text column)
15
+ # e.g. {image_path: report} means use the 'report' column for the 'image_path' modality
16
+ autointerp_text_map: {image_path: report}
17
+
18
+ # Prompt templates - use {samples} and {interpretation} as placeholders
19
+ full_reports_prompt: |
20
+ === Task ===
21
+ You are an expert radiologist and diagnostic data scientist. Your task is to distill the most accurate Interpretation or Description that characterizes a specific cohort of patients compared to a reference set of controls. The primary objective is to produce a simple, high-performing explanation that maximizes discriminative accuracy on unseen samples.
22
+
23
+ === Data provided ===
24
+ 1. REFERENCE PATIENTS (CONTROLS): A sample of patients who DO NOT possess the feature. These represent the background baseline where the characteristic in question is generally absent.
25
+ 2. COHORT PATIENTS: Patients who strongly activate this specific feature, listed in order of activation strength.
26
+
27
+ === Core Directives ===
28
+ Your goal is to define the "boundary" or "signal" that is present in the Cohort but absent or significantly different in the Reference Patients.
29
+
30
+ 1. Contrastive Filtering: Focus on findings with a clear disparity in prevalence or intensity between the two groups. If a finding appears frequently in both the cohort and the reference set, it is highly unlikely to be the characterizing feature.
31
+ 2. Concept over Literalism: Prioritize the underlying clinical or radiological concept. While you should note the linguistic style (e.g., if the reports are consistently "vague" or "specific"), do not be misled by synonyms; "renal calculus" and "kidney stone" represent the same anchor.
32
+ 3. Discriminative Balance: Aim for a level of abstraction that is general enough to apply to the vast majority of the cohort (not just the top few examples), but specific enough that it does not apply to the reference set.
33
+ 4. Chain of Thought Reasoning: You must begin by verbalizing your thought process. Analyze the patterns across the *entire* cohort, cross-reference them against the reference patients to identify unique signals, and verify your interpretation's accuracy against both groups to ensure it maximizes discriminative performance. You should expect to get at least 85% accuracy on the discriminative task, and test this on your own data before providing your final interpretation.
34
+
35
+ === Logical Guidelines ===
36
+ - Pure Description: Focus strictly on characterizing the feature as a finding. Do not provide clinical definitions or background explanations of the pathologies themselves (e.g., avoid "This represents X, which is defined as...").
37
+ - Atomic Simplicity: Simple, single atomic explanations are preferred. Aim for the most concise "unit" of clinical meaning that captures the cohort's essence.
38
+ - The "Or" Exception: Strictly avoid "A or B" structures in general. However, if a singular unifying concept cannot be found and an "OR" is strictly necessary to achieve high discriminative performance, you may use it sparingly.
39
+ - The "And" Clause: You may use "and" if the feature is a conjunctive relationship between two elements that consistently appear together.
40
+ - Discriminative Power: Your interpretation must act as a precise rule that allows a human (or another AI) to accurately classify a mixed/shuffled pile of reports into "Cohort" and "Reference" groups.
41
+
42
+ === Formatting Rules ===
43
+ You must first provide your reasoning and analysis. Once your analysis is complete, you must provide your final interpretation as the very last line of your response, starting with an asterisk:
44
+ * This cohort is characterized by [your specific, synthesized, and discriminative description of the feature].
45
+
46
+ The asterisk is essential for the final line. Do not include any text after the starred sentence.
47
+
48
+ === Data ===
49
+ [REFERENCE PATIENTS (CONTROLS)]
50
+ {reference_patients}
51
+
52
+ [COHORT PATIENTS (FEATURE ACTIVATIONS)]
53
+ {patient_data}
54
+
55
+ discriminative_autointerp_score_prompt: |
56
+ Task: Distinguish samples (randomly ordered) based on those that do and do not belong to the group characterized by the group characteristics.
57
+
58
+ Medical Findings Reports:
59
+ {patient_data}
60
+
61
+ Group characteristics:
62
+ {interpretation}
63
+
64
+ Analyze the findings reports one by one, reasoning on whether or not the characteristics apply to the report. Use the following format for each report:
65
+ {{i}}: 1 (if belongs to the group) or 0 (if does not belong to the group). We have designed the list so that exactly half of the reports belong to the group, and half do not. Therefore, you should use the characteristics to differentiate each of the reports into two distinct groups, rather than finding perfect matches.
66
+
67
+ You need to be as accurate as possible when assigning 1s and 0s. We will score the accuracy of your response at the end.
68
+
69
+ === Format ===:
70
+ It is crucial that you present your final explanation in the following format, started with an asterisk:
71
+ * PREDICTIONS:
72
+ 1. 1 (if the sample belongs to the group) or 0 (if it does not belong to the group)
73
+ 2. 1 (if the sample belongs to the group) or 0 (if it does not belong to the group)
74
+ ...
75
+ {{n}}. 1 (if the sample belongs to the group) or 0 (if it does not belong to the group)
@@ -0,0 +1,6 @@
1
+ # @package _global_.autointerp.llm
2
+ provider: gemini_api
3
+ gemini_api_key: null # Set in user config or via override
4
+ model: gemini-flash-lite-latest
5
+ temperature: 0.0
6
+ max_tokens: 1024
@@ -0,0 +1,7 @@
1
+ # @package _global_.autointerp.llm
2
+ provider: vertex
3
+ vertex_project: null # Set in user config or via override
4
+ vertex_location: us-central1
5
+ model: gemini-2.5-pro
6
+ temperature: 0.0
7
+ max_tokens: 32768
@@ -0,0 +1,17 @@
1
+ # Glyph Main Configuration
2
+ # This file defines the default configuration structure and references to config groups
3
+
4
+ defaults:
5
+ - data: default
6
+ - sae: default
7
+ - phewas: default
8
+ - autointerp: default
9
+ - metrics: default
10
+ - visualization: default
11
+ - logging: none
12
+ - _self_
13
+
14
+ # Top-level settings
15
+ seed: 42
16
+ device: auto # Options: auto, cuda, cpu
17
+ path_to_embeddings: null
@@ -0,0 +1,33 @@
1
+ # @package _global_.data
2
+ # Default data configuration
3
+
4
+ # Data paths
5
+ data_path: ??? # Path to data directory (optional, if files are in one directory)
6
+ metadata: ??? # Path to metadata CSV file (required)
7
+
8
+ # Column names in metadata CSV
9
+ input_cols: ??? # List of input column names (features/modalities)
10
+ outcome_cols: ??? # List of outcome column names (target variables)
11
+
12
+ # Embedding models
13
+ # Maps metadata column names to model names
14
+ # Example: {'image_path': 'resnet18'}
15
+ embedding_map: null
16
+ embedding_model_kwargs: null # Optional per-modality kwargs: {col_name: {key: val}}
17
+
18
+ # Output paths
19
+ cache_dir: ./cache # Directory for caching intermediate results
20
+ output_dir: ./outputs # Directory for saving final outputs
21
+
22
+ # DataLoader settings
23
+ batch_size: 32 # Batch size for training
24
+ num_workers: 4 # Number of workers for data loading
25
+ shuffle_train: true # Whether to shuffle training data
26
+
27
+ # Train/val/test split ratios (must sum to 1.0)
28
+ split_ratios: [0.7, 0.15, 0.15] # [train, val, test]
29
+
30
+ # AutoInterp text column mapping
31
+ # Maps modality column names to text columns for LLM-based interpretation
32
+ # e.g. {image_path: report} uses the 'report' column for image modality autointerp
33
+ autointerp_text_map: null
@@ -0,0 +1,49 @@
1
+ # Developer Configuration Template for MechSci
2
+ # Copy this file to dev.yaml and modify for experiments
3
+
4
+ # SAE Configuration
5
+ sae:
6
+ # TopK values to try (number of active features per sample)
7
+ top_ks:
8
+ - 5
9
+ - 10
10
+ - 20
11
+ - 40
12
+
13
+ # Matryoshka dictionary sizes (nested feature dictionaries)
14
+ matryoshka:
15
+ - 128
16
+ - 512
17
+ - 2048
18
+ - 8192
19
+
20
+ # Training hyperparameters
21
+ learning_rate: 0.0003
22
+ max_steps: 50000
23
+ sparsity_coefficient: 0.001
24
+ use_ghost_grads: true
25
+ aux_scale: 1.0
26
+ dead_feature_threshold: 1.0e-8
27
+
28
+ # PheWAS Configuration
29
+ phewas:
30
+ # Minimum number of feature activations required for hypothesis testing
31
+ min_activations: 25
32
+
33
+ # AutoInterp selection criteria
34
+ autointerp_selection_criteria: top_20
35
+
36
+ # Device Configuration
37
+ device: auto # Options: auto, cuda, cpu
38
+
39
+ # Optional: Path to precomputed embeddings (skips embedding extraction)
40
+ path_to_embeddings: null
41
+
42
+ # Optional: Random seed for reproducibility
43
+ seed: 42
44
+
45
+ # Optional: Logging
46
+ logging:
47
+ use_wandb: false
48
+ wandb_project: mechsci
49
+ log_frequency: 100
@@ -0,0 +1,6 @@
1
+ # @package _global_.logging
2
+ # No logging configuration (local only)
3
+
4
+ use_wandb: false
5
+ wandb_project: glyph
6
+ log_frequency: 100