sagaconf 1.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sagaconf-1.0.1/LICENSE +21 -0
- sagaconf-1.0.1/PKG-INFO +231 -0
- sagaconf-1.0.1/README.md +210 -0
- sagaconf-1.0.1/pyproject.toml +34 -0
- sagaconf-1.0.1/sagaconf/__init__.py +1 -0
- sagaconf-1.0.1/sagaconf/__main__.py +3 -0
- sagaconf-1.0.1/sagaconf/_chromhmm.py +55 -0
- sagaconf-1.0.1/sagaconf/_cluster_matching.py +242 -0
- sagaconf-1.0.1/sagaconf/_reproducibility.py +631 -0
- sagaconf-1.0.1/sagaconf/_utils.py +120 -0
- sagaconf-1.0.1/sagaconf/cli.py +358 -0
- sagaconf-1.0.1/sagaconf/granul.py +227 -0
- sagaconf-1.0.1/sagaconf/overall.py +324 -0
- sagaconf-1.0.1/sagaconf/reports.py +1522 -0
- sagaconf-1.0.1/sagaconf/run.py +186 -0
- sagaconf-1.0.1/sagaconf.egg-info/PKG-INFO +231 -0
- sagaconf-1.0.1/sagaconf.egg-info/SOURCES.txt +21 -0
- sagaconf-1.0.1/sagaconf.egg-info/dependency_links.txt +1 -0
- sagaconf-1.0.1/sagaconf.egg-info/entry_points.txt +2 -0
- sagaconf-1.0.1/sagaconf.egg-info/requires.txt +10 -0
- sagaconf-1.0.1/sagaconf.egg-info/top_level.txt +1 -0
- sagaconf-1.0.1/setup.cfg +4 -0
- sagaconf-1.0.1/tests/test_golden.py +58 -0
sagaconf-1.0.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2021-2026 Mehdi Foroozandeh
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
sagaconf-1.0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sagaconf
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: Calibrated confidence scores (r-values) for chromatin state annotations from SAGA methods.
|
|
5
|
+
Author: Mehdi Foroozandeh
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/mehdiforoozandeh/SAGAconf
|
|
8
|
+
Requires-Python: >=3.9
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Dist: numpy<2,>=1.21
|
|
12
|
+
Requires-Dist: pandas>=1.3
|
|
13
|
+
Requires-Dist: scipy>=1.7
|
|
14
|
+
Requires-Dist: scikit-learn>=0.24
|
|
15
|
+
Requires-Dist: matplotlib>=3.4
|
|
16
|
+
Requires-Dist: seaborn>=0.11
|
|
17
|
+
Requires-Dist: pybedtools>=0.9
|
|
18
|
+
Provides-Extra: test
|
|
19
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+
<picture>
|
|
23
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/mehdiforoozandeh/SAGAconf/main/docs/graphical_abstract_dark.svg">
|
|
24
|
+
<img alt="SAGAconf compares two replicate chromatin state annotations, gives each genomic bin an r-value, and keeps the reproducible calls." src="https://raw.githubusercontent.com/mehdiforoozandeh/SAGAconf/main/docs/graphical_abstract_light.svg">
|
|
25
|
+
</picture>
|
|
26
|
+
|
|
27
|
+
# SAGAconf
|
|
28
|
+
|
|
29
|
+
[](https://pypi.org/project/sagaconf/)
|
|
30
|
+
[](LICENSE)
|
|
31
|
+
[](https://doi.org/10.1101/gr.278343.123)
|
|
32
|
+
|
|
33
|
+
SAGAconf gives a calibrated confidence score to each call in a chromatin state annotation.
|
|
34
|
+
Segmentation and genome annotation (SAGA) methods, such as ChromHMM and Segway, label every
|
|
35
|
+
genomic bin with a chromatin state. Many of these labels do not reproduce across replicates.
|
|
36
|
+
SAGAconf compares a base annotation with a verification annotation and gives each bin an
|
|
37
|
+
**r-value**: a score for how well its state call reproduces. You can then keep only the
|
|
38
|
+
reproducible calls for downstream analysis. SAGAconf works with any SAGA method, because it
|
|
39
|
+
reads only the posterior probability matrix.
|
|
40
|
+
|
|
41
|
+
- Paper: Foroozandeh Shahraki, Farahbod and Libbrecht, *Robust chromatin state annotation*,
|
|
42
|
+
[Genome Research 34(3):469–483, 2024](https://doi.org/10.1101/gr.278343.123).
|
|
43
|
+
- Blog post: [Reproducibility as a measure of confidence](https://medium.com/@mehdiforoozandehsh/reproducibility-as-a-measure-of-confidence-7454b72f984e).
|
|
44
|
+
|
|
45
|
+
## Install
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install sagaconf
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
SAGAconf needs Python 3.9 or later. The `active-regions` mode also needs
|
|
52
|
+
[bedtools](https://bedtools.readthedocs.io/) on your `PATH`.
|
|
53
|
+
|
|
54
|
+
## Quick start
|
|
55
|
+
|
|
56
|
+
Parse the posteriors of each annotation, then compare the two parsed files:
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
sagaconf parse --saga chmm base_posteriors/ 200 base/
|
|
60
|
+
sagaconf parse --saga chmm verif_posteriors/ 200 verif/
|
|
61
|
+
sagaconf run base/parsed_posterior.bed verif/parsed_posterior.bed results/
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
`results/r_values.bed` holds one r-value per bin. `results/confident_segments.bed` holds the
|
|
65
|
+
reproducible subset of the base annotation (this file needs mnemonics; see
|
|
66
|
+
[Known issues](#known-issues)).
|
|
67
|
+
|
|
68
|
+
To run the full example on ChromHMM's sample data, run [`example/run.sh`](example/run.sh).
|
|
69
|
+
|
|
70
|
+
## Input
|
|
71
|
+
|
|
72
|
+
SAGAconf compares two annotations of the same genome:
|
|
73
|
+
|
|
74
|
+
- The **base** annotation is the one you want to score.
|
|
75
|
+
- The **verification** annotation comes from a replicate. It can come from different data and
|
|
76
|
+
a different model (setting S1 in the paper), the same model trained on both replicates (S2),
|
|
77
|
+
or the same data with a different random initialization (S3).
|
|
78
|
+
|
|
79
|
+
`sagaconf run` reads one **parsed posterior** file per annotation. The file has one row per
|
|
80
|
+
genomic bin and `K + 3` columns: `chr`, `start`, `end`, and one posterior probability for
|
|
81
|
+
each of the `K` states. The file can be BED (tab-separated) or CSV. `sagaconf parse` writes
|
|
82
|
+
this file from the output of ChromHMM or Segway:
|
|
83
|
+
|
|
84
|
+
- **ChromHMM:** run `LearnModel` with `-printposterior`. Give `sagaconf parse` the directory
|
|
85
|
+
of `*_posterior.txt` files (one file per cell type and chromosome).
|
|
86
|
+
- **Segway:** give `sagaconf parse` the directory of `posterior<i>.bedGraph` files (one file
|
|
87
|
+
per state).
|
|
88
|
+
|
|
89
|
+
## Commands
|
|
90
|
+
|
|
91
|
+
### `sagaconf parse`
|
|
92
|
+
|
|
93
|
+
```
|
|
94
|
+
sagaconf parse --saga {chmm,segway} [--out-format {bed,csv}] POSTERIORDIR RESOLUTION OUTDIR
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
| Argument | Meaning |
|
|
98
|
+
|---|---|
|
|
99
|
+
| `POSTERIORDIR` | Directory with the SAGA model's posterior files. |
|
|
100
|
+
| `RESOLUTION` | Bin size of the SAGA model, in bp. |
|
|
101
|
+
| `OUTDIR` | Directory for `parsed_posterior.bed` (or `.csv`). SAGAconf creates it if it does not exist. |
|
|
102
|
+
| `--saga` | The SAGA model that wrote the posteriors: `chmm` or `segway`. |
|
|
103
|
+
| `--out-format` | `bed` (default) or `csv`. |
|
|
104
|
+
|
|
105
|
+
### `sagaconf run`
|
|
106
|
+
|
|
107
|
+
```
|
|
108
|
+
sagaconf run BASE VERIF OUTDIR [--mode MODE] [options]
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
| Option | Default | Meaning |
|
|
112
|
+
|---|---|---|
|
|
113
|
+
| `--mode` | `full` | The analysis to run. See the table below. |
|
|
114
|
+
| `-bm`, `--base-mnemonics FILE` | none | State names for the base annotation. |
|
|
115
|
+
| `-vm`, `--verif-mnemonics FILE` | none | State names for the verification annotation. |
|
|
116
|
+
| `-s`, `--chr21-only` | off | Analyse chr21 only, for a quick check. |
|
|
117
|
+
| `-w`, `--window-size BP` | 1000 | Window around each bin, in bp. |
|
|
118
|
+
| `-to`, `--iou-threshold` | 0.75 | IoU overlap that makes two states correspond. |
|
|
119
|
+
| `-tr`, `--repr-threshold` | 0.8 | r-value that makes a bin reproduced (α in the paper). |
|
|
120
|
+
| `-k`, `--merge-k K` | none | Merge the base states down to `K` states. |
|
|
121
|
+
| `--ccre-file FILE` | none | cCRE BED file. The `active-regions` mode needs it. |
|
|
122
|
+
| `--meuleman-file FILE` | none | Meuleman et al. DHS index for the `active-regions` mode. If you omit it, SAGAconf skips that step. |
|
|
123
|
+
| `-v`, `--verbose` | off | Report the analysis steps that fail. |
|
|
124
|
+
|
|
125
|
+
| Mode | What it writes |
|
|
126
|
+
|---|---|
|
|
127
|
+
| `full` | All analyses: r-values, confident segments, overlap, calibration, granularity and misalignment reports. |
|
|
128
|
+
| `quick` | The essential reports only. |
|
|
129
|
+
| `rvalues` | `r_values.bed` only. |
|
|
130
|
+
| `celltype` | The per-annotation analyses (overlap, calibration, granularity, misalignment). |
|
|
131
|
+
| `seglength` | Reproducibility as a function of position in a segment. |
|
|
132
|
+
| `merge` | The reports after merging the base states down to `-k` states. |
|
|
133
|
+
| `active-regions` | r-values genome-wide and in cCREs (and in Meuleman DHSs if you give the file). |
|
|
134
|
+
|
|
135
|
+
`python -m sagaconf` runs the same command as `sagaconf`.
|
|
136
|
+
|
|
137
|
+
### Mnemonics file
|
|
138
|
+
|
|
139
|
+
A mnemonics file gives each state a biological name. It is a tab-separated file with the
|
|
140
|
+
header `old` and `new`. The `old` column holds the state number, in the column order of the
|
|
141
|
+
parsed posterior file. The numbers can start at 0 or at 1. The `new` column holds the name.
|
|
142
|
+
SAGAconf shortens each name to its first 4 characters (and to 3 characters after an
|
|
143
|
+
underscore), so `Enhancer_low` becomes `Enha_low`.
|
|
144
|
+
[`example/example_mnemonics.txt`](example/example_mnemonics.txt) is an example:
|
|
145
|
+
|
|
146
|
+
```
|
|
147
|
+
old new
|
|
148
|
+
1 Enhancer_low
|
|
149
|
+
2 Enhancer
|
|
150
|
+
3 Promoter_flanking
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
## Output
|
|
154
|
+
|
|
155
|
+
`sagaconf run` writes these files in `OUTDIR` in `full` mode:
|
|
156
|
+
|
|
157
|
+
| File | Content |
|
|
158
|
+
|---|---|
|
|
159
|
+
| `r_values.bed` | The r-value of each bin: `chr`, `start`, `end`, `MAP` (the most probable state), `r_value`. |
|
|
160
|
+
| `r_values_UCSC_GenomeBrowser.bed` | The r-values as a UCSC Genome Browser track. |
|
|
161
|
+
| `r_values_report.txt` | The average r-value, genome-wide and per state. |
|
|
162
|
+
| `rval_hist` | Histograms of r-values per state. |
|
|
163
|
+
| `confident_segments.bed` | The reproducible subset of the base annotation, one row per bin. |
|
|
164
|
+
| `confident_segments_dense.bed` | The same subset, with adjacent bins of one state merged into segments. |
|
|
165
|
+
| `ratio_robust.txt` | The fraction of bins that are reproduced, genome-wide and per state. |
|
|
166
|
+
| `overlap_ratio.txt` | The naive overlap per state. |
|
|
167
|
+
| `NMI.txt` | Mutual information, with and without the posteriors. |
|
|
168
|
+
| `coverages1.txt`, `coverages2.txt` | The genome coverage of each state in the base and verification annotations. |
|
|
169
|
+
| `heatmap`, `heatmap_w` | IoU overlap between the states of the two annotations, with `w = 0` and with `w > 0`. |
|
|
170
|
+
| `binned_posterior_heatmap` | IoU overlap against the binned posteriors of the base annotation. |
|
|
171
|
+
| `granularity`, `barplot`, `AUC_mAUC.txt` | The state-merging curve and the area under it (auSMC) per state. |
|
|
172
|
+
| `len_bound`, `len_bound_overall` | Overlap as a function of `w`, per state and genome-wide. |
|
|
173
|
+
| `calib/` | Posterior calibration curves per state. |
|
|
174
|
+
| `Dist_vs_Corresp/`, `Dist_vs_Corresp_3/` | Overlap and correspondence as a function of `w`. |
|
|
175
|
+
|
|
176
|
+
Each plot is written as PDF and SVG. Each `.txt` file next to a plot holds the plotted values.
|
|
177
|
+
|
|
178
|
+
## Known issues
|
|
179
|
+
|
|
180
|
+
These issues come from the version of SAGAconf that the paper used. This release keeps them,
|
|
181
|
+
so that its results match that version exactly. Fixes will come in later releases.
|
|
182
|
+
|
|
183
|
+
- In `full` mode, `-k` does not merge states.
|
|
184
|
+
- In `full` mode without mnemonics, SAGAconf does not write the genome-wide results
|
|
185
|
+
(including `confident_segments.bed` and the UCSC track). Give `-bm` and `-vm` to get them.
|
|
186
|
+
- In `merge` mode, SAGAconf can stop with a `KeyError` after it writes the merge reports.
|
|
187
|
+
|
|
188
|
+
## Legacy interface
|
|
189
|
+
|
|
190
|
+
The original scripts still work, with the same flags and the same output:
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
python SAGAconf_parser.py --saga chmm POSTERIORDIR 200 OUTDIR
|
|
194
|
+
python SAGAconf.py [-q | --r_only | --ct_only | ...] BASE VERIF OUTDIR
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
`sagaconf run` also accepts the original flag names (`--r_only`, `--windowsize`,
|
|
198
|
+
`--base_mnemonics`, and so on). In the original scripts, the `--active_regions` mode reads
|
|
199
|
+
`src/biointerpret/GRCh38-cCREs.bed` and `src/biointerpret/Meuleman.tsv` relative to the
|
|
200
|
+
working directory.
|
|
201
|
+
|
|
202
|
+
The code as the paper used it is on the [`legacy`](https://github.com/mehdiforoozandeh/SAGAconf/tree/legacy)
|
|
203
|
+
branch (tag `v0-legacy`). That branch also keeps the scripts that produced the paper's figures.
|
|
204
|
+
|
|
205
|
+
## Tests
|
|
206
|
+
|
|
207
|
+
[`tests/golden/`](tests/golden) checks that this version gives byte-identical output to the
|
|
208
|
+
legacy version. It runs both versions on the same input and compares every file they write,
|
|
209
|
+
and everything they print. [`tests/golden/gate.py`](tests/golden/gate.py) describes the method.
|
|
210
|
+
The check passes on synthetic data and on the paper's GM12878 ChromHMM and Segway annotations.
|
|
211
|
+
|
|
212
|
+
## Citation
|
|
213
|
+
|
|
214
|
+
```bibtex
|
|
215
|
+
@article{foroozandeh2024robust,
|
|
216
|
+
title = {Robust chromatin state annotation},
|
|
217
|
+
author = {Foroozandeh Shahraki, Mehdi and Farahbod, Marjan and Libbrecht, Maxwell W.},
|
|
218
|
+
journal = {Genome Research},
|
|
219
|
+
volume = {34},
|
|
220
|
+
number = {3},
|
|
221
|
+
pages = {469--483},
|
|
222
|
+
year = {2024},
|
|
223
|
+
doi = {10.1101/gr.278343.123}
|
|
224
|
+
}
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+

|
|
228
|
+
|
|
229
|
+
## License
|
|
230
|
+
|
|
231
|
+
SAGAconf is available under the [MIT License](LICENSE).
|
sagaconf-1.0.1/README.md
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
<picture>
|
|
2
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/mehdiforoozandeh/SAGAconf/main/docs/graphical_abstract_dark.svg">
|
|
3
|
+
<img alt="SAGAconf compares two replicate chromatin state annotations, gives each genomic bin an r-value, and keeps the reproducible calls." src="https://raw.githubusercontent.com/mehdiforoozandeh/SAGAconf/main/docs/graphical_abstract_light.svg">
|
|
4
|
+
</picture>
|
|
5
|
+
|
|
6
|
+
# SAGAconf
|
|
7
|
+
|
|
8
|
+
[](https://pypi.org/project/sagaconf/)
|
|
9
|
+
[](LICENSE)
|
|
10
|
+
[](https://doi.org/10.1101/gr.278343.123)
|
|
11
|
+
|
|
12
|
+
SAGAconf gives a calibrated confidence score to each call in a chromatin state annotation.
|
|
13
|
+
Segmentation and genome annotation (SAGA) methods, such as ChromHMM and Segway, label every
|
|
14
|
+
genomic bin with a chromatin state. Many of these labels do not reproduce across replicates.
|
|
15
|
+
SAGAconf compares a base annotation with a verification annotation and gives each bin an
|
|
16
|
+
**r-value**: a score for how well its state call reproduces. You can then keep only the
|
|
17
|
+
reproducible calls for downstream analysis. SAGAconf works with any SAGA method, because it
|
|
18
|
+
reads only the posterior probability matrix.
|
|
19
|
+
|
|
20
|
+
- Paper: Foroozandeh Shahraki, Farahbod and Libbrecht, *Robust chromatin state annotation*,
|
|
21
|
+
[Genome Research 34(3):469–483, 2024](https://doi.org/10.1101/gr.278343.123).
|
|
22
|
+
- Blog post: [Reproducibility as a measure of confidence](https://medium.com/@mehdiforoozandehsh/reproducibility-as-a-measure-of-confidence-7454b72f984e).
|
|
23
|
+
|
|
24
|
+
## Install
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install sagaconf
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
SAGAconf needs Python 3.9 or later. The `active-regions` mode also needs
|
|
31
|
+
[bedtools](https://bedtools.readthedocs.io/) on your `PATH`.
|
|
32
|
+
|
|
33
|
+
## Quick start
|
|
34
|
+
|
|
35
|
+
Parse the posteriors of each annotation, then compare the two parsed files:
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
sagaconf parse --saga chmm base_posteriors/ 200 base/
|
|
39
|
+
sagaconf parse --saga chmm verif_posteriors/ 200 verif/
|
|
40
|
+
sagaconf run base/parsed_posterior.bed verif/parsed_posterior.bed results/
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`results/r_values.bed` holds one r-value per bin. `results/confident_segments.bed` holds the
|
|
44
|
+
reproducible subset of the base annotation (this file needs mnemonics; see
|
|
45
|
+
[Known issues](#known-issues)).
|
|
46
|
+
|
|
47
|
+
To run the full example on ChromHMM's sample data, run [`example/run.sh`](example/run.sh).
|
|
48
|
+
|
|
49
|
+
## Input
|
|
50
|
+
|
|
51
|
+
SAGAconf compares two annotations of the same genome:
|
|
52
|
+
|
|
53
|
+
- The **base** annotation is the one you want to score.
|
|
54
|
+
- The **verification** annotation comes from a replicate. It can come from different data and
|
|
55
|
+
a different model (setting S1 in the paper), the same model trained on both replicates (S2),
|
|
56
|
+
or the same data with a different random initialization (S3).
|
|
57
|
+
|
|
58
|
+
`sagaconf run` reads one **parsed posterior** file per annotation. The file has one row per
|
|
59
|
+
genomic bin and `K + 3` columns: `chr`, `start`, `end`, and one posterior probability for
|
|
60
|
+
each of the `K` states. The file can be BED (tab-separated) or CSV. `sagaconf parse` writes
|
|
61
|
+
this file from the output of ChromHMM or Segway:
|
|
62
|
+
|
|
63
|
+
- **ChromHMM:** run `LearnModel` with `-printposterior`. Give `sagaconf parse` the directory
|
|
64
|
+
of `*_posterior.txt` files (one file per cell type and chromosome).
|
|
65
|
+
- **Segway:** give `sagaconf parse` the directory of `posterior<i>.bedGraph` files (one file
|
|
66
|
+
per state).
|
|
67
|
+
|
|
68
|
+
## Commands
|
|
69
|
+
|
|
70
|
+
### `sagaconf parse`
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
sagaconf parse --saga {chmm,segway} [--out-format {bed,csv}] POSTERIORDIR RESOLUTION OUTDIR
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
| Argument | Meaning |
|
|
77
|
+
|---|---|
|
|
78
|
+
| `POSTERIORDIR` | Directory with the SAGA model's posterior files. |
|
|
79
|
+
| `RESOLUTION` | Bin size of the SAGA model, in bp. |
|
|
80
|
+
| `OUTDIR` | Directory for `parsed_posterior.bed` (or `.csv`). SAGAconf creates it if it does not exist. |
|
|
81
|
+
| `--saga` | The SAGA model that wrote the posteriors: `chmm` or `segway`. |
|
|
82
|
+
| `--out-format` | `bed` (default) or `csv`. |
|
|
83
|
+
|
|
84
|
+
### `sagaconf run`
|
|
85
|
+
|
|
86
|
+
```
|
|
87
|
+
sagaconf run BASE VERIF OUTDIR [--mode MODE] [options]
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
| Option | Default | Meaning |
|
|
91
|
+
|---|---|---|
|
|
92
|
+
| `--mode` | `full` | The analysis to run. See the table below. |
|
|
93
|
+
| `-bm`, `--base-mnemonics FILE` | none | State names for the base annotation. |
|
|
94
|
+
| `-vm`, `--verif-mnemonics FILE` | none | State names for the verification annotation. |
|
|
95
|
+
| `-s`, `--chr21-only` | off | Analyse chr21 only, for a quick check. |
|
|
96
|
+
| `-w`, `--window-size BP` | 1000 | Window around each bin, in bp. |
|
|
97
|
+
| `-to`, `--iou-threshold` | 0.75 | IoU overlap that makes two states correspond. |
|
|
98
|
+
| `-tr`, `--repr-threshold` | 0.8 | r-value that makes a bin reproduced (α in the paper). |
|
|
99
|
+
| `-k`, `--merge-k K` | none | Merge the base states down to `K` states. |
|
|
100
|
+
| `--ccre-file FILE` | none | cCRE BED file. The `active-regions` mode needs it. |
|
|
101
|
+
| `--meuleman-file FILE` | none | Meuleman et al. DHS index for the `active-regions` mode. If you omit it, SAGAconf skips that step. |
|
|
102
|
+
| `-v`, `--verbose` | off | Report the analysis steps that fail. |
|
|
103
|
+
|
|
104
|
+
| Mode | What it writes |
|
|
105
|
+
|---|---|
|
|
106
|
+
| `full` | All analyses: r-values, confident segments, overlap, calibration, granularity and misalignment reports. |
|
|
107
|
+
| `quick` | The essential reports only. |
|
|
108
|
+
| `rvalues` | `r_values.bed` only. |
|
|
109
|
+
| `celltype` | The per-annotation analyses (overlap, calibration, granularity, misalignment). |
|
|
110
|
+
| `seglength` | Reproducibility as a function of position in a segment. |
|
|
111
|
+
| `merge` | The reports after merging the base states down to `-k` states. |
|
|
112
|
+
| `active-regions` | r-values genome-wide and in cCREs (and in Meuleman DHSs if you give the file). |
|
|
113
|
+
|
|
114
|
+
`python -m sagaconf` runs the same command as `sagaconf`.
|
|
115
|
+
|
|
116
|
+
### Mnemonics file
|
|
117
|
+
|
|
118
|
+
A mnemonics file gives each state a biological name. It is a tab-separated file with the
|
|
119
|
+
header `old` and `new`. The `old` column holds the state number, in the column order of the
|
|
120
|
+
parsed posterior file. The numbers can start at 0 or at 1. The `new` column holds the name.
|
|
121
|
+
SAGAconf shortens each name to its first 4 characters (and to 3 characters after an
|
|
122
|
+
underscore), so `Enhancer_low` becomes `Enha_low`.
|
|
123
|
+
[`example/example_mnemonics.txt`](example/example_mnemonics.txt) is an example:
|
|
124
|
+
|
|
125
|
+
```
|
|
126
|
+
old new
|
|
127
|
+
1 Enhancer_low
|
|
128
|
+
2 Enhancer
|
|
129
|
+
3 Promoter_flanking
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## Output
|
|
133
|
+
|
|
134
|
+
`sagaconf run` writes these files in `OUTDIR` in `full` mode:
|
|
135
|
+
|
|
136
|
+
| File | Content |
|
|
137
|
+
|---|---|
|
|
138
|
+
| `r_values.bed` | The r-value of each bin: `chr`, `start`, `end`, `MAP` (the most probable state), `r_value`. |
|
|
139
|
+
| `r_values_UCSC_GenomeBrowser.bed` | The r-values as a UCSC Genome Browser track. |
|
|
140
|
+
| `r_values_report.txt` | The average r-value, genome-wide and per state. |
|
|
141
|
+
| `rval_hist` | Histograms of r-values per state. |
|
|
142
|
+
| `confident_segments.bed` | The reproducible subset of the base annotation, one row per bin. |
|
|
143
|
+
| `confident_segments_dense.bed` | The same subset, with adjacent bins of one state merged into segments. |
|
|
144
|
+
| `ratio_robust.txt` | The fraction of bins that are reproduced, genome-wide and per state. |
|
|
145
|
+
| `overlap_ratio.txt` | The naive overlap per state. |
|
|
146
|
+
| `NMI.txt` | Mutual information, with and without the posteriors. |
|
|
147
|
+
| `coverages1.txt`, `coverages2.txt` | The genome coverage of each state in the base and verification annotations. |
|
|
148
|
+
| `heatmap`, `heatmap_w` | IoU overlap between the states of the two annotations, with `w = 0` and with `w > 0`. |
|
|
149
|
+
| `binned_posterior_heatmap` | IoU overlap against the binned posteriors of the base annotation. |
|
|
150
|
+
| `granularity`, `barplot`, `AUC_mAUC.txt` | The state-merging curve and the area under it (auSMC) per state. |
|
|
151
|
+
| `len_bound`, `len_bound_overall` | Overlap as a function of `w`, per state and genome-wide. |
|
|
152
|
+
| `calib/` | Posterior calibration curves per state. |
|
|
153
|
+
| `Dist_vs_Corresp/`, `Dist_vs_Corresp_3/` | Overlap and correspondence as a function of `w`. |
|
|
154
|
+
|
|
155
|
+
Each plot is written as PDF and SVG. Each `.txt` file next to a plot holds the plotted values.
|
|
156
|
+
|
|
157
|
+
## Known issues
|
|
158
|
+
|
|
159
|
+
These issues come from the version of SAGAconf that the paper used. This release keeps them,
|
|
160
|
+
so that its results match that version exactly. Fixes will come in later releases.
|
|
161
|
+
|
|
162
|
+
- In `full` mode, `-k` does not merge states.
|
|
163
|
+
- In `full` mode without mnemonics, SAGAconf does not write the genome-wide results
|
|
164
|
+
(including `confident_segments.bed` and the UCSC track). Give `-bm` and `-vm` to get them.
|
|
165
|
+
- In `merge` mode, SAGAconf can stop with a `KeyError` after it writes the merge reports.
|
|
166
|
+
|
|
167
|
+
## Legacy interface
|
|
168
|
+
|
|
169
|
+
The original scripts still work, with the same flags and the same output:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
python SAGAconf_parser.py --saga chmm POSTERIORDIR 200 OUTDIR
|
|
173
|
+
python SAGAconf.py [-q | --r_only | --ct_only | ...] BASE VERIF OUTDIR
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
`sagaconf run` also accepts the original flag names (`--r_only`, `--windowsize`,
|
|
177
|
+
`--base_mnemonics`, and so on). In the original scripts, the `--active_regions` mode reads
|
|
178
|
+
`src/biointerpret/GRCh38-cCREs.bed` and `src/biointerpret/Meuleman.tsv` relative to the
|
|
179
|
+
working directory.
|
|
180
|
+
|
|
181
|
+
The code as the paper used it is on the [`legacy`](https://github.com/mehdiforoozandeh/SAGAconf/tree/legacy)
|
|
182
|
+
branch (tag `v0-legacy`). That branch also keeps the scripts that produced the paper's figures.
|
|
183
|
+
|
|
184
|
+
## Tests
|
|
185
|
+
|
|
186
|
+
[`tests/golden/`](tests/golden) checks that this version gives byte-identical output to the
|
|
187
|
+
legacy version. It runs both versions on the same input and compares every file they write,
|
|
188
|
+
and everything they print. [`tests/golden/gate.py`](tests/golden/gate.py) describes the method.
|
|
189
|
+
The check passes on synthetic data and on the paper's GM12878 ChromHMM and Segway annotations.
|
|
190
|
+
|
|
191
|
+
## Citation
|
|
192
|
+
|
|
193
|
+
```bibtex
|
|
194
|
+
@article{foroozandeh2024robust,
|
|
195
|
+
title = {Robust chromatin state annotation},
|
|
196
|
+
author = {Foroozandeh Shahraki, Mehdi and Farahbod, Marjan and Libbrecht, Maxwell W.},
|
|
197
|
+
journal = {Genome Research},
|
|
198
|
+
volume = {34},
|
|
199
|
+
number = {3},
|
|
200
|
+
pages = {469--483},
|
|
201
|
+
year = {2024},
|
|
202
|
+
doi = {10.1101/gr.278343.123}
|
|
203
|
+
}
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+

|
|
207
|
+
|
|
208
|
+
## License
|
|
209
|
+
|
|
210
|
+
SAGAconf is available under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "sagaconf"
|
|
7
|
+
version = "1.0.1"
|
|
8
|
+
description = "Calibrated confidence scores (r-values) for chromatin state annotations from SAGA methods."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
authors = [{ name = "Mehdi Foroozandeh" }]
|
|
13
|
+
requires-python = ">=3.9"
|
|
14
|
+
dependencies = [
|
|
15
|
+
"numpy>=1.21,<2",
|
|
16
|
+
"pandas>=1.3",
|
|
17
|
+
"scipy>=1.7",
|
|
18
|
+
"scikit-learn>=0.24",
|
|
19
|
+
"matplotlib>=3.4",
|
|
20
|
+
"seaborn>=0.11",
|
|
21
|
+
"pybedtools>=0.9",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
[project.optional-dependencies]
|
|
25
|
+
test = ["pytest>=7"]
|
|
26
|
+
|
|
27
|
+
[project.scripts]
|
|
28
|
+
sagaconf = "sagaconf.cli:main"
|
|
29
|
+
|
|
30
|
+
[project.urls]
|
|
31
|
+
Homepage = "https://github.com/mehdiforoozandeh/SAGAconf"
|
|
32
|
+
|
|
33
|
+
[tool.setuptools.packages.find]
|
|
34
|
+
include = ["sagaconf"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import pandas as pd
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def chrhmm_initialize_bin(chrom, numbins, res):
|
|
6
|
+
empty_bins = []
|
|
7
|
+
next_start = 0
|
|
8
|
+
for _ in range(numbins):
|
|
9
|
+
empty_bins.append(
|
|
10
|
+
[chrom, next_start, int(next_start+res)])
|
|
11
|
+
next_start = int(next_start+res)
|
|
12
|
+
empty_bins = pd.DataFrame(empty_bins, columns=['chr', 'start', 'end'])
|
|
13
|
+
empty_bins['start'] = empty_bins['start'].astype("int32")
|
|
14
|
+
empty_bins['end'] = empty_bins['end'].astype("int32")
|
|
15
|
+
return empty_bins
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def read_posterior_file(filepath):
|
|
19
|
+
with open(filepath,'r') as posteriorfile:
|
|
20
|
+
lines = posteriorfile.readlines()
|
|
21
|
+
vals = []
|
|
22
|
+
for il in range(len(lines)):
|
|
23
|
+
ilth_vals = lines[il].split('\t')
|
|
24
|
+
ilth_vals[-1] = ilth_vals[-1].replace("\n","")
|
|
25
|
+
vals.append(ilth_vals)
|
|
26
|
+
vals = pd.DataFrame(vals[2:], columns=["posterior{}".format(i.replace("E","")) for i in vals[1]])
|
|
27
|
+
vals = vals.astype("float32")
|
|
28
|
+
return vals
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def ChrHMM_read_posteriordir(posteriordir, resolution=200):
|
|
32
|
+
'''
|
|
33
|
+
for each file in posteriordir
|
|
34
|
+
Initialize emptybins based on chromsizes
|
|
35
|
+
fill in the posterior values for each slot
|
|
36
|
+
return DF
|
|
37
|
+
'''
|
|
38
|
+
ls = os.listdir(posteriordir)
|
|
39
|
+
to_parse = []
|
|
40
|
+
for f in ls:
|
|
41
|
+
# if rep in f:
|
|
42
|
+
to_parse.append(f)
|
|
43
|
+
|
|
44
|
+
parsed_posteriors = {}
|
|
45
|
+
for f in to_parse:
|
|
46
|
+
fileinfo = f.split("_")
|
|
47
|
+
if "chr" in fileinfo[-2]:
|
|
48
|
+
posteriors = read_posterior_file(posteriordir + '/' + f)
|
|
49
|
+
bins = chrhmm_initialize_bin(fileinfo[-2], len(posteriors), resolution)
|
|
50
|
+
posteriors = pd.concat([bins, posteriors], axis=1)
|
|
51
|
+
parsed_posteriors[fileinfo[-2]] = posteriors
|
|
52
|
+
|
|
53
|
+
parsed_posteriors = pd.concat([parsed_posteriors[c] for c in sorted(list(parsed_posteriors.keys()))], axis=0)
|
|
54
|
+
parsed_posteriors = parsed_posteriors.reset_index(drop=True)
|
|
55
|
+
return parsed_posteriors
|