sirna-data-grabber 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sirna_data_grabber-0.1.0/LICENSE +30 -0
- sirna_data_grabber-0.1.0/NOTICE.md +50 -0
- sirna_data_grabber-0.1.0/PKG-INFO +183 -0
- sirna_data_grabber-0.1.0/README.md +155 -0
- sirna_data_grabber-0.1.0/pyproject.toml +68 -0
- sirna_data_grabber-0.1.0/setup.cfg +4 -0
- sirna_data_grabber-0.1.0/src/sirna_data/__init__.py +29 -0
- sirna_data_grabber-0.1.0/src/sirna_data/fetch/__init__.py +22 -0
- sirna_data_grabber-0.1.0/src/sirna_data/fetch/cli.py +70 -0
- sirna_data_grabber-0.1.0/src/sirna_data/fetch/cmsirnadb.py +118 -0
- sirna_data_grabber-0.1.0/src/sirna_data/fetch/monopoli.py +102 -0
- sirna_data_grabber-0.1.0/src/sirna_data/fetch/shabalina.py +201 -0
- sirna_data_grabber-0.1.0/src/sirna_data/fetch/sirna_efficacy.py +88 -0
- sirna_data_grabber-0.1.0/src/sirna_data/ncbi_fetch.py +80 -0
- sirna_data_grabber-0.1.0/src/sirna_data/raw_loader.py +504 -0
- sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/PKG-INFO +183 -0
- sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/SOURCES.txt +22 -0
- sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/dependency_links.txt +1 -0
- sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/entry_points.txt +2 -0
- sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/requires.txt +11 -0
- sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/top_level.txt +1 -0
- sirna_data_grabber-0.1.0/tests/test_fetch_cli.py +88 -0
- sirna_data_grabber-0.1.0/tests/test_ncbi_fetch.py +102 -0
- sirna_data_grabber-0.1.0/tests/test_raw_loader.py +282 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Brandon Walker
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
NOTE: This license applies only to the code in this repository (the
|
|
26
|
+
`sirna_data` Python package, scripts/, and tests/). It does NOT apply to the
|
|
27
|
+
datasets in data/raw/, which are redistributed under their own original
|
|
28
|
+
licenses -- several of which restrict commercial use. See NOTICE.md and
|
|
29
|
+
data/DATA_SOURCES.md before using the data itself outside non-commercial
|
|
30
|
+
research.
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# NOTICE — data licenses
|
|
2
|
+
|
|
3
|
+
This repo's code (`sirna_data`, `tests/`) is MIT licensed (see
|
|
4
|
+
[`LICENSE`](LICENSE)) and may be used, modified, and redistributed freely,
|
|
5
|
+
including commercially.
|
|
6
|
+
|
|
7
|
+
**The datasets in `data/raw/` are separate from the code and are NOT MIT
|
|
8
|
+
licensed.** Each was fetched from its original publisher and is redistributed
|
|
9
|
+
here under that publisher's own terms. The MIT license on the code does not
|
|
10
|
+
extend to the data, and using this permissively-licensed loader to read the
|
|
11
|
+
data does not lift the data's own restrictions. Most sources below are
|
|
12
|
+
**non-commercial only** — read this table before using the data for anything
|
|
13
|
+
beyond non-commercial research, and see
|
|
14
|
+
[`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for full terms and provenance
|
|
15
|
+
per source.
|
|
16
|
+
|
|
17
|
+
## Sources loaded by `load_records()`
|
|
18
|
+
|
|
19
|
+
| Source | File(s) | License | Commercial use? |
|
|
20
|
+
|---|---|---|---|
|
|
21
|
+
| siRNAEfficacyDB (Zhang et al. 2024) | `sirna_efficacy.csv` | CC BY-NC | **No** — non-commercial only |
|
|
22
|
+
| Monopoli et al. 2023 | `monopoli_extra.csv` | CC BY 4.0 | Yes, with attribution |
|
|
23
|
+
| Shabalina et al. 2006 | `shabalina_extra.csv` | CC BY 2.0 | Yes, with attribution |
|
|
24
|
+
| CMsiRNAdb (He et al. 2026) | `cmsirnadb_full_raw.tsv` | CC BY-NC-ND 4.0 | **No** — non-commercial only, and the "ND" term means only the original unmodified file may be redistributed (see below) |
|
|
25
|
+
| NCBI RefSeq/GenBank transcripts | `*_transcripts.fasta` | Public domain | Yes, unrestricted |
|
|
26
|
+
|
|
27
|
+
**CMsiRNAdb's "No Derivatives" term**: this repo ships only the untouched
|
|
28
|
+
original `cmsirnadb_full_raw.tsv` download. All filtering, collapsing, and
|
|
29
|
+
transformation happens in code at load time
|
|
30
|
+
(`_load_cmsirnadb_records`/`_load_cmsirnadb_full_records` in
|
|
31
|
+
`src/sirna_data/raw_loader.py`), not as a precomputed derivative file — so no
|
|
32
|
+
adaptation of CMsiRNAdb's data is redistributed, only the original plus code
|
|
33
|
+
that anyone can run themselves. See `data/CMSIRNADB_FULL_RETRIEVAL.md`.
|
|
34
|
+
|
|
35
|
+
## Other files present in `data/raw/` but not used by anything in `sirna_data`
|
|
36
|
+
|
|
37
|
+
These were investigated as candidate sources (see
|
|
38
|
+
[`data/DATA_SOURCE_LEDGER.md`](data/DATA_SOURCE_LEDGER.md)) and kept for
|
|
39
|
+
reference/provenance, but nothing in `sirna_data` reads them:
|
|
40
|
+
|
|
41
|
+
| File(s) | Source | License |
|
|
42
|
+
|---|---|---|
|
|
43
|
+
| `sirecords_efficacy.csv`, `sirecords_new_only.csv` | siRecords (04/28/05 release, via Internet Archive) | **Restricted, not established for redistribution.** Researched: the database's own paper (Ren et al. 2009, NAR 37:D146-D149) is CC BY-NC licensed as an *article*, but its DATA ACCESS section says bulk copies were only ever given to "academic users" who emailed the authors directly — not published as an open download. The data here was recovered from an Internet Archive snapshot, not that channel, so no license actually covers this copy. See `data/sirecords_overlap_analysis.md` for the full writeup and sources. Kept locally but excluded from git (`.gitignore`). |
|
|
44
|
+
|
|
45
|
+
## If you're not sure whether your use is covered
|
|
46
|
+
|
|
47
|
+
None of the above is legal advice. If your use case isn't clearly
|
|
48
|
+
non-commercial research, check the original source's license directly (links
|
|
49
|
+
in `data/DATA_SOURCES.md`) or contact the original authors before relying on
|
|
50
|
+
this data.
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sirna-data-grabber
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Reusable loader for the siRNA knockdown-efficacy dataset (siRNAEfficacyDB + supplementary sources), plus the fetchers that built it.
|
|
5
|
+
Author-email: Brandon Walker <brandon.walker@ucsf.edu>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/BrandonWalk/sirna-data-grabber
|
|
8
|
+
Project-URL: Repository, https://github.com/BrandonWalk/sirna-data-grabber
|
|
9
|
+
Project-URL: Issues, https://github.com/BrandonWalk/sirna-data-grabber/issues
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
License-File: NOTICE.md
|
|
18
|
+
Requires-Dist: pandas>=2.0
|
|
19
|
+
Requires-Dist: requests>=2.32
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest>=8.0; extra == "test"
|
|
22
|
+
Provides-Extra: lint
|
|
23
|
+
Requires-Dist: ruff>=0.6; extra == "lint"
|
|
24
|
+
Requires-Dist: mypy>=1.10; extra == "lint"
|
|
25
|
+
Requires-Dist: pandas-stubs; extra == "lint"
|
|
26
|
+
Requires-Dist: types-requests; extra == "lint"
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# sirna-data-grabber
|
|
30
|
+
|
|
31
|
+
A standalone siRNA knockdown-efficacy dataset: the raw data files, full
|
|
32
|
+
provenance/license documentation, and a small reusable Python package
|
|
33
|
+
(`sirna_data`) for loading it -- and, since `pip install sirna-data-grabber`
|
|
34
|
+
alone can't ship most of this non-commercial data, a bundled `sirna-data-fetch`
|
|
35
|
+
command that re-fetches it from its original sources. Any project that wants
|
|
36
|
+
this dataset can depend on this repo (or just the PyPI package) rather than
|
|
37
|
+
vendoring a copy of the data or the loading code.
|
|
38
|
+
|
|
39
|
+
## License
|
|
40
|
+
|
|
41
|
+
**The code in this repo (`sirna_data`, `tests/`) is MIT licensed** — see
|
|
42
|
+
[`LICENSE`](LICENSE). Use it, modify it, ship it commercially, whatever you
|
|
43
|
+
want.
|
|
44
|
+
|
|
45
|
+
**The data in `data/raw/` is NOT covered by that license.** It's redistributed
|
|
46
|
+
under each original source's own terms, and most of those sources are
|
|
47
|
+
**non-commercial only** (CC BY-NC / CC BY-NC-ND). Loading the data with this
|
|
48
|
+
permissively-licensed code does not lift those restrictions — you still have
|
|
49
|
+
to comply with them separately. See [`NOTICE.md`](NOTICE.md) for the
|
|
50
|
+
per-source summary and [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for
|
|
51
|
+
full terms before using the data itself, especially commercially.
|
|
52
|
+
|
|
53
|
+
## What's here
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
LICENSE MIT license -- covers the code only, not data/raw/
|
|
57
|
+
NOTICE.md per-source data license summary (see License section above)
|
|
58
|
+
data/
|
|
59
|
+
raw/ fetched CSVs + FASTA transcripts (the actual dataset)
|
|
60
|
+
DATA_SOURCES.md full provenance + license terms for every source
|
|
61
|
+
DATA_SOURCE_LEDGER.md audit: what's trainable, what's not, and why
|
|
62
|
+
CMSIRNADB_FULL_RETRIEVAL.md detail on the CMsiRNAdb full-database retrieval
|
|
63
|
+
DEMETER2_README.txt upstream release notes for DepMap DEMETER2 (investigated, not included -- see FUNCTIONAL_GENOMICS_SCREENS.md)
|
|
64
|
+
FUNCTIONAL_GENOMICS_SCREENS.md notes on functional-genomics screen sources considered
|
|
65
|
+
POTENTIAL_DATA_SOURCES.md landscape of sources investigated
|
|
66
|
+
sirecords_overlap_analysis.md siRecords overlap/dedup analysis
|
|
67
|
+
data_source_ledger.csv machine-readable companion to DATA_SOURCE_LEDGER.md
|
|
68
|
+
*.png figures referenced by the docs above
|
|
69
|
+
src/sirna_data/
|
|
70
|
+
raw_loader.py load + merge every source into SiRNARecord rows
|
|
71
|
+
ncbi_fetch.py fetch a gene's RefSeq mRNA transcript by symbol
|
|
72
|
+
__init__.py public API
|
|
73
|
+
fetch/ sirna-data-fetch CLI + per-source fetchers (see Install below)
|
|
74
|
+
cli.py `sirna-data-fetch` entry point ([project.scripts])
|
|
75
|
+
sirna_efficacy.py siRNAEfficacyDB + NCBI -> sirna_efficacy.csv, mrna_transcripts.fasta
|
|
76
|
+
monopoli.py Monopoli et al. 2023 supplementary data -> monopoli_*
|
|
77
|
+
shabalina.py Shabalina et al. 2006 supplementary data -> shabalina_*
|
|
78
|
+
cmsirnadb.py CMsiRNAdb + NCBI -> cmsirnadb_full_raw.tsv, cmsirnadb*_transcripts.fasta
|
|
79
|
+
tests/
|
|
80
|
+
test_raw_loader.py unit tests for raw_loader.py (fixtures, no real data needed)
|
|
81
|
+
test_ncbi_fetch.py unit tests for ncbi_fetch.py (mocked HTTP calls)
|
|
82
|
+
conftest.py shared pytest fixtures
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Start with [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for what's in the
|
|
86
|
+
dataset and where it came from; [`data/DATA_SOURCE_LEDGER.md`](data/DATA_SOURCE_LEDGER.md)
|
|
87
|
+
for the bottom-line audit (6,577 trainable records across 87 genes, 4
|
|
88
|
+
sources — 16,178 records / 97 genes if the optional CMsiRNAdb full-database
|
|
89
|
+
retrieval is also included). Primary source is **siRNAEfficacyDB** (Zhang
|
|
90
|
+
et al. 2024, CC BY-NC); see the docs for the rest and their individual
|
|
91
|
+
license terms before reusing this data outside this project.
|
|
92
|
+
|
|
93
|
+
## Install
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
python3 -m venv .venv && source .venv/bin/activate
|
|
97
|
+
pip install -e .
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
This installs the `sirna_data` package in editable mode, so it resolves
|
|
101
|
+
`data/raw/` relative to this checkout automatically. If you copy the `data/`
|
|
102
|
+
folder somewhere else, point at it explicitly instead:
|
|
103
|
+
|
|
104
|
+
```
|
|
105
|
+
export SIRNA_DATA_DIR=/path/to/data/raw
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Re-fetching the raw data from scratch (not required if `data/raw/` already
|
|
109
|
+
has the files) uses the `sirna-data-fetch` command, installed automatically
|
|
110
|
+
with the package -- no extras needed:
|
|
111
|
+
|
|
112
|
+
```
|
|
113
|
+
sirna-data-fetch --dest data/raw
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
This also means a plain `pip install sirna-data-grabber` from PyPI (with no
|
|
117
|
+
git checkout at all) can reconstruct the full dataset itself:
|
|
118
|
+
|
|
119
|
+
```
|
|
120
|
+
pip install sirna-data-grabber
|
|
121
|
+
sirna-data-fetch --dest ./my_data
|
|
122
|
+
export SIRNA_DATA_DIR=./my_data
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
`sirna-data-fetch --only sirna_efficacy monopoli` fetches a subset instead of
|
|
126
|
+
all four sources; see `sirna-data-fetch --help`.
|
|
127
|
+
|
|
128
|
+
## Usage
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
from sirna_data import load_records, fetch_mrna_by_gene
|
|
132
|
+
|
|
133
|
+
records = load_records() # list[SiRNARecord]
|
|
134
|
+
print(len(records), "records across", len({r.gene for r in records}), "genes")
|
|
135
|
+
|
|
136
|
+
r = records[0]
|
|
137
|
+
r.guide_seq # siRNA antisense strand
|
|
138
|
+
r.mrna_window # local mRNA context around the real target site
|
|
139
|
+
r.label # experimental %knockdown / %inhibition
|
|
140
|
+
r.source # provenance, e.g. "siRNAEfficacyDB"
|
|
141
|
+
|
|
142
|
+
# Look up any gene's RefSeq transcript live from NCBI:
|
|
143
|
+
transcript = fetch_mrna_by_gene("TP53")
|
|
144
|
+
transcript.accession, transcript.sequence
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
`load_records()` takes `include_monopoli` / `include_shabalina` /
|
|
148
|
+
`include_cmsirnadb` / `include_cmsirnadb_full` flags to exclude any
|
|
149
|
+
supplementary source and use only the primary siRNAEfficacyDB set.
|
|
150
|
+
|
|
151
|
+
## Using this from another project
|
|
152
|
+
|
|
153
|
+
Install as a sibling checkout in editable mode:
|
|
154
|
+
|
|
155
|
+
```
|
|
156
|
+
pip install -e ../sirna-data-grabber
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
That gives you `import sirna_data` with no other coupling — this repo only
|
|
160
|
+
depends on pandas and requests, and knows nothing about any particular
|
|
161
|
+
downstream model or feature-engineering pipeline.
|
|
162
|
+
|
|
163
|
+
## Tests
|
|
164
|
+
|
|
165
|
+
```
|
|
166
|
+
pip install -e ".[test]"
|
|
167
|
+
pytest
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Tests run entirely against small in-memory/tmp-dir fixtures (see
|
|
171
|
+
`tests/conftest.py`) and mocked HTTP calls, so they don't touch the real
|
|
172
|
+
dataset or the network.
|
|
173
|
+
|
|
174
|
+
## Linting and type checking
|
|
175
|
+
|
|
176
|
+
```
|
|
177
|
+
pip install -e ".[lint]"
|
|
178
|
+
ruff check .
|
|
179
|
+
mypy
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
Both run in CI on every pull request (`.github/workflows/tests.yml`), alongside
|
|
183
|
+
the test matrix.
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
# sirna-data-grabber
|
|
2
|
+
|
|
3
|
+
A standalone siRNA knockdown-efficacy dataset: the raw data files, full
|
|
4
|
+
provenance/license documentation, and a small reusable Python package
|
|
5
|
+
(`sirna_data`) for loading it -- and, since `pip install sirna-data-grabber`
|
|
6
|
+
alone can't ship most of this non-commercial data, a bundled `sirna-data-fetch`
|
|
7
|
+
command that re-fetches it from its original sources. Any project that wants
|
|
8
|
+
this dataset can depend on this repo (or just the PyPI package) rather than
|
|
9
|
+
vendoring a copy of the data or the loading code.
|
|
10
|
+
|
|
11
|
+
## License
|
|
12
|
+
|
|
13
|
+
**The code in this repo (`sirna_data`, `tests/`) is MIT licensed** — see
|
|
14
|
+
[`LICENSE`](LICENSE). Use it, modify it, ship it commercially, whatever you
|
|
15
|
+
want.
|
|
16
|
+
|
|
17
|
+
**The data in `data/raw/` is NOT covered by that license.** It's redistributed
|
|
18
|
+
under each original source's own terms, and most of those sources are
|
|
19
|
+
**non-commercial only** (CC BY-NC / CC BY-NC-ND). Loading the data with this
|
|
20
|
+
permissively-licensed code does not lift those restrictions — you still have
|
|
21
|
+
to comply with them separately. See [`NOTICE.md`](NOTICE.md) for the
|
|
22
|
+
per-source summary and [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for
|
|
23
|
+
full terms before using the data itself, especially commercially.
|
|
24
|
+
|
|
25
|
+
## What's here
|
|
26
|
+
|
|
27
|
+
```
|
|
28
|
+
LICENSE MIT license -- covers the code only, not data/raw/
|
|
29
|
+
NOTICE.md per-source data license summary (see License section above)
|
|
30
|
+
data/
|
|
31
|
+
raw/ fetched CSVs + FASTA transcripts (the actual dataset)
|
|
32
|
+
DATA_SOURCES.md full provenance + license terms for every source
|
|
33
|
+
DATA_SOURCE_LEDGER.md audit: what's trainable, what's not, and why
|
|
34
|
+
CMSIRNADB_FULL_RETRIEVAL.md detail on the CMsiRNAdb full-database retrieval
|
|
35
|
+
DEMETER2_README.txt upstream release notes for DepMap DEMETER2 (investigated, not included -- see FUNCTIONAL_GENOMICS_SCREENS.md)
|
|
36
|
+
FUNCTIONAL_GENOMICS_SCREENS.md notes on functional-genomics screen sources considered
|
|
37
|
+
POTENTIAL_DATA_SOURCES.md landscape of sources investigated
|
|
38
|
+
sirecords_overlap_analysis.md siRecords overlap/dedup analysis
|
|
39
|
+
data_source_ledger.csv machine-readable companion to DATA_SOURCE_LEDGER.md
|
|
40
|
+
*.png figures referenced by the docs above
|
|
41
|
+
src/sirna_data/
|
|
42
|
+
raw_loader.py load + merge every source into SiRNARecord rows
|
|
43
|
+
ncbi_fetch.py fetch a gene's RefSeq mRNA transcript by symbol
|
|
44
|
+
__init__.py public API
|
|
45
|
+
fetch/ sirna-data-fetch CLI + per-source fetchers (see Install below)
|
|
46
|
+
cli.py `sirna-data-fetch` entry point ([project.scripts])
|
|
47
|
+
sirna_efficacy.py siRNAEfficacyDB + NCBI -> sirna_efficacy.csv, mrna_transcripts.fasta
|
|
48
|
+
monopoli.py Monopoli et al. 2023 supplementary data -> monopoli_*
|
|
49
|
+
shabalina.py Shabalina et al. 2006 supplementary data -> shabalina_*
|
|
50
|
+
cmsirnadb.py CMsiRNAdb + NCBI -> cmsirnadb_full_raw.tsv, cmsirnadb*_transcripts.fasta
|
|
51
|
+
tests/
|
|
52
|
+
test_raw_loader.py unit tests for raw_loader.py (fixtures, no real data needed)
|
|
53
|
+
test_ncbi_fetch.py unit tests for ncbi_fetch.py (mocked HTTP calls)
|
|
54
|
+
conftest.py shared pytest fixtures
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Start with [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for what's in the
|
|
58
|
+
dataset and where it came from; [`data/DATA_SOURCE_LEDGER.md`](data/DATA_SOURCE_LEDGER.md)
|
|
59
|
+
for the bottom-line audit (6,577 trainable records across 87 genes, 4
|
|
60
|
+
sources — 16,178 records / 97 genes if the optional CMsiRNAdb full-database
|
|
61
|
+
retrieval is also included). Primary source is **siRNAEfficacyDB** (Zhang
|
|
62
|
+
et al. 2024, CC BY-NC); see the docs for the rest and their individual
|
|
63
|
+
license terms before reusing this data outside this project.
|
|
64
|
+
|
|
65
|
+
## Install
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
python3 -m venv .venv && source .venv/bin/activate
|
|
69
|
+
pip install -e .
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
This installs the `sirna_data` package in editable mode, so it resolves
|
|
73
|
+
`data/raw/` relative to this checkout automatically. If you copy the `data/`
|
|
74
|
+
folder somewhere else, point at it explicitly instead:
|
|
75
|
+
|
|
76
|
+
```
|
|
77
|
+
export SIRNA_DATA_DIR=/path/to/data/raw
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Re-fetching the raw data from scratch (not required if `data/raw/` already
|
|
81
|
+
has the files) uses the `sirna-data-fetch` command, installed automatically
|
|
82
|
+
with the package -- no extras needed:
|
|
83
|
+
|
|
84
|
+
```
|
|
85
|
+
sirna-data-fetch --dest data/raw
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
This also means a plain `pip install sirna-data-grabber` from PyPI (with no
|
|
89
|
+
git checkout at all) can reconstruct the full dataset itself:
|
|
90
|
+
|
|
91
|
+
```
|
|
92
|
+
pip install sirna-data-grabber
|
|
93
|
+
sirna-data-fetch --dest ./my_data
|
|
94
|
+
export SIRNA_DATA_DIR=./my_data
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
`sirna-data-fetch --only sirna_efficacy monopoli` fetches a subset instead of
|
|
98
|
+
all four sources; see `sirna-data-fetch --help`.
|
|
99
|
+
|
|
100
|
+
## Usage
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from sirna_data import load_records, fetch_mrna_by_gene
|
|
104
|
+
|
|
105
|
+
records = load_records() # list[SiRNARecord]
|
|
106
|
+
print(len(records), "records across", len({r.gene for r in records}), "genes")
|
|
107
|
+
|
|
108
|
+
r = records[0]
|
|
109
|
+
r.guide_seq # siRNA antisense strand
|
|
110
|
+
r.mrna_window # local mRNA context around the real target site
|
|
111
|
+
r.label # experimental %knockdown / %inhibition
|
|
112
|
+
r.source # provenance, e.g. "siRNAEfficacyDB"
|
|
113
|
+
|
|
114
|
+
# Look up any gene's RefSeq transcript live from NCBI:
|
|
115
|
+
transcript = fetch_mrna_by_gene("TP53")
|
|
116
|
+
transcript.accession, transcript.sequence
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
`load_records()` takes `include_monopoli` / `include_shabalina` /
|
|
120
|
+
`include_cmsirnadb` / `include_cmsirnadb_full` flags to exclude any
|
|
121
|
+
supplementary source and use only the primary siRNAEfficacyDB set.
|
|
122
|
+
|
|
123
|
+
## Using this from another project
|
|
124
|
+
|
|
125
|
+
Install as a sibling checkout in editable mode:
|
|
126
|
+
|
|
127
|
+
```
|
|
128
|
+
pip install -e ../sirna-data-grabber
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
That gives you `import sirna_data` with no other coupling — this repo only
|
|
132
|
+
depends on pandas and requests, and knows nothing about any particular
|
|
133
|
+
downstream model or feature-engineering pipeline.
|
|
134
|
+
|
|
135
|
+
## Tests
|
|
136
|
+
|
|
137
|
+
```
|
|
138
|
+
pip install -e ".[test]"
|
|
139
|
+
pytest
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Tests run entirely against small in-memory/tmp-dir fixtures (see
|
|
143
|
+
`tests/conftest.py`) and mocked HTTP calls, so they don't touch the real
|
|
144
|
+
dataset or the network.
|
|
145
|
+
|
|
146
|
+
## Linting and type checking
|
|
147
|
+
|
|
148
|
+
```
|
|
149
|
+
pip install -e ".[lint]"
|
|
150
|
+
ruff check .
|
|
151
|
+
mypy
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
Both run in CI on every pull request (`.github/workflows/tests.yml`), alongside
|
|
155
|
+
the test matrix.
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "sirna-data-grabber"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Reusable loader for the siRNA knockdown-efficacy dataset (siRNAEfficacyDB + supplementary sources), plus the fetchers that built it."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Brandon Walker", email = "brandon.walker@ucsf.edu" },
|
|
14
|
+
]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Intended Audience :: Science/Research",
|
|
19
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
20
|
+
]
|
|
21
|
+
dependencies = [
|
|
22
|
+
"pandas>=2.0",
|
|
23
|
+
"requests>=2.32",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[project.urls]
|
|
27
|
+
Homepage = "https://github.com/BrandonWalk/sirna-data-grabber"
|
|
28
|
+
Repository = "https://github.com/BrandonWalk/sirna-data-grabber"
|
|
29
|
+
Issues = "https://github.com/BrandonWalk/sirna-data-grabber/issues"
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
test = ["pytest>=8.0"]
|
|
33
|
+
lint = ["ruff>=0.6", "mypy>=1.10", "pandas-stubs", "types-requests"]
|
|
34
|
+
|
|
35
|
+
# Installed automatically -- `sirna-data-fetch` fetches every source
|
|
36
|
+
# load_records() reads into a local directory (data/DATA_SOURCES.md), so a
|
|
37
|
+
# plain `pip install sirna-data-grabber` can reconstruct the dataset without
|
|
38
|
+
# cloning the git repo. No extra dependencies: everything under
|
|
39
|
+
# sirna_data/fetch/ only needs pandas (already required above) plus the
|
|
40
|
+
# standard library.
|
|
41
|
+
[project.scripts]
|
|
42
|
+
sirna-data-fetch = "sirna_data.fetch.cli:main"
|
|
43
|
+
|
|
44
|
+
[tool.setuptools.packages.find]
|
|
45
|
+
where = ["src"]
|
|
46
|
+
|
|
47
|
+
[tool.pytest.ini_options]
|
|
48
|
+
testpaths = ["tests"]
|
|
49
|
+
addopts = "-ra"
|
|
50
|
+
|
|
51
|
+
[tool.ruff]
|
|
52
|
+
line-length = 100
|
|
53
|
+
target-version = "py310"
|
|
54
|
+
|
|
55
|
+
[tool.ruff.lint]
|
|
56
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
57
|
+
|
|
58
|
+
[tool.ruff.lint.per-file-ignores]
|
|
59
|
+
# long single-line TSV fixture rows are more readable unwrapped
|
|
60
|
+
"tests/conftest.py" = ["E501"]
|
|
61
|
+
|
|
62
|
+
[tool.mypy]
|
|
63
|
+
python_version = "3.10"
|
|
64
|
+
files = ["src"]
|
|
65
|
+
ignore_missing_imports = true
|
|
66
|
+
warn_unused_ignores = true
|
|
67
|
+
warn_redundant_casts = true
|
|
68
|
+
check_untyped_defs = true
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""sirna_data: reusable loader for the siRNA knockdown-efficacy dataset.
|
|
2
|
+
|
|
3
|
+
from sirna_data import load_records, SiRNARecord, fetch_mrna_by_gene
|
|
4
|
+
|
|
5
|
+
`load_records()` returns the full merged, provenance-documented dataset
|
|
6
|
+
(siRNAEfficacyDB + supplementary sources) as a list of `SiRNARecord`, each
|
|
7
|
+
pairing an siRNA guide with its real local mRNA target context and
|
|
8
|
+
experimentally measured knockdown label. See ../../data/DATA_SOURCES.md for
|
|
9
|
+
where every row comes from.
|
|
10
|
+
|
|
11
|
+
`fetch_mrna_by_gene()` looks up a gene's RefSeq mRNA transcript live from
|
|
12
|
+
NCBI, for callers that only have a gene symbol.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from .ncbi_fetch import FetchedTranscript, GeneNotFoundError, fetch_mrna_by_gene
|
|
17
|
+
from .raw_loader import DATA_DIR, SiRNARecord, load_records, read_fasta
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"load_records",
|
|
21
|
+
"SiRNARecord",
|
|
22
|
+
"read_fasta",
|
|
23
|
+
"DATA_DIR",
|
|
24
|
+
"fetch_mrna_by_gene",
|
|
25
|
+
"FetchedTranscript",
|
|
26
|
+
"GeneNotFoundError",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Fetch the raw data `sirna_data.load_records()` reads, from each source's
|
|
2
|
+
original location, into a local directory -- so a plain `pip install
|
|
3
|
+
sirna-data-grabber` install can reconstruct the dataset without cloning the
|
|
4
|
+
git repo.
|
|
5
|
+
|
|
6
|
+
from sirna_data.fetch import cmsirnadb, monopoli, shabalina, sirna_efficacy
|
|
7
|
+
sirna_efficacy.fetch(Path("./my_data"))
|
|
8
|
+
|
|
9
|
+
Or from the command line (installed automatically, no extras required --
|
|
10
|
+
every fetcher here only needs pandas, already a core dependency, plus the
|
|
11
|
+
standard library):
|
|
12
|
+
|
|
13
|
+
sirna-data-fetch --dest ./my_data
|
|
14
|
+
export SIRNA_DATA_DIR=./my_data
|
|
15
|
+
|
|
16
|
+
This only covers the four sources `load_records()` actually reads
|
|
17
|
+
(siRNAEfficacyDB, Monopoli 2023, Shabalina 2006, CMsiRNAdb). See
|
|
18
|
+
../../../data/DATA_SOURCES.md for full provenance/license notes per source,
|
|
19
|
+
and NOTICE.md before using the fetched data commercially -- most of it is
|
|
20
|
+
non-commercial only.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Command-line entry point: `sirna-data-fetch` (registered via
|
|
2
|
+
[project.scripts] in pyproject.toml). Fetches every source
|
|
3
|
+
`sirna_data.load_records()` reads, from its original location, into a local
|
|
4
|
+
directory.
|
|
5
|
+
|
|
6
|
+
sirna-data-fetch --dest ./my_data
|
|
7
|
+
export SIRNA_DATA_DIR=./my_data
|
|
8
|
+
|
|
9
|
+
This is installed automatically with `pip install sirna-data-grabber` --
|
|
10
|
+
no extras needed, since every fetcher here only needs pandas (already a
|
|
11
|
+
core dependency) plus the standard library.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import argparse
|
|
16
|
+
import os
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
from . import cmsirnadb, monopoli, shabalina, sirna_efficacy
|
|
20
|
+
|
|
21
|
+
# Order matters only for display; each fetcher is independent.
|
|
22
|
+
SOURCES = {
|
|
23
|
+
"sirna_efficacy": sirna_efficacy.fetch,
|
|
24
|
+
"monopoli": monopoli.fetch,
|
|
25
|
+
"shabalina": shabalina.fetch,
|
|
26
|
+
"cmsirnadb": cmsirnadb.fetch,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def main() -> None:
|
|
31
|
+
parser = argparse.ArgumentParser(
|
|
32
|
+
prog="sirna-data-fetch",
|
|
33
|
+
description=(
|
|
34
|
+
"Fetch the siRNA knockdown-efficacy dataset (siRNAEfficacyDB + "
|
|
35
|
+
"supplementary sources) from its original sources into a local "
|
|
36
|
+
"directory, for use with sirna_data.load_records()."
|
|
37
|
+
),
|
|
38
|
+
)
|
|
39
|
+
parser.add_argument(
|
|
40
|
+
"--dest",
|
|
41
|
+
type=Path,
|
|
42
|
+
default=Path(os.environ.get("SIRNA_DATA_DIR", "data/raw")),
|
|
43
|
+
help=(
|
|
44
|
+
"Directory to write the fetched files into "
|
|
45
|
+
"(default: $SIRNA_DATA_DIR if set, else ./data/raw)."
|
|
46
|
+
),
|
|
47
|
+
)
|
|
48
|
+
parser.add_argument(
|
|
49
|
+
"--only",
|
|
50
|
+
nargs="+",
|
|
51
|
+
choices=sorted(SOURCES),
|
|
52
|
+
metavar="SOURCE",
|
|
53
|
+
help=f"Fetch only these sources instead of all four ({', '.join(sorted(SOURCES))}).",
|
|
54
|
+
)
|
|
55
|
+
args = parser.parse_args()
|
|
56
|
+
|
|
57
|
+
dest: Path = args.dest.resolve()
|
|
58
|
+
dest.mkdir(parents=True, exist_ok=True)
|
|
59
|
+
names = args.only or sorted(SOURCES)
|
|
60
|
+
|
|
61
|
+
for name in names:
|
|
62
|
+
print(f"=== {name} ===")
|
|
63
|
+
SOURCES[name](dest)
|
|
64
|
+
print()
|
|
65
|
+
|
|
66
|
+
print(f"Done. To use this data:\n export SIRNA_DATA_DIR={dest}")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
if __name__ == "__main__":
|
|
70
|
+
main()
|