sirna-data-grabber 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. sirna_data_grabber-0.1.0/LICENSE +30 -0
  2. sirna_data_grabber-0.1.0/NOTICE.md +50 -0
  3. sirna_data_grabber-0.1.0/PKG-INFO +183 -0
  4. sirna_data_grabber-0.1.0/README.md +155 -0
  5. sirna_data_grabber-0.1.0/pyproject.toml +68 -0
  6. sirna_data_grabber-0.1.0/setup.cfg +4 -0
  7. sirna_data_grabber-0.1.0/src/sirna_data/__init__.py +29 -0
  8. sirna_data_grabber-0.1.0/src/sirna_data/fetch/__init__.py +22 -0
  9. sirna_data_grabber-0.1.0/src/sirna_data/fetch/cli.py +70 -0
  10. sirna_data_grabber-0.1.0/src/sirna_data/fetch/cmsirnadb.py +118 -0
  11. sirna_data_grabber-0.1.0/src/sirna_data/fetch/monopoli.py +102 -0
  12. sirna_data_grabber-0.1.0/src/sirna_data/fetch/shabalina.py +201 -0
  13. sirna_data_grabber-0.1.0/src/sirna_data/fetch/sirna_efficacy.py +88 -0
  14. sirna_data_grabber-0.1.0/src/sirna_data/ncbi_fetch.py +80 -0
  15. sirna_data_grabber-0.1.0/src/sirna_data/raw_loader.py +504 -0
  16. sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/PKG-INFO +183 -0
  17. sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/SOURCES.txt +22 -0
  18. sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/dependency_links.txt +1 -0
  19. sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/entry_points.txt +2 -0
  20. sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/requires.txt +11 -0
  21. sirna_data_grabber-0.1.0/src/sirna_data_grabber.egg-info/top_level.txt +1 -0
  22. sirna_data_grabber-0.1.0/tests/test_fetch_cli.py +88 -0
  23. sirna_data_grabber-0.1.0/tests/test_ncbi_fetch.py +102 -0
  24. sirna_data_grabber-0.1.0/tests/test_raw_loader.py +282 -0
@@ -0,0 +1,30 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Brandon Walker
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
22
+
23
+ ---
24
+
25
+ NOTE: This license applies only to the code in this repository (the
26
+ `sirna_data` Python package, scripts/, and tests/). It does NOT apply to the
27
+ datasets in data/raw/, which are redistributed under their own original
28
+ licenses -- several of which restrict commercial use. See NOTICE.md and
29
+ data/DATA_SOURCES.md before using the data itself outside non-commercial
30
+ research.
@@ -0,0 +1,50 @@
1
+ # NOTICE — data licenses
2
+
3
+ This repo's code (`sirna_data`, `tests/`) is MIT licensed (see
4
+ [`LICENSE`](LICENSE)) and may be used, modified, and redistributed freely,
5
+ including commercially.
6
+
7
+ **The datasets in `data/raw/` are separate from the code and are NOT MIT
8
+ licensed.** Each was fetched from its original publisher and is redistributed
9
+ here under that publisher's own terms. The MIT license on the code does not
10
+ extend to the data, and using this permissively-licensed loader to read the
11
+ data does not lift the data's own restrictions. Most sources below are
12
+ **non-commercial only** — read this table before using the data for anything
13
+ beyond non-commercial research, and see
14
+ [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for full terms and provenance
15
+ per source.
16
+
17
+ ## Sources loaded by `load_records()`
18
+
19
+ | Source | File(s) | License | Commercial use? |
20
+ |---|---|---|---|
21
+ | siRNAEfficacyDB (Zhang et al. 2024) | `sirna_efficacy.csv` | CC BY-NC | **No** — non-commercial only |
22
+ | Monopoli et al. 2023 | `monopoli_extra.csv` | CC BY 4.0 | Yes, with attribution |
23
+ | Shabalina et al. 2006 | `shabalina_extra.csv` | CC BY 2.0 | Yes, with attribution |
24
+ | CMsiRNAdb (He et al. 2026) | `cmsirnadb_full_raw.tsv` | CC BY-NC-ND 4.0 | **No** — non-commercial only, and the "ND" term means only the original unmodified file may be redistributed (see below) |
25
+ | NCBI RefSeq/GenBank transcripts | `*_transcripts.fasta` | Public domain | Yes, unrestricted |
26
+
27
+ **CMsiRNAdb's "No Derivatives" term**: this repo ships only the untouched
28
+ original `cmsirnadb_full_raw.tsv` download. All filtering, collapsing, and
29
+ transformation happens in code at load time
30
+ (`_load_cmsirnadb_records`/`_load_cmsirnadb_full_records` in
31
+ `src/sirna_data/raw_loader.py`), not as a precomputed derivative file — so no
32
+ adaptation of CMsiRNAdb's data is redistributed, only the original plus code
33
+ that anyone can run themselves. See `data/CMSIRNADB_FULL_RETRIEVAL.md`.
34
+
35
+ ## Other files present in `data/raw/` but not used by anything in `sirna_data`
36
+
37
+ These were investigated as candidate sources (see
38
+ [`data/DATA_SOURCE_LEDGER.md`](data/DATA_SOURCE_LEDGER.md)) and kept for
39
+ reference/provenance, but nothing in `sirna_data` reads them:
40
+
41
+ | File(s) | Source | License |
42
+ |---|---|---|
43
+ | `sirecords_efficacy.csv`, `sirecords_new_only.csv` | siRecords (04/28/05 release, via Internet Archive) | **Restricted, not established for redistribution.** Researched: the database's own paper (Ren et al. 2009, NAR 37:D146-D149) is CC BY-NC licensed as an *article*, but its DATA ACCESS section says bulk copies were only ever given to "academic users" who emailed the authors directly — not published as an open download. The data here was recovered from an Internet Archive snapshot, not that channel, so no license actually covers this copy. See `data/sirecords_overlap_analysis.md` for the full writeup and sources. Kept locally but excluded from git (`.gitignore`). |
44
+
45
+ ## If you're not sure whether your use is covered
46
+
47
+ None of the above is legal advice. If your use case isn't clearly
48
+ non-commercial research, check the original source's license directly (links
49
+ in `data/DATA_SOURCES.md`) or contact the original authors before relying on
50
+ this data.
@@ -0,0 +1,183 @@
1
+ Metadata-Version: 2.4
2
+ Name: sirna-data-grabber
3
+ Version: 0.1.0
4
+ Summary: Reusable loader for the siRNA knockdown-efficacy dataset (siRNAEfficacyDB + supplementary sources), plus the fetchers that built it.
5
+ Author-email: Brandon Walker <brandon.walker@ucsf.edu>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/BrandonWalk/sirna-data-grabber
8
+ Project-URL: Repository, https://github.com/BrandonWalk/sirna-data-grabber
9
+ Project-URL: Issues, https://github.com/BrandonWalk/sirna-data-grabber/issues
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
14
+ Requires-Python: >=3.10
15
+ Description-Content-Type: text/markdown
16
+ License-File: LICENSE
17
+ License-File: NOTICE.md
18
+ Requires-Dist: pandas>=2.0
19
+ Requires-Dist: requests>=2.32
20
+ Provides-Extra: test
21
+ Requires-Dist: pytest>=8.0; extra == "test"
22
+ Provides-Extra: lint
23
+ Requires-Dist: ruff>=0.6; extra == "lint"
24
+ Requires-Dist: mypy>=1.10; extra == "lint"
25
+ Requires-Dist: pandas-stubs; extra == "lint"
26
+ Requires-Dist: types-requests; extra == "lint"
27
+ Dynamic: license-file
28
+
29
+ # sirna-data-grabber
30
+
31
+ A standalone siRNA knockdown-efficacy dataset: the raw data files, full
32
+ provenance/license documentation, and a small reusable Python package
33
+ (`sirna_data`) for loading it -- and, since `pip install sirna-data-grabber`
34
+ alone can't ship most of this non-commercial data, a bundled `sirna-data-fetch`
35
+ command that re-fetches it from its original sources. Any project that wants
36
+ this dataset can depend on this repo (or just the PyPI package) rather than
37
+ vendoring a copy of the data or the loading code.
38
+
39
+ ## License
40
+
41
+ **The code in this repo (`sirna_data`, `tests/`) is MIT licensed** — see
42
+ [`LICENSE`](LICENSE). Use it, modify it, ship it commercially, whatever you
43
+ want.
44
+
45
+ **The data in `data/raw/` is NOT covered by that license.** It's redistributed
46
+ under each original source's own terms, and most of those sources are
47
+ **non-commercial only** (CC BY-NC / CC BY-NC-ND). Loading the data with this
48
+ permissively-licensed code does not lift those restrictions — you still have
49
+ to comply with them separately. See [`NOTICE.md`](NOTICE.md) for the
50
+ per-source summary and [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for
51
+ full terms before using the data itself, especially commercially.
52
+
53
+ ## What's here
54
+
55
+ ```
56
+ LICENSE MIT license -- covers the code only, not data/raw/
57
+ NOTICE.md per-source data license summary (see License section above)
58
+ data/
59
+ raw/ fetched CSVs + FASTA transcripts (the actual dataset)
60
+ DATA_SOURCES.md full provenance + license terms for every source
61
+ DATA_SOURCE_LEDGER.md audit: what's trainable, what's not, and why
62
+ CMSIRNADB_FULL_RETRIEVAL.md detail on the CMsiRNAdb full-database retrieval
63
+ DEMETER2_README.txt upstream release notes for DepMap DEMETER2 (investigated, not included -- see FUNCTIONAL_GENOMICS_SCREENS.md)
64
+ FUNCTIONAL_GENOMICS_SCREENS.md notes on functional-genomics screen sources considered
65
+ POTENTIAL_DATA_SOURCES.md landscape of sources investigated
66
+ sirecords_overlap_analysis.md siRecords overlap/dedup analysis
67
+ data_source_ledger.csv machine-readable companion to DATA_SOURCE_LEDGER.md
68
+ *.png figures referenced by the docs above
69
+ src/sirna_data/
70
+ raw_loader.py load + merge every source into SiRNARecord rows
71
+ ncbi_fetch.py fetch a gene's RefSeq mRNA transcript by symbol
72
+ __init__.py public API
73
+ fetch/ sirna-data-fetch CLI + per-source fetchers (see Install below)
74
+ cli.py `sirna-data-fetch` entry point ([project.scripts])
75
+ sirna_efficacy.py siRNAEfficacyDB + NCBI -> sirna_efficacy.csv, mrna_transcripts.fasta
76
+ monopoli.py Monopoli et al. 2023 supplementary data -> monopoli_*
77
+ shabalina.py Shabalina et al. 2006 supplementary data -> shabalina_*
78
+ cmsirnadb.py CMsiRNAdb + NCBI -> cmsirnadb_full_raw.tsv, cmsirnadb*_transcripts.fasta
79
+ tests/
80
+ test_raw_loader.py unit tests for raw_loader.py (fixtures, no real data needed)
81
+ test_ncbi_fetch.py unit tests for ncbi_fetch.py (mocked HTTP calls)
82
+ conftest.py shared pytest fixtures
83
+ ```
84
+
85
+ Start with [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for what's in the
86
+ dataset and where it came from; [`data/DATA_SOURCE_LEDGER.md`](data/DATA_SOURCE_LEDGER.md)
87
+ for the bottom-line audit (6,577 trainable records across 87 genes, 4
88
+ sources — 16,178 records / 97 genes if the optional CMsiRNAdb full-database
89
+ retrieval is also included). Primary source is **siRNAEfficacyDB** (Zhang
90
+ et al. 2024, CC BY-NC); see the docs for the rest and their individual
91
+ license terms before reusing this data outside this project.
92
+
93
+ ## Install
94
+
95
+ ```
96
+ python3 -m venv .venv && source .venv/bin/activate
97
+ pip install -e .
98
+ ```
99
+
100
+ This installs the `sirna_data` package in editable mode, so it resolves
101
+ `data/raw/` relative to this checkout automatically. If you copy the `data/`
102
+ folder somewhere else, point at it explicitly instead:
103
+
104
+ ```
105
+ export SIRNA_DATA_DIR=/path/to/data/raw
106
+ ```
107
+
108
+ Re-fetching the raw data from scratch (not required if `data/raw/` already
109
+ has the files) uses the `sirna-data-fetch` command, installed automatically
110
+ with the package -- no extras needed:
111
+
112
+ ```
113
+ sirna-data-fetch --dest data/raw
114
+ ```
115
+
116
+ This also means a plain `pip install sirna-data-grabber` from PyPI (with no
117
+ git checkout at all) can reconstruct the full dataset itself:
118
+
119
+ ```
120
+ pip install sirna-data-grabber
121
+ sirna-data-fetch --dest ./my_data
122
+ export SIRNA_DATA_DIR=./my_data
123
+ ```
124
+
125
+ `sirna-data-fetch --only sirna_efficacy monopoli` fetches a subset instead of
126
+ all four sources; see `sirna-data-fetch --help`.
127
+
128
+ ## Usage
129
+
130
+ ```python
131
+ from sirna_data import load_records, fetch_mrna_by_gene
132
+
133
+ records = load_records() # list[SiRNARecord]
134
+ print(len(records), "records across", len({r.gene for r in records}), "genes")
135
+
136
+ r = records[0]
137
+ r.guide_seq # siRNA antisense strand
138
+ r.mrna_window # local mRNA context around the real target site
139
+ r.label # experimental %knockdown / %inhibition
140
+ r.source # provenance, e.g. "siRNAEfficacyDB"
141
+
142
+ # Look up any gene's RefSeq transcript live from NCBI:
143
+ transcript = fetch_mrna_by_gene("TP53")
144
+ transcript.accession, transcript.sequence
145
+ ```
146
+
147
+ `load_records()` takes `include_monopoli` / `include_shabalina` /
148
+ `include_cmsirnadb` / `include_cmsirnadb_full` flags to exclude any
149
+ supplementary source and use only the primary siRNAEfficacyDB set.
150
+
151
+ ## Using this from another project
152
+
153
+ Install as a sibling checkout in editable mode:
154
+
155
+ ```
156
+ pip install -e ../sirna-data-grabber
157
+ ```
158
+
159
+ That gives you `import sirna_data` with no other coupling — this repo only
160
+ depends on pandas and requests, and knows nothing about any particular
161
+ downstream model or feature-engineering pipeline.
162
+
163
+ ## Tests
164
+
165
+ ```
166
+ pip install -e ".[test]"
167
+ pytest
168
+ ```
169
+
170
+ Tests run entirely against small in-memory/tmp-dir fixtures (see
171
+ `tests/conftest.py`) and mocked HTTP calls, so they don't touch the real
172
+ dataset or the network.
173
+
174
+ ## Linting and type checking
175
+
176
+ ```
177
+ pip install -e ".[lint]"
178
+ ruff check .
179
+ mypy
180
+ ```
181
+
182
+ Both run in CI on every pull request (`.github/workflows/tests.yml`), alongside
183
+ the test matrix.
@@ -0,0 +1,155 @@
1
+ # sirna-data-grabber
2
+
3
+ A standalone siRNA knockdown-efficacy dataset: the raw data files, full
4
+ provenance/license documentation, and a small reusable Python package
5
+ (`sirna_data`) for loading it -- and, since `pip install sirna-data-grabber`
6
+ alone can't ship most of this non-commercial data, a bundled `sirna-data-fetch`
7
+ command that re-fetches it from its original sources. Any project that wants
8
+ this dataset can depend on this repo (or just the PyPI package) rather than
9
+ vendoring a copy of the data or the loading code.
10
+
11
+ ## License
12
+
13
+ **The code in this repo (`sirna_data`, `tests/`) is MIT licensed** — see
14
+ [`LICENSE`](LICENSE). Use it, modify it, ship it commercially, whatever you
15
+ want.
16
+
17
+ **The data in `data/raw/` is NOT covered by that license.** It's redistributed
18
+ under each original source's own terms, and most of those sources are
19
+ **non-commercial only** (CC BY-NC / CC BY-NC-ND). Loading the data with this
20
+ permissively-licensed code does not lift those restrictions — you still have
21
+ to comply with them separately. See [`NOTICE.md`](NOTICE.md) for the
22
+ per-source summary and [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for
23
+ full terms before using the data itself, especially commercially.
24
+
25
+ ## What's here
26
+
27
+ ```
28
+ LICENSE MIT license -- covers the code only, not data/raw/
29
+ NOTICE.md per-source data license summary (see License section above)
30
+ data/
31
+ raw/ fetched CSVs + FASTA transcripts (the actual dataset)
32
+ DATA_SOURCES.md full provenance + license terms for every source
33
+ DATA_SOURCE_LEDGER.md audit: what's trainable, what's not, and why
34
+ CMSIRNADB_FULL_RETRIEVAL.md detail on the CMsiRNAdb full-database retrieval
35
+ DEMETER2_README.txt upstream release notes for DepMap DEMETER2 (investigated, not included -- see FUNCTIONAL_GENOMICS_SCREENS.md)
36
+ FUNCTIONAL_GENOMICS_SCREENS.md notes on functional-genomics screen sources considered
37
+ POTENTIAL_DATA_SOURCES.md landscape of sources investigated
38
+ sirecords_overlap_analysis.md siRecords overlap/dedup analysis
39
+ data_source_ledger.csv machine-readable companion to DATA_SOURCE_LEDGER.md
40
+ *.png figures referenced by the docs above
41
+ src/sirna_data/
42
+ raw_loader.py load + merge every source into SiRNARecord rows
43
+ ncbi_fetch.py fetch a gene's RefSeq mRNA transcript by symbol
44
+ __init__.py public API
45
+ fetch/ sirna-data-fetch CLI + per-source fetchers (see Install below)
46
+ cli.py `sirna-data-fetch` entry point ([project.scripts])
47
+ sirna_efficacy.py siRNAEfficacyDB + NCBI -> sirna_efficacy.csv, mrna_transcripts.fasta
48
+ monopoli.py Monopoli et al. 2023 supplementary data -> monopoli_*
49
+ shabalina.py Shabalina et al. 2006 supplementary data -> shabalina_*
50
+ cmsirnadb.py CMsiRNAdb + NCBI -> cmsirnadb_full_raw.tsv, cmsirnadb*_transcripts.fasta
51
+ tests/
52
+ test_raw_loader.py unit tests for raw_loader.py (fixtures, no real data needed)
53
+ test_ncbi_fetch.py unit tests for ncbi_fetch.py (mocked HTTP calls)
54
+ conftest.py shared pytest fixtures
55
+ ```
56
+
57
+ Start with [`data/DATA_SOURCES.md`](data/DATA_SOURCES.md) for what's in the
58
+ dataset and where it came from; [`data/DATA_SOURCE_LEDGER.md`](data/DATA_SOURCE_LEDGER.md)
59
+ for the bottom-line audit (6,577 trainable records across 87 genes, 4
60
+ sources — 16,178 records / 97 genes if the optional CMsiRNAdb full-database
61
+ retrieval is also included). Primary source is **siRNAEfficacyDB** (Zhang
62
+ et al. 2024, CC BY-NC); see the docs for the rest and their individual
63
+ license terms before reusing this data outside this project.
64
+
65
+ ## Install
66
+
67
+ ```
68
+ python3 -m venv .venv && source .venv/bin/activate
69
+ pip install -e .
70
+ ```
71
+
72
+ This installs the `sirna_data` package in editable mode, so it resolves
73
+ `data/raw/` relative to this checkout automatically. If you copy the `data/`
74
+ folder somewhere else, point at it explicitly instead:
75
+
76
+ ```
77
+ export SIRNA_DATA_DIR=/path/to/data/raw
78
+ ```
79
+
80
+ Re-fetching the raw data from scratch (not required if `data/raw/` already
81
+ has the files) uses the `sirna-data-fetch` command, installed automatically
82
+ with the package -- no extras needed:
83
+
84
+ ```
85
+ sirna-data-fetch --dest data/raw
86
+ ```
87
+
88
+ This also means a plain `pip install sirna-data-grabber` from PyPI (with no
89
+ git checkout at all) can reconstruct the full dataset itself:
90
+
91
+ ```
92
+ pip install sirna-data-grabber
93
+ sirna-data-fetch --dest ./my_data
94
+ export SIRNA_DATA_DIR=./my_data
95
+ ```
96
+
97
+ `sirna-data-fetch --only sirna_efficacy monopoli` fetches a subset instead of
98
+ all four sources; see `sirna-data-fetch --help`.
99
+
100
+ ## Usage
101
+
102
+ ```python
103
+ from sirna_data import load_records, fetch_mrna_by_gene
104
+
105
+ records = load_records() # list[SiRNARecord]
106
+ print(len(records), "records across", len({r.gene for r in records}), "genes")
107
+
108
+ r = records[0]
109
+ r.guide_seq # siRNA antisense strand
110
+ r.mrna_window # local mRNA context around the real target site
111
+ r.label # experimental %knockdown / %inhibition
112
+ r.source # provenance, e.g. "siRNAEfficacyDB"
113
+
114
+ # Look up any gene's RefSeq transcript live from NCBI:
115
+ transcript = fetch_mrna_by_gene("TP53")
116
+ transcript.accession, transcript.sequence
117
+ ```
118
+
119
+ `load_records()` takes `include_monopoli` / `include_shabalina` /
120
+ `include_cmsirnadb` / `include_cmsirnadb_full` flags to exclude any
121
+ supplementary source and use only the primary siRNAEfficacyDB set.
122
+
123
+ ## Using this from another project
124
+
125
+ Install as a sibling checkout in editable mode:
126
+
127
+ ```
128
+ pip install -e ../sirna-data-grabber
129
+ ```
130
+
131
+ That gives you `import sirna_data` with no other coupling — this repo only
132
+ depends on pandas and requests, and knows nothing about any particular
133
+ downstream model or feature-engineering pipeline.
134
+
135
+ ## Tests
136
+
137
+ ```
138
+ pip install -e ".[test]"
139
+ pytest
140
+ ```
141
+
142
+ Tests run entirely against small in-memory/tmp-dir fixtures (see
143
+ `tests/conftest.py`) and mocked HTTP calls, so they don't touch the real
144
+ dataset or the network.
145
+
146
+ ## Linting and type checking
147
+
148
+ ```
149
+ pip install -e ".[lint]"
150
+ ruff check .
151
+ mypy
152
+ ```
153
+
154
+ Both run in CI on every pull request (`.github/workflows/tests.yml`), alongside
155
+ the test matrix.
@@ -0,0 +1,68 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "sirna-data-grabber"
7
+ version = "0.1.0"
8
+ description = "Reusable loader for the siRNA knockdown-efficacy dataset (siRNAEfficacyDB + supplementary sources), plus the fetchers that built it."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = { text = "MIT" }
12
+ authors = [
13
+ { name = "Brandon Walker", email = "brandon.walker@ucsf.edu" },
14
+ ]
15
+ classifiers = [
16
+ "License :: OSI Approved :: MIT License",
17
+ "Programming Language :: Python :: 3",
18
+ "Intended Audience :: Science/Research",
19
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
20
+ ]
21
+ dependencies = [
22
+ "pandas>=2.0",
23
+ "requests>=2.32",
24
+ ]
25
+
26
+ [project.urls]
27
+ Homepage = "https://github.com/BrandonWalk/sirna-data-grabber"
28
+ Repository = "https://github.com/BrandonWalk/sirna-data-grabber"
29
+ Issues = "https://github.com/BrandonWalk/sirna-data-grabber/issues"
30
+
31
+ [project.optional-dependencies]
32
+ test = ["pytest>=8.0"]
33
+ lint = ["ruff>=0.6", "mypy>=1.10", "pandas-stubs", "types-requests"]
34
+
35
+ # Installed automatically -- `sirna-data-fetch` fetches every source
36
+ # load_records() reads into a local directory (data/DATA_SOURCES.md), so a
37
+ # plain `pip install sirna-data-grabber` can reconstruct the dataset without
38
+ # cloning the git repo. No extra dependencies: everything under
39
+ # sirna_data/fetch/ only needs pandas (already required above) plus the
40
+ # standard library.
41
+ [project.scripts]
42
+ sirna-data-fetch = "sirna_data.fetch.cli:main"
43
+
44
+ [tool.setuptools.packages.find]
45
+ where = ["src"]
46
+
47
+ [tool.pytest.ini_options]
48
+ testpaths = ["tests"]
49
+ addopts = "-ra"
50
+
51
+ [tool.ruff]
52
+ line-length = 100
53
+ target-version = "py310"
54
+
55
+ [tool.ruff.lint]
56
+ select = ["E", "F", "I", "UP", "B"]
57
+
58
+ [tool.ruff.lint.per-file-ignores]
59
+ # long single-line TSV fixture rows are more readable unwrapped
60
+ "tests/conftest.py" = ["E501"]
61
+
62
+ [tool.mypy]
63
+ python_version = "3.10"
64
+ files = ["src"]
65
+ ignore_missing_imports = true
66
+ warn_unused_ignores = true
67
+ warn_redundant_casts = true
68
+ check_untyped_defs = true
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,29 @@
1
+ """sirna_data: reusable loader for the siRNA knockdown-efficacy dataset.
2
+
3
+ from sirna_data import load_records, SiRNARecord, fetch_mrna_by_gene
4
+
5
+ `load_records()` returns the full merged, provenance-documented dataset
6
+ (siRNAEfficacyDB + supplementary sources) as a list of `SiRNARecord`, each
7
+ pairing an siRNA guide with its real local mRNA target context and
8
+ experimentally measured knockdown label. See ../../data/DATA_SOURCES.md for
9
+ where every row comes from.
10
+
11
+ `fetch_mrna_by_gene()` looks up a gene's RefSeq mRNA transcript live from
12
+ NCBI, for callers that only have a gene symbol.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ from .ncbi_fetch import FetchedTranscript, GeneNotFoundError, fetch_mrna_by_gene
17
+ from .raw_loader import DATA_DIR, SiRNARecord, load_records, read_fasta
18
+
19
+ __all__ = [
20
+ "load_records",
21
+ "SiRNARecord",
22
+ "read_fasta",
23
+ "DATA_DIR",
24
+ "fetch_mrna_by_gene",
25
+ "FetchedTranscript",
26
+ "GeneNotFoundError",
27
+ ]
28
+
29
+ __version__ = "0.1.0"
@@ -0,0 +1,22 @@
1
+ """Fetch the raw data `sirna_data.load_records()` reads, from each source's
2
+ original location, into a local directory -- so a plain `pip install
3
+ sirna-data-grabber` install can reconstruct the dataset without cloning the
4
+ git repo.
5
+
6
+ from sirna_data.fetch import cmsirnadb, monopoli, shabalina, sirna_efficacy
7
+ sirna_efficacy.fetch(Path("./my_data"))
8
+
9
+ Or from the command line (installed automatically, no extras required --
10
+ every fetcher here only needs pandas, already a core dependency, plus the
11
+ standard library):
12
+
13
+ sirna-data-fetch --dest ./my_data
14
+ export SIRNA_DATA_DIR=./my_data
15
+
16
+ This only covers the four sources `load_records()` actually reads
17
+ (siRNAEfficacyDB, Monopoli 2023, Shabalina 2006, CMsiRNAdb). See
18
+ ../../../data/DATA_SOURCES.md for full provenance/license notes per source,
19
+ and NOTICE.md before using the fetched data commercially -- most of it is
20
+ non-commercial only.
21
+ """
22
+ from __future__ import annotations
@@ -0,0 +1,70 @@
1
+ """Command-line entry point: `sirna-data-fetch` (registered via
2
+ [project.scripts] in pyproject.toml). Fetches every source
3
+ `sirna_data.load_records()` reads, from its original location, into a local
4
+ directory.
5
+
6
+ sirna-data-fetch --dest ./my_data
7
+ export SIRNA_DATA_DIR=./my_data
8
+
9
+ This is installed automatically with `pip install sirna-data-grabber` --
10
+ no extras needed, since every fetcher here only needs pandas (already a
11
+ core dependency) plus the standard library.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import argparse
16
+ import os
17
+ from pathlib import Path
18
+
19
+ from . import cmsirnadb, monopoli, shabalina, sirna_efficacy
20
+
21
+ # Order matters only for display; each fetcher is independent.
22
+ SOURCES = {
23
+ "sirna_efficacy": sirna_efficacy.fetch,
24
+ "monopoli": monopoli.fetch,
25
+ "shabalina": shabalina.fetch,
26
+ "cmsirnadb": cmsirnadb.fetch,
27
+ }
28
+
29
+
30
+ def main() -> None:
31
+ parser = argparse.ArgumentParser(
32
+ prog="sirna-data-fetch",
33
+ description=(
34
+ "Fetch the siRNA knockdown-efficacy dataset (siRNAEfficacyDB + "
35
+ "supplementary sources) from its original sources into a local "
36
+ "directory, for use with sirna_data.load_records()."
37
+ ),
38
+ )
39
+ parser.add_argument(
40
+ "--dest",
41
+ type=Path,
42
+ default=Path(os.environ.get("SIRNA_DATA_DIR", "data/raw")),
43
+ help=(
44
+ "Directory to write the fetched files into "
45
+ "(default: $SIRNA_DATA_DIR if set, else ./data/raw)."
46
+ ),
47
+ )
48
+ parser.add_argument(
49
+ "--only",
50
+ nargs="+",
51
+ choices=sorted(SOURCES),
52
+ metavar="SOURCE",
53
+ help=f"Fetch only these sources instead of all four ({', '.join(sorted(SOURCES))}).",
54
+ )
55
+ args = parser.parse_args()
56
+
57
+ dest: Path = args.dest.resolve()
58
+ dest.mkdir(parents=True, exist_ok=True)
59
+ names = args.only or sorted(SOURCES)
60
+
61
+ for name in names:
62
+ print(f"=== {name} ===")
63
+ SOURCES[name](dest)
64
+ print()
65
+
66
+ print(f"Done. To use this data:\n export SIRNA_DATA_DIR={dest}")
67
+
68
+
69
+ if __name__ == "__main__":
70
+ main()