gnomad-api-cache 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gnomad_api_cache/__init__.py +35 -0
- gnomad_api_cache/__main__.py +8 -0
- gnomad_api_cache/_utils.py +14 -0
- gnomad_api_cache/adapters/__init__.py +0 -0
- gnomad_api_cache/adapters/vcf_adapter.py +147 -0
- gnomad_api_cache/cache.py +328 -0
- gnomad_api_cache/cli.py +284 -0
- gnomad_api_cache/client.py +20 -0
- gnomad_api_cache/export.py +319 -0
- gnomad_api_cache/fetch.py +149 -0
- gnomad_api_cache/keys.py +78 -0
- gnomad_api_cache/query.py +305 -0
- gnomad_api_cache-0.1.0.dist-info/METADATA +115 -0
- gnomad_api_cache-0.1.0.dist-info/RECORD +17 -0
- gnomad_api_cache-0.1.0.dist-info/WHEEL +4 -0
- gnomad_api_cache-0.1.0.dist-info/entry_points.txt +2 -0
- gnomad_api_cache-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
from collections.abc import Callable, Sequence
|
|
5
|
+
from dataclasses import dataclass, replace
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import requests
|
|
10
|
+
from tqdm import tqdm
|
|
11
|
+
|
|
12
|
+
from gnomad_api_cache._utils import _chunked
|
|
13
|
+
from gnomad_api_cache.cache import VariantCache
|
|
14
|
+
from gnomad_api_cache.client import post_gnomad
|
|
15
|
+
from gnomad_api_cache.keys import VariantKey
|
|
16
|
+
from gnomad_api_cache.query import (
|
|
17
|
+
MAX_BATCH_SIZE,
|
|
18
|
+
build_mitochondrial_query,
|
|
19
|
+
build_variant_query,
|
|
20
|
+
parse_batch_response,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
# Documented limit is 10 requests per IP per 60s.
|
|
24
|
+
REQUEST_DELAY_SECONDS = 8
|
|
25
|
+
|
|
26
|
+
QueryBuilder = Callable[[list[str]], tuple[str, dict[str, str]]]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class FetchSummary:
|
|
31
|
+
"""Fetch run summary. Returned by fetch_into and VariantCache.fetch."""
|
|
32
|
+
|
|
33
|
+
requested: int = 0
|
|
34
|
+
already_cached: int = 0
|
|
35
|
+
fetched: int = 0
|
|
36
|
+
not_found: int = 0
|
|
37
|
+
errors: int = 0
|
|
38
|
+
failed_batches: int = 0
|
|
39
|
+
|
|
40
|
+
def __str__(self) -> str:
|
|
41
|
+
return (
|
|
42
|
+
f"{self.requested} requested, {self.already_cached} already cached, "
|
|
43
|
+
f"{self.fetched} fetched, {self.not_found} not in gnomAD, "
|
|
44
|
+
f"{self.errors} errored"
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _fetch_batches(
|
|
49
|
+
cache: VariantCache,
|
|
50
|
+
keys: Sequence[VariantKey],
|
|
51
|
+
build_query: QueryBuilder,
|
|
52
|
+
progress: tqdm,
|
|
53
|
+
delay: float,
|
|
54
|
+
) -> FetchSummary:
|
|
55
|
+
"""Query `keys` in batches, depositing each response as it arrives.
|
|
56
|
+
|
|
57
|
+
Returns the tally for these batches only; the caller merges it."""
|
|
58
|
+
fetched = not_found = errors = failed_batches = 0
|
|
59
|
+
for batch in _chunked(keys, MAX_BATCH_SIZE):
|
|
60
|
+
ids = [v.id for v in batch]
|
|
61
|
+
try:
|
|
62
|
+
query, variables = build_query(ids)
|
|
63
|
+
payload = post_gnomad(query, variables)
|
|
64
|
+
records = parse_batch_response(payload, ids)
|
|
65
|
+
cache.put_many(records)
|
|
66
|
+
except (requests.RequestException, KeyError) as e:
|
|
67
|
+
cache.mark_errors(batch)
|
|
68
|
+
errors += len(batch)
|
|
69
|
+
failed_batches += 1
|
|
70
|
+
tqdm.write(f"batch of {len(batch)} failed, recorded for retry: {e}")
|
|
71
|
+
else:
|
|
72
|
+
# A null record means gnomAD has no such variant, not a failure.
|
|
73
|
+
found = sum(1 for record in records.values() if record is not None)
|
|
74
|
+
fetched += found
|
|
75
|
+
not_found += len(records) - found
|
|
76
|
+
progress.update(len(batch))
|
|
77
|
+
time.sleep(delay)
|
|
78
|
+
return FetchSummary(
|
|
79
|
+
fetched=fetched,
|
|
80
|
+
not_found=not_found,
|
|
81
|
+
errors=errors,
|
|
82
|
+
failed_batches=failed_batches,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def fetch_into(
|
|
87
|
+
cache: VariantCache,
|
|
88
|
+
variants: list[VariantKey],
|
|
89
|
+
retry_errors: bool = True,
|
|
90
|
+
retry_not_found: bool = False,
|
|
91
|
+
delay: float = REQUEST_DELAY_SECONDS,
|
|
92
|
+
) -> FetchSummary:
|
|
93
|
+
"""Populate an open cache with gnomAD records for `variants`.
|
|
94
|
+
|
|
95
|
+
The cache is left open: it belongs to the caller, who may well want to
|
|
96
|
+
export from it next.
|
|
97
|
+
"""
|
|
98
|
+
variants_to_fetch = cache.needs_query(
|
|
99
|
+
variants,
|
|
100
|
+
retry_errors=retry_errors,
|
|
101
|
+
retry_not_found=retry_not_found,
|
|
102
|
+
)
|
|
103
|
+
summary = FetchSummary(
|
|
104
|
+
requested=len(variants),
|
|
105
|
+
already_cached=len(variants) - len(variants_to_fetch),
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
if not variants_to_fetch:
|
|
109
|
+
tqdm.write("All variants already cached.")
|
|
110
|
+
return summary
|
|
111
|
+
|
|
112
|
+
nuclear = [v for v in variants_to_fetch if not v.is_mitochondrial]
|
|
113
|
+
mitochondrial = [v for v in variants_to_fetch if v.is_mitochondrial]
|
|
114
|
+
|
|
115
|
+
with tqdm(
|
|
116
|
+
total=len(variants_to_fetch), unit="variant", desc="Querying gnomAD"
|
|
117
|
+
) as progress:
|
|
118
|
+
nuclear_tally = _fetch_batches(
|
|
119
|
+
cache, nuclear, build_variant_query, progress, delay
|
|
120
|
+
)
|
|
121
|
+
mito_tally = _fetch_batches(
|
|
122
|
+
cache, mitochondrial, build_mitochondrial_query, progress, delay
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
summary = replace(
|
|
126
|
+
summary,
|
|
127
|
+
fetched=nuclear_tally.fetched + mito_tally.fetched,
|
|
128
|
+
not_found=nuclear_tally.not_found + mito_tally.not_found,
|
|
129
|
+
errors=nuclear_tally.errors + mito_tally.errors,
|
|
130
|
+
failed_batches=nuclear_tally.failed_batches + mito_tally.failed_batches,
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
if summary.failed_batches:
|
|
134
|
+
tqdm.write(
|
|
135
|
+
f"{summary.failed_batches} batch(es) failed and were recorded as "
|
|
136
|
+
f"errors; re-run to retry them."
|
|
137
|
+
)
|
|
138
|
+
return summary
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def fetch_gnomad(
|
|
142
|
+
variants: list[VariantKey],
|
|
143
|
+
cache_file: str | Path,
|
|
144
|
+
**kwargs: Any,
|
|
145
|
+
) -> FetchSummary:
|
|
146
|
+
"""Populate a cache file with gnomAD records for `variants`,
|
|
147
|
+
creating the cache if it does not exist."""
|
|
148
|
+
with VariantCache(cache_file) as cache:
|
|
149
|
+
return fetch_into(cache, variants, **kwargs)
|
gnomad_api_cache/keys.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
CANONICAL_CHROMS = frozenset([str(i) for i in range(1, 23)] + ["X", "Y", "M"])
|
|
7
|
+
|
|
8
|
+
_ALLELE_RE = re.compile(r"^[ACGTN]+$")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class InvalidVariantError(ValueError):
|
|
12
|
+
"""Raised when a record cannot be expressed as a gnomAD variant ID."""
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def normalize_chrom(chrom: str) -> str:
|
|
16
|
+
"""Canonicalize a contig name to the form gnomAD expects."""
|
|
17
|
+
c = str(chrom).strip()
|
|
18
|
+
|
|
19
|
+
# strip "chr" prefix
|
|
20
|
+
if c.lower().startswith("chr"):
|
|
21
|
+
c = c[3:]
|
|
22
|
+
|
|
23
|
+
# convert "MT" to "M"
|
|
24
|
+
c = c.upper()
|
|
25
|
+
if c == "MT":
|
|
26
|
+
c = "M"
|
|
27
|
+
|
|
28
|
+
# remove leading zeros from numeric chromosomes
|
|
29
|
+
if c.isdigit():
|
|
30
|
+
c = str(int(c)) # "01" -> "1"
|
|
31
|
+
|
|
32
|
+
return c
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class VariantKey:
|
|
37
|
+
chrom: str
|
|
38
|
+
pos: int
|
|
39
|
+
ref: str
|
|
40
|
+
alt: str
|
|
41
|
+
|
|
42
|
+
@property
|
|
43
|
+
def id(self) -> str:
|
|
44
|
+
return f"{self.chrom}-{self.pos}-{self.ref}-{self.alt}"
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def is_mitochondrial(self) -> bool:
|
|
48
|
+
return self.chrom == "M"
|
|
49
|
+
|
|
50
|
+
@classmethod
|
|
51
|
+
def from_parts(cls, chrom: str, pos: int | str, ref: str, alt: str) -> VariantKey:
|
|
52
|
+
"""Build a key from raw record fields, canonicalizing and validating."""
|
|
53
|
+
c = normalize_chrom(chrom)
|
|
54
|
+
r = str(ref).strip().upper()
|
|
55
|
+
a = str(alt).strip().upper()
|
|
56
|
+
|
|
57
|
+
if c not in CANONICAL_CHROMS:
|
|
58
|
+
raise InvalidVariantError(f"non-canonical contig: {chrom!r}")
|
|
59
|
+
try:
|
|
60
|
+
p = int(pos)
|
|
61
|
+
except (TypeError, ValueError):
|
|
62
|
+
raise InvalidVariantError(f"non-integer position: {pos!r}") from None
|
|
63
|
+
if p < 1:
|
|
64
|
+
raise InvalidVariantError(f"position must be 1-based: {pos!r}")
|
|
65
|
+
if not _ALLELE_RE.match(r):
|
|
66
|
+
raise InvalidVariantError(f"non-ACGTN ref allele: {ref!r}")
|
|
67
|
+
if not _ALLELE_RE.match(a):
|
|
68
|
+
raise InvalidVariantError(f"non-ACGTN alt allele: {alt!r}")
|
|
69
|
+
|
|
70
|
+
return cls(chrom=c, pos=p, ref=r, alt=a)
|
|
71
|
+
|
|
72
|
+
@classmethod
|
|
73
|
+
def from_id(cls, variant_id: str) -> VariantKey:
|
|
74
|
+
"""Parse a "chrom-pos-ref-alt" string, e.g. from a text file or the cache."""
|
|
75
|
+
parts = str(variant_id).strip().split("-")
|
|
76
|
+
if len(parts) != 4:
|
|
77
|
+
raise InvalidVariantError(f"expected chrom-pos-ref-alt, got {variant_id!r}")
|
|
78
|
+
return cls.from_parts(*parts)
|
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
"""GraphQL query construction for the gnomAD API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
DEFAULT_DATASET = "gnomad_r4"
|
|
6
|
+
|
|
7
|
+
# Server-enforced: MAX_QUERY_COST is 25 and a root `variant` field costs 1.
|
|
8
|
+
MAX_BATCH_SIZE = 25
|
|
9
|
+
|
|
10
|
+
# Bump whenever the fragments below change, so the cache can tell which stored
|
|
11
|
+
# records predate the new fields and re-fetch only those.
|
|
12
|
+
QUERY_VERSION = 1
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# --- Fragments ------------------------------------------------------------
|
|
16
|
+
#
|
|
17
|
+
# A fragment is a named, reusable selection set. `...VariantDetail` splices it
|
|
18
|
+
# in. Declaring it once is what keeps the request small no matter how many
|
|
19
|
+
# variants the batch carries.
|
|
20
|
+
|
|
21
|
+
# Frequency block shared by the `exome` and `genome` fields, which have the
|
|
22
|
+
# same type. `joint` is a *different* type with fewer fields, so it is spelled
|
|
23
|
+
# out separately in VariantDetail below.
|
|
24
|
+
SEQUENCING_DATA_FRAGMENT = """
|
|
25
|
+
fragment SeqData on VariantDetailsSequencingTypeData {
|
|
26
|
+
ac
|
|
27
|
+
an
|
|
28
|
+
homozygote_count
|
|
29
|
+
hemizygote_count
|
|
30
|
+
filters
|
|
31
|
+
flags
|
|
32
|
+
faf95 { popmax popmax_population }
|
|
33
|
+
faf99 { popmax popmax_population }
|
|
34
|
+
populations { id ac an homozygote_count hemizygote_count }
|
|
35
|
+
local_ancestry_populations { id ac an }
|
|
36
|
+
age_distribution {
|
|
37
|
+
het { bin_edges bin_freq n_smaller n_larger }
|
|
38
|
+
hom { bin_edges bin_freq n_smaller n_larger }
|
|
39
|
+
}
|
|
40
|
+
quality_metrics {
|
|
41
|
+
allele_balance { alt { bin_edges bin_freq n_smaller n_larger } }
|
|
42
|
+
genotype_depth {
|
|
43
|
+
all { bin_edges bin_freq n_smaller n_larger }
|
|
44
|
+
alt { bin_edges bin_freq n_smaller n_larger }
|
|
45
|
+
}
|
|
46
|
+
genotype_quality {
|
|
47
|
+
all { bin_edges bin_freq n_smaller n_larger }
|
|
48
|
+
alt { bin_edges bin_freq n_smaller n_larger }
|
|
49
|
+
}
|
|
50
|
+
site_quality_metrics { metric value }
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
# Everything VariantDetails exposes that is worth caching. Deliberately omitted:
|
|
56
|
+
# af, ac_hom, ac_hemi -- deprecated; derive from ac/an and *_count
|
|
57
|
+
# va -- GA4GH restatement of the frequency data already here;
|
|
58
|
+
# roughly doubles the payload for no new information
|
|
59
|
+
VARIANT_FRAGMENT = """
|
|
60
|
+
fragment VariantDetail on VariantDetails {
|
|
61
|
+
variant_id
|
|
62
|
+
reference_genome
|
|
63
|
+
chrom
|
|
64
|
+
pos
|
|
65
|
+
ref
|
|
66
|
+
alt
|
|
67
|
+
caid
|
|
68
|
+
rsids
|
|
69
|
+
colocated_variants
|
|
70
|
+
flags
|
|
71
|
+
|
|
72
|
+
coverage {
|
|
73
|
+
exome { mean median over_1 over_5 over_10 over_15 over_20 over_25 over_30 over_50 over_100 }
|
|
74
|
+
genome { mean median over_1 over_5 over_10 over_15 over_20 over_25 over_30 over_50 over_100 }
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
exome { ...SeqData }
|
|
78
|
+
genome { ...SeqData }
|
|
79
|
+
|
|
80
|
+
joint {
|
|
81
|
+
ac
|
|
82
|
+
an
|
|
83
|
+
homozygote_count
|
|
84
|
+
hemizygote_count
|
|
85
|
+
filters
|
|
86
|
+
faf95 { popmax popmax_population }
|
|
87
|
+
faf99 { popmax popmax_population }
|
|
88
|
+
populations { id ac an homozygote_count hemizygote_count }
|
|
89
|
+
age_distribution {
|
|
90
|
+
het { bin_edges bin_freq n_smaller n_larger }
|
|
91
|
+
hom { bin_edges bin_freq n_smaller n_larger }
|
|
92
|
+
}
|
|
93
|
+
quality_metrics { site_quality_metrics { metric value } }
|
|
94
|
+
freq_comparison_stats {
|
|
95
|
+
contingency_table_test { p_value odds_ratio }
|
|
96
|
+
cochran_mantel_haenszel_test { chisq p_value }
|
|
97
|
+
stat_union { p_value stat_test_name gen_ancs }
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
transcript_consequences {
|
|
102
|
+
consequence_terms
|
|
103
|
+
domains
|
|
104
|
+
gene_id
|
|
105
|
+
gene_version
|
|
106
|
+
gene_symbol
|
|
107
|
+
hgvs
|
|
108
|
+
hgvsc
|
|
109
|
+
hgvsp
|
|
110
|
+
is_canonical
|
|
111
|
+
is_mane_select
|
|
112
|
+
is_mane_select_version
|
|
113
|
+
lof
|
|
114
|
+
lof_flags
|
|
115
|
+
lof_filter
|
|
116
|
+
major_consequence
|
|
117
|
+
polyphen_prediction
|
|
118
|
+
sift_prediction
|
|
119
|
+
refseq_id
|
|
120
|
+
refseq_version
|
|
121
|
+
transcript_id
|
|
122
|
+
transcript_version
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
in_silico_predictors { id value flags }
|
|
126
|
+
lof_curations { gene_id gene_version gene_symbol verdict flags project }
|
|
127
|
+
multi_nucleotide_variants {
|
|
128
|
+
combined_variant_id
|
|
129
|
+
changes_amino_acids
|
|
130
|
+
n_individuals
|
|
131
|
+
other_constituent_snvs
|
|
132
|
+
}
|
|
133
|
+
non_coding_constraint { chrom start stop element_id possible observed expected oe z }
|
|
134
|
+
vrs {
|
|
135
|
+
_id
|
|
136
|
+
type
|
|
137
|
+
location {
|
|
138
|
+
_id
|
|
139
|
+
type
|
|
140
|
+
sequence_id
|
|
141
|
+
interval { type start { type value } end { type value } }
|
|
142
|
+
}
|
|
143
|
+
state { type sequence }
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
# Mitochondrial variants live behind a different root field with a different
|
|
149
|
+
# type. `variant(...)` returns null for them, which would otherwise be cached
|
|
150
|
+
# as "not in gnomAD". Note the heteroplasmy-aware counts: there is no single
|
|
151
|
+
# `ac`, because a mitochondrial call is homoplasmic or heteroplasmic.
|
|
152
|
+
MITOCHONDRIAL_FRAGMENT = """
|
|
153
|
+
fragment MitoDetail on MitochondrialVariantDetails {
|
|
154
|
+
variant_id
|
|
155
|
+
reference_genome
|
|
156
|
+
pos
|
|
157
|
+
ref
|
|
158
|
+
alt
|
|
159
|
+
rsids
|
|
160
|
+
flags
|
|
161
|
+
filters
|
|
162
|
+
an
|
|
163
|
+
ac_hom
|
|
164
|
+
ac_het
|
|
165
|
+
ac_hom_mnv
|
|
166
|
+
excluded_ac
|
|
167
|
+
max_heteroplasmy
|
|
168
|
+
heteroplasmy_distribution { bin_edges bin_freq n_smaller n_larger }
|
|
169
|
+
haplogroup_defining
|
|
170
|
+
haplogroups { id an ac_het ac_hom faf faf_hom }
|
|
171
|
+
populations { id an ac_het ac_hom heteroplasmy_distribution { bin_edges bin_freq } }
|
|
172
|
+
mitotip_score
|
|
173
|
+
mitotip_trna_prediction
|
|
174
|
+
pon_ml_probability_of_pathogenicity
|
|
175
|
+
pon_mt_trna_prediction
|
|
176
|
+
age_distribution {
|
|
177
|
+
het { bin_edges bin_freq n_smaller n_larger }
|
|
178
|
+
hom { bin_edges bin_freq n_smaller n_larger }
|
|
179
|
+
}
|
|
180
|
+
site_quality_metrics { name value }
|
|
181
|
+
genotype_quality_metrics { name all { bin_edges bin_freq } alt { bin_edges bin_freq } }
|
|
182
|
+
genotype_quality_filters { name filtered { bin_edges bin_freq } }
|
|
183
|
+
transcript_consequences {
|
|
184
|
+
consequence_terms
|
|
185
|
+
gene_id
|
|
186
|
+
gene_symbol
|
|
187
|
+
hgvsc
|
|
188
|
+
hgvsp
|
|
189
|
+
is_canonical
|
|
190
|
+
is_mane_select
|
|
191
|
+
lof
|
|
192
|
+
lof_flags
|
|
193
|
+
lof_filter
|
|
194
|
+
major_consequence
|
|
195
|
+
transcript_id
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
"""
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def alias_for(index: int) -> str:
|
|
202
|
+
"""Alias used for the Nth variant in a batch.
|
|
203
|
+
|
|
204
|
+
Every field in a GraphQL response is keyed by its name, so 25 `variant`
|
|
205
|
+
fields would collide. Aliasing renames each one in the response.
|
|
206
|
+
"""
|
|
207
|
+
return f"v{index}"
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def build_variant_query(
|
|
211
|
+
variant_ids: list[str],
|
|
212
|
+
dataset: str = DEFAULT_DATASET,
|
|
213
|
+
) -> tuple[str, dict[str, str]]:
|
|
214
|
+
"""Build one batched query for up to MAX_BATCH_SIZE nuclear variants.
|
|
215
|
+
|
|
216
|
+
Returns (query_text, variables), to be POSTed as
|
|
217
|
+
{"query": query_text, "variables": variables}.
|
|
218
|
+
"""
|
|
219
|
+
if not variant_ids:
|
|
220
|
+
raise ValueError("no variant ids given")
|
|
221
|
+
if len(variant_ids) > MAX_BATCH_SIZE:
|
|
222
|
+
raise ValueError(
|
|
223
|
+
f"{len(variant_ids)} variants exceeds the server's per-request " +
|
|
224
|
+
f"cost ceiling of {MAX_BATCH_SIZE}"
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
# One "$vN: String!" per variant, plus the dataset. Declaring the dataset as
|
|
228
|
+
# a variable keeps the query text identical for every batch of a given size,
|
|
229
|
+
# which is friendlier to the server's response cache.
|
|
230
|
+
declarations = ", ".join(
|
|
231
|
+
["$dataset: DatasetId!"]
|
|
232
|
+
+ [f"${alias_for(i)}: String!" for i in range(len(variant_ids))]
|
|
233
|
+
)
|
|
234
|
+
selections = "\n".join(
|
|
235
|
+
f" {alias_for(i)}: variant(variantId: ${alias_for(i)}, dataset: $dataset)" +
|
|
236
|
+
f" {{ ...VariantDetail }}" # noqa: F541
|
|
237
|
+
for i in range(len(variant_ids))
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
query = (
|
|
241
|
+
f"query VariantBatch({declarations}) {{\n{selections}\n}}\n"
|
|
242
|
+
f"{VARIANT_FRAGMENT}{SEQUENCING_DATA_FRAGMENT}"
|
|
243
|
+
)
|
|
244
|
+
variables: dict[str, str] = {"dataset": dataset}
|
|
245
|
+
for i, vid in enumerate(variant_ids):
|
|
246
|
+
variables[alias_for(i)] = vid
|
|
247
|
+
|
|
248
|
+
return query, variables
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def build_mitochondrial_query(
|
|
252
|
+
variant_ids: list[str],
|
|
253
|
+
dataset: str = DEFAULT_DATASET,
|
|
254
|
+
) -> tuple[str, dict[str, str]]:
|
|
255
|
+
"""Same as build_variant_query, for chrM variants."""
|
|
256
|
+
if not variant_ids:
|
|
257
|
+
raise ValueError("no variant ids given")
|
|
258
|
+
if len(variant_ids) > MAX_BATCH_SIZE:
|
|
259
|
+
raise ValueError(
|
|
260
|
+
f"{len(variant_ids)} variants exceeds the server's per-request " +
|
|
261
|
+
f"cost ceiling of {MAX_BATCH_SIZE}"
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
declarations = ", ".join(
|
|
265
|
+
["$dataset: DatasetId!"]
|
|
266
|
+
+ [f"${alias_for(i)}: String!" for i in range(len(variant_ids))]
|
|
267
|
+
)
|
|
268
|
+
selections = "\n".join(
|
|
269
|
+
f" {alias_for(i)}: mitochondrial_variant" +
|
|
270
|
+
f"(variant_id: ${alias_for(i)}, dataset: $dataset) {{ ...MitoDetail }}"
|
|
271
|
+
for i in range(len(variant_ids))
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
query = (
|
|
275
|
+
f"query MitoBatch({declarations}) {{\n{selections}\n}}\n"
|
|
276
|
+
f"{MITOCHONDRIAL_FRAGMENT}"
|
|
277
|
+
)
|
|
278
|
+
variables: dict[str, str] = {"dataset": dataset}
|
|
279
|
+
for i, vid in enumerate(variant_ids):
|
|
280
|
+
variables[alias_for(i)] = vid
|
|
281
|
+
|
|
282
|
+
return query, variables
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def parse_batch_response(
|
|
286
|
+
payload: dict,
|
|
287
|
+
variant_ids: list[str],
|
|
288
|
+
) -> dict[str, dict | None]:
|
|
289
|
+
"""Map a response body back onto the variant ids that produced it.
|
|
290
|
+
|
|
291
|
+
Returns {variant_id: record_or_None}, where None means the server had no
|
|
292
|
+
such variant.
|
|
293
|
+
|
|
294
|
+
Raises KeyError if the body has no `data` at all, which indicates a
|
|
295
|
+
transport- or query-level failure the caller should retry rather than
|
|
296
|
+
interpret as 25 absent variants.
|
|
297
|
+
"""
|
|
298
|
+
data = payload.get("data")
|
|
299
|
+
if data is None:
|
|
300
|
+
raise KeyError("response contained no 'data'")
|
|
301
|
+
|
|
302
|
+
return {
|
|
303
|
+
vid: data.get(alias_for(i))
|
|
304
|
+
for i, vid in enumerate(variant_ids)
|
|
305
|
+
}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: gnomad-api-cache
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Fetch and cache gnomAD annotations for VCF variants.
|
|
5
|
+
License-File: LICENSE
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Requires-Dist: cyvcf2<0.35,>=0.34.0
|
|
8
|
+
Requires-Dist: pyarrow<26,>=25.0.0
|
|
9
|
+
Requires-Dist: requests<3,>=2.32
|
|
10
|
+
Requires-Dist: tqdm<5,>=4.66
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# gnomad-api-cache
|
|
14
|
+
|
|
15
|
+
Fetch gnomAD annotations for the variants in a VCF, cache them in SQLite, and
|
|
16
|
+
export to various output formats (parquet, csv, tsv, json).
|
|
17
|
+
|
|
18
|
+
The public gnomAD API allows roughly 10 requests (of 25 variants each) per minute.
|
|
19
|
+
Annotating a cohort twice — or exporting a second format — fetches from the locally
|
|
20
|
+
built cache rather than requerying.
|
|
21
|
+
|
|
22
|
+
Re-running into an existing cache fetches only what is missing.
|
|
23
|
+
|
|
24
|
+
## Install
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install gnomad-api-cache
|
|
28
|
+
|
|
29
|
+
# or with uv
|
|
30
|
+
uv pip install gnomad-api-cache # into the active environment
|
|
31
|
+
uv add gnomad-api-cache # into a project
|
|
32
|
+
uv tool install gnomad-api-cache # just the CLI
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Requires Python 3.11 or newer.
|
|
36
|
+
|
|
37
|
+
## Usage
|
|
38
|
+
|
|
39
|
+
### Command line
|
|
40
|
+
|
|
41
|
+
Inputs are any VCF-like file (VCF, BCF, or bgzipped VCF) and a SQLite database to store the cache. The output retrieves all variants from the cache and writes them to a file in the specified format.
|
|
42
|
+
```bash
|
|
43
|
+
gnomad-api-cache -i cohort.vcf.gz -c gnomad.sqlite -o annotations.parquet
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
The output format is inferred from the file extension (`.csv`, `.tsv`,
|
|
47
|
+
`.parquet`, `.json`, `.jsonl`) and can be forced with `-f`:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
gnomad-api-cache -i cohort.vcf.gz -c gnomad.sqlite -o annotations.txt -f tsv
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Omit `-o` to populate the cache without writing a table:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
gnomad-api-cache -i cohort.vcf.gz -c gnomad.sqlite
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
`python -m gnomad_api_cache` accepts the same arguments. Run
|
|
60
|
+
`gnomad-api-cache --help` for the full list, including `--dataset`,
|
|
61
|
+
`--include-populations`, and `--require-build`.
|
|
62
|
+
|
|
63
|
+
### Python
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
from gnomad_api_cache import VariantCache
|
|
67
|
+
|
|
68
|
+
with VariantCache("gnomad.sqlite") as cache:
|
|
69
|
+
summary = cache.fetch_vcf("cohort.vcf.gz")
|
|
70
|
+
print(summary) # 1234 requested, 0 already cached, 1200 fetched, ...
|
|
71
|
+
|
|
72
|
+
cache.to_parquet("annotations.parquet")
|
|
73
|
+
cache.to_csv("annotations.csv")
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Pass a `filter_function(cyvcf2.Variant) -> bool` to decide which variants are worth querying:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from cyvcf2 import Variant
|
|
80
|
+
from gnomad_api_cache import VariantCache
|
|
81
|
+
|
|
82
|
+
def rare_only(variant: Variant) -> bool:
|
|
83
|
+
af = variant.INFO.get("gnomad41_exome_AF", 0)
|
|
84
|
+
return float(0 if af == "." else af) <= 0.05
|
|
85
|
+
|
|
86
|
+
with VariantCache("gnomad.sqlite") as cache:
|
|
87
|
+
cache.fetch_vcf("cohort.vcf.gz", filter_function=rare_only)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
An open cache is a read-only `Mapping` keyed by `chrom-pos-ref-alt`, so cached
|
|
91
|
+
records are available without another request:
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
record = cache["1-55051215-G-A"] # None if gnomAD has no such variant
|
|
95
|
+
print(len(cache), cache.status_counts())
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
## Acknowledgement & Citation
|
|
99
|
+
|
|
100
|
+
This is an unofficial client. It queries the public gnomAD GraphQL API at
|
|
101
|
+
<https://gnomad.broadinstitute.org/api> and is not affiliated with or endorsed
|
|
102
|
+
by the Broad Institute or the gnomAD project. Please use the shared API
|
|
103
|
+
considerately — the default delay between requests is set to stay within the
|
|
104
|
+
documented rate limit.
|
|
105
|
+
|
|
106
|
+
If you use gnomAD data obtained through this tool, cite the current flagship
|
|
107
|
+
gnomAD paper. As of the latest release, it is as follows (v4 Preprint, Vancouver):
|
|
108
|
+
|
|
109
|
+
> Guez J, Goodrich JK, Moldovan MA, Chao KR, Kar P, Panchal R, Wilson MW, Laricchia KM, Rohlicek G, Biba D, Marten D. Integrating 730,947 exome sequences with clinical literature improves gene discovery. Medrxiv. 2026 Mar 25. <https://doi.org/10.64898/2026.03.23.26349081>
|
|
110
|
+
|
|
111
|
+
gnomAD data use terms: <https://gnomad.broadinstitute.org/terms>
|
|
112
|
+
|
|
113
|
+
## License
|
|
114
|
+
|
|
115
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
gnomad_api_cache/__init__.py,sha256=jQuyaHkYkXBy_C40vdi8mLlvt9PkfUiwLrJxSwZm2UE,911
|
|
2
|
+
gnomad_api_cache/__main__.py,sha256=d_KM_g0AGvoSLaWuR3WHcJSEgadDyM2bISLxFdScsc4,176
|
|
3
|
+
gnomad_api_cache/_utils.py,sha256=7hQhd6HAvDcrwZcBv5kM-GHoiGi7YCfigF0le9SjgPk,372
|
|
4
|
+
gnomad_api_cache/cache.py,sha256=pcah2J1CiErkM8WKRDkbvzx7nplhxwO4evGntYJkJEw,11763
|
|
5
|
+
gnomad_api_cache/cli.py,sha256=yUJyjViyIwbYlsxLU0emSgTPP5waZQK-CWFxRgCpunc,8839
|
|
6
|
+
gnomad_api_cache/client.py,sha256=uZhpHeh0cp_eZAhGjBeClg4i1P-_fU5asAaUefUVoLg,566
|
|
7
|
+
gnomad_api_cache/export.py,sha256=8-QQkvSz-F8_-PPl3tUj3fCLYLW3RM52JhNPDr5gkP0,11277
|
|
8
|
+
gnomad_api_cache/fetch.py,sha256=P2JuQxEj20RfDU9VL87UkGW8_EeTBt1We9hhQ7nc3Xc,4659
|
|
9
|
+
gnomad_api_cache/keys.py,sha256=c26ulJEyPcSo2-7902fkobe6i17uWLvRCmRyvHio_so,2316
|
|
10
|
+
gnomad_api_cache/query.py,sha256=x703RuA7TKkIEOAnCUuM1UuATWQuoEDRKwSn-pXNvh0,8860
|
|
11
|
+
gnomad_api_cache/adapters/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
12
|
+
gnomad_api_cache/adapters/vcf_adapter.py,sha256=JjY1GRGOOvA_7LxAVcobaKO1TWggGiRcaIm6JkLPbN0,4350
|
|
13
|
+
gnomad_api_cache-0.1.0.dist-info/METADATA,sha256=3PgyiQRm6648Cd_y_z0iZ5zZ6Sso6R-A5Y4hUf3szV8,3785
|
|
14
|
+
gnomad_api_cache-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
15
|
+
gnomad_api_cache-0.1.0.dist-info/entry_points.txt,sha256=kb9VVmRxneYatXKPPkcp-ebFqh1zZ8aolHGNv1XOC0Y,63
|
|
16
|
+
gnomad_api_cache-0.1.0.dist-info/licenses/LICENSE,sha256=04X5_8K92nIndcQhx7eO3jBWvIMsBdAgrIoE-T1t3hQ,1065
|
|
17
|
+
gnomad_api_cache-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 limenode
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|