gnomad-api-cache 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/.github/workflows/workflow.yml +2 -1
  2. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/PKG-INFO +33 -9
  3. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/README.md +32 -8
  4. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/pixi.toml +0 -1
  5. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/pyproject.toml +6 -1
  6. gnomad_api_cache-0.2.0/src/gnomad_api_cache/__init__.py +89 -0
  7. gnomad_api_cache-0.2.0/src/gnomad_api_cache/adapters/_collect.py +57 -0
  8. gnomad_api_cache-0.2.0/src/gnomad_api_cache/adapters/record_adapter.py +267 -0
  9. gnomad_api_cache-0.2.0/src/gnomad_api_cache/adapters/vcf_adapter.py +165 -0
  10. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/cache.py +7 -25
  11. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/cli.py +0 -2
  12. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/fetch.py +8 -13
  13. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/keys.py +23 -8
  14. gnomad_api_cache-0.1.0/src/gnomad_api_cache/__init__.py +0 -35
  15. gnomad_api_cache-0.1.0/src/gnomad_api_cache/adapters/vcf_adapter.py +0 -147
  16. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/.gitattributes +0 -0
  17. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/.gitignore +0 -0
  18. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/LICENSE +0 -0
  19. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/pixi.lock +0 -0
  20. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/pyrightconfig.json +0 -0
  21. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/__main__.py +0 -0
  22. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/_utils.py +0 -0
  23. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/adapters/__init__.py +0 -0
  24. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/client.py +0 -0
  25. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/export.py +0 -0
  26. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/src/gnomad_api_cache/query.py +0 -0
  27. {gnomad_api_cache-0.1.0 → gnomad_api_cache-0.2.0}/typings/cyvcf2/__init__.pyi +0 -0
@@ -66,7 +66,7 @@ jobs:
66
66
  run: |
67
67
  gnomad-api-cache --version
68
68
  python -m gnomad_api_cache --help > /dev/null
69
- python -c "from gnomad_api_cache import VariantCache, read_vcf, fetch_gnomad"
69
+ python -c "from gnomad_api_cache import VariantCache, read_vcf, fetch_into, to_variant_keys"
70
70
 
71
71
  publish-testpypi:
72
72
  name: Publish to TestPyPI
@@ -88,6 +88,7 @@ jobs:
88
88
  uses: pypa/gh-action-pypi-publish@release/v1
89
89
  with:
90
90
  repository-url: https://test.pypi.org/legacy/
91
+ skip-existing: true
91
92
 
92
93
  publish-pypi:
93
94
  name: Publish to PyPI
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gnomad-api-cache
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: Fetch and cache gnomAD annotations for VCF variants.
5
5
  License-File: LICENSE
6
6
  Requires-Python: >=3.11
@@ -65,34 +65,58 @@ gnomad-api-cache -i cohort.vcf.gz -c gnomad.sqlite
65
65
  ```python
66
66
  from gnomad_api_cache import VariantCache
67
67
 
68
+ variant_ids = ["1-55051215-G-A", "M-8993-T-C", ...]
69
+
68
70
  with VariantCache("gnomad.sqlite") as cache:
69
- summary = cache.fetch_vcf("cohort.vcf.gz")
70
- print(summary) # 1234 requested, 0 already cached, 1200 fetched, ...
71
+ summary = cache.fetch(variant_ids)
72
+ print(summary) # 1234 requested, 234 already cached, 1000 fetched, ...
71
73
 
72
74
  cache.to_parquet("annotations.parquet")
73
75
  cache.to_csv("annotations.csv")
74
76
  ```
75
77
 
76
- Pass a `filter_function(cyvcf2.Variant) -> bool` to decide which variants are worth querying:
78
+ `cache.fetch` takes in gnomAD id strings (hyphen- or colon-separated),
79
+ `(chrom, pos, ref, alt)` tuples, row dicts, a dataframe, or a VCF path.
80
+
81
+ ```python
82
+ with VariantCache("gnomad.sqlite") as cache:
83
+ cache.fetch(df) # any pandas/polars/duckdb frame with chrom/pos/ref/alt columns
84
+ cache.fetch("cohort.vcf.gz")
85
+ ```
86
+
87
+ Unusable rows are skipped and tallied; pass `on_error="raise"` to fail on the first one instead.
88
+
89
+ Reading the VCF yourself lets you filter before querying.
90
+ Define a function of the form `filter_function(cyvcf2.Variant) -> bool`
91
+ and pass it to `read_vcf` to only fetch variants that pass the filter:
77
92
 
78
93
  ```python
79
94
  from cyvcf2 import Variant
80
- from gnomad_api_cache import VariantCache
95
+ from gnomad_api_cache import VariantCache, read_vcf
81
96
 
82
97
  def rare_only(variant: Variant) -> bool:
83
98
  af = variant.INFO.get("gnomad41_exome_AF", 0)
84
99
  return float(0 if af == "." else af) <= 0.05
85
100
 
86
101
  with VariantCache("gnomad.sqlite") as cache:
87
- cache.fetch_vcf("cohort.vcf.gz", filter_function=rare_only)
102
+ cache.fetch(read_vcf("cohort.vcf.gz", filter_function=rare_only))
88
103
  ```
89
104
 
90
105
  An open cache is a read-only `Mapping` keyed by `chrom-pos-ref-alt`, so cached
91
- records are available without another request:
106
+ records can be retrieved using dictionary-like syntax:
107
+
108
+ ```python
109
+ record = cache["1-55051215-G-A"] # dict, or None if gnomAD has no such variant
110
+ ```
111
+
112
+ `to_variant_keys` builds those keys from any input `fetch` takes, so lookup keys
113
+ do not need to be manually constructed:
92
114
 
93
115
  ```python
94
- record = cache["1-55051215-G-A"] # None if gnomAD has no such variant
95
- print(len(cache), cache.status_counts())
116
+ from gnomad_api_cache import to_variant_keys
117
+
118
+ for key in to_variant_keys([("chr1", 55051215, "G", "A")]):
119
+ record = cache[key.id] # key.id == "1-55051215-G-A"
96
120
  ```
97
121
 
98
122
  ## Acknowledgement & Citation
@@ -53,34 +53,58 @@ gnomad-api-cache -i cohort.vcf.gz -c gnomad.sqlite
53
53
  ```python
54
54
  from gnomad_api_cache import VariantCache
55
55
 
56
+ variant_ids = ["1-55051215-G-A", "M-8993-T-C", ...]
57
+
56
58
  with VariantCache("gnomad.sqlite") as cache:
57
- summary = cache.fetch_vcf("cohort.vcf.gz")
58
- print(summary) # 1234 requested, 0 already cached, 1200 fetched, ...
59
+ summary = cache.fetch(variant_ids)
60
+ print(summary) # 1234 requested, 234 already cached, 1000 fetched, ...
59
61
 
60
62
  cache.to_parquet("annotations.parquet")
61
63
  cache.to_csv("annotations.csv")
62
64
  ```
63
65
 
64
- Pass a `filter_function(cyvcf2.Variant) -> bool` to decide which variants are worth querying:
66
+ `cache.fetch` takes in gnomAD id strings (hyphen- or colon-separated),
67
+ `(chrom, pos, ref, alt)` tuples, row dicts, a dataframe, or a VCF path.
68
+
69
+ ```python
70
+ with VariantCache("gnomad.sqlite") as cache:
71
+ cache.fetch(df) # any pandas/polars/duckdb frame with chrom/pos/ref/alt columns
72
+ cache.fetch("cohort.vcf.gz")
73
+ ```
74
+
75
+ Unusable rows are skipped and tallied; pass `on_error="raise"` to fail on the first one instead.
76
+
77
+ Reading the VCF yourself lets you filter before querying.
78
+ Define a function of the form `filter_function(cyvcf2.Variant) -> bool`
79
+ and pass it to `read_vcf` to only fetch variants that pass the filter:
65
80
 
66
81
  ```python
67
82
  from cyvcf2 import Variant
68
- from gnomad_api_cache import VariantCache
83
+ from gnomad_api_cache import VariantCache, read_vcf
69
84
 
70
85
  def rare_only(variant: Variant) -> bool:
71
86
  af = variant.INFO.get("gnomad41_exome_AF", 0)
72
87
  return float(0 if af == "." else af) <= 0.05
73
88
 
74
89
  with VariantCache("gnomad.sqlite") as cache:
75
- cache.fetch_vcf("cohort.vcf.gz", filter_function=rare_only)
90
+ cache.fetch(read_vcf("cohort.vcf.gz", filter_function=rare_only))
76
91
  ```
77
92
 
78
93
  An open cache is a read-only `Mapping` keyed by `chrom-pos-ref-alt`, so cached
79
- records are available without another request:
94
+ records can be retrieved using dictionary-like syntax:
95
+
96
+ ```python
97
+ record = cache["1-55051215-G-A"] # dict, or None if gnomAD has no such variant
98
+ ```
99
+
100
+ `to_variant_keys` builds those keys from any input `fetch` takes, so lookup keys
101
+ do not need to be manually constructed:
80
102
 
81
103
  ```python
82
- record = cache["1-55051215-G-A"] # None if gnomAD has no such variant
83
- print(len(cache), cache.status_counts())
104
+ from gnomad_api_cache import to_variant_keys
105
+
106
+ for key in to_variant_keys([("chr1", 55051215, "G", "A")]):
107
+ record = cache[key.id] # key.id == "1-55051215-G-A"
84
108
  ```
85
109
 
86
110
  ## Acknowledgement & Citation
@@ -2,7 +2,6 @@
2
2
  channels = ["conda-forge", "bioconda"]
3
3
  name = "gnomad_api_cache"
4
4
  platforms = ["linux-64"]
5
- version = "0.1.0"
6
5
 
7
6
  [tasks]
8
7
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "gnomad-api-cache"
3
- version = "0.1.0"
3
+ dynamic = ["version"]
4
4
  description = "Fetch and cache gnomAD annotations for VCF variants."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -25,5 +25,10 @@ dev = [
25
25
  requires = ["hatchling"]
26
26
  build-backend = "hatchling.build"
27
27
 
28
+ # Single source of truth for the version. Read at build time, so __version__
29
+ # stays a plain literal and `--version` costs no imports.
30
+ [tool.hatch.version]
31
+ path = "src/gnomad_api_cache/__init__.py"
32
+
28
33
  [tool.hatch.build.targets.wheel]
29
34
  packages = ["src/gnomad_api_cache"]
@@ -0,0 +1,89 @@
1
+ """Fetch and cache gnomAD annotations for VCF variants.
2
+
3
+ from gnomad_api_cache import VariantCache, read_vcf
4
+
5
+ with VariantCache("gnomad.sqlite") as cache:
6
+ print(cache.fetch("cohort.vcf.gz"))
7
+ cache.to_parquet("annotations.parquet")
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from importlib import import_module
13
+ from typing import TYPE_CHECKING, Any
14
+
15
+ __version__ = "0.2.0"
16
+
17
+ # Public name -> the module that defines it.
18
+ _LAZY_IMPORTS: dict[str, str] = {
19
+ "BuildMismatchError": "gnomad_api_cache.adapters.vcf_adapter",
20
+ "iter_variant_keys": "gnomad_api_cache.adapters.vcf_adapter",
21
+ "read_variants": "gnomad_api_cache.adapters.vcf_adapter",
22
+ "read_vcf": "gnomad_api_cache.adapters.vcf_adapter",
23
+ "read_arrow": "gnomad_api_cache.adapters.record_adapter",
24
+ "read_dicts": "gnomad_api_cache.adapters.record_adapter",
25
+ "read_ids": "gnomad_api_cache.adapters.record_adapter",
26
+ "read_tuples": "gnomad_api_cache.adapters.record_adapter",
27
+ "to_variant_keys": "gnomad_api_cache.adapters.record_adapter",
28
+ "VariantCache": "gnomad_api_cache.cache",
29
+ "FetchSummary": "gnomad_api_cache.fetch",
30
+ "fetch_into": "gnomad_api_cache.fetch",
31
+ "InvalidVariantError": "gnomad_api_cache.keys",
32
+ "VariantKey": "gnomad_api_cache.keys",
33
+ "DEFAULT_DATASET": "gnomad_api_cache.query",
34
+ "QUERY_VERSION": "gnomad_api_cache.query",
35
+ }
36
+
37
+ if TYPE_CHECKING:
38
+ # Type checkers do not follow __getattr__ well enough to give these real
39
+ # types, so state them here. Never executed at runtime.
40
+ from gnomad_api_cache.adapters.record_adapter import (
41
+ read_arrow,
42
+ read_dicts,
43
+ read_ids,
44
+ read_tuples,
45
+ to_variant_keys,
46
+ )
47
+ from gnomad_api_cache.adapters.vcf_adapter import (
48
+ BuildMismatchError,
49
+ iter_variant_keys,
50
+ read_variants,
51
+ read_vcf,
52
+ )
53
+ from gnomad_api_cache.cache import VariantCache
54
+ from gnomad_api_cache.fetch import FetchSummary, fetch_into
55
+ from gnomad_api_cache.keys import InvalidVariantError, VariantKey
56
+ from gnomad_api_cache.query import DEFAULT_DATASET, QUERY_VERSION
57
+
58
+
59
+ def __getattr__(name: str) -> Any:
60
+ module_name = _LAZY_IMPORTS.get(name)
61
+ if module_name is None:
62
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
63
+ value = getattr(import_module(module_name), name)
64
+ globals()[name] = value
65
+ return value
66
+
67
+
68
+ def __dir__() -> list[str]:
69
+ return sorted([*__all__, "__version__"])
70
+
71
+
72
+ __all__ = [
73
+ "DEFAULT_DATASET",
74
+ "QUERY_VERSION",
75
+ "BuildMismatchError",
76
+ "FetchSummary",
77
+ "InvalidVariantError",
78
+ "VariantCache",
79
+ "VariantKey",
80
+ "fetch_into",
81
+ "iter_variant_keys",
82
+ "read_arrow",
83
+ "read_dicts",
84
+ "read_ids",
85
+ "read_tuples",
86
+ "read_variants",
87
+ "read_vcf",
88
+ "to_variant_keys",
89
+ ]
@@ -0,0 +1,57 @@
1
+ """The row -> VariantKey loop."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from collections import Counter
7
+ from collections.abc import Callable, Iterable, Iterator
8
+ from typing import Any
9
+
10
+ from gnomad_api_cache.keys import InvalidVariantError, VariantKey
11
+
12
+ log = logging.getLogger(__name__)
13
+
14
+ ON_ERROR_SKIP = "skip"
15
+ ON_ERROR_RAISE = "raise"
16
+ ON_ERROR_CHOICES = (ON_ERROR_SKIP, ON_ERROR_RAISE)
17
+
18
+
19
+ def collect(
20
+ rows: Iterable[Any],
21
+ to_keys: Callable[[Any], VariantKey],
22
+ *,
23
+ on_error: str = ON_ERROR_SKIP,
24
+ label: str | None = None,
25
+ ) -> Iterator[VariantKey]:
26
+ """Build keys from rows, tally and logging what could not be built successfully."""
27
+ if on_error not in ON_ERROR_CHOICES:
28
+ raise ValueError(
29
+ f"on_error must be one of {ON_ERROR_CHOICES}, got {on_error!r}"
30
+ )
31
+
32
+ skipped: Counter[str] = Counter()
33
+ yielded = 0
34
+
35
+ for row in rows:
36
+ try:
37
+ key = to_keys(row)
38
+ except InvalidVariantError as exc:
39
+ if on_error == ON_ERROR_RAISE:
40
+ raise
41
+ skipped[exc.reason] += 1
42
+ log.debug("skipping %r -- %s", row, exc)
43
+ continue
44
+ yielded += 1
45
+ yield key
46
+
47
+ prefix = f"{label}: " if label else ""
48
+ if skipped:
49
+ log.info(
50
+ "%s%d keys, %d rows skipped (%s)",
51
+ prefix,
52
+ yielded,
53
+ sum(skipped.values()),
54
+ ", ".join(f"{k}={v}" for k, v in sorted(skipped.items())),
55
+ )
56
+ else:
57
+ log.info("%s%d keys", prefix, yielded)
@@ -0,0 +1,267 @@
1
+ """Build VariantKey objects from in-memory Python data.
2
+
3
+ Covers gnomAD id strings, 4-tuples, row mappings, and any dataframe exposing
4
+ the Arrow PyCapsule interface (pandas >= 2.2, polars >= 1.3, duckdb relations,
5
+ pyarrow tables). `to_variant_keys` handles all cases.
6
+
7
+ Each reader supplies callables to the collect() function in _collect.py.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections.abc import Iterable, Iterator, Mapping, Sequence
13
+ from itertools import chain
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+ from gnomad_api_cache.adapters._collect import ON_ERROR_SKIP, collect
18
+ from gnomad_api_cache.keys import InvalidVariantError, VariantKey
19
+
20
+ FIELDS = ("chrom", "pos", "ref", "alt")
21
+
22
+ # Common headers for chrom, pos, ref, alt.
23
+ _FIELD_BY_ALIAS: Mapping[str, str] = {
24
+ "chr": "chrom",
25
+ "chrom": "chrom",
26
+ "chromosome": "chrom",
27
+ "contig": "chrom",
28
+ "pos": "pos",
29
+ "position": "pos",
30
+ "start_1based": "pos",
31
+ "ref": "ref",
32
+ "reference": "ref",
33
+ "ref_allele": "ref",
34
+ "reference_allele": "ref",
35
+ "alt": "alt",
36
+ "alternate": "alt",
37
+ "alt_allele": "alt",
38
+ "alternate_allele": "alt",
39
+ }
40
+
41
+ # Interval-frame columns are 0-based half-open (BED, bioframe, pyranges) while
42
+ # gnomAD positions are 1-based. Treat them as an error to avoid off-by-one mistakes.
43
+ _INTERVAL_ALIASES = frozenset({"start", "chromstart", "end", "chromend", "stop"})
44
+
45
+ _VCF_SUFFIXES = (".vcf", ".vcf.gz", ".vcf.bgz", ".bcf", ".bcf.gz")
46
+
47
+
48
+ def _looks_like_vcf_path(text: str) -> bool:
49
+ return text.lower().endswith(_VCF_SUFFIXES)
50
+
51
+
52
+ # --- row -> key functions ------------------------------------------------
53
+
54
+
55
+ def _key_from_key(row: Any) -> VariantKey:
56
+ if not isinstance(row, VariantKey):
57
+ raise InvalidVariantError(f"not a VariantKey: {row!r}", "malformed_row")
58
+ return row
59
+
60
+
61
+ def _key_from_id(row: Any) -> VariantKey:
62
+ return VariantKey.from_id(str(row))
63
+
64
+
65
+ def _key_from_tuple(row: Any) -> VariantKey:
66
+ if isinstance(row, (str, bytes)) or not isinstance(row, Sequence):
67
+ raise InvalidVariantError(
68
+ f"not a (chrom, pos, ref, alt) row: {row!r}", "malformed_row"
69
+ )
70
+ if len(row) != 4:
71
+ raise InvalidVariantError(
72
+ f"expected 4 fields (chrom, pos, ref, alt), got {len(row)}: {row!r}",
73
+ "malformed_row",
74
+ )
75
+ chrom, pos, ref, alt = row
76
+ return VariantKey.from_parts(str(chrom), pos, str(ref), str(alt))
77
+
78
+
79
+ def _key_from_mapping(
80
+ row: Any,
81
+ columns: Mapping[str, str] | None = None,
82
+ ) -> VariantKey:
83
+ if not isinstance(row, Mapping):
84
+ raise InvalidVariantError(f"not a mapping: {row!r}", "malformed_row")
85
+
86
+ if columns is not None:
87
+ try:
88
+ values = {field: row[columns[field]] for field in FIELDS}
89
+ except KeyError as exc:
90
+ raise InvalidVariantError(
91
+ f"column {exc.args[0]!r} missing from row: {sorted(row)}",
92
+ "missing_fields",
93
+ ) from None
94
+ else:
95
+ # Lowercase the row's keys
96
+ lowered = {str(key).lower(): value for key, value in row.items()}
97
+ values = {}
98
+ for alias, field in _FIELD_BY_ALIAS.items():
99
+ if field not in values and alias in lowered:
100
+ values[field] = lowered[alias]
101
+
102
+ missing = [field for field in FIELDS if field not in values]
103
+ if missing:
104
+ if "pos" in missing and _INTERVAL_ALIASES & lowered.keys():
105
+ raise InvalidVariantError(
106
+ "interval columns (start/end) are 0-based half-open while "
107
+ "gnomAD positions are 1-based; supply a 1-based 'pos' "
108
+ f"column instead: {sorted(row)}",
109
+ "interval_columns",
110
+ )
111
+ raise InvalidVariantError(
112
+ f"no column for {missing} in row: {sorted(row)}", "missing_fields"
113
+ )
114
+
115
+ return VariantKey.from_parts(
116
+ str(values["chrom"]), values["pos"], str(values["ref"]), str(values["alt"])
117
+ )
118
+
119
+
120
+ # --- readers -------------------------------------------------------------
121
+
122
+
123
+ def iter_ids(
124
+ variant_ids: Iterable[Any],
125
+ *,
126
+ on_error: str = ON_ERROR_SKIP,
127
+ label: str | None = None,
128
+ ) -> Iterator[VariantKey]:
129
+ """Yield keys from "chrom-pos-ref-alt" (or colon-separated) id strings."""
130
+ return collect(variant_ids, _key_from_id, on_error=on_error, label=label)
131
+
132
+
133
+ def iter_tuples(
134
+ variants: Iterable[Any],
135
+ *,
136
+ on_error: str = ON_ERROR_SKIP,
137
+ label: str | None = None,
138
+ ) -> Iterator[VariantKey]:
139
+ """Yield keys from (chrom, pos, ref, alt) rows, e.g. df.itertuples()."""
140
+ return collect(variants, _key_from_tuple, on_error=on_error, label=label)
141
+
142
+
143
+ def iter_dicts(
144
+ variants: Iterable[Any],
145
+ *,
146
+ columns: Mapping[str, str] | None = None,
147
+ on_error: str = ON_ERROR_SKIP,
148
+ label: str | None = None,
149
+ ) -> Iterator[VariantKey]:
150
+ """Yield keys from row mappings, e.g. df.to_dict("records").
151
+
152
+ Column names are matched case-insensitively against a table of common
153
+ spellings; `columns` overrides that with an explicit
154
+ {"chrom": ..., "pos": ..., "ref": ..., "alt": ...} mapping.
155
+ """
156
+ return collect(
157
+ variants,
158
+ lambda row: _key_from_mapping(row, columns),
159
+ on_error=on_error,
160
+ label=label,
161
+ )
162
+
163
+
164
+ def iter_arrow(
165
+ source: Any,
166
+ *,
167
+ columns: Mapping[str, str] | None = None,
168
+ on_error: str = ON_ERROR_SKIP,
169
+ label: str | None = None,
170
+ ) -> Iterator[VariantKey]:
171
+ """Yield keys from any object implementing the Arrow PyCapsule interface."""
172
+ import pyarrow
173
+
174
+ def _rows() -> Iterator[dict[str, Any]]:
175
+ # Stream batch by batch so a large frame is never fully materialized
176
+ # as Python objects.
177
+ reader = pyarrow.RecordBatchReader.from_stream(source)
178
+ for batch in reader:
179
+ yield from batch.to_pylist()
180
+
181
+ return iter_dicts(_rows(), columns=columns, on_error=on_error, label=label)
182
+
183
+
184
+ def read_ids(variant_ids: Iterable[Any], **kwargs: Any) -> list[VariantKey]:
185
+ """Eager convenience wrapper around iter_ids."""
186
+ return list(iter_ids(variant_ids, **kwargs))
187
+
188
+
189
+ def read_tuples(variants: Iterable[Any], **kwargs: Any) -> list[VariantKey]:
190
+ """Eager convenience wrapper around iter_tuples."""
191
+ return list(iter_tuples(variants, **kwargs))
192
+
193
+
194
+ def read_dicts(variants: Iterable[Any], **kwargs: Any) -> list[VariantKey]:
195
+ """Eager convenience wrapper around iter_dicts."""
196
+ return list(iter_dicts(variants, **kwargs))
197
+
198
+
199
+ def read_arrow(source: Any, **kwargs: Any) -> list[VariantKey]:
200
+ """Eager convenience wrapper around iter_arrow."""
201
+ return list(iter_arrow(source, **kwargs))
202
+
203
+
204
+ # --- dispatch ------------------------------------------------------------
205
+
206
+
207
+ def to_variant_keys(
208
+ source: Any,
209
+ *,
210
+ columns: Mapping[str, str] | None = None,
211
+ on_error: str = ON_ERROR_SKIP,
212
+ ) -> list[VariantKey]:
213
+ """Coerce any supported input into a list of VariantKey.
214
+
215
+ Accepts a VariantKey, a VCF path (str/Path ending in a VCF suffix), a
216
+ single id string, a dataframe exposing the Arrow PyCapsule interface, a
217
+ single row mapping, or an iterable of ids / 4-tuples / row mappings /
218
+ cyvcf2 Variant records.
219
+ """
220
+ if isinstance(source, VariantKey):
221
+ return [source]
222
+
223
+ if isinstance(source, Path) or (
224
+ isinstance(source, str) and _looks_like_vcf_path(source)
225
+ ):
226
+ from gnomad_api_cache.adapters.vcf_adapter import read_vcf
227
+
228
+ return read_vcf(source, on_error=on_error)
229
+
230
+ if isinstance(source, str):
231
+ return [VariantKey.from_id(source)]
232
+
233
+ if hasattr(source, "__arrow_c_stream__"):
234
+ return read_arrow(source, columns=columns, on_error=on_error)
235
+
236
+ # A dataframe predating the PyCapsule interface. Intercept it before the
237
+ # generic iterable path, where iterating would yield column names.
238
+ if hasattr(source, "to_dict") and hasattr(source, "columns"):
239
+ return read_dicts(
240
+ source.to_dict("records"), columns=columns, on_error=on_error
241
+ )
242
+
243
+ if isinstance(source, Mapping):
244
+ return read_dicts([source], columns=columns, on_error=on_error)
245
+
246
+ if not isinstance(source, Iterable):
247
+ raise TypeError(f"cannot build variant keys from {type(source).__name__}")
248
+
249
+ # Peek at the first row to pick a reader, then put it back.
250
+ iterator = iter(source)
251
+ try:
252
+ first = next(iterator)
253
+ except StopIteration:
254
+ return []
255
+ rows = chain([first], iterator)
256
+
257
+ if isinstance(first, VariantKey):
258
+ return list(collect(rows, _key_from_key, on_error=on_error))
259
+ if isinstance(first, (str, bytes)):
260
+ return read_ids(rows, on_error=on_error)
261
+ if isinstance(first, Mapping):
262
+ return read_dicts(rows, columns=columns, on_error=on_error)
263
+ if hasattr(first, "ALT") and hasattr(first, "CHROM"):
264
+ from gnomad_api_cache.adapters.vcf_adapter import read_variants
265
+
266
+ return read_variants(rows, on_error=on_error)
267
+ return read_tuples(rows, on_error=on_error)
@@ -0,0 +1,165 @@
1
+ """Read a VCF, or already-loaded cyvcf2 records, into VariantKey objects."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from collections.abc import Callable, Iterable, Iterator
7
+ from pathlib import Path
8
+
9
+ from cyvcf2 import VCF, Variant
10
+
11
+ from gnomad_api_cache.adapters._collect import ON_ERROR_SKIP, collect
12
+ from gnomad_api_cache.keys import InvalidVariantError, VariantKey
13
+
14
+ log = logging.getLogger(__name__)
15
+
16
+ # Use chr1 to detect build
17
+ CHR1_LENGTH_BY_BUILD = {248956422: "GRCh38", 249250621: "GRCh37"}
18
+
19
+ AltRow = tuple[Variant, str]
20
+
21
+
22
+ class BuildMismatchError(RuntimeError):
23
+ """Raised when a VCF is not on the build the target gnomAD dataset uses."""
24
+
25
+
26
+ def detect_build(vcf: VCF) -> str | None:
27
+ """Return "GRCh38", "GRCh37", or None if the header doesn't say."""
28
+ lengths: dict[str, int] = dict(zip(vcf.seqnames, vcf.seqlens))
29
+ chr1 = lengths.get("chr1") or lengths.get("1")
30
+ if chr1 is None:
31
+ return None
32
+ return CHR1_LENGTH_BY_BUILD.get(chr1)
33
+
34
+
35
+ def is_normalized(vcf: VCF) -> bool:
36
+ """True if the header shows bcftools norm was run."""
37
+ return any(
38
+ line.startswith("##bcftools_norm") for line in vcf.raw_header.split("\n")
39
+ )
40
+
41
+
42
+ def _alt_syntax_reason(alt: str) -> str | None:
43
+ """Name the VCF ALT spelling that cannot become a gnomAD ID, or None."""
44
+ if alt.startswith("<"):
45
+ return "symbolic" # <DEL>, <NON_REF>, <*>
46
+ if "[" in alt or "]" in alt:
47
+ return "breakend"
48
+ if alt == "*":
49
+ return "spanning_deletion"
50
+ return None
51
+
52
+
53
+ def _iter_alt_rows(
54
+ records: Iterable[Variant],
55
+ filter_function: Callable[[Variant], bool] | None,
56
+ ) -> Iterator[AltRow]:
57
+ """Fan records out to one (record, alt) row per ALT allele."""
58
+ for record in records:
59
+ if filter_function is not None and not filter_function(record):
60
+ continue
61
+ for alt in record.ALT: # empty list when ALT is "."
62
+ yield record, alt
63
+
64
+
65
+ def _key_from_alt_row(row: AltRow) -> VariantKey:
66
+ record, alt = row
67
+ reason = _alt_syntax_reason(alt)
68
+ if reason is not None:
69
+ raise InvalidVariantError(f"unusable ALT allele: {alt!r}", reason)
70
+ return VariantKey.from_parts(record.CHROM, record.POS, record.REF, alt)
71
+
72
+
73
+ def iter_variant_keys_from_records(
74
+ records: Iterable[Variant],
75
+ *,
76
+ filter_function: Callable[[Variant], bool] | None = None,
77
+ on_error: str = ON_ERROR_SKIP,
78
+ label: str | None = None,
79
+ ) -> Iterator[VariantKey]:
80
+ """Yield one VariantKey per ALT allele of already-loaded cyvcf2 records."""
81
+ return collect(
82
+ _iter_alt_rows(records, filter_function),
83
+ _key_from_alt_row,
84
+ on_error=on_error,
85
+ label=label,
86
+ )
87
+
88
+
89
+ def iter_variant_keys(
90
+ path: str | Path,
91
+ *,
92
+ require_build: str | None = "GRCh38",
93
+ warn_unnormalized: bool = True,
94
+ filter_function: Callable[[Variant], bool] | None = None,
95
+ on_error: str = ON_ERROR_SKIP,
96
+ ) -> Iterator[VariantKey]:
97
+ """Yield one VariantKey per ALT allele of every record in `path`."""
98
+ vcf = VCF(str(path))
99
+ try:
100
+ build = detect_build(vcf)
101
+ if require_build is not None:
102
+ if build is None:
103
+ log.warning(
104
+ "%s: could not determine build from header; assuming %s",
105
+ path,
106
+ require_build,
107
+ )
108
+ elif build != require_build:
109
+ raise BuildMismatchError(
110
+ f"{path}: VCF is {build}, expected {require_build}. "
111
+ + "Lift over first, or target a matching gnomAD dataset."
112
+ )
113
+
114
+ if warn_unnormalized and not is_normalized(vcf):
115
+ log.warning(
116
+ "%s: no ##bcftools_norm header. Unnormalized indels will silently "
117
+ + "miss in gnomAD. Run: bcftools norm -f <ref.fa> -m -any",
118
+ path,
119
+ )
120
+
121
+ yield from iter_variant_keys_from_records(
122
+ vcf,
123
+ filter_function=filter_function,
124
+ on_error=on_error,
125
+ label=str(path),
126
+ )
127
+ finally:
128
+ # try/finally because a caller may abandon this generator part-way.
129
+ vcf.close()
130
+
131
+
132
+ def read_vcf(
133
+ path: str | Path,
134
+ *,
135
+ require_build: str | None = "GRCh38",
136
+ warn_unnormalized: bool = True,
137
+ filter_function: Callable[[Variant], bool] | None = None,
138
+ on_error: str = ON_ERROR_SKIP,
139
+ ) -> list[VariantKey]:
140
+ """Eager convenience wrapper around iter_variant_keys."""
141
+ return list(
142
+ iter_variant_keys(
143
+ path,
144
+ require_build=require_build,
145
+ warn_unnormalized=warn_unnormalized,
146
+ filter_function=filter_function,
147
+ on_error=on_error,
148
+ )
149
+ )
150
+
151
+
152
+ def read_variants(
153
+ variants: Iterable[Variant],
154
+ *,
155
+ filter_function: Callable[[Variant], bool] | None = None,
156
+ on_error: str = ON_ERROR_SKIP,
157
+ ) -> list[VariantKey]:
158
+ """Eager convenience wrapper around iter_variant_keys_from_records."""
159
+ return list(
160
+ iter_variant_keys_from_records(
161
+ variants,
162
+ filter_function=filter_function,
163
+ on_error=on_error,
164
+ )
165
+ )
@@ -26,10 +26,9 @@ from __future__ import annotations
26
26
  import json
27
27
  import sqlite3
28
28
  import zlib
29
- from collections.abc import Callable, Iterator, Mapping, Sequence
29
+ from collections.abc import Iterator, Mapping, Sequence
30
30
  from pathlib import Path
31
31
  from typing import TYPE_CHECKING, Any, Self
32
- from cyvcf2 import Variant
33
32
 
34
33
  if TYPE_CHECKING:
35
34
  from gnomad_api_cache.fetch import FetchSummary
@@ -300,29 +299,12 @@ class VariantCache(Mapping[str, "dict[str, Any] | None"]):
300
299
  )
301
300
  }
302
301
 
303
- def fetch(self, variants: list[VariantKey], **kwargs) -> FetchSummary:
304
- """Fetch missing records into this cache. See fetch.fetch_into."""
302
+ def fetch(self, variants: Any, **kwargs) -> FetchSummary:
303
+ """Fetch missing records into this cache.
304
+
305
+ Accepts VariantKey objects, gnomAD id strings, (chrom, pos, ref, alt)
306
+ tuples, row mappings, a dataframe, or a VCF path. See fetch.fetch_into.
307
+ """
305
308
  from gnomad_api_cache import fetch as _fetch
306
309
 
307
310
  return _fetch.fetch_into(self, variants, **kwargs)
308
-
309
- def fetch_vcf(
310
- self,
311
- vcf_path: str | Path,
312
- retry_errors: bool = True,
313
- retry_not_found: bool = False,
314
- filter_function: Callable[[Variant], bool] | None = None,
315
- ) -> FetchSummary:
316
- """Read a VCF and fetch missing records into this cache."""
317
- from gnomad_api_cache import fetch as _fetch
318
- from gnomad_api_cache.adapters import vcf_adapter
319
-
320
- return _fetch.fetch_into(
321
- self,
322
- vcf_adapter.read_vcf(
323
- vcf_path,
324
- filter_function=filter_function
325
- ),
326
- retry_errors=retry_errors,
327
- retry_not_found=retry_not_found,
328
- )
@@ -218,8 +218,6 @@ def main(argv: Sequence[str] | None = None) -> int:
218
218
  None if args.require_build.lower() == ANY_BUILD else args.require_build
219
219
  )
220
220
 
221
- # Imported here rather than at module scope so --help and --version stay
222
- # fast and do not require cyvcf2 to be importable.
223
221
  from gnomad_api_cache.adapters.vcf_adapter import BuildMismatchError, read_vcf
224
222
  from gnomad_api_cache.cache import VariantCache
225
223
 
@@ -3,13 +3,13 @@ from __future__ import annotations
3
3
  import time
4
4
  from collections.abc import Callable, Sequence
5
5
  from dataclasses import dataclass, replace
6
- from pathlib import Path
7
6
  from typing import Any
8
7
 
9
8
  import requests
10
9
  from tqdm import tqdm
11
10
 
12
11
  from gnomad_api_cache._utils import _chunked
12
+ from gnomad_api_cache.adapters.record_adapter import to_variant_keys
13
13
  from gnomad_api_cache.cache import VariantCache
14
14
  from gnomad_api_cache.client import post_gnomad
15
15
  from gnomad_api_cache.keys import VariantKey
@@ -85,16 +85,22 @@ def _fetch_batches(
85
85
 
86
86
  def fetch_into(
87
87
  cache: VariantCache,
88
- variants: list[VariantKey],
88
+ variants: Any,
89
89
  retry_errors: bool = True,
90
90
  retry_not_found: bool = False,
91
91
  delay: float = REQUEST_DELAY_SECONDS,
92
92
  ) -> FetchSummary:
93
93
  """Populate an open cache with gnomAD records for `variants`.
94
94
 
95
+ `variants` is anything record_adapter.to_variant_keys accepts: VariantKey
96
+ objects, id strings, 4-tuples, row mappings, a dataframe, or a VCF path.
97
+ Coercing here rather than in each caller means every entry point takes the
98
+ same inputs.
99
+
95
100
  The cache is left open: it belongs to the caller, who may well want to
96
101
  export from it next.
97
102
  """
103
+ variants = to_variant_keys(variants)
98
104
  variants_to_fetch = cache.needs_query(
99
105
  variants,
100
106
  retry_errors=retry_errors,
@@ -136,14 +142,3 @@ def fetch_into(
136
142
  f"errors; re-run to retry them."
137
143
  )
138
144
  return summary
139
-
140
-
141
- def fetch_gnomad(
142
- variants: list[VariantKey],
143
- cache_file: str | Path,
144
- **kwargs: Any,
145
- ) -> FetchSummary:
146
- """Populate a cache file with gnomAD records for `variants`,
147
- creating the cache if it does not exist."""
148
- with VariantCache(cache_file) as cache:
149
- return fetch_into(cache, variants, **kwargs)
@@ -11,6 +11,10 @@ _ALLELE_RE = re.compile(r"^[ACGTN]+$")
11
11
  class InvalidVariantError(ValueError):
12
12
  """Raised when a record cannot be expressed as a gnomAD variant ID."""
13
13
 
14
+ def __init__(self, message: str, reason: str = "invalid") -> None:
15
+ super().__init__(message)
16
+ self.reason = reason
17
+
14
18
 
15
19
  def normalize_chrom(chrom: str) -> str:
16
20
  """Canonicalize a contig name to the form gnomAD expects."""
@@ -55,24 +59,35 @@ class VariantKey:
55
59
  a = str(alt).strip().upper()
56
60
 
57
61
  if c not in CANONICAL_CHROMS:
58
- raise InvalidVariantError(f"non-canonical contig: {chrom!r}")
62
+ raise InvalidVariantError(
63
+ f"non-canonical contig: {chrom!r}", "non_canonical_contig"
64
+ )
59
65
  try:
60
66
  p = int(pos)
61
67
  except (TypeError, ValueError):
62
- raise InvalidVariantError(f"non-integer position: {pos!r}") from None
68
+ raise InvalidVariantError(
69
+ f"non-integer position: {pos!r}", "non_integer_position"
70
+ ) from None
63
71
  if p < 1:
64
- raise InvalidVariantError(f"position must be 1-based: {pos!r}")
72
+ raise InvalidVariantError(
73
+ f"position must be 1-based: {pos!r}", "non_positive_position"
74
+ )
65
75
  if not _ALLELE_RE.match(r):
66
- raise InvalidVariantError(f"non-ACGTN ref allele: {ref!r}")
76
+ raise InvalidVariantError(f"non-ACGTN ref allele: {ref!r}", "non_acgtn_ref")
67
77
  if not _ALLELE_RE.match(a):
68
- raise InvalidVariantError(f"non-ACGTN alt allele: {alt!r}")
78
+ raise InvalidVariantError(f"non-ACGTN alt allele: {alt!r}", "non_acgtn_alt")
69
79
 
70
80
  return cls(chrom=c, pos=p, ref=r, alt=a)
71
81
 
72
82
  @classmethod
73
83
  def from_id(cls, variant_id: str) -> VariantKey:
74
- """Parse a "chrom-pos-ref-alt" string, e.g. from a text file or the cache."""
75
- parts = str(variant_id).strip().split("-")
84
+ """Parse a "chrom-pos-ref-alt" or "chrom:pos:ref:alt" ID string into a VariantKey."""
85
+ text = str(variant_id).strip()
86
+ parts = text.split("-")
87
+ if len(parts) != 4 and ":" in text:
88
+ parts = text.split(":")
76
89
  if len(parts) != 4:
77
- raise InvalidVariantError(f"expected chrom-pos-ref-alt, got {variant_id!r}")
90
+ raise InvalidVariantError(
91
+ f"expected chrom-pos-ref-alt, got {variant_id!r}", "malformed_id"
92
+ )
78
93
  return cls.from_parts(*parts)
@@ -1,35 +0,0 @@
1
- """Fetch and cache gnomAD annotations for VCF variants.
2
-
3
- from gnomad_api_cache import VariantCache, read_vcf
4
-
5
- with VariantCache("gnomad.sqlite") as cache:
6
- print(cache.fetch_vcf("cohort.vcf.gz"))
7
- cache.to_parquet("annotations.parquet")
8
- """
9
-
10
- from __future__ import annotations
11
-
12
- from gnomad_api_cache.adapters.vcf_adapter import (
13
- BuildMismatchError,
14
- iter_variant_keys,
15
- read_vcf,
16
- )
17
- from gnomad_api_cache.cache import VariantCache
18
- from gnomad_api_cache.fetch import FetchSummary, fetch_gnomad
19
- from gnomad_api_cache.keys import InvalidVariantError, VariantKey
20
- from gnomad_api_cache.query import DEFAULT_DATASET, QUERY_VERSION
21
-
22
- __version__ = "0.1.0"
23
-
24
- __all__ = [
25
- "DEFAULT_DATASET",
26
- "QUERY_VERSION",
27
- "BuildMismatchError",
28
- "FetchSummary",
29
- "InvalidVariantError",
30
- "VariantCache",
31
- "VariantKey",
32
- "fetch_gnomad",
33
- "iter_variant_keys",
34
- "read_vcf",
35
- ]
@@ -1,147 +0,0 @@
1
- """Read a VCF into VariantKey objects."""
2
-
3
- from __future__ import annotations
4
-
5
- import logging
6
- from collections import Counter
7
- from collections.abc import Callable, Iterator
8
- from pathlib import Path
9
-
10
- from cyvcf2 import VCF, Variant
11
-
12
- from gnomad_api_cache.keys import (
13
- CANONICAL_CHROMS,
14
- InvalidVariantError,
15
- VariantKey,
16
- normalize_chrom,
17
- )
18
-
19
- log = logging.getLogger(__name__)
20
-
21
- # Use chr1 to detect build
22
- CHR1_LENGTH_BY_BUILD = {248956422: "GRCh38", 249250621: "GRCh37"}
23
-
24
-
25
- class BuildMismatchError(RuntimeError):
26
- """Raised when a VCF is not on the build the target gnomAD dataset uses."""
27
-
28
-
29
- def detect_build(vcf: VCF) -> str | None:
30
- """Return "GRCh38", "GRCh37", or None if the header doesn't say."""
31
- lengths: dict[str, int] = dict(zip(vcf.seqnames, vcf.seqlens))
32
- chr1 = lengths.get("chr1") or lengths.get("1")
33
- if chr1 is None:
34
- return None
35
- return CHR1_LENGTH_BY_BUILD.get(chr1)
36
-
37
-
38
- def is_normalized(vcf: VCF) -> bool:
39
- """True if the header shows bcftools norm was run."""
40
- return any(
41
- line.startswith("##bcftools_norm") for line in vcf.raw_header.split("\n")
42
- )
43
-
44
-
45
- def _skip_reason(alt: str, chrom: str) -> str | None:
46
- """Return why this allele can't become a gnomAD ID, or None if it can."""
47
- if alt.startswith("<"):
48
- return "symbolic" # <DEL>, <NON_REF>, <*>
49
- if "[" in alt or "]" in alt:
50
- return "breakend"
51
- if alt == "*":
52
- return "spanning_deletion"
53
- if normalize_chrom(chrom) not in CANONICAL_CHROMS:
54
- return "non_canonical_contig" # *_random, chrUn_*, HLA-*
55
- return None
56
-
57
-
58
- def iter_variant_keys(
59
- path: str | Path,
60
- *,
61
- require_build: str | None = "GRCh38",
62
- warn_unnormalized: bool = True,
63
- filter_function: Callable[[Variant], bool] | None = None,
64
- ) -> Iterator[VariantKey]:
65
- """Yield one VariantKey per ALT allele of every record in `path`."""
66
- vcf = VCF(str(path))
67
-
68
- build = detect_build(vcf)
69
- if require_build is not None:
70
- if build is None:
71
- log.warning(
72
- "%s: could not determine build from header; assuming %s",
73
- path,
74
- require_build,
75
- )
76
- elif build != require_build:
77
- raise BuildMismatchError(
78
- f"{path}: VCF is {build}, expected {require_build}. "
79
- + "Lift over first, or target a matching gnomAD dataset."
80
- )
81
-
82
- if warn_unnormalized and not is_normalized(vcf):
83
- log.warning(
84
- "%s: no ##bcftools_norm header. Unnormalized indels will silently "
85
- + "miss in gnomAD. Run: bcftools norm -f <ref.fa> -m -any",
86
- path,
87
- )
88
-
89
- skipped: Counter[str] = Counter()
90
- yielded = 0
91
-
92
- # main loop
93
- for record in vcf:
94
- if filter_function is not None and not filter_function(record):
95
- continue
96
-
97
- for alt in record.ALT: # empty list when ALT is "."
98
- reason = _skip_reason(alt, record.CHROM)
99
- if reason is not None:
100
- skipped[reason] += 1
101
- continue
102
- try:
103
- key = VariantKey.from_parts(record.CHROM, record.POS, record.REF, alt)
104
- except InvalidVariantError as e:
105
- skipped["invalid"] += 1
106
- log.debug(
107
- "skipping %s:%s %s>%s -- %s",
108
- record.CHROM,
109
- record.POS,
110
- record.REF,
111
- alt,
112
- e,
113
- )
114
- continue
115
- yielded += 1
116
- yield key
117
-
118
- vcf.close()
119
-
120
- if skipped:
121
- log.info(
122
- "%s: %d keys, %d alleles skipped (%s)",
123
- path,
124
- yielded,
125
- sum(skipped.values()),
126
- ", ".join(f"{k}={v}" for k, v in sorted(skipped.items())),
127
- )
128
- else:
129
- log.info("%s: %d keys", path, yielded)
130
-
131
-
132
- def read_vcf(
133
- path: str | Path,
134
- *,
135
- require_build: str | None = "GRCh38",
136
- warn_unnormalized: bool = True,
137
- filter_function: Callable[[Variant], bool] | None = None,
138
- ) -> list[VariantKey]:
139
- """Eager convenience wrapper around iter_variant_keys."""
140
- return list(
141
- iter_variant_keys(
142
- path,
143
- require_build=require_build,
144
- warn_unnormalized=warn_unnormalized,
145
- filter_function=filter_function
146
- )
147
- )