afquery 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {afquery-0.3.0 → afquery-0.3.2}/CONTRIBUTING.md +1 -1
- afquery-0.3.2/PKG-INFO +256 -0
- afquery-0.3.2/README.md +226 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/getting-started/installation.md +5 -5
- {afquery-0.3.0 → afquery-0.3.2}/docs/getting-started/preprocessing.md +1 -1
- {afquery-0.3.0 → afquery-0.3.2}/mkdocs.yml +2 -2
- {afquery-0.3.0 → afquery-0.3.2}/recipes/afquery/meta.yaml +1 -1
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/_version.py +2 -2
- afquery-0.3.0/PKG-INFO +0 -145
- afquery-0.3.0/README.md +0 -115
- {afquery-0.3.0 → afquery-0.3.2}/.dockerignore +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/.github/workflows/ci.yml +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/.github/workflows/docs.yml +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/.github/workflows/release.yml +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/.gitignore +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/Dockerfile +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/LICENSE +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/.gitignore +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/INSTALL.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/README.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/Snakefile +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/.gitignore +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/01a_assign_samples.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/01b_write_manifests.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/02_build_databases.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/03_compute_metrics.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/04_classify_acmg.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/05_plot_figures.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/README.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/Snakefile +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/__init__.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/beds/SureSelect_v5.bed +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/beds/SureSelect_v6.bed +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/beds/SureSelect_v7.bed +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/capture_kit/config.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/config.yaml +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/envs/benchmark.yaml +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/.gitignore +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/01_prepare_data.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/02_query_scaling.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/03_build.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/04_annotate.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/05_vs_bcftools.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/06_plot.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/README.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/Snakefile +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/collect_annotate.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/collect_bcftools.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/collect_build_perf.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/collect_prepare.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/collect_query_scaling.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/config.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/performance/config_smoke.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/shared/__init__.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/shared/config.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/shared/rules/download_1kg.smk +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/benchmarks/shared/utils.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/benchmarking.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/coverage-evidence.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/debugging-results.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/filter-pass-tracking.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/multi-cohort-strategies.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/performance.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/pipeline-integration.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/advanced/ploidy-and-sex-chroms.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/assets/img/gap2_mixed_technologies.png +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/faq.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/getting-started/concepts.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/getting-started/motivation.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/getting-started/quickstart.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/getting-started/tutorial.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/getting-started/understanding-output.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/annotate-vcf.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/create-database.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/dump-export.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/manifest-format.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/query.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/sample-filtering.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/update-database.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/guides/variant-info.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/index.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/reference/cli.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/reference/data-model.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/reference/glossary.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/reference/python-api.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/scripts/gen_gap2_figure.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/stylesheets/extra.css +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/troubleshooting.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/use-cases/acmg-use-cases.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/use-cases/clinical-prioritization.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/use-cases/cohort-stratification.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/use-cases/population-specific-af.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/use-cases/pseudo-controls.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/use-cases/sex-specific-af.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/docs/use-cases/technology-integration.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/examples/demo/README.md +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/examples/demo/create_demo_data.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/pyproject.toml +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/resources/normalize_vcf.sh +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/__init__.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/annotate.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/benchmark.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/bitmaps.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/capture.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/cli.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/cli.pyr +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/constants.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/database.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/dump.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/models.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/ploidy.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/__init__.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/build.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/compact.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/ingest.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/manifest.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/regions.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/synth.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/preprocess/update.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/query.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/src/afquery/variant_info.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/conftest.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/annotate_input.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/annotate_multi_bucket.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/annotate_multi_chrom.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/beds/wes_kit_a.bed +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/beds/wes_kit_b.bed +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/expected_results.json +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/manifest.tsv +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S00.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S01.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S02.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S03.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S04.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S05.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S06.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S07.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S08.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/data/vcfs/S09.vcf +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_annotate.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_batch.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_benchmark.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_bitmaps.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_capture.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_cli.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_cli_docs_consistency.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_compact.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_constants.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_dump.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_haploid_stats.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_info.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_no_coverage.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_pass_filter.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_ploidy.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_preprocess.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_query.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_sample_filter.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_synth.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_synthetic_stats.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_update.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_update_metadata.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_variant_info.py +0 -0
- {afquery-0.3.0 → afquery-0.3.2}/tests/test_warnings.py +0 -0
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## Documentation deployment
|
|
4
4
|
|
|
5
|
-
The published site at <https://
|
|
5
|
+
The published site at <https://babelomics.github.io/afquery/> is built with [MkDocs](https://www.mkdocs.org/) (Material theme) and versioned with [mike](https://github.com/jimporter/mike). Source content lives under `docs/` and is configured by `mkdocs.yml`.
|
|
6
6
|
|
|
7
7
|
### CI workflows
|
|
8
8
|
|
afquery-0.3.2/PKG-INFO
ADDED
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: afquery
|
|
3
|
+
Version: 0.3.2
|
|
4
|
+
Summary: Genomic allele frequency query engine with bitmap-encoded genotypes
|
|
5
|
+
License: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
11
|
+
Requires-Python: >=3.10
|
|
12
|
+
Requires-Dist: click>=8.1
|
|
13
|
+
Requires-Dist: cyvcf2>=0.30
|
|
14
|
+
Requires-Dist: duckdb>=0.10
|
|
15
|
+
Requires-Dist: pyarrow>=14.0
|
|
16
|
+
Requires-Dist: pyranges<0.2,>=0.1.2
|
|
17
|
+
Requires-Dist: pyroaring>=0.4.8
|
|
18
|
+
Requires-Dist: tqdm>=4.60
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest-cov>=5.0; extra == 'dev'
|
|
21
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
22
|
+
Provides-Extra: docs
|
|
23
|
+
Requires-Dist: matplotlib>=3.8; extra == 'docs'
|
|
24
|
+
Requires-Dist: mike>=2.0; extra == 'docs'
|
|
25
|
+
Requires-Dist: mkdocs-material>=9.5; extra == 'docs'
|
|
26
|
+
Requires-Dist: mkdocs-print-site-plugin>=2.6.1; extra == 'docs'
|
|
27
|
+
Requires-Dist: mkdocstrings[python]>=0.25; extra == 'docs'
|
|
28
|
+
Requires-Dist: pymdown-extensions>=10.7; extra == 'docs'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# AFQuery
|
|
32
|
+
|
|
33
|
+
[](https://github.com/babelomics/afquery/actions/workflows/ci.yml)
|
|
34
|
+
[](https://codecov.io/gh/babelomics/afquery)
|
|
35
|
+
[](https://babelomics.github.io/afquery/)
|
|
36
|
+
<br>
|
|
37
|
+
[](https://pypi.org/project/afquery/)
|
|
38
|
+
[](https://anaconda.org/bioconda/afquery)
|
|
39
|
+
[](https://github.com/babelomics/afquery/pkgs/container/afquery)
|
|
40
|
+
[](https://pypi.org/project/afquery/)
|
|
41
|
+
[](https://github.com/babelomics/afquery/blob/master/LICENSE)
|
|
42
|
+
|
|
43
|
+
**Fast, capture-aware allele frequency queries on local genomic cohorts — without re-scanning VCFs.**
|
|
44
|
+
|
|
45
|
+
AFQuery is a bitmap-indexed engine that recomputes AC/AN/AF for arbitrary subcohorts (by phenotype, sex, sequencing technology, or any combination) in tens of milliseconds, independently of cohort size. It accounts for capture-kit heterogeneity, ploidy on sex chromosomes, and FILTER/coverage evidence, and runs locally as a file-based system (Parquet + SQLite) with no server or cluster required.
|
|
46
|
+
|
|
47
|
+
[Full documentation →](https://babelomics.github.io/afquery/)
|
|
48
|
+
|
|
49
|
+
---
|
|
50
|
+
|
|
51
|
+
## Headline results
|
|
52
|
+
|
|
53
|
+
- **~14 ms point queries**, constant from 1,000 to 50,000 samples (O(1) scaling).
|
|
54
|
+
- **33× faster than bcftools** for full-chromosome bulk export at 2,504 samples; **R² > 0.99999** AF concordance over 1.1M common variants.
|
|
55
|
+
- **~25,000 variants/s** VCF annotation on 4 cores — a typical exome annotates in **~1 second**.
|
|
56
|
+
- **Up to 45× reduction** in toward-pathogenic ACMG classification errors vs. naive AN on mixed-capture-kit cohorts.
|
|
57
|
+
- **9.5×–13.4× storage compression** vs. input single-sample VCFs.
|
|
58
|
+
|
|
59
|
+
## Why AFQuery
|
|
60
|
+
|
|
61
|
+
Local allele frequencies are a core input to ACMG/AMP variant classification (criteria BA1, BS1, PM2), and global resources like gnomAD systematically underrepresent local ancestries and disease-enriched institutional cohorts. Computing AF on your own cohort sounds simple — until the cohort mixes WGS, WES, and panels, or several versions of the same capture kit. Successive Agilent SureSelect versions (v5/v6/v7), for example, only share ~57% of their targets on chromosome 22. Naive AN counting then *inflates* AN at positions outside some kits' targets, *deflates* AF, and systematically shifts variants toward "pathogenic" by ACMG criteria.
|
|
62
|
+
|
|
63
|
+
AFQuery solves this with per-position, per-technology capture-aware AN, ploidy-aware sex chromosome handling, and an explicit `N_NO_COVERAGE` channel that separates trusted hom-ref from "we cannot tell". Queries on subcohorts are answered by intersecting precomputed Roaring Bitmaps, so latency is constant in cohort size.
|
|
64
|
+
|
|
65
|
+
## When to use AFQuery
|
|
66
|
+
|
|
67
|
+
- You need allele frequencies for phenotype-defined or arbitrarily filtered subcohorts.
|
|
68
|
+
- Your cohort mixes sequencing technologies (WGS, WES, panels, multiple capture-kit versions).
|
|
69
|
+
- You want fast, repeated, interactive queries instead of one-off VCF re-scans.
|
|
70
|
+
- You need a local, reproducible workflow — no cloud, no Spark cluster.
|
|
71
|
+
|
|
72
|
+
## Features
|
|
73
|
+
|
|
74
|
+
- **Constant-time subcohort queries** — bitmap intersections at query time; no per-query VCF re-scan.
|
|
75
|
+
- **Capture-aware AN** — per-position eligibility from each technology's BED, eliminating systematic AF bias when mixing WGS / WES / panels and kit versions.
|
|
76
|
+
- **Ploidy-aware sex chromosomes** — correct AN on chrX PAR / non-PAR, chrY, chrM, by sample sex.
|
|
77
|
+
- **Coverage evidence model** — `N_NO_COVERAGE` separates trusted hom-ref from samples lacking sufficient evidence; query-time gates (`--min-pass`, `--min-observed`, `--min-quality-evidence`) keep AF conservative.
|
|
78
|
+
- **ACMG-compatible AC/AN/AF** — per-standard definitions, exposed as both query output and VCF INFO fields.
|
|
79
|
+
- **Flexible metadata filtering** — arbitrary phenotype labels (ICD-10, HPO, OMIM, custom tags), inclusion or exclusion (`^` prefix), combined with sex and technology.
|
|
80
|
+
- **Parallel VCF annotation** — multi-threaded; adds `AFQUERY_AC/AN/AF/N_HET/N_HOM_ALT/N_HOM_REF/N_FAIL/N_NO_COVERAGE` INFO fields.
|
|
81
|
+
- **Bulk CSV export** — per-variant frequencies with optional disaggregation by sex, technology, or phenotype.
|
|
82
|
+
- **Incremental updates** — add or remove samples, edit phenotype/sex metadata, compact storage, without full rebuilds.
|
|
83
|
+
- **Audit changelog** — every database operation is logged with timestamps and operator notes.
|
|
84
|
+
- **Database validation** — `afquery check` with scripted exit codes.
|
|
85
|
+
- **Serverless** — Parquet + SQLite on disk; no daemon, no Java, no Spark.
|
|
86
|
+
|
|
87
|
+
## Installation
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
# PyPI
|
|
91
|
+
pip install afquery
|
|
92
|
+
|
|
93
|
+
# Bioconda
|
|
94
|
+
conda install -c bioconda afquery
|
|
95
|
+
|
|
96
|
+
# Docker (linux/amd64, linux/arm64)
|
|
97
|
+
docker pull ghcr.io/babelomics/afquery:latest
|
|
98
|
+
|
|
99
|
+
# From source
|
|
100
|
+
git clone https://github.com/babelomics/afquery.git
|
|
101
|
+
cd afquery
|
|
102
|
+
pip install -e .
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Requires Python ≥ 3.10. Core dependencies: `pyroaring`, `pyarrow`, `duckdb`, `pyranges`, `cyvcf2`, `click`, `tqdm`.
|
|
106
|
+
|
|
107
|
+
## Quickstart
|
|
108
|
+
|
|
109
|
+
### 1. Prepare a manifest
|
|
110
|
+
|
|
111
|
+
One row per sample (TSV, header required):
|
|
112
|
+
|
|
113
|
+
```tsv
|
|
114
|
+
sample_name vcf_path sex tech_name phenotype_codes
|
|
115
|
+
SAMP_001 /data/vcfs/SAMP_001.vcf.gz female wgs E11.9,I10
|
|
116
|
+
SAMP_002 /data/vcfs/SAMP_002.vcf.gz male wes_v6 E11.9
|
|
117
|
+
SAMP_003 /data/vcfs/SAMP_003.vcf.gz female panel_card I42.0
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
For every non-WGS technology, place a 3-column BED in `--bed-dir` named `<tech_name>.bed`.
|
|
121
|
+
|
|
122
|
+
### 2. Build the database
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
afquery create-db \
|
|
126
|
+
--manifest manifest.tsv \
|
|
127
|
+
--output-dir ./db/ \
|
|
128
|
+
--genome-build GRCh38 \
|
|
129
|
+
--bed-dir ./beds/
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
### 3. Query
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
# Single locus
|
|
136
|
+
afquery query --db ./db/ --locus chr1:925952
|
|
137
|
+
|
|
138
|
+
# Locus filtered to a phenotype-and-sex subcohort
|
|
139
|
+
afquery query --db ./db/ --locus chr1:925952 --phenotype E11.9 --sex female
|
|
140
|
+
|
|
141
|
+
# Genomic region
|
|
142
|
+
afquery query --db ./db/ --region chr1:900000-1000000
|
|
143
|
+
|
|
144
|
+
# Batch from file (chrom pos [ref [alt]] per line)
|
|
145
|
+
afquery query --db ./db/ --from-file variants.tsv
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
### 4. Inspect carriers of a variant
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
afquery variant-info --db ./db/ --locus chr17:43093454
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
### 5. Annotate a VCF
|
|
155
|
+
|
|
156
|
+
```bash
|
|
157
|
+
afquery annotate \
|
|
158
|
+
--db ./db/ \
|
|
159
|
+
--input patient.vcf \
|
|
160
|
+
--output patient.annotated.vcf \
|
|
161
|
+
--threads 4
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
### 6. Export
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
# Region export, disaggregated by sex
|
|
168
|
+
afquery dump --db ./db/ --chrom chr17 --start 43044292 --end 43170327 \
|
|
169
|
+
--output brca1.csv --by-sex
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
### 7. Update
|
|
173
|
+
|
|
174
|
+
```bash
|
|
175
|
+
afquery update-db --db ./db/ --add-samples new_batch.tsv
|
|
176
|
+
afquery update-db --db ./db/ --update-sample SAMP_007 --set-phenotype I42.0
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
## Output fields
|
|
180
|
+
|
|
181
|
+
Every query and annotated VCF reports:
|
|
182
|
+
|
|
183
|
+
| Field | Meaning |
|
|
184
|
+
|---|---|
|
|
185
|
+
| `AC` | Alt-allele count over eligible samples (FILTER=PASS) |
|
|
186
|
+
| `AN` | Total alleles considered, ploidy- and capture-aware |
|
|
187
|
+
| `AF` | `AC / AN` |
|
|
188
|
+
| `N_HET` | Heterozygous PASS carriers |
|
|
189
|
+
| `N_HOM_ALT` | Homozygous-alt PASS carriers |
|
|
190
|
+
| `N_HOM_REF` | Samples trusted as homozygous reference |
|
|
191
|
+
| `N_FAIL` | Carriers with FILTER ≠ PASS (excluded from AC/AN) |
|
|
192
|
+
| `N_NO_COVERAGE` | Non-carriers on a partially-covered tech without sufficient evidence to call hom-ref |
|
|
193
|
+
|
|
194
|
+
VCF INFO field names use the `AFQUERY_` prefix (e.g. `AFQUERY_AF`, `AFQUERY_N_NO_COVERAGE`).
|
|
195
|
+
|
|
196
|
+
## CLI commands
|
|
197
|
+
|
|
198
|
+
| Command | Purpose |
|
|
199
|
+
|---|---|
|
|
200
|
+
| `create-db` | Build a database from a manifest of single-sample VCFs |
|
|
201
|
+
| `query` | Point / region / batch AF queries |
|
|
202
|
+
| `variant-info` | List samples carrying a variant, with metadata |
|
|
203
|
+
| `annotate` | Annotate a VCF with cohort `AFQUERY_*` INFO fields |
|
|
204
|
+
| `dump` | Bulk CSV export, optionally disaggregated |
|
|
205
|
+
| `update-db` | Add / remove samples, edit metadata, compact |
|
|
206
|
+
| `info` | Show database metadata, sample list, changelog |
|
|
207
|
+
| `check` | Validate database integrity (scripted exit code) |
|
|
208
|
+
| `version show` / `version set` | Inspect or set the database version label |
|
|
209
|
+
| `benchmark` | Run synthetic or on-database performance benchmarks |
|
|
210
|
+
|
|
211
|
+
See the [CLI reference](https://babelomics.github.io/afquery/reference/cli/) for all options.
|
|
212
|
+
|
|
213
|
+
## How it works
|
|
214
|
+
|
|
215
|
+
AFQuery indexes each variant as three Roaring Bitmaps — heterozygous PASS carriers, homozygous-alt PASS carriers, and FILTER≠PASS carriers — stored in Apache Parquet, partitioned by chromosome and 1-Mbp positional buckets. Sample metadata (sex, technology, phenotype) is precomputed as bitmaps in SQLite. A query resolves its sample filter into a single candidate bitmap by intersection/difference in microseconds, then intersects it against each variant's genotype bitmaps to compute AC. AN is computed per position from the same candidate bitmap, restricted to samples whose technology actually covers the position (via BED-derived capture indices) and adjusted for ploidy on sex chromosomes. See the [data model reference](https://babelomics.github.io/afquery/reference/data-model/) for details.
|
|
216
|
+
|
|
217
|
+
## How AFQuery compares
|
|
218
|
+
|
|
219
|
+
| | AFQuery | bcftools | GATK GenomicsDB | Hail |
|
|
220
|
+
|---|---|---|---|---|
|
|
221
|
+
| Capture-aware AN | **Yes** | No | No | No |
|
|
222
|
+
| Metadata filtering | **Arbitrary labels** | No | No | Custom code |
|
|
223
|
+
| Ploidy-aware sex chromosomes | **Yes** | Manual | No | Manual |
|
|
224
|
+
| Dynamic subcohort queries | **Yes** | No | Limited | Requires code |
|
|
225
|
+
| FILTER / coverage tracking | **Per variant** | Manual | No | Manual |
|
|
226
|
+
| Incremental updates | **Yes** | No | **Yes** | No |
|
|
227
|
+
| Infrastructure required | **None** | **None** | Java/server | Spark cluster |
|
|
228
|
+
|
|
229
|
+
### Benchmarks vs. bcftools (1000 Genomes Phase 3, n = 2,504, chr22)
|
|
230
|
+
|
|
231
|
+
| Workload | AFQuery | bcftools | Speedup |
|
|
232
|
+
|---|---|---|---|
|
|
233
|
+
| Full-chromosome AC/AN/AF export | ~7.0 s | ~3.8 min | **~33×** |
|
|
234
|
+
| AF concordance over 1,106,181 common variants | — | — | **R² > 0.99999** |
|
|
235
|
+
|
|
236
|
+
Point-query latency on AFQuery is ~14 ms and constant from 1K to 50K samples (median over 50 replicates, warm cache).
|
|
237
|
+
|
|
238
|
+
## Documentation
|
|
239
|
+
|
|
240
|
+
- [Full documentation](https://babelomics.github.io/afquery/)
|
|
241
|
+
- [Getting started](https://babelomics.github.io/afquery/getting-started/)
|
|
242
|
+
- [Why local allele frequencies?](https://babelomics.github.io/afquery/getting-started/motivation/)
|
|
243
|
+
- [Manifest format](https://babelomics.github.io/afquery/guides/manifest-format/)
|
|
244
|
+
- [CLI reference](https://babelomics.github.io/afquery/reference/cli/)
|
|
245
|
+
- [FAQ](https://babelomics.github.io/afquery/faq/)
|
|
246
|
+
|
|
247
|
+
## Citation
|
|
248
|
+
|
|
249
|
+
If you use AFQuery in your work, please cite:
|
|
250
|
+
|
|
251
|
+
> AFQuery: fast, capture-aware allele frequency queries on local genomic cohorts.
|
|
252
|
+
> *(manuscript in preparation)*
|
|
253
|
+
|
|
254
|
+
## License
|
|
255
|
+
|
|
256
|
+
[MIT](LICENSE)
|
afquery-0.3.2/README.md
ADDED
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
# AFQuery
|
|
2
|
+
|
|
3
|
+
[](https://github.com/babelomics/afquery/actions/workflows/ci.yml)
|
|
4
|
+
[](https://codecov.io/gh/babelomics/afquery)
|
|
5
|
+
[](https://babelomics.github.io/afquery/)
|
|
6
|
+
<br>
|
|
7
|
+
[](https://pypi.org/project/afquery/)
|
|
8
|
+
[](https://anaconda.org/bioconda/afquery)
|
|
9
|
+
[](https://github.com/babelomics/afquery/pkgs/container/afquery)
|
|
10
|
+
[](https://pypi.org/project/afquery/)
|
|
11
|
+
[](https://github.com/babelomics/afquery/blob/master/LICENSE)
|
|
12
|
+
|
|
13
|
+
**Fast, capture-aware allele frequency queries on local genomic cohorts — without re-scanning VCFs.**
|
|
14
|
+
|
|
15
|
+
AFQuery is a bitmap-indexed engine that recomputes AC/AN/AF for arbitrary subcohorts (by phenotype, sex, sequencing technology, or any combination) in tens of milliseconds, independently of cohort size. It accounts for capture-kit heterogeneity, ploidy on sex chromosomes, and FILTER/coverage evidence, and runs locally as a file-based system (Parquet + SQLite) with no server or cluster required.
|
|
16
|
+
|
|
17
|
+
[Full documentation →](https://babelomics.github.io/afquery/)
|
|
18
|
+
|
|
19
|
+
---
|
|
20
|
+
|
|
21
|
+
## Headline results
|
|
22
|
+
|
|
23
|
+
- **~14 ms point queries**, constant from 1,000 to 50,000 samples (O(1) scaling).
|
|
24
|
+
- **33× faster than bcftools** for full-chromosome bulk export at 2,504 samples; **R² > 0.99999** AF concordance over 1.1M common variants.
|
|
25
|
+
- **~25,000 variants/s** VCF annotation on 4 cores — a typical exome annotates in **~1 second**.
|
|
26
|
+
- **Up to 45× reduction** in toward-pathogenic ACMG classification errors vs. naive AN on mixed-capture-kit cohorts.
|
|
27
|
+
- **9.5×–13.4× storage compression** vs. input single-sample VCFs.
|
|
28
|
+
|
|
29
|
+
## Why AFQuery
|
|
30
|
+
|
|
31
|
+
Local allele frequencies are a core input to ACMG/AMP variant classification (criteria BA1, BS1, PM2), and global resources like gnomAD systematically underrepresent local ancestries and disease-enriched institutional cohorts. Computing AF on your own cohort sounds simple — until the cohort mixes WGS, WES, and panels, or several versions of the same capture kit. Successive Agilent SureSelect versions (v5/v6/v7), for example, only share ~57% of their targets on chromosome 22. Naive AN counting then *inflates* AN at positions outside some kits' targets, *deflates* AF, and systematically shifts variants toward "pathogenic" by ACMG criteria.
|
|
32
|
+
|
|
33
|
+
AFQuery solves this with per-position, per-technology capture-aware AN, ploidy-aware sex chromosome handling, and an explicit `N_NO_COVERAGE` channel that separates trusted hom-ref from "we cannot tell". Queries on subcohorts are answered by intersecting precomputed Roaring Bitmaps, so latency is constant in cohort size.
|
|
34
|
+
|
|
35
|
+
## When to use AFQuery
|
|
36
|
+
|
|
37
|
+
- You need allele frequencies for phenotype-defined or arbitrarily filtered subcohorts.
|
|
38
|
+
- Your cohort mixes sequencing technologies (WGS, WES, panels, multiple capture-kit versions).
|
|
39
|
+
- You want fast, repeated, interactive queries instead of one-off VCF re-scans.
|
|
40
|
+
- You need a local, reproducible workflow — no cloud, no Spark cluster.
|
|
41
|
+
|
|
42
|
+
## Features
|
|
43
|
+
|
|
44
|
+
- **Constant-time subcohort queries** — bitmap intersections at query time; no per-query VCF re-scan.
|
|
45
|
+
- **Capture-aware AN** — per-position eligibility from each technology's BED, eliminating systematic AF bias when mixing WGS / WES / panels and kit versions.
|
|
46
|
+
- **Ploidy-aware sex chromosomes** — correct AN on chrX PAR / non-PAR, chrY, chrM, by sample sex.
|
|
47
|
+
- **Coverage evidence model** — `N_NO_COVERAGE` separates trusted hom-ref from samples lacking sufficient evidence; query-time gates (`--min-pass`, `--min-observed`, `--min-quality-evidence`) keep AF conservative.
|
|
48
|
+
- **ACMG-compatible AC/AN/AF** — per-standard definitions, exposed as both query output and VCF INFO fields.
|
|
49
|
+
- **Flexible metadata filtering** — arbitrary phenotype labels (ICD-10, HPO, OMIM, custom tags), inclusion or exclusion (`^` prefix), combined with sex and technology.
|
|
50
|
+
- **Parallel VCF annotation** — multi-threaded; adds `AFQUERY_AC/AN/AF/N_HET/N_HOM_ALT/N_HOM_REF/N_FAIL/N_NO_COVERAGE` INFO fields.
|
|
51
|
+
- **Bulk CSV export** — per-variant frequencies with optional disaggregation by sex, technology, or phenotype.
|
|
52
|
+
- **Incremental updates** — add or remove samples, edit phenotype/sex metadata, compact storage, without full rebuilds.
|
|
53
|
+
- **Audit changelog** — every database operation is logged with timestamps and operator notes.
|
|
54
|
+
- **Database validation** — `afquery check` with scripted exit codes.
|
|
55
|
+
- **Serverless** — Parquet + SQLite on disk; no daemon, no Java, no Spark.
|
|
56
|
+
|
|
57
|
+
## Installation
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
# PyPI
|
|
61
|
+
pip install afquery
|
|
62
|
+
|
|
63
|
+
# Bioconda
|
|
64
|
+
conda install -c bioconda afquery
|
|
65
|
+
|
|
66
|
+
# Docker (linux/amd64, linux/arm64)
|
|
67
|
+
docker pull ghcr.io/babelomics/afquery:latest
|
|
68
|
+
|
|
69
|
+
# From source
|
|
70
|
+
git clone https://github.com/babelomics/afquery.git
|
|
71
|
+
cd afquery
|
|
72
|
+
pip install -e .
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Requires Python ≥ 3.10. Core dependencies: `pyroaring`, `pyarrow`, `duckdb`, `pyranges`, `cyvcf2`, `click`, `tqdm`.
|
|
76
|
+
|
|
77
|
+
## Quickstart
|
|
78
|
+
|
|
79
|
+
### 1. Prepare a manifest
|
|
80
|
+
|
|
81
|
+
One row per sample (TSV, header required):
|
|
82
|
+
|
|
83
|
+
```tsv
|
|
84
|
+
sample_name vcf_path sex tech_name phenotype_codes
|
|
85
|
+
SAMP_001 /data/vcfs/SAMP_001.vcf.gz female wgs E11.9,I10
|
|
86
|
+
SAMP_002 /data/vcfs/SAMP_002.vcf.gz male wes_v6 E11.9
|
|
87
|
+
SAMP_003 /data/vcfs/SAMP_003.vcf.gz female panel_card I42.0
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
For every non-WGS technology, place a 3-column BED in `--bed-dir` named `<tech_name>.bed`.
|
|
91
|
+
|
|
92
|
+
### 2. Build the database
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
afquery create-db \
|
|
96
|
+
--manifest manifest.tsv \
|
|
97
|
+
--output-dir ./db/ \
|
|
98
|
+
--genome-build GRCh38 \
|
|
99
|
+
--bed-dir ./beds/
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
### 3. Query
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
# Single locus
|
|
106
|
+
afquery query --db ./db/ --locus chr1:925952
|
|
107
|
+
|
|
108
|
+
# Locus filtered to a phenotype-and-sex subcohort
|
|
109
|
+
afquery query --db ./db/ --locus chr1:925952 --phenotype E11.9 --sex female
|
|
110
|
+
|
|
111
|
+
# Genomic region
|
|
112
|
+
afquery query --db ./db/ --region chr1:900000-1000000
|
|
113
|
+
|
|
114
|
+
# Batch from file (chrom pos [ref [alt]] per line)
|
|
115
|
+
afquery query --db ./db/ --from-file variants.tsv
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### 4. Inspect carriers of a variant
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
afquery variant-info --db ./db/ --locus chr17:43093454
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### 5. Annotate a VCF
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
afquery annotate \
|
|
128
|
+
--db ./db/ \
|
|
129
|
+
--input patient.vcf \
|
|
130
|
+
--output patient.annotated.vcf \
|
|
131
|
+
--threads 4
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
### 6. Export
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
# Region export, disaggregated by sex
|
|
138
|
+
afquery dump --db ./db/ --chrom chr17 --start 43044292 --end 43170327 \
|
|
139
|
+
--output brca1.csv --by-sex
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
### 7. Update
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
afquery update-db --db ./db/ --add-samples new_batch.tsv
|
|
146
|
+
afquery update-db --db ./db/ --update-sample SAMP_007 --set-phenotype I42.0
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## Output fields
|
|
150
|
+
|
|
151
|
+
Every query and annotated VCF reports:
|
|
152
|
+
|
|
153
|
+
| Field | Meaning |
|
|
154
|
+
|---|---|
|
|
155
|
+
| `AC` | Alt-allele count over eligible samples (FILTER=PASS) |
|
|
156
|
+
| `AN` | Total alleles considered, ploidy- and capture-aware |
|
|
157
|
+
| `AF` | `AC / AN` |
|
|
158
|
+
| `N_HET` | Heterozygous PASS carriers |
|
|
159
|
+
| `N_HOM_ALT` | Homozygous-alt PASS carriers |
|
|
160
|
+
| `N_HOM_REF` | Samples trusted as homozygous reference |
|
|
161
|
+
| `N_FAIL` | Carriers with FILTER ≠ PASS (excluded from AC/AN) |
|
|
162
|
+
| `N_NO_COVERAGE` | Non-carriers on a partially-covered tech without sufficient evidence to call hom-ref |
|
|
163
|
+
|
|
164
|
+
VCF INFO field names use the `AFQUERY_` prefix (e.g. `AFQUERY_AF`, `AFQUERY_N_NO_COVERAGE`).
|
|
165
|
+
|
|
166
|
+
## CLI commands
|
|
167
|
+
|
|
168
|
+
| Command | Purpose |
|
|
169
|
+
|---|---|
|
|
170
|
+
| `create-db` | Build a database from a manifest of single-sample VCFs |
|
|
171
|
+
| `query` | Point / region / batch AF queries |
|
|
172
|
+
| `variant-info` | List samples carrying a variant, with metadata |
|
|
173
|
+
| `annotate` | Annotate a VCF with cohort `AFQUERY_*` INFO fields |
|
|
174
|
+
| `dump` | Bulk CSV export, optionally disaggregated |
|
|
175
|
+
| `update-db` | Add / remove samples, edit metadata, compact |
|
|
176
|
+
| `info` | Show database metadata, sample list, changelog |
|
|
177
|
+
| `check` | Validate database integrity (scripted exit code) |
|
|
178
|
+
| `version show` / `version set` | Inspect or set the database version label |
|
|
179
|
+
| `benchmark` | Run synthetic or on-database performance benchmarks |
|
|
180
|
+
|
|
181
|
+
See the [CLI reference](https://babelomics.github.io/afquery/reference/cli/) for all options.
|
|
182
|
+
|
|
183
|
+
## How it works
|
|
184
|
+
|
|
185
|
+
AFQuery indexes each variant as three Roaring Bitmaps — heterozygous PASS carriers, homozygous-alt PASS carriers, and FILTER≠PASS carriers — stored in Apache Parquet, partitioned by chromosome and 1-Mbp positional buckets. Sample metadata (sex, technology, phenotype) is precomputed as bitmaps in SQLite. A query resolves its sample filter into a single candidate bitmap by intersection/difference in microseconds, then intersects it against each variant's genotype bitmaps to compute AC. AN is computed per position from the same candidate bitmap, restricted to samples whose technology actually covers the position (via BED-derived capture indices) and adjusted for ploidy on sex chromosomes. See the [data model reference](https://babelomics.github.io/afquery/reference/data-model/) for details.
|
|
186
|
+
|
|
187
|
+
## How AFQuery compares
|
|
188
|
+
|
|
189
|
+
| | AFQuery | bcftools | GATK GenomicsDB | Hail |
|
|
190
|
+
|---|---|---|---|---|
|
|
191
|
+
| Capture-aware AN | **Yes** | No | No | No |
|
|
192
|
+
| Metadata filtering | **Arbitrary labels** | No | No | Custom code |
|
|
193
|
+
| Ploidy-aware sex chromosomes | **Yes** | Manual | No | Manual |
|
|
194
|
+
| Dynamic subcohort queries | **Yes** | No | Limited | Requires code |
|
|
195
|
+
| FILTER / coverage tracking | **Per variant** | Manual | No | Manual |
|
|
196
|
+
| Incremental updates | **Yes** | No | **Yes** | No |
|
|
197
|
+
| Infrastructure required | **None** | **None** | Java/server | Spark cluster |
|
|
198
|
+
|
|
199
|
+
### Benchmarks vs. bcftools (1000 Genomes Phase 3, n = 2,504, chr22)
|
|
200
|
+
|
|
201
|
+
| Workload | AFQuery | bcftools | Speedup |
|
|
202
|
+
|---|---|---|---|
|
|
203
|
+
| Full-chromosome AC/AN/AF export | ~7.0 s | ~3.8 min | **~33×** |
|
|
204
|
+
| AF concordance over 1,106,181 common variants | — | — | **R² > 0.99999** |
|
|
205
|
+
|
|
206
|
+
Point-query latency on AFQuery is ~14 ms and constant from 1K to 50K samples (median over 50 replicates, warm cache).
|
|
207
|
+
|
|
208
|
+
## Documentation
|
|
209
|
+
|
|
210
|
+
- [Full documentation](https://babelomics.github.io/afquery/)
|
|
211
|
+
- [Getting started](https://babelomics.github.io/afquery/getting-started/)
|
|
212
|
+
- [Why local allele frequencies?](https://babelomics.github.io/afquery/getting-started/motivation/)
|
|
213
|
+
- [Manifest format](https://babelomics.github.io/afquery/guides/manifest-format/)
|
|
214
|
+
- [CLI reference](https://babelomics.github.io/afquery/reference/cli/)
|
|
215
|
+
- [FAQ](https://babelomics.github.io/afquery/faq/)
|
|
216
|
+
|
|
217
|
+
## Citation
|
|
218
|
+
|
|
219
|
+
If you use AFQuery in your work, please cite:
|
|
220
|
+
|
|
221
|
+
> AFQuery: fast, capture-aware allele frequency queries on local genomic cohorts.
|
|
222
|
+
> *(manuscript in preparation)*
|
|
223
|
+
|
|
224
|
+
## License
|
|
225
|
+
|
|
226
|
+
[MIT](LICENSE)
|
|
@@ -16,7 +16,7 @@ pip install afquery
|
|
|
16
16
|
To install the latest development version directly from GitHub:
|
|
17
17
|
|
|
18
18
|
```bash
|
|
19
|
-
pip install git+https://github.com/
|
|
19
|
+
pip install git+https://github.com/babelomics/afquery.git
|
|
20
20
|
```
|
|
21
21
|
|
|
22
22
|
---
|
|
@@ -38,10 +38,10 @@ mamba install bioconda::afquery
|
|
|
38
38
|
|
|
39
39
|
## Docker
|
|
40
40
|
|
|
41
|
-
Official images are published to [GitHub Container Registry](https://github.com/
|
|
41
|
+
Official images are published to [GitHub Container Registry](https://github.com/babelomics/afquery/pkgs/container/afquery) for every release.
|
|
42
42
|
|
|
43
43
|
```bash
|
|
44
|
-
docker pull ghcr.io/
|
|
44
|
+
docker pull ghcr.io/babelomics/afquery:latest
|
|
45
45
|
```
|
|
46
46
|
|
|
47
47
|
Run the CLI by mounting a local directory with your database:
|
|
@@ -49,7 +49,7 @@ Run the CLI by mounting a local directory with your database:
|
|
|
49
49
|
```bash
|
|
50
50
|
docker run --rm \
|
|
51
51
|
-v /path/to/db:/db \
|
|
52
|
-
ghcr.io/
|
|
52
|
+
ghcr.io/babelomics/afquery:latest \
|
|
53
53
|
query --db /db --locus chr1:925952 --phenotype E11.9
|
|
54
54
|
```
|
|
55
55
|
|
|
@@ -61,7 +61,7 @@ docker run --rm \
|
|
|
61
61
|
## From Source
|
|
62
62
|
|
|
63
63
|
```bash
|
|
64
|
-
git clone https://github.com/
|
|
64
|
+
git clone https://github.com/babelomics/afquery.git
|
|
65
65
|
cd afquery
|
|
66
66
|
pip install -e .
|
|
67
67
|
```
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
AFQuery ingests single-sample, normalized VCF files. While AFQuery itself does not perform VCF normalization, the accuracy of your allele frequency estimates depends on the quality and consistency of your input VCFs. This page explains why normalization matters and provides a reference pipeline.
|
|
4
4
|
|
|
5
|
-
A ready-to-use normalization script is provided at [`resources/normalize_vcf.sh`](https://github.com/
|
|
5
|
+
A ready-to-use normalization script is provided at [`resources/normalize_vcf.sh`](https://github.com/babelomics/afquery/blob/master/resources/normalize_vcf.sh).
|
|
6
6
|
|
|
7
7
|
|
|
8
8
|
---
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
site_name: AFQuery
|
|
2
2
|
site_description: Fast, file-based genomic allele frequency queries for large cohorts
|
|
3
|
-
site_url: https://
|
|
4
|
-
repo_url: https://github.com/
|
|
3
|
+
site_url: https://babelomics.github.io/afquery/
|
|
4
|
+
repo_url: https://github.com/babelomics/afquery
|
|
5
5
|
repo_name: afquery
|
|
6
6
|
edit_uri: edit/master/docs/
|
|
7
7
|
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.3.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 3,
|
|
21
|
+
__version__ = version = '0.3.2'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 3, 2)
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|