afquery 0.4.0__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {afquery-0.4.0 → afquery-0.4.2}/PKG-INFO +2 -2
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/ploidy-and-sex-chroms.md +2 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/update-database.md +11 -3
- {afquery-0.4.0 → afquery-0.4.2}/docs/reference/cli.md +13 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/reference/data-model.md +16 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/troubleshooting.md +1 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/_version.py +2 -2
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/annotate.py +7 -6
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/benchmark.py +11 -24
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/cli.py +12 -7
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/dump.py +9 -16
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/build.py +4 -5
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/compact.py +7 -10
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/update.py +202 -118
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/query.py +19 -24
- afquery-0.4.2/src/afquery/storage.py +177 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/expected_results.json +2 -2
- afquery-0.4.2/tests/oracle.py +195 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_cli.py +23 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_haploid_stats.py +30 -0
- afquery-0.4.2/tests/test_invariants.py +186 -0
- afquery-0.4.2/tests/test_oracle_consistency.py +164 -0
- afquery-0.4.2/tests/test_storage.py +252 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_update.py +54 -0
- afquery-0.4.2/tests/test_update_partitioned.py +355 -0
- {afquery-0.4.0 → afquery-0.4.2}/.dockerignore +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/.github/workflows/ci.yml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/.github/workflows/docs.yml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/.github/workflows/release.yml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/.gitignore +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/CONTRIBUTING.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/Dockerfile +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/LICENSE +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/README.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/.gitignore +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/INSTALL.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/README.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/Snakefile +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/.gitignore +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/01a_assign_samples.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/01b_write_manifests.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/02_build_databases.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/03_compute_metrics.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/04_classify_acmg.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/05_plot_figures.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/README.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/Snakefile +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/__init__.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/beds/SureSelect_v5.bed +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/beds/SureSelect_v6.bed +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/beds/SureSelect_v7.bed +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/config.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/config.yaml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/envs/benchmark.yaml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/.gitignore +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/01_prepare_data.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/02_query_scaling.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/03_build.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/04_annotate.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/05_vs_bcftools.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/06_plot.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/README.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/Snakefile +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_annotate.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_bcftools.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_build_perf.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_prepare.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_query_scaling.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/config.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/config_smoke.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/__init__.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/config.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/rules/download_1kg.smk +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/utils.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/benchmarking.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/coverage-evidence.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/debugging-results.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/filter-pass-tracking.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/multi-cohort-strategies.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/performance.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/pipeline-integration.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/assets/img/gap2_mixed_technologies.png +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/faq.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/concepts.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/installation.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/motivation.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/preprocessing.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/quickstart.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/tutorial.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/understanding-output.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/annotate-vcf.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/create-database.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/dump-export.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/manifest-format.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/query.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/sample-filtering.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/guides/variant-info.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/index.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/reference/glossary.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/reference/python-api.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/scripts/gen_gap2_figure.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/stylesheets/extra.css +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/acmg-use-cases.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/clinical-prioritization.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/cohort-stratification.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/population-specific-af.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/pseudo-controls.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/sex-specific-af.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/technology-integration.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/examples/demo/README.md +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/examples/demo/create_demo_data.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/mkdocs.yml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/pyproject.toml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/recipes/afquery/meta.yaml +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/resources/normalize_vcf.sh +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/__init__.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/bitmaps.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/capture.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/cli.pyr +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/constants.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/database.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/models.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/ploidy.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/__init__.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/ingest.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/manifest.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/regions.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/synth.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/src/afquery/variant_info.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/conftest.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/annotate_input.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/annotate_multi_bucket.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/annotate_multi_chrom.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/beds/wes_kit_a.bed +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/beds/wes_kit_b.bed +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/beds/wes_kit_nochr.bed +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/manifest.tsv +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S00.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S01.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S02.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S03.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S04.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S05.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S06.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S07.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S08.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S09.vcf +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_annotate.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_batch.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_benchmark.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_bitmaps.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_capture.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_cli_docs_consistency.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_compact.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_constants.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_dump.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_info.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_no_coverage.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_pass_filter.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_ploidy.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_preprocess.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_query.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_sample_filter.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_synth.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_synthetic_stats.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_update_metadata.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_variant_info.py +0 -0
- {afquery-0.4.0 → afquery-0.4.2}/tests/test_warnings.py +0 -0
|
@@ -19,6 +19,8 @@ AFQuery computes ploidy-aware AN for sex chromosomes (chrX, chrY) and the mitoch
|
|
|
19
19
|
|
|
20
20
|
For each eligible sample at a given position, AFQuery adds the appropriate ploidy count to AN based on the sample's sex and the chromosome/position.
|
|
21
21
|
|
|
22
|
+
A sample contributing ploidy 0 is not eligible at that position at all: a female has no chrY to genotype, so she is neither a carrier nor homozygous reference there. On chrY, `n_eligible`, `N_HOM_REF` and `AN` therefore count males only.
|
|
23
|
+
|
|
22
24
|
---
|
|
23
25
|
|
|
24
26
|
## Pseudoautosomal Regions (PAR)
|
|
@@ -38,6 +38,8 @@ afquery update-db \
|
|
|
38
38
|
|
|
39
39
|
The new manifest follows the same format as the original (see [Manifest Format](manifest-format.md)). New samples are assigned monotonically increasing sample IDs.
|
|
40
40
|
|
|
41
|
+
New variants are merged into the per-chromosome bucket files the database already uses; buckets are created on demand, and no new top-level Parquet file is produced. See [Data Model](../reference/data-model.md#storage-layouts).
|
|
42
|
+
|
|
41
43
|
To add multiple manifests at once:
|
|
42
44
|
|
|
43
45
|
```bash
|
|
@@ -67,9 +69,15 @@ decisions are comparable across batches.
|
|
|
67
69
|
When new carriers push a partially-covered tech above the `--min-covered`
|
|
68
70
|
threshold at positions that were previously below it, those positions are
|
|
69
71
|
re-evaluated and their non-carrier samples once again count as `N_HOM_REF`
|
|
70
|
-
instead of `N_NO_COVERAGE`.
|
|
71
|
-
|
|
72
|
-
|
|
72
|
+
instead of `N_NO_COVERAGE`. Because that value derives from the tech bitmaps of
|
|
73
|
+
the whole cohort rather than from which file received rows, the recomputation
|
|
74
|
+
spans **every bucket of every chromosome the database holds**, not only the ones
|
|
75
|
+
the new samples carry variants on. Skipping the rest would leave an added WES
|
|
76
|
+
sample counted as `N_HOM_REF` everywhere else its capture BED reaches. Only a
|
|
77
|
+
batch that puts a sample into a capture-based technology can move those bitmaps,
|
|
78
|
+
so a WGS-only batch still visits nothing beyond its own chromosomes; adding a
|
|
79
|
+
WES sample to a large database rewrites broadly and takes minutes rather than
|
|
80
|
+
seconds.
|
|
73
81
|
|
|
74
82
|
VCFs added via `update-db` should preserve `FORMAT/DP` and `FORMAT/GQ` (the
|
|
75
83
|
bundled `resources/normalize_vcf.sh` does so by default). Samples without
|
|
@@ -4,6 +4,19 @@ All AFQuery commands follow the pattern `afquery <command> [OPTIONS]`.
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## Global options
|
|
8
|
+
|
|
9
|
+
| Option | Description |
|
|
10
|
+
|--------|-------------|
|
|
11
|
+
| `--version` | Print the AFQuery version and exit |
|
|
12
|
+
| `--help` | Show help for the command and exit |
|
|
13
|
+
|
|
14
|
+
`afquery --version` prints the installed package version (e.g. `afquery 0.4.2`),
|
|
15
|
+
taken from the Git release tag at build time. In an editable install
|
|
16
|
+
(`pip install -e .`) it reflects the last build, not later local commits.
|
|
17
|
+
|
|
18
|
+
---
|
|
19
|
+
|
|
7
20
|
## create-db
|
|
8
21
|
|
|
9
22
|
Build a new AFQuery database from a manifest of single-sample VCFs.
|
|
@@ -22,6 +22,18 @@ This page documents the on-disk layout of an AFQuery database, including file fo
|
|
|
22
22
|
└── wes_v2.pkl
|
|
23
23
|
```
|
|
24
24
|
|
|
25
|
+
### Storage layouts
|
|
26
|
+
|
|
27
|
+
`create-db` always writes the bucketed layout shown above. A single-file-per-chromosome
|
|
28
|
+
layout (`variants/chr1.parquet`, with no bucket directory) is also readable, and appears
|
|
29
|
+
in small hand-built databases and in test fixtures. `update-db` merges into whichever
|
|
30
|
+
layout a chromosome already uses, and creates buckets for a chromosome new to the
|
|
31
|
+
database.
|
|
32
|
+
|
|
33
|
+
A chromosome must never have both. `afquery check` reports that as an error, because
|
|
34
|
+
queries resolve the bucket directory first and would silently ignore the flat file — and
|
|
35
|
+
every sample stored only in it.
|
|
36
|
+
|
|
25
37
|
---
|
|
26
38
|
|
|
27
39
|
## manifest.json
|
|
@@ -156,6 +168,10 @@ Variants are partitioned into 1-Mbp buckets:
|
|
|
156
168
|
bucket_id = pos // 1_000_000
|
|
157
169
|
```
|
|
158
170
|
|
|
171
|
+
`update-db --add-samples` writes new positions into the bucket that owns
|
|
172
|
+
them, creating `bucket_N.parquet` when a batch extends a chromosome past its
|
|
173
|
+
previous last bucket.
|
|
174
|
+
|
|
159
175
|
!!! warning "DuckDB integer arithmetic"
|
|
160
176
|
When computing bucket IDs in DuckDB SQL, always use the integer-division operator:
|
|
161
177
|
```sql
|
|
@@ -116,6 +116,7 @@ afquery info --db ./db/ --samples | grep SAMP
|
|
|
116
116
|
| `Missing Parquet for chromosome chr3` | Re-run `create-db` or investigate incomplete build |
|
|
117
117
|
| `Manifest mismatch: expected N samples, found M` | Database may be partially updated; re-run `update-db` |
|
|
118
118
|
| `Capture file missing for wes_v1` | BED file was not provided at build time; rebuild with `--bed-dir` |
|
|
119
|
+
| `chr1: both variants/chr1/ ... and variants/chr1.parquet exist` | Samples added by a pre-0.4.1 `update-db` are invisible to queries; see [Samples Added by update-db Are Missing From Queries](#samples-added-by-update-db-are-missing-from-queries) |
|
|
119
120
|
|
|
120
121
|
---
|
|
121
122
|
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.4.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 4,
|
|
21
|
+
__version__ = version = '0.4.2'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 4, 2)
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
|
@@ -3,6 +3,7 @@ import os
|
|
|
3
3
|
import warnings
|
|
4
4
|
import duckdb
|
|
5
5
|
|
|
6
|
+
from . import storage
|
|
6
7
|
from .bitmaps import deserialize
|
|
7
8
|
from .constants import normalize_chrom
|
|
8
9
|
from .models import AfqueryWarning, SampleFilter
|
|
@@ -48,12 +49,12 @@ def _compute_chunk_annotations(
|
|
|
48
49
|
n_bitmap_cols = 5 if engine._has_coverage_data else 3
|
|
49
50
|
variant_data: dict[tuple[int, str, str], tuple] = {}
|
|
50
51
|
_db = Path(db_path)
|
|
51
|
-
bucket_start = bucket_id *
|
|
52
|
-
bucket_end = (bucket_id + 1) *
|
|
52
|
+
bucket_start = bucket_id * storage.BUCKET_SIZE
|
|
53
|
+
bucket_end = (bucket_id + 1) * storage.BUCKET_SIZE - 1
|
|
53
54
|
cols = ", ".join(engine._bitmap_cols(with_pos=True))
|
|
54
55
|
|
|
55
|
-
if chrom
|
|
56
|
-
parquet_file = _db / "variants"
|
|
56
|
+
if storage.chrom_layout(_db / "variants", chrom) == storage.PARTITIONED:
|
|
57
|
+
parquet_file = storage.bucket_path(_db / "variants", chrom, bucket_id)
|
|
57
58
|
if valid_positions and parquet_file.exists():
|
|
58
59
|
con = duckdb.connect()
|
|
59
60
|
placeholders = ", ".join("?" * len(valid_positions))
|
|
@@ -67,7 +68,7 @@ def _compute_chunk_annotations(
|
|
|
67
68
|
pos, ref, alt = row[0], row[1], row[2]
|
|
68
69
|
variant_data[(pos, ref, alt)] = tuple(bytes(b) for b in row[3:3 + n_bitmap_cols])
|
|
69
70
|
else:
|
|
70
|
-
parquet_file = _db / "variants"
|
|
71
|
+
parquet_file = storage.flat_path(_db / "variants", chrom)
|
|
71
72
|
if valid_positions and parquet_file.exists():
|
|
72
73
|
con = duckdb.connect()
|
|
73
74
|
rows = con.execute(
|
|
@@ -185,7 +186,7 @@ def annotate_vcf(
|
|
|
185
186
|
for variant in vcf:
|
|
186
187
|
|
|
187
188
|
norm = normalize_chrom(variant.CHROM)
|
|
188
|
-
bucket = variant.POS
|
|
189
|
+
bucket = storage.bucket_id(variant.POS)
|
|
189
190
|
key = (norm, bucket)
|
|
190
191
|
if key not in variant_buffers:
|
|
191
192
|
work_order.append(key)
|
|
@@ -6,6 +6,7 @@ from pathlib import Path
|
|
|
6
6
|
|
|
7
7
|
import pyarrow.parquet as pq
|
|
8
8
|
|
|
9
|
+
from . import storage
|
|
9
10
|
from .database import Database
|
|
10
11
|
|
|
11
12
|
|
|
@@ -116,32 +117,18 @@ def _find_test_variants(
|
|
|
116
117
|
return []
|
|
117
118
|
|
|
118
119
|
results: list[tuple[str, int, str, str]] = []
|
|
119
|
-
for entry in
|
|
120
|
+
for entry in storage.iter_variant_parquets(variants_dir):
|
|
120
121
|
if len(results) >= n:
|
|
121
122
|
break
|
|
122
|
-
if entry.
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
))
|
|
132
|
-
elif entry.is_dir():
|
|
133
|
-
chrom = entry.name
|
|
134
|
-
for bucket in sorted(entry.glob("bucket_*.parquet")):
|
|
135
|
-
if len(results) >= n:
|
|
136
|
-
break
|
|
137
|
-
tbl = pq.read_table(str(bucket), columns=["pos", "ref", "alt"])
|
|
138
|
-
for row in range(min(len(tbl), n - len(results))):
|
|
139
|
-
results.append((
|
|
140
|
-
chrom,
|
|
141
|
-
int(tbl["pos"][row].as_py()),
|
|
142
|
-
str(tbl["ref"][row].as_py()),
|
|
143
|
-
str(tbl["alt"][row].as_py()),
|
|
144
|
-
))
|
|
123
|
+
chrom = entry.stem if entry.parent == variants_dir else entry.parent.name
|
|
124
|
+
tbl = pq.read_table(str(entry), columns=["pos", "ref", "alt"])
|
|
125
|
+
for row in range(min(len(tbl), n - len(results))):
|
|
126
|
+
results.append((
|
|
127
|
+
chrom,
|
|
128
|
+
int(tbl["pos"][row].as_py()),
|
|
129
|
+
str(tbl["ref"][row].as_py()),
|
|
130
|
+
str(tbl["alt"][row].as_py()),
|
|
131
|
+
))
|
|
145
132
|
return results
|
|
146
133
|
|
|
147
134
|
|
|
@@ -4,6 +4,8 @@ import sys
|
|
|
4
4
|
|
|
5
5
|
import click
|
|
6
6
|
|
|
7
|
+
from afquery import __version__
|
|
8
|
+
|
|
7
9
|
from .database import Database
|
|
8
10
|
|
|
9
11
|
|
|
@@ -168,15 +170,18 @@ def _print_carriers(carriers, variant_key, fmt: str) -> None:
|
|
|
168
170
|
click.echo(fmt_row.format(*row))
|
|
169
171
|
|
|
170
172
|
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
"""AFQuery: bitmap-indexed allele frequency engine for local genomic cohorts.
|
|
173
|
+
_CLI_HELP = f"""\
|
|
174
|
+
AFQuery v{__version__} — bitmap-indexed allele frequency engine for local genomic cohorts.
|
|
174
175
|
|
|
175
|
-
|
|
176
|
-
|
|
176
|
+
Enables fast AC/AN/AF queries on user-defined subcohorts (phenotype, sex,
|
|
177
|
+
technology) without rescanning VCFs.
|
|
178
|
+
"""
|
|
177
179
|
|
|
178
|
-
|
|
179
|
-
|
|
180
|
+
|
|
181
|
+
@click.group(help=_CLI_HELP)
|
|
182
|
+
@click.version_option(version=__version__, prog_name="afquery", message="%(prog)s %(version)s")
|
|
183
|
+
def cli():
|
|
184
|
+
pass
|
|
180
185
|
|
|
181
186
|
|
|
182
187
|
@cli.command()
|
|
@@ -7,6 +7,7 @@ from pathlib import Path
|
|
|
7
7
|
|
|
8
8
|
import duckdb
|
|
9
9
|
|
|
10
|
+
from . import storage
|
|
10
11
|
from .bitmaps import deserialize
|
|
11
12
|
from .constants import normalize_chrom, ALL_CHROMS
|
|
12
13
|
from .models import SampleFilter
|
|
@@ -14,7 +15,7 @@ from .ploidy import split_ploidy
|
|
|
14
15
|
|
|
15
16
|
logger = logging.getLogger(__name__)
|
|
16
17
|
|
|
17
|
-
BUCKET_SIZE =
|
|
18
|
+
BUCKET_SIZE = storage.BUCKET_SIZE
|
|
18
19
|
|
|
19
20
|
|
|
20
21
|
def _build_groups(engine, base_sf, by_sex, by_tech, by_phenotype, all_groups):
|
|
@@ -136,8 +137,8 @@ def _dump_bucket_worker(
|
|
|
136
137
|
bucket_end = (bucket_id + 1) * BUCKET_SIZE - 1
|
|
137
138
|
|
|
138
139
|
# Resolve parquet path and WHERE clause
|
|
139
|
-
if chrom
|
|
140
|
-
parquet_file = _db / "variants"
|
|
140
|
+
if storage.chrom_layout(_db / "variants", chrom) == storage.PARTITIONED:
|
|
141
|
+
parquet_file = storage.bucket_path(_db / "variants", chrom, bucket_id)
|
|
141
142
|
if not parquet_file.exists():
|
|
142
143
|
return []
|
|
143
144
|
where_parts = []
|
|
@@ -150,7 +151,7 @@ def _dump_bucket_worker(
|
|
|
150
151
|
params.append(pos_end)
|
|
151
152
|
where_clause = ("WHERE " + " AND ".join(where_parts)) if where_parts else ""
|
|
152
153
|
else:
|
|
153
|
-
parquet_file = _db / "variants"
|
|
154
|
+
parquet_file = storage.flat_path(_db / "variants", chrom)
|
|
154
155
|
if not parquet_file.exists():
|
|
155
156
|
return []
|
|
156
157
|
range_start = max(bucket_start, pos_start) if pos_start is not None else bucket_start
|
|
@@ -326,23 +327,15 @@ def dump_database(
|
|
|
326
327
|
# All chroms that have data
|
|
327
328
|
available = set()
|
|
328
329
|
for chrom in ALL_CHROMS:
|
|
329
|
-
if chrom
|
|
330
|
-
available.add(chrom)
|
|
331
|
-
elif (variants_dir / f"{chrom}.parquet").exists():
|
|
330
|
+
if storage.variant_parquet_glob(variants_dir, chrom) is not None:
|
|
332
331
|
available.add(chrom)
|
|
333
332
|
chroms = [c for c in ALL_CHROMS if c in available]
|
|
334
333
|
|
|
335
334
|
# Build work units: (chrom, bucket_id) in genomic order
|
|
336
335
|
work_units: list[tuple[str, int]] = []
|
|
337
336
|
for chrom in chroms:
|
|
338
|
-
if chrom
|
|
339
|
-
|
|
340
|
-
bucket_files = sorted(
|
|
341
|
-
chrom_dir.glob("bucket_*.parquet"),
|
|
342
|
-
key=lambda p: int(p.stem.split("_")[1]),
|
|
343
|
-
)
|
|
344
|
-
for bf in bucket_files:
|
|
345
|
-
bid = int(bf.stem.split("_")[1])
|
|
337
|
+
if storage.chrom_layout(variants_dir, chrom) == storage.PARTITIONED:
|
|
338
|
+
for bid in storage.existing_bucket_ids(variants_dir, chrom):
|
|
346
339
|
# Filter by region if specified
|
|
347
340
|
if pos_start is not None and (bid + 1) * BUCKET_SIZE - 1 < pos_start:
|
|
348
341
|
continue
|
|
@@ -350,7 +343,7 @@ def dump_database(
|
|
|
350
343
|
continue
|
|
351
344
|
work_units.append((chrom, bid))
|
|
352
345
|
else:
|
|
353
|
-
flat_path = variants_dir
|
|
346
|
+
flat_path = storage.flat_path(variants_dir, chrom)
|
|
354
347
|
if not flat_path.exists():
|
|
355
348
|
continue
|
|
356
349
|
bucket_ids = _discover_flat_buckets(flat_path, pos_start, pos_end)
|
|
@@ -14,12 +14,13 @@ import pyarrow as pa
|
|
|
14
14
|
import pyarrow.parquet as pq
|
|
15
15
|
from pyroaring import BitMap
|
|
16
16
|
|
|
17
|
+
from .. import storage
|
|
17
18
|
from ..bitmaps import serialize
|
|
18
19
|
from ..constants import ALL_CHROMS
|
|
19
20
|
|
|
20
21
|
logger = logging.getLogger(__name__)
|
|
21
22
|
|
|
22
|
-
BUCKET_SIZE =
|
|
23
|
+
BUCKET_SIZE = storage.BUCKET_SIZE
|
|
23
24
|
|
|
24
25
|
PARQUET_SCHEMA = pa.schema([
|
|
25
26
|
("pos", pa.uint32()),
|
|
@@ -685,11 +686,9 @@ def build_all_parquets(
|
|
|
685
686
|
for chrom in valid_chroms:
|
|
686
687
|
if resume:
|
|
687
688
|
if partitioned:
|
|
688
|
-
|
|
689
|
-
done = (os.path.isdir(chrom_dir) and
|
|
690
|
-
bool(glob_module.glob(os.path.join(chrom_dir, "bucket_*.parquet"))))
|
|
689
|
+
done = bool(storage.existing_bucket_ids(variants_dir, chrom))
|
|
691
690
|
else:
|
|
692
|
-
done =
|
|
691
|
+
done = storage.flat_path(variants_dir, chrom).exists()
|
|
693
692
|
if done:
|
|
694
693
|
skipped_chroms.append(chrom)
|
|
695
694
|
continue
|
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
import glob as glob_module
|
|
2
1
|
import json
|
|
3
2
|
import logging
|
|
4
3
|
import os
|
|
@@ -11,6 +10,7 @@ import pyarrow as pa
|
|
|
11
10
|
import pyarrow.parquet as pq
|
|
12
11
|
from pyroaring import BitMap
|
|
13
12
|
|
|
13
|
+
from .. import storage
|
|
14
14
|
from ..bitmaps import deserialize, serialize
|
|
15
15
|
from .build import PARQUET_SCHEMA
|
|
16
16
|
|
|
@@ -43,13 +43,7 @@ def compact_database(db_path: Path) -> dict:
|
|
|
43
43
|
active_ids = BitMap([r[0] for r in rows])
|
|
44
44
|
|
|
45
45
|
# Collect all parquet files (flat + partitioned buckets)
|
|
46
|
-
all_parquets: list[Path] =
|
|
47
|
-
for f in sorted(variants_dir.glob("*.parquet")):
|
|
48
|
-
all_parquets.append(f)
|
|
49
|
-
for chrom_dir in sorted(variants_dir.iterdir()):
|
|
50
|
-
if chrom_dir.is_dir():
|
|
51
|
-
for f in sorted(chrom_dir.glob("bucket_*.parquet")):
|
|
52
|
-
all_parquets.append(f)
|
|
46
|
+
all_parquets: list[Path] = list(storage.iter_variant_parquets(variants_dir))
|
|
53
47
|
|
|
54
48
|
logger.info("[compact] Compacting %d Parquet file(s) against %d active sample(s)...",
|
|
55
49
|
len(all_parquets), len(active_ids))
|
|
@@ -115,8 +109,11 @@ def compact_database(db_path: Path) -> dict:
|
|
|
115
109
|
logger.debug(" [compact] %s: no changes", parquet_file.name)
|
|
116
110
|
continue
|
|
117
111
|
|
|
118
|
-
# Build new table with kept rows and updated bitmaps
|
|
119
|
-
|
|
112
|
+
# Build new table with kept rows and updated bitmaps.
|
|
113
|
+
# The index type is spelled out: a bare [] makes pyarrow infer a null
|
|
114
|
+
# array, which has no take kernel, and every row of a file can legitimately
|
|
115
|
+
# be dropped when the removed samples were the only carriers in it.
|
|
116
|
+
orig_keep = table.take(pa.array(keep_indices, type=pa.int64()))
|
|
120
117
|
new_table = pa.table(
|
|
121
118
|
{
|
|
122
119
|
"pos": orig_keep["pos"],
|