afquery 0.4.0__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (168) hide show
  1. {afquery-0.4.0 → afquery-0.4.2}/PKG-INFO +2 -2
  2. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/ploidy-and-sex-chroms.md +2 -0
  3. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/update-database.md +11 -3
  4. {afquery-0.4.0 → afquery-0.4.2}/docs/reference/cli.md +13 -0
  5. {afquery-0.4.0 → afquery-0.4.2}/docs/reference/data-model.md +16 -0
  6. {afquery-0.4.0 → afquery-0.4.2}/docs/troubleshooting.md +1 -0
  7. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/_version.py +2 -2
  8. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/annotate.py +7 -6
  9. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/benchmark.py +11 -24
  10. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/cli.py +12 -7
  11. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/dump.py +9 -16
  12. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/build.py +4 -5
  13. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/compact.py +7 -10
  14. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/update.py +202 -118
  15. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/query.py +19 -24
  16. afquery-0.4.2/src/afquery/storage.py +177 -0
  17. {afquery-0.4.0 → afquery-0.4.2}/tests/data/expected_results.json +2 -2
  18. afquery-0.4.2/tests/oracle.py +195 -0
  19. {afquery-0.4.0 → afquery-0.4.2}/tests/test_cli.py +23 -0
  20. {afquery-0.4.0 → afquery-0.4.2}/tests/test_haploid_stats.py +30 -0
  21. afquery-0.4.2/tests/test_invariants.py +186 -0
  22. afquery-0.4.2/tests/test_oracle_consistency.py +164 -0
  23. afquery-0.4.2/tests/test_storage.py +252 -0
  24. {afquery-0.4.0 → afquery-0.4.2}/tests/test_update.py +54 -0
  25. afquery-0.4.2/tests/test_update_partitioned.py +355 -0
  26. {afquery-0.4.0 → afquery-0.4.2}/.dockerignore +0 -0
  27. {afquery-0.4.0 → afquery-0.4.2}/.github/workflows/ci.yml +0 -0
  28. {afquery-0.4.0 → afquery-0.4.2}/.github/workflows/docs.yml +0 -0
  29. {afquery-0.4.0 → afquery-0.4.2}/.github/workflows/release.yml +0 -0
  30. {afquery-0.4.0 → afquery-0.4.2}/.gitignore +0 -0
  31. {afquery-0.4.0 → afquery-0.4.2}/CONTRIBUTING.md +0 -0
  32. {afquery-0.4.0 → afquery-0.4.2}/Dockerfile +0 -0
  33. {afquery-0.4.0 → afquery-0.4.2}/LICENSE +0 -0
  34. {afquery-0.4.0 → afquery-0.4.2}/README.md +0 -0
  35. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/.gitignore +0 -0
  36. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/INSTALL.md +0 -0
  37. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/README.md +0 -0
  38. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/Snakefile +0 -0
  39. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/.gitignore +0 -0
  40. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/01a_assign_samples.py +0 -0
  41. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/01b_write_manifests.py +0 -0
  42. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/02_build_databases.py +0 -0
  43. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/03_compute_metrics.py +0 -0
  44. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/04_classify_acmg.py +0 -0
  45. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/05_plot_figures.py +0 -0
  46. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/README.md +0 -0
  47. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/Snakefile +0 -0
  48. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/__init__.py +0 -0
  49. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/beds/SureSelect_v5.bed +0 -0
  50. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/beds/SureSelect_v6.bed +0 -0
  51. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/beds/SureSelect_v7.bed +0 -0
  52. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/capture_kit/config.py +0 -0
  53. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/config.yaml +0 -0
  54. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/envs/benchmark.yaml +0 -0
  55. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/.gitignore +0 -0
  56. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/01_prepare_data.py +0 -0
  57. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/02_query_scaling.py +0 -0
  58. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/03_build.py +0 -0
  59. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/04_annotate.py +0 -0
  60. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/05_vs_bcftools.py +0 -0
  61. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/06_plot.py +0 -0
  62. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/README.md +0 -0
  63. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/Snakefile +0 -0
  64. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_annotate.py +0 -0
  65. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_bcftools.py +0 -0
  66. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_build_perf.py +0 -0
  67. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_prepare.py +0 -0
  68. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/collect_query_scaling.py +0 -0
  69. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/config.py +0 -0
  70. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/performance/config_smoke.py +0 -0
  71. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/__init__.py +0 -0
  72. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/config.py +0 -0
  73. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/rules/download_1kg.smk +0 -0
  74. {afquery-0.4.0 → afquery-0.4.2}/benchmarks/shared/utils.py +0 -0
  75. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/benchmarking.md +0 -0
  76. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/coverage-evidence.md +0 -0
  77. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/debugging-results.md +0 -0
  78. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/filter-pass-tracking.md +0 -0
  79. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/multi-cohort-strategies.md +0 -0
  80. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/performance.md +0 -0
  81. {afquery-0.4.0 → afquery-0.4.2}/docs/advanced/pipeline-integration.md +0 -0
  82. {afquery-0.4.0 → afquery-0.4.2}/docs/assets/img/gap2_mixed_technologies.png +0 -0
  83. {afquery-0.4.0 → afquery-0.4.2}/docs/faq.md +0 -0
  84. {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/concepts.md +0 -0
  85. {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/installation.md +0 -0
  86. {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/motivation.md +0 -0
  87. {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/preprocessing.md +0 -0
  88. {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/quickstart.md +0 -0
  89. {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/tutorial.md +0 -0
  90. {afquery-0.4.0 → afquery-0.4.2}/docs/getting-started/understanding-output.md +0 -0
  91. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/annotate-vcf.md +0 -0
  92. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/create-database.md +0 -0
  93. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/dump-export.md +0 -0
  94. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/manifest-format.md +0 -0
  95. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/query.md +0 -0
  96. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/sample-filtering.md +0 -0
  97. {afquery-0.4.0 → afquery-0.4.2}/docs/guides/variant-info.md +0 -0
  98. {afquery-0.4.0 → afquery-0.4.2}/docs/index.md +0 -0
  99. {afquery-0.4.0 → afquery-0.4.2}/docs/reference/glossary.md +0 -0
  100. {afquery-0.4.0 → afquery-0.4.2}/docs/reference/python-api.md +0 -0
  101. {afquery-0.4.0 → afquery-0.4.2}/docs/scripts/gen_gap2_figure.py +0 -0
  102. {afquery-0.4.0 → afquery-0.4.2}/docs/stylesheets/extra.css +0 -0
  103. {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/acmg-use-cases.md +0 -0
  104. {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/clinical-prioritization.md +0 -0
  105. {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/cohort-stratification.md +0 -0
  106. {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/population-specific-af.md +0 -0
  107. {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/pseudo-controls.md +0 -0
  108. {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/sex-specific-af.md +0 -0
  109. {afquery-0.4.0 → afquery-0.4.2}/docs/use-cases/technology-integration.md +0 -0
  110. {afquery-0.4.0 → afquery-0.4.2}/examples/demo/README.md +0 -0
  111. {afquery-0.4.0 → afquery-0.4.2}/examples/demo/create_demo_data.py +0 -0
  112. {afquery-0.4.0 → afquery-0.4.2}/mkdocs.yml +0 -0
  113. {afquery-0.4.0 → afquery-0.4.2}/pyproject.toml +0 -0
  114. {afquery-0.4.0 → afquery-0.4.2}/recipes/afquery/meta.yaml +0 -0
  115. {afquery-0.4.0 → afquery-0.4.2}/resources/normalize_vcf.sh +0 -0
  116. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/__init__.py +0 -0
  117. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/bitmaps.py +0 -0
  118. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/capture.py +0 -0
  119. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/cli.pyr +0 -0
  120. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/constants.py +0 -0
  121. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/database.py +0 -0
  122. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/models.py +0 -0
  123. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/ploidy.py +0 -0
  124. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/__init__.py +0 -0
  125. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/ingest.py +0 -0
  126. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/manifest.py +0 -0
  127. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/regions.py +0 -0
  128. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/preprocess/synth.py +0 -0
  129. {afquery-0.4.0 → afquery-0.4.2}/src/afquery/variant_info.py +0 -0
  130. {afquery-0.4.0 → afquery-0.4.2}/tests/conftest.py +0 -0
  131. {afquery-0.4.0 → afquery-0.4.2}/tests/data/annotate_input.vcf +0 -0
  132. {afquery-0.4.0 → afquery-0.4.2}/tests/data/annotate_multi_bucket.vcf +0 -0
  133. {afquery-0.4.0 → afquery-0.4.2}/tests/data/annotate_multi_chrom.vcf +0 -0
  134. {afquery-0.4.0 → afquery-0.4.2}/tests/data/beds/wes_kit_a.bed +0 -0
  135. {afquery-0.4.0 → afquery-0.4.2}/tests/data/beds/wes_kit_b.bed +0 -0
  136. {afquery-0.4.0 → afquery-0.4.2}/tests/data/beds/wes_kit_nochr.bed +0 -0
  137. {afquery-0.4.0 → afquery-0.4.2}/tests/data/manifest.tsv +0 -0
  138. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S00.vcf +0 -0
  139. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S01.vcf +0 -0
  140. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S02.vcf +0 -0
  141. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S03.vcf +0 -0
  142. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S04.vcf +0 -0
  143. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S05.vcf +0 -0
  144. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S06.vcf +0 -0
  145. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S07.vcf +0 -0
  146. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S08.vcf +0 -0
  147. {afquery-0.4.0 → afquery-0.4.2}/tests/data/vcfs/S09.vcf +0 -0
  148. {afquery-0.4.0 → afquery-0.4.2}/tests/test_annotate.py +0 -0
  149. {afquery-0.4.0 → afquery-0.4.2}/tests/test_batch.py +0 -0
  150. {afquery-0.4.0 → afquery-0.4.2}/tests/test_benchmark.py +0 -0
  151. {afquery-0.4.0 → afquery-0.4.2}/tests/test_bitmaps.py +0 -0
  152. {afquery-0.4.0 → afquery-0.4.2}/tests/test_capture.py +0 -0
  153. {afquery-0.4.0 → afquery-0.4.2}/tests/test_cli_docs_consistency.py +0 -0
  154. {afquery-0.4.0 → afquery-0.4.2}/tests/test_compact.py +0 -0
  155. {afquery-0.4.0 → afquery-0.4.2}/tests/test_constants.py +0 -0
  156. {afquery-0.4.0 → afquery-0.4.2}/tests/test_dump.py +0 -0
  157. {afquery-0.4.0 → afquery-0.4.2}/tests/test_info.py +0 -0
  158. {afquery-0.4.0 → afquery-0.4.2}/tests/test_no_coverage.py +0 -0
  159. {afquery-0.4.0 → afquery-0.4.2}/tests/test_pass_filter.py +0 -0
  160. {afquery-0.4.0 → afquery-0.4.2}/tests/test_ploidy.py +0 -0
  161. {afquery-0.4.0 → afquery-0.4.2}/tests/test_preprocess.py +0 -0
  162. {afquery-0.4.0 → afquery-0.4.2}/tests/test_query.py +0 -0
  163. {afquery-0.4.0 → afquery-0.4.2}/tests/test_sample_filter.py +0 -0
  164. {afquery-0.4.0 → afquery-0.4.2}/tests/test_synth.py +0 -0
  165. {afquery-0.4.0 → afquery-0.4.2}/tests/test_synthetic_stats.py +0 -0
  166. {afquery-0.4.0 → afquery-0.4.2}/tests/test_update_metadata.py +0 -0
  167. {afquery-0.4.0 → afquery-0.4.2}/tests/test_variant_info.py +0 -0
  168. {afquery-0.4.0 → afquery-0.4.2}/tests/test_warnings.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: afquery
3
- Version: 0.4.0
3
+ Version: 0.4.2
4
4
  Summary: Genomic allele frequency query engine with bitmap-encoded genotypes
5
5
  License: MIT
6
6
  License-File: LICENSE
@@ -19,6 +19,8 @@ AFQuery computes ploidy-aware AN for sex chromosomes (chrX, chrY) and the mitoch
19
19
 
20
20
  For each eligible sample at a given position, AFQuery adds the appropriate ploidy count to AN based on the sample's sex and the chromosome/position.
21
21
 
22
+ A sample contributing ploidy 0 is not eligible at that position at all: a female has no chrY to genotype, so she is neither a carrier nor homozygous reference there. On chrY, `n_eligible`, `N_HOM_REF` and `AN` therefore count males only.
23
+
22
24
  ---
23
25
 
24
26
  ## Pseudoautosomal Regions (PAR)
@@ -38,6 +38,8 @@ afquery update-db \
38
38
 
39
39
  The new manifest follows the same format as the original (see [Manifest Format](manifest-format.md)). New samples are assigned monotonically increasing sample IDs.
40
40
 
41
+ New variants are merged into the per-chromosome bucket files the database already uses; buckets are created on demand, and no new top-level Parquet file is produced. See [Data Model](../reference/data-model.md#storage-layouts).
42
+
41
43
  To add multiple manifests at once:
42
44
 
43
45
  ```bash
@@ -67,9 +69,15 @@ decisions are comparable across batches.
67
69
  When new carriers push a partially-covered tech above the `--min-covered`
68
70
  threshold at positions that were previously below it, those positions are
69
71
  re-evaluated and their non-carrier samples once again count as `N_HOM_REF`
70
- instead of `N_NO_COVERAGE`. The recomputation runs only for chromosomes
71
- touched by the new samples; existing rows on other chromosomes are not
72
- rewritten.
72
+ instead of `N_NO_COVERAGE`. Because that value derives from the tech bitmaps of
73
+ the whole cohort rather than from which file received rows, the recomputation
74
+ spans **every bucket of every chromosome the database holds**, not only the ones
75
+ the new samples carry variants on. Skipping the rest would leave an added WES
76
+ sample counted as `N_HOM_REF` everywhere else its capture BED reaches. Only a
77
+ batch that puts a sample into a capture-based technology can move those bitmaps,
78
+ so a WGS-only batch still visits nothing beyond its own chromosomes; adding a
79
+ WES sample to a large database rewrites broadly and takes minutes rather than
80
+ seconds.
73
81
 
74
82
  VCFs added via `update-db` should preserve `FORMAT/DP` and `FORMAT/GQ` (the
75
83
  bundled `resources/normalize_vcf.sh` does so by default). Samples without
@@ -4,6 +4,19 @@ All AFQuery commands follow the pattern `afquery <command> [OPTIONS]`.
4
4
 
5
5
  ---
6
6
 
7
+ ## Global options
8
+
9
+ | Option | Description |
10
+ |--------|-------------|
11
+ | `--version` | Print the AFQuery version and exit |
12
+ | `--help` | Show help for the command and exit |
13
+
14
+ `afquery --version` prints the installed package version (e.g. `afquery 0.4.2`),
15
+ taken from the Git release tag at build time. In an editable install
16
+ (`pip install -e .`) it reflects the last build, not later local commits.
17
+
18
+ ---
19
+
7
20
  ## create-db
8
21
 
9
22
  Build a new AFQuery database from a manifest of single-sample VCFs.
@@ -22,6 +22,18 @@ This page documents the on-disk layout of an AFQuery database, including file fo
22
22
  └── wes_v2.pkl
23
23
  ```
24
24
 
25
+ ### Storage layouts
26
+
27
+ `create-db` always writes the bucketed layout shown above. A single-file-per-chromosome
28
+ layout (`variants/chr1.parquet`, with no bucket directory) is also readable, and appears
29
+ in small hand-built databases and in test fixtures. `update-db` merges into whichever
30
+ layout a chromosome already uses, and creates buckets for a chromosome new to the
31
+ database.
32
+
33
+ A chromosome must never have both. `afquery check` reports that as an error, because
34
+ queries resolve the bucket directory first and would silently ignore the flat file — and
35
+ every sample stored only in it.
36
+
25
37
  ---
26
38
 
27
39
  ## manifest.json
@@ -156,6 +168,10 @@ Variants are partitioned into 1-Mbp buckets:
156
168
  bucket_id = pos // 1_000_000
157
169
  ```
158
170
 
171
+ `update-db --add-samples` writes new positions into the bucket that owns
172
+ them, creating `bucket_N.parquet` when a batch extends a chromosome past its
173
+ previous last bucket.
174
+
159
175
  !!! warning "DuckDB integer arithmetic"
160
176
  When computing bucket IDs in DuckDB SQL, always use the integer-division operator:
161
177
  ```sql
@@ -116,6 +116,7 @@ afquery info --db ./db/ --samples | grep SAMP
116
116
  | `Missing Parquet for chromosome chr3` | Re-run `create-db` or investigate incomplete build |
117
117
  | `Manifest mismatch: expected N samples, found M` | Database may be partially updated; re-run `update-db` |
118
118
  | `Capture file missing for wes_v1` | BED file was not provided at build time; rebuild with `--bed-dir` |
119
+ | `chr1: both variants/chr1/ ... and variants/chr1.parquet exist` | Samples added by a pre-0.4.1 `update-db` are invisible to queries; see [Samples Added by update-db Are Missing From Queries](#samples-added-by-update-db-are-missing-from-queries) |
119
120
 
120
121
  ---
121
122
 
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.4.0'
22
- __version_tuple__ = version_tuple = (0, 4, 0)
21
+ __version__ = version = '0.4.2'
22
+ __version_tuple__ = version_tuple = (0, 4, 2)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -3,6 +3,7 @@ import os
3
3
  import warnings
4
4
  import duckdb
5
5
 
6
+ from . import storage
6
7
  from .bitmaps import deserialize
7
8
  from .constants import normalize_chrom
8
9
  from .models import AfqueryWarning, SampleFilter
@@ -48,12 +49,12 @@ def _compute_chunk_annotations(
48
49
  n_bitmap_cols = 5 if engine._has_coverage_data else 3
49
50
  variant_data: dict[tuple[int, str, str], tuple] = {}
50
51
  _db = Path(db_path)
51
- bucket_start = bucket_id * 1_000_000
52
- bucket_end = (bucket_id + 1) * 1_000_000 - 1
52
+ bucket_start = bucket_id * storage.BUCKET_SIZE
53
+ bucket_end = (bucket_id + 1) * storage.BUCKET_SIZE - 1
53
54
  cols = ", ".join(engine._bitmap_cols(with_pos=True))
54
55
 
55
- if chrom in engine._partitioned_chroms:
56
- parquet_file = _db / "variants" / chrom / f"bucket_{bucket_id}.parquet"
56
+ if storage.chrom_layout(_db / "variants", chrom) == storage.PARTITIONED:
57
+ parquet_file = storage.bucket_path(_db / "variants", chrom, bucket_id)
57
58
  if valid_positions and parquet_file.exists():
58
59
  con = duckdb.connect()
59
60
  placeholders = ", ".join("?" * len(valid_positions))
@@ -67,7 +68,7 @@ def _compute_chunk_annotations(
67
68
  pos, ref, alt = row[0], row[1], row[2]
68
69
  variant_data[(pos, ref, alt)] = tuple(bytes(b) for b in row[3:3 + n_bitmap_cols])
69
70
  else:
70
- parquet_file = _db / "variants" / f"{chrom}.parquet"
71
+ parquet_file = storage.flat_path(_db / "variants", chrom)
71
72
  if valid_positions and parquet_file.exists():
72
73
  con = duckdb.connect()
73
74
  rows = con.execute(
@@ -185,7 +186,7 @@ def annotate_vcf(
185
186
  for variant in vcf:
186
187
 
187
188
  norm = normalize_chrom(variant.CHROM)
188
- bucket = variant.POS // 1_000_000
189
+ bucket = storage.bucket_id(variant.POS)
189
190
  key = (norm, bucket)
190
191
  if key not in variant_buffers:
191
192
  work_order.append(key)
@@ -6,6 +6,7 @@ from pathlib import Path
6
6
 
7
7
  import pyarrow.parquet as pq
8
8
 
9
+ from . import storage
9
10
  from .database import Database
10
11
 
11
12
 
@@ -116,32 +117,18 @@ def _find_test_variants(
116
117
  return []
117
118
 
118
119
  results: list[tuple[str, int, str, str]] = []
119
- for entry in sorted(variants_dir.iterdir()):
120
+ for entry in storage.iter_variant_parquets(variants_dir):
120
121
  if len(results) >= n:
121
122
  break
122
- if entry.suffix == ".parquet":
123
- chrom = entry.stem
124
- tbl = pq.read_table(str(entry), columns=["pos", "ref", "alt"])
125
- for row in range(min(len(tbl), n - len(results))):
126
- results.append((
127
- chrom,
128
- int(tbl["pos"][row].as_py()),
129
- str(tbl["ref"][row].as_py()),
130
- str(tbl["alt"][row].as_py()),
131
- ))
132
- elif entry.is_dir():
133
- chrom = entry.name
134
- for bucket in sorted(entry.glob("bucket_*.parquet")):
135
- if len(results) >= n:
136
- break
137
- tbl = pq.read_table(str(bucket), columns=["pos", "ref", "alt"])
138
- for row in range(min(len(tbl), n - len(results))):
139
- results.append((
140
- chrom,
141
- int(tbl["pos"][row].as_py()),
142
- str(tbl["ref"][row].as_py()),
143
- str(tbl["alt"][row].as_py()),
144
- ))
123
+ chrom = entry.stem if entry.parent == variants_dir else entry.parent.name
124
+ tbl = pq.read_table(str(entry), columns=["pos", "ref", "alt"])
125
+ for row in range(min(len(tbl), n - len(results))):
126
+ results.append((
127
+ chrom,
128
+ int(tbl["pos"][row].as_py()),
129
+ str(tbl["ref"][row].as_py()),
130
+ str(tbl["alt"][row].as_py()),
131
+ ))
145
132
  return results
146
133
 
147
134
 
@@ -4,6 +4,8 @@ import sys
4
4
 
5
5
  import click
6
6
 
7
+ from afquery import __version__
8
+
7
9
  from .database import Database
8
10
 
9
11
 
@@ -168,15 +170,18 @@ def _print_carriers(carriers, variant_key, fmt: str) -> None:
168
170
  click.echo(fmt_row.format(*row))
169
171
 
170
172
 
171
- @click.group()
172
- def cli():
173
- """AFQuery: bitmap-indexed allele frequency engine for local genomic cohorts.
173
+ _CLI_HELP = f"""\
174
+ AFQuery v{__version__} — bitmap-indexed allele frequency engine for local genomic cohorts.
174
175
 
175
- Enables fast AC/AN/AF queries on user-defined subcohorts (phenotype, sex,
176
- technology) without rescanning VCFs.
176
+ Enables fast AC/AN/AF queries on user-defined subcohorts (phenotype, sex,
177
+ technology) without rescanning VCFs.
178
+ """
177
179
 
178
- Commands: query, variant-info, annotate, dump, info, version, create-db, update-db, check, benchmark
179
- """
180
+
181
+ @click.group(help=_CLI_HELP)
182
+ @click.version_option(version=__version__, prog_name="afquery", message="%(prog)s %(version)s")
183
+ def cli():
184
+ pass
180
185
 
181
186
 
182
187
  @cli.command()
@@ -7,6 +7,7 @@ from pathlib import Path
7
7
 
8
8
  import duckdb
9
9
 
10
+ from . import storage
10
11
  from .bitmaps import deserialize
11
12
  from .constants import normalize_chrom, ALL_CHROMS
12
13
  from .models import SampleFilter
@@ -14,7 +15,7 @@ from .ploidy import split_ploidy
14
15
 
15
16
  logger = logging.getLogger(__name__)
16
17
 
17
- BUCKET_SIZE = 1_000_000
18
+ BUCKET_SIZE = storage.BUCKET_SIZE
18
19
 
19
20
 
20
21
  def _build_groups(engine, base_sf, by_sex, by_tech, by_phenotype, all_groups):
@@ -136,8 +137,8 @@ def _dump_bucket_worker(
136
137
  bucket_end = (bucket_id + 1) * BUCKET_SIZE - 1
137
138
 
138
139
  # Resolve parquet path and WHERE clause
139
- if chrom in engine._partitioned_chroms:
140
- parquet_file = _db / "variants" / chrom / f"bucket_{bucket_id}.parquet"
140
+ if storage.chrom_layout(_db / "variants", chrom) == storage.PARTITIONED:
141
+ parquet_file = storage.bucket_path(_db / "variants", chrom, bucket_id)
141
142
  if not parquet_file.exists():
142
143
  return []
143
144
  where_parts = []
@@ -150,7 +151,7 @@ def _dump_bucket_worker(
150
151
  params.append(pos_end)
151
152
  where_clause = ("WHERE " + " AND ".join(where_parts)) if where_parts else ""
152
153
  else:
153
- parquet_file = _db / "variants" / f"{chrom}.parquet"
154
+ parquet_file = storage.flat_path(_db / "variants", chrom)
154
155
  if not parquet_file.exists():
155
156
  return []
156
157
  range_start = max(bucket_start, pos_start) if pos_start is not None else bucket_start
@@ -326,23 +327,15 @@ def dump_database(
326
327
  # All chroms that have data
327
328
  available = set()
328
329
  for chrom in ALL_CHROMS:
329
- if chrom in engine._partitioned_chroms:
330
- available.add(chrom)
331
- elif (variants_dir / f"{chrom}.parquet").exists():
330
+ if storage.variant_parquet_glob(variants_dir, chrom) is not None:
332
331
  available.add(chrom)
333
332
  chroms = [c for c in ALL_CHROMS if c in available]
334
333
 
335
334
  # Build work units: (chrom, bucket_id) in genomic order
336
335
  work_units: list[tuple[str, int]] = []
337
336
  for chrom in chroms:
338
- if chrom in engine._partitioned_chroms:
339
- chrom_dir = variants_dir / chrom
340
- bucket_files = sorted(
341
- chrom_dir.glob("bucket_*.parquet"),
342
- key=lambda p: int(p.stem.split("_")[1]),
343
- )
344
- for bf in bucket_files:
345
- bid = int(bf.stem.split("_")[1])
337
+ if storage.chrom_layout(variants_dir, chrom) == storage.PARTITIONED:
338
+ for bid in storage.existing_bucket_ids(variants_dir, chrom):
346
339
  # Filter by region if specified
347
340
  if pos_start is not None and (bid + 1) * BUCKET_SIZE - 1 < pos_start:
348
341
  continue
@@ -350,7 +343,7 @@ def dump_database(
350
343
  continue
351
344
  work_units.append((chrom, bid))
352
345
  else:
353
- flat_path = variants_dir / f"{chrom}.parquet"
346
+ flat_path = storage.flat_path(variants_dir, chrom)
354
347
  if not flat_path.exists():
355
348
  continue
356
349
  bucket_ids = _discover_flat_buckets(flat_path, pos_start, pos_end)
@@ -14,12 +14,13 @@ import pyarrow as pa
14
14
  import pyarrow.parquet as pq
15
15
  from pyroaring import BitMap
16
16
 
17
+ from .. import storage
17
18
  from ..bitmaps import serialize
18
19
  from ..constants import ALL_CHROMS
19
20
 
20
21
  logger = logging.getLogger(__name__)
21
22
 
22
- BUCKET_SIZE = 1_000_000
23
+ BUCKET_SIZE = storage.BUCKET_SIZE
23
24
 
24
25
  PARQUET_SCHEMA = pa.schema([
25
26
  ("pos", pa.uint32()),
@@ -685,11 +686,9 @@ def build_all_parquets(
685
686
  for chrom in valid_chroms:
686
687
  if resume:
687
688
  if partitioned:
688
- chrom_dir = os.path.join(variants_dir, chrom)
689
- done = (os.path.isdir(chrom_dir) and
690
- bool(glob_module.glob(os.path.join(chrom_dir, "bucket_*.parquet"))))
689
+ done = bool(storage.existing_bucket_ids(variants_dir, chrom))
691
690
  else:
692
- done = os.path.exists(os.path.join(variants_dir, f"{chrom}.parquet"))
691
+ done = storage.flat_path(variants_dir, chrom).exists()
693
692
  if done:
694
693
  skipped_chroms.append(chrom)
695
694
  continue
@@ -1,4 +1,3 @@
1
- import glob as glob_module
2
1
  import json
3
2
  import logging
4
3
  import os
@@ -11,6 +10,7 @@ import pyarrow as pa
11
10
  import pyarrow.parquet as pq
12
11
  from pyroaring import BitMap
13
12
 
13
+ from .. import storage
14
14
  from ..bitmaps import deserialize, serialize
15
15
  from .build import PARQUET_SCHEMA
16
16
 
@@ -43,13 +43,7 @@ def compact_database(db_path: Path) -> dict:
43
43
  active_ids = BitMap([r[0] for r in rows])
44
44
 
45
45
  # Collect all parquet files (flat + partitioned buckets)
46
- all_parquets: list[Path] = []
47
- for f in sorted(variants_dir.glob("*.parquet")):
48
- all_parquets.append(f)
49
- for chrom_dir in sorted(variants_dir.iterdir()):
50
- if chrom_dir.is_dir():
51
- for f in sorted(chrom_dir.glob("bucket_*.parquet")):
52
- all_parquets.append(f)
46
+ all_parquets: list[Path] = list(storage.iter_variant_parquets(variants_dir))
53
47
 
54
48
  logger.info("[compact] Compacting %d Parquet file(s) against %d active sample(s)...",
55
49
  len(all_parquets), len(active_ids))
@@ -115,8 +109,11 @@ def compact_database(db_path: Path) -> dict:
115
109
  logger.debug(" [compact] %s: no changes", parquet_file.name)
116
110
  continue
117
111
 
118
- # Build new table with kept rows and updated bitmaps
119
- orig_keep = table.take(keep_indices)
112
+ # Build new table with kept rows and updated bitmaps.
113
+ # The index type is spelled out: a bare [] makes pyarrow infer a null
114
+ # array, which has no take kernel, and every row of a file can legitimately
115
+ # be dropped when the removed samples were the only carriers in it.
116
+ orig_keep = table.take(pa.array(keep_indices, type=pa.int64()))
120
117
  new_table = pa.table(
121
118
  {
122
119
  "pos": orig_keep["pos"],