rust-annovar 0.1.0.beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/Cargo.toml ADDED
@@ -0,0 +1,33 @@
1
+ [package]
2
+ name = "rust-annovar"
3
+ version = "0.1.0-beta.1"
4
+ edition = "2024"
5
+ license = "MIT OR Apache-2.0"
6
+ description = "A fast, auditable Rust implementation of core ANNOVAR workflows"
7
+ repository = "https://github.com/ydlongtao/RustAnnovar"
8
+ homepage = "https://github.com/ydlongtao/RustAnnovar"
9
+ readme = "README.md"
10
+ rust-version = "1.85"
11
+ keywords = ["annovar", "bioinformatics", "genomics", "variant-annotation"]
12
+ categories = ["science", "command-line-utilities"]
13
+
14
+ [lib]
15
+ name = "rust_annovar"
16
+ path = "src/lib.rs"
17
+
18
+ [[bin]]
19
+ name = "rust-annovar"
20
+ path = "src/main.rs"
21
+
22
+ [dependencies]
23
+ anyhow = "1"
24
+ clap = { version = "4", features = ["derive"] }
25
+ flate2 = "1"
26
+ rayon = "1"
27
+ serde = { version = "1", features = ["derive"] }
28
+ serde_json = "1"
29
+ sha2 = "0.10"
30
+ ureq = { version = "2", default-features = false, features = ["tls"] }
31
+
32
+ [dev-dependencies]
33
+ tempfile = "3"
data/LICENSE-APACHE ADDED
@@ -0,0 +1,15 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ Licensed under the Apache License, Version 2.0 (the "License");
6
+ you may not use this file except in compliance with the License.
7
+ You may obtain a copy of the License at
8
+
9
+ http://www.apache.org/licenses/LICENSE-2.0
10
+
11
+ Unless required by applicable law or agreed to in writing, software
12
+ distributed under the License is distributed on an "AS IS" BASIS,
13
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
+ See the License for the specific language governing permissions and
15
+ limitations under the License.
data/LICENSE-MIT ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 RustAnnovar contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
data/README.md ADDED
@@ -0,0 +1,298 @@
1
+ # RustAnnovar
2
+
3
+ [![CI](https://github.com/ydlongtao/RustAnnovar/actions/workflows/ci.yml/badge.svg)](https://github.com/ydlongtao/RustAnnovar/actions/workflows/ci.yml)
4
+ [![Open Beta](https://img.shields.io/badge/status-open%20beta-orange)](https://github.com/ydlongtao/RustAnnovar/issues)
5
+ [![Rust](https://img.shields.io/badge/Rust-stable-000000?logo=rust)](https://www.rust-lang.org/)
6
+ [![License](https://img.shields.io/badge/license-MIT%20OR%20Apache--2.0-blue)](LICENSE-MIT)
7
+
8
+ **English** | [简体中文](README.zh-CN.md)
9
+
10
+ **A variant annotation engine written in Rust, with support for ANNOVAR database formats.**
11
+
12
+ RustAnnovar provides a native command-line interface and Rust library for annotating genomic variants. It reads supported existing `humandb` files and combines three core operations: exact allele matching, genomic interval overlap, and transcript consequence calculation. Development prioritizes human hg19/hg38 workflows.
13
+
14
+ ## Important statements
15
+
16
+ > [!WARNING]
17
+ > **RustAnnovar is currently in open beta.** It is intended for evaluation, compatibility testing, and research workflow development. It is not yet a complete replacement for ANNOVAR or a clinically validated tool. Validate results with established tools before relying on them for consequential decisions.
18
+
19
+ - **Compatibility is a goal, not a guarantee.** Selected SNV consequences, generic filter matches, and GFF3 overlap results have been checked against a local ANNOVAR baseline. Complex indels, complete HGVS notation, ncRNA classification, transcript ordering, and specialized database protocols remain incomplete.
20
+ - **Independent implementation.** RustAnnovar is not affiliated with or endorsed by ANNOVAR. The annotation engine does not invoke Perl.
21
+ - **Bring your own databases.** ANNOVAR scripts and registered databases are not distributed here. Obtain databases separately and comply with their licenses. The bundled demonstration data are entirely synthetic.
22
+ - **Performance results have a limited scope.** The measurements below describe specific local workloads; they do not establish equivalent results or a universal speedup on real WES/WGS datasets.
23
+
24
+ ## Features
25
+
26
+ - Read VCF, gzip-compressed VCF, and AVinput; split multiallelic VCF records.
27
+ - Match variants by chromosome, coordinates, reference allele, and alternate allele.
28
+ - Query supported BED-style, UCSC-style, and GFF3 region files.
29
+ - Read refGene-style transcript models and calculate coding SNV consequences with transcript FASTA.
30
+ - Combine databases into TSV or CSV output, or add annotations to VCF INFO while preserving sample columns.
31
+ - Build optional 1 Mb block indexes for plain-text filter databases.
32
+ - Annotate input rows in parallel while retaining input order.
33
+ - Extract reference sequences and select TSV rows by exact field value.
34
+
35
+ ## Runtime comparison
36
+
37
+ Local measurements used an **Apple M1 (8 cores), macOS 26.5, Perl 5.34.1**, the same hg19 refGene database and transcript FASTA, and an optimized Rust release build. Each workload was warmed up before collecting the median wall-clock time over five or seven runs. Timings include process startup, database loading, annotation, and output writing.
38
+
39
+ | Workload | Original ANNOVAR (Perl) | Rust implementation | Speedup |
40
+ |---|---:|---:|---:|
41
+ | 13 SNVs from the bundled ANNOVAR example | 2.40 s | 0.29 s | **8.28×** |
42
+ | 26,000 SNV rows, default settings | 4.68 s | 0.37 s | **12.65×** |
43
+ | 26,000 SNV rows, single thread | 4.70 s | 0.47 s | **10.00×** |
44
+ | 21 variants against 25,688 GFF3 regions | 0.14 s | 0.01 s | **Approximately 14×** |
45
+
46
+ A separate memory measurement on the 26,000-row workload reported maximum resident memory of **400 MiB for Perl** and **360.3 MiB for Rust**, approximately 9.9% lower.
47
+
48
+ **How to interpret these numbers:**
49
+
50
+ - The 26,000-row input repeats 13 SNVs 2,000 times. It is a repeated-record workload, not 26,000 independent variants or a representative WES sample.
51
+ - The checked functional and coding SNV fields agree in this example, allowing transcript-order differences. Complete output equivalence has not been established: some UTR/splice `GeneDetail` fields differ.
52
+ - GFF3 hit contents agree in the example, but Perl emits only hits and Rust emits all input rows. The short runtime and 0.01-second timer resolution make the ratio approximate.
53
+ - A separate 50,000-position synthetic test exposed classification and gene-field differences. Its speedup must not be presented as an equivalent-output benchmark.
54
+ - These are historical measurements from the initial implementation, not a benchmark automatically rerun for every commit. See the [detailed benchmark report (Chinese)](docs/BENCHMARK_2026-09-14.md) for environment and compatibility findings.
55
+
56
+ ## Installation
57
+
58
+ ### Requirements
59
+
60
+ Use the **current stable Rust toolchain** and a native C compiler/linker when building from source. Public CI checks Linux and macOS. A Windows release build workflow is also configured; check release assets for actually available binaries.
61
+
62
+ Perl is not required to run RustAnnovar. Real annotation requires your own matching database files; the quick-start example does not.
63
+
64
+ ### Option 1: Install from GitHub
65
+
66
+ If Rust is not installed, install its toolchain using rustup on Linux/macOS:
67
+
68
+ ```bash
69
+ curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh
70
+ source "$HOME/.cargo/env"
71
+ ```
72
+
73
+ Install the current repository version:
74
+
75
+ ```bash
76
+ cargo install --git https://github.com/ydlongtao/RustAnnovar.git --locked
77
+ rust-annovar --version
78
+ ```
79
+
80
+ For a fixed open-beta version, add `--tag v0.1.0-beta.1`. Cargo installs the executable in `~/.cargo/bin`; ensure that directory is in your `PATH`.
81
+
82
+ ### Option 2: Build from source
83
+
84
+ ```bash
85
+ git clone https://github.com/ydlongtao/RustAnnovar.git
86
+ cd RustAnnovar
87
+ cargo build --release --locked
88
+ ./target/release/rust-annovar --version
89
+
90
+ # Optional: install the local checkout into ~/.cargo/bin
91
+ cargo install --path . --locked
92
+ ```
93
+
94
+ ### Option 3: Install from RubyGems
95
+
96
+ The RubyGems package contains the Rust source and compiles the executable during installation. It therefore requires Cargo and a native linker; Ruby is used only for packaging and launching the compiled binary.
97
+
98
+ ```bash
99
+ gem install rust-annovar --pre
100
+ rust-annovar --version
101
+ ```
102
+
103
+ Version `0.1.0.beta.1` is a prerelease, so `--pre` is required until a stable version is published. The gem version uses RubyGems notation (`0.1.0.beta.1`), while the executable reports the Cargo version (`0.1.0-beta.1`).
104
+
105
+ ### Option 4: Download a release binary
106
+
107
+ Visit [Releases](https://github.com/ydlongtao/RustAnnovar/releases) and choose an asset matching your operating system and processor. The initial `v0.1.0-beta.1` release includes an Apple Silicon macOS archive and `SHA256SUMS`. Other platforms can build from source if no matching asset is available.
108
+
109
+ After downloading both files to the same directory on macOS:
110
+
111
+ ```bash
112
+ shasum -a 256 -c SHA256SUMS
113
+ tar -xzf RustAnnovar-v0.1.0-beta.1-aarch64-apple-darwin.tar.gz
114
+ ./RustAnnovar-v0.1.0-beta.1-aarch64-apple-darwin/rust-annovar --version
115
+ ```
116
+
117
+ ## Quick start
118
+
119
+ Clone the repository if you installed only the executable; example files are located in the source checkout. Run the following from the repository root after installing `rust-annovar`:
120
+
121
+ ```bash
122
+ rust-annovar table examples/demo.vcf examples/humandb \
123
+ --build hg38 \
124
+ --protocol demo \
125
+ --operation f \
126
+ --vcf-input \
127
+ --output demo.multianno.tsv \
128
+ --vcf-output demo.annotated.vcf
129
+
130
+ cat demo.multianno.tsv
131
+ ```
132
+
133
+ If you built without installing, replace `rust-annovar` with `./target/release/rust-annovar` in all commands.
134
+
135
+ Expected table:
136
+
137
+ ```text
138
+ Chr Start End Ref Alt CLNSIG.demo SOURCE.demo
139
+ 1 10 10 A C Pathogenic Synthetic_demo
140
+ 1 25 25 G A . .
141
+ ```
142
+
143
+ The actual file is tab-delimited. `Pathogenic` is a fabricated demonstration label, not a clinical assertion about this position. The second record has no database match. The annotated VCF retains the original sample genotype columns.
144
+
145
+ ## Usage guide
146
+
147
+ ### 1. Prepare matching databases
148
+
149
+ For `table`, files are resolved as `<database-directory>/<build>_<protocol>.txt`. Gene annotation also looks for `<build>_<protocol>Mrna.fa`.
150
+
151
+ ```text
152
+ humandb/
153
+ ├── hg38_refGene.txt
154
+ ├── hg38_refGeneMrna.fa
155
+ ├── hg38_cytoBand.txt
156
+ └── hg38_clinvar.txt
157
+ ```
158
+
159
+ Use the same genome assembly for inputs and databases. `--build` selects filenames; it does not perform liftover or verify assembly identity. Substitute the actual protocol names installed on your machine, including version suffixes. The example filenames do not imply that every database release or specialized schema has been validated.
160
+
161
+ Gene annotation expects a **transcript FASTA** with matching transcript identifiers. The `sequence` command instead takes a **genomic reference FASTA**. Without transcript sequences, coding consequences may be unavailable.
162
+
163
+ ### 2. Convert VCF to AVinput
164
+
165
+ ```bash
166
+ rust-annovar convert sample.vcf.gz \
167
+ --include-info \
168
+ --output sample.avinput
169
+ ```
170
+
171
+ Conversion splits alternate alleles and removes common VCF indel anchor bases. It does not establish full reference-aware normalization compatibility across databases.
172
+
173
+ AVinput uses five required fields: `Chr Start End Ref Alt`, followed by optional extra columns. Substitution/deletion coordinates are one-based and inclusive. Insertions use `-` as the reference allele and an insertion anchor coordinate. Internally, RustAnnovar uses zero-based, half-open intervals with separate insertion handling.
174
+
175
+ ### 3. Annotate one database
176
+
177
+ **Filter annotation** matches the complete variant key:
178
+
179
+ ```bash
180
+ rust-annovar annotate sample.avinput humandb/hg38_clinvar.txt \
181
+ --operation filter --protocol clinvar \
182
+ --output sample.clinvar.tsv
183
+ ```
184
+
185
+ Generic filter files use `Chr`, `Start`, `End`, `Ref`, `Alt`, then annotation columns. Query and database alleles must use compatible representations.
186
+
187
+ **Region annotation** reports interval overlaps:
188
+
189
+ ```bash
190
+ rust-annovar annotate sample.avinput humandb/hg38_cytoBand.txt \
191
+ --operation region --protocol cytoBand \
192
+ --output sample.cytoband.tsv
193
+ ```
194
+
195
+ BED-style regions use zero-based, half-open coordinates; GFF3 uses one-based, inclusive coordinates. GFF3 support here is for region overlap, not GFF3 transcript-model ingestion.
196
+
197
+ **Gene annotation** reads refGene-style models:
198
+
199
+ ```bash
200
+ rust-annovar annotate sample.avinput humandb/hg38_refGene.txt \
201
+ --operation gene --protocol refGene \
202
+ --fasta humandb/hg38_refGeneMrna.fa \
203
+ --output sample.refgene.tsv
204
+ ```
205
+
206
+ The main columns are `Func.refGene`, `Gene.refGene`, `GeneDetail.refGene`, `ExonicFunc.refGene`, and `AAChange.refGene`. Add `--vcf-input` when passing a VCF directly to `annotate`.
207
+
208
+ ### 4. Combine gene, region, and filter annotations
209
+
210
+ ```bash
211
+ rust-annovar table sample.vcf humandb \
212
+ --build hg38 \
213
+ --protocol refGene,cytoBand,clinvar \
214
+ --operation g,r,f \
215
+ --vcf-input \
216
+ --output sample.hg38_multianno.tsv \
217
+ --vcf-output sample.hg38_multianno.vcf
218
+ ```
219
+
220
+ Protocols and operations must correspond one-to-one. `g`, `r`, and `f` mean gene, region, and filter. Database columns follow the requested protocol order. Missing values default to `.` and can be changed with `--nastring`.
221
+
222
+ Use `--csv` for CSV table output. Use `--vcf-output` together with `--vcf-input` for an annotated VCF. Added INFO fields use `FA_<protocol>` identifiers; this schema differs from original ANNOVAR VCF output. Original sample and FORMAT columns are retained.
223
+
224
+ ### 5. Manage filter database indexes
225
+
226
+ ```bash
227
+ rust-annovar db index humandb/hg38_dbnsfp.txt --kind filter
228
+ rust-annovar db check humandb/hg38_dbnsfp.fai.json
229
+ rust-annovar db list humandb --build hg38
230
+ ```
231
+
232
+ The default sidecar records 1 Mb block byte ranges and source metadata. Queries can load relevant ranges from plain-text filter files. Missing or detected-stale indexes and gzip files fall back to full loading. Rebuild indexes after changing or relocating a database. Source checks use size, modification time, and a prefix hash, not a full-file integrity check.
233
+
234
+ `db download <URL> <OUTPUT> --sha256 <EXPECTED_SHA256>` downloads a file from a supplied public URL and optionally validates its full checksum. It does not implement the registered ANNOVAR download catalog.
235
+
236
+ ### 6. Extract sequences and filter tables
237
+
238
+ ```bash
239
+ rust-annovar sequence regions.avinput reference.fa --output regions.fa
240
+
241
+ rust-annovar reduce sample.hg38_multianno.tsv \
242
+ --column Func.refGene --equals exonic \
243
+ --output sample.exonic.tsv
244
+ ```
245
+
246
+ Sequence extraction currently loads the reference FASTA into memory. `reduce` accepts a tab-delimited table and performs exact string equality; it is not a numeric threshold or expression engine.
247
+
248
+ `coding-change` is a convenience entry point for gene annotation with the same arguments as `annotate`. It is not a complete replacement for the original `coding_change.pl` protein FASTA workflow.
249
+
250
+ ### 7. Control parallelism
251
+
252
+ ```bash
253
+ RAYON_NUM_THREADS=1 rust-annovar table sample.avinput humandb \
254
+ --build hg38 --protocol refGene --operation g \
255
+ --output sample.single-thread.tsv
256
+ ```
257
+
258
+ Set `RAYON_NUM_THREADS` to the desired worker count. Without it, Rayon selects the thread pool size automatically. Input and result tables are currently held in memory, so large WGS workloads still require memory planning and validation.
259
+
260
+ ## Command reference
261
+
262
+ | Command | Purpose |
263
+ |---|---|
264
+ | `convert` | Convert VCF or gzip VCF to AVinput |
265
+ | `annotate` | Annotate with one gene, region, or filter database |
266
+ | `table` | Combine databases into TSV, CSV, and optionally VCF |
267
+ | `db index/check/list/download` | Manage local database metadata and downloads |
268
+ | `sequence` | Extract genomic FASTA intervals |
269
+ | `coding-change` | Run gene consequence annotation |
270
+ | `reduce` | Select TSV rows by exact column value |
271
+
272
+ Run `rust-annovar --help` or `rust-annovar <command> --help` for available options.
273
+
274
+ ## Validation and feedback
275
+
276
+ Run public checks from the source directory:
277
+
278
+ ```bash
279
+ cargo fmt -- --check
280
+ cargo clippy --all-targets --all-features -- -D warnings
281
+ cargo test --all-features
282
+ ```
283
+
284
+ Optional compatibility tests require the registered installation at **`annovar/` in the repository root**, including its bundled example and hg19 databases:
285
+
286
+ ```bash
287
+ cargo test --test annovar_compat -- --ignored
288
+ ```
289
+
290
+ These tests check selected fields against saved Perl outputs and expected example hits; they do not validate every ANNOVAR feature. To regenerate the saved baseline, use `ANNOVAR_HOME=/path/to/annovar scripts/capture_perl_baseline.sh`. The test loader itself does not read `ANNOVAR_HOME`.
291
+
292
+ See [compatibility status (Chinese)](docs/COMPATIBILITY.md). Report reproducible differences through [GitHub Issues](https://github.com/ydlongtao/RustAnnovar/issues), including the software version, genome build, database version, commands, a minimal input, and expected versus observed output. Do not include restricted databases, credentials, or identifiable genomic data.
293
+
294
+ ## License and acknowledgments
295
+
296
+ The project source is offered under [MIT](LICENSE-MIT) or [Apache-2.0](LICENSE-APACHE), at your option. This does not extend to third-party databases or ANNOVAR software.
297
+
298
+ Repository presentation was inspired by [Huang-lab/fastVEP](https://github.com/Huang-lab/fastVEP). We acknowledge the ANNOVAR authors and the variant annotation community for the formats and resources underlying compatibility evaluation.
data/bin/rust-annovar ADDED
@@ -0,0 +1,6 @@
1
+ #!/usr/bin/env ruby
2
+ # frozen_string_literal: true
3
+
4
+ require "rust_annovar"
5
+
6
+ exec(RustAnnovar.executable, *ARGV)
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "rbconfig"
4
+ require "shellwords"
5
+
6
+ manifest = File.expand_path("../../Cargo.toml", __dir__)
7
+ cargo = ENV.fetch("CARGO", "cargo")
8
+ unless system(cargo, "--version", out: File::NULL)
9
+ abort "Cargo is required to install rust-annovar. Install Rust from https://rustup.rs/"
10
+ end
11
+
12
+ manifest_arg = Shellwords.escape(manifest)
13
+ cargo_arg = Shellwords.escape(cargo)
14
+ executable = "rust-annovar#{RbConfig::CONFIG.fetch("EXEEXT", "")}"
15
+ target_dir = File.expand_path("target", __dir__)
16
+ install_dir = File.join(RbConfig::CONFIG.fetch("sitearchdir"), "rust_annovar")
17
+ ruby_arg = Shellwords.escape(RbConfig.ruby)
18
+ target_arg = Shellwords.escape(target_dir)
19
+ install_arg = Shellwords.escape(install_dir)
20
+ binary_arg = Shellwords.escape(File.join(target_dir, "release", executable))
21
+
22
+ File.write(
23
+ "Makefile",
24
+ <<~MAKEFILE
25
+ all:
26
+ \tCARGO_TARGET_DIR=#{target_arg} #{cargo_arg} build --release --locked --manifest-path #{manifest_arg}
27
+
28
+ install:
29
+ \t#{ruby_arg} -rfileutils -e 'FileUtils.mkdir_p(ARGV.fetch(0)); FileUtils.install(ARGV.fetch(1), ARGV.fetch(0), mode: 0755)' #{install_arg} #{binary_arg}
30
+
31
+ clean:
32
+ \t#{ruby_arg} -rfileutils -e 'FileUtils.rm_rf(ARGV.fetch(0))' #{target_arg}
33
+ MAKEFILE
34
+ )
@@ -0,0 +1,5 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RustAnnovar
4
+ VERSION = "0.1.0.beta.1"
5
+ end
@@ -0,0 +1,35 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "rbconfig"
4
+ require_relative "rust_annovar/version"
5
+
6
+ module RustAnnovar
7
+ class Error < StandardError; end
8
+
9
+ def self.executable
10
+ specification = Gem.loaded_specs.fetch("rust-annovar")
11
+ suffix = RbConfig::CONFIG.fetch("EXEEXT", "")
12
+ executable_name = "rust-annovar#{suffix}"
13
+ candidates = [
14
+ File.join(specification.extension_dir, "rust_annovar", executable_name),
15
+ File.join(RbConfig::CONFIG.fetch("sitearchdir"), "rust_annovar", executable_name),
16
+ File.join(specification.full_gem_path, "ext", "rust_annovar", "target", "release", executable_name)
17
+ ]
18
+ candidates.concat(
19
+ Dir.glob(
20
+ File.join(
21
+ specification.base_dir,
22
+ "extensions",
23
+ "**",
24
+ specification.full_name,
25
+ "rust_annovar",
26
+ executable_name
27
+ )
28
+ )
29
+ )
30
+ path = candidates.find { |candidate| File.file?(candidate) && File.executable?(candidate) }
31
+ return path if path
32
+
33
+ raise Error, "compiled rust-annovar executable was not found in the RubyGems extension directories"
34
+ end
35
+ end