altar-identity 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- altar_identity-0.1.0/.gitignore +32 -0
- altar_identity-0.1.0/LICENSE +21 -0
- altar_identity-0.1.0/PKG-INFO +114 -0
- altar_identity-0.1.0/README.md +89 -0
- altar_identity-0.1.0/altar_identity/__init__.py +75 -0
- altar_identity-0.1.0/altar_identity/digest.py +186 -0
- altar_identity-0.1.0/altar_identity/py.typed +0 -0
- altar_identity-0.1.0/altar_identity/tsv.py +144 -0
- altar_identity-0.1.0/altar_identity/variant.py +209 -0
- altar_identity-0.1.0/pyproject.toml +50 -0
- altar_identity-0.1.0/tests/test_digest.py +209 -0
- altar_identity-0.1.0/tests/test_tsv.py +244 -0
- altar_identity-0.1.0/tests/test_variant.py +131 -0
- altar_identity-0.1.0/tests/test_version.py +11 -0
- altar_identity-0.1.0/tests/variant_identity_cases.json +614 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
*.pyc
|
|
2
|
+
*~
|
|
3
|
+
**/__pycache__/*
|
|
4
|
+
*.swp
|
|
5
|
+
.vscode/
|
|
6
|
+
.idea/
|
|
7
|
+
.DS_Store
|
|
8
|
+
.env*
|
|
9
|
+
.mypy_cache/
|
|
10
|
+
.ruff_cache/
|
|
11
|
+
.pytest_cache/
|
|
12
|
+
.coverage
|
|
13
|
+
htmlcov/
|
|
14
|
+
coverage.xml
|
|
15
|
+
pytest.xml
|
|
16
|
+
.hypothesis/
|
|
17
|
+
.python-version
|
|
18
|
+
.venv/
|
|
19
|
+
.venv-*/
|
|
20
|
+
*.egg-info/
|
|
21
|
+
build/
|
|
22
|
+
dist/
|
|
23
|
+
site/
|
|
24
|
+
*.sqlite3
|
|
25
|
+
*.log
|
|
26
|
+
|
|
27
|
+
# Generated variant indexes and datasets. The small canonical gene table is tracked.
|
|
28
|
+
altar/altar/variants/data/ccres.dnatree
|
|
29
|
+
altar/altar/variants/data/region_annotations.parquet
|
|
30
|
+
altar/altar/variants/data/variants.pkl.gz
|
|
31
|
+
altar/altar/variants/data/raw/
|
|
32
|
+
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Riya Sinha
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: altar-identity
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Dependency-free variant and content identity shared by Altar and its model runtimes
|
|
5
|
+
Project-URL: Documentation, https://kundajelab.github.io/altar/
|
|
6
|
+
Project-URL: Issues, https://github.com/kundajelab/altar/issues
|
|
7
|
+
Project-URL: Repository, https://github.com/kundajelab/altar
|
|
8
|
+
Author: Riya Sinha
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: bioinformatics,genomics,variant-identity
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.9
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# altar-identity
|
|
27
|
+
|
|
28
|
+
`altar-identity` holds the identity rules that Altar core and every Altar model runtime must agree on. It uses
|
|
29
|
+
only the Python standard library and supports Python 3.9 and later, so a runtime pinned to an older
|
|
30
|
+
TensorFlow or PyTorch stack runs the same code as Altar core instead of keeping a copy.
|
|
31
|
+
|
|
32
|
+
Most users do not install it directly: `altar` depends on it and re-exports the variant names from
|
|
33
|
+
`altar.variants`, `altar.models`, and `altar.sources`.
|
|
34
|
+
|
|
35
|
+
## Variant identity
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from altar_identity import VariantKey, canonical_chromosome, canonical_variant_id
|
|
39
|
+
|
|
40
|
+
canonical_chromosome("MT") # "chrM"
|
|
41
|
+
canonical_variant_id("1", 10, "a", "t") # "chr1:10:A:T"
|
|
42
|
+
VariantKey.require_canonical("chr1:10:A:T") # rejects aliases such as "1:10:A:T"
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
A key is `chromosome:position:REF:ALT` with a one-based position, and every field is ASCII. Surrounding ASCII
|
|
46
|
+
whitespace is trimmed from each field; other whitespace and non-ASCII text raise `VariantIdentityError`, a
|
|
47
|
+
`ValueError`, rather than being folded (an Arabic-Indic or full-width digit never becomes `1`).
|
|
48
|
+
|
|
49
|
+
- **Chromosome.** The `chr` prefix is optional and case-insensitive. Primary chromosomes are normalized in any
|
|
50
|
+
case (`1`, `chr01`, `CHR1` → `chr1`; `M`, `MT` → `chrM`). Every other contig keeps its exact spelling after
|
|
51
|
+
the prefix (`CHRUn_KI270302v1` → `chrUn_KI270302v1`), because reference contig names are case-sensitive and
|
|
52
|
+
the key must match the name in the FASTA. The name uses the VCF contig-name characters: letters, digits, and
|
|
53
|
+
`!#$%&*+./;=?@^_|~-`, which exclude `:`.
|
|
54
|
+
- **Position.** A positive `int`. In text (`VariantKey.parse`, variant files, and `parse_position` for other
|
|
55
|
+
readers) it is ASCII digits `[0-9]+`: `+5`, `1_000`, `1e3`, and non-ASCII digits are rejected, although
|
|
56
|
+
Python's `int()` accepts some of them.
|
|
57
|
+
- **Alleles.** Uppercased, then `[ACGTN]+`, the VCF base alphabet for REF and a concrete ALT. Symbolic
|
|
58
|
+
(`<DEL>`), `*`, `.`, breakend, IUPAC-ambiguity and `-` alleles are rejected, and REF must differ from ALT.
|
|
59
|
+
`N` is allowed in the key; reference validation decides whether a keyed variant can be scored.
|
|
60
|
+
|
|
61
|
+
The key does not left-align or trim indels, and it does not carry the genome build.
|
|
62
|
+
|
|
63
|
+
## Variant files
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
from altar_identity import batched, read_variants
|
|
67
|
+
|
|
68
|
+
for batch in batched(read_variants("variants.tsv"), 1024):
|
|
69
|
+
...
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Altar writes the variants for a model container as a headerless, tab-separated UTF-8 file with the columns
|
|
73
|
+
`chr`, `pos`, `ref`, `alt`, and `variant_id`. `read_variants` also accepts four columns, or any label in the
|
|
74
|
+
fifth, and never uses the fifth column. Fields are never quoted: a `"` is an ordinary character, and a field
|
|
75
|
+
cannot hold a tab or line break. A leading UTF-8 byte-order mark is ignored. The reader yields canonical
|
|
76
|
+
`VariantKey` values, skips blank lines, and raises `VariantFileError` naming the file and line for a malformed
|
|
77
|
+
row. By default a repeated `variant_id` is an
|
|
78
|
+
error; pass `duplicates="skip"` to keep the first occurrence or `duplicates="allow"` to yield every row.
|
|
79
|
+
`read_variant_rows` yields `VariantRow(line, key)` records instead, for callers that apply their own policy
|
|
80
|
+
(for example, SNVs only) and need to report the offending line. An empty or blank-only file yields nothing.
|
|
81
|
+
|
|
82
|
+
## Content identity
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from altar_identity import sha256_file, verify_file
|
|
86
|
+
|
|
87
|
+
digest = sha256_file("weights.h5") # "sha256:<64 lowercase hex>"
|
|
88
|
+
verify_file("weights.h5", digest, label="weights")
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
A digest is `sha256:` followed by 64 lowercase hexadecimal digits and names the bytes of one regular file.
|
|
92
|
+
`verify_file` raises `DigestMismatchError` when the bytes differ, and also when the path is missing or is a
|
|
93
|
+
directory. Symlinks are followed. `parse_sha256_digest` and `is_sha256_digest` check the exact spelling, and
|
|
94
|
+
`SHA256_DIGEST_PATTERN` is the same rule as an unanchored regular expression for schemas that embed it.
|
|
95
|
+
|
|
96
|
+
A file that many tasks read, such as a reference genome on a shared volume, need not be hashed by every task.
|
|
97
|
+
`verify_file(..., trust_record=True, record=True)` accepts a file whose verification record is current and
|
|
98
|
+
writes a record after a successful hash. The record is a one-line file at `<path>.sha256-verified`:
|
|
99
|
+
|
|
100
|
+
```text
|
|
101
|
+
verified-file/1 sha256:<64 hex> size=<bytes> mtime=<seconds> ctime=<seconds> inode=<number>
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Rewriting, replacing, truncating, or appending to the file changes a recorded value and invalidates the record.
|
|
105
|
+
`has_verification_record`, `write_verification_record`, and `verification_record_line` are the primitives Altar
|
|
106
|
+
core's `verify_file_digest`, its staging backends, and the runtimes share. Anyone who can write the storage
|
|
107
|
+
can also forge a record, so do not trust records where bytes first arrive, such as a download.
|
|
108
|
+
|
|
109
|
+
## Development
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
uv run --isolated --no-project --python 3.9 --with pytest --with-editable identity \
|
|
113
|
+
python -m pytest identity/tests
|
|
114
|
+
```
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# altar-identity
|
|
2
|
+
|
|
3
|
+
`altar-identity` holds the identity rules that Altar core and every Altar model runtime must agree on. It uses
|
|
4
|
+
only the Python standard library and supports Python 3.9 and later, so a runtime pinned to an older
|
|
5
|
+
TensorFlow or PyTorch stack runs the same code as Altar core instead of keeping a copy.
|
|
6
|
+
|
|
7
|
+
Most users do not install it directly: `altar` depends on it and re-exports the variant names from
|
|
8
|
+
`altar.variants`, `altar.models`, and `altar.sources`.
|
|
9
|
+
|
|
10
|
+
## Variant identity
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
from altar_identity import VariantKey, canonical_chromosome, canonical_variant_id
|
|
14
|
+
|
|
15
|
+
canonical_chromosome("MT") # "chrM"
|
|
16
|
+
canonical_variant_id("1", 10, "a", "t") # "chr1:10:A:T"
|
|
17
|
+
VariantKey.require_canonical("chr1:10:A:T") # rejects aliases such as "1:10:A:T"
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
A key is `chromosome:position:REF:ALT` with a one-based position, and every field is ASCII. Surrounding ASCII
|
|
21
|
+
whitespace is trimmed from each field; other whitespace and non-ASCII text raise `VariantIdentityError`, a
|
|
22
|
+
`ValueError`, rather than being folded (an Arabic-Indic or full-width digit never becomes `1`).
|
|
23
|
+
|
|
24
|
+
- **Chromosome.** The `chr` prefix is optional and case-insensitive. Primary chromosomes are normalized in any
|
|
25
|
+
case (`1`, `chr01`, `CHR1` → `chr1`; `M`, `MT` → `chrM`). Every other contig keeps its exact spelling after
|
|
26
|
+
the prefix (`CHRUn_KI270302v1` → `chrUn_KI270302v1`), because reference contig names are case-sensitive and
|
|
27
|
+
the key must match the name in the FASTA. The name uses the VCF contig-name characters: letters, digits, and
|
|
28
|
+
`!#$%&*+./;=?@^_|~-`, which exclude `:`.
|
|
29
|
+
- **Position.** A positive `int`. In text (`VariantKey.parse`, variant files, and `parse_position` for other
|
|
30
|
+
readers) it is ASCII digits `[0-9]+`: `+5`, `1_000`, `1e3`, and non-ASCII digits are rejected, although
|
|
31
|
+
Python's `int()` accepts some of them.
|
|
32
|
+
- **Alleles.** Uppercased, then `[ACGTN]+`, the VCF base alphabet for REF and a concrete ALT. Symbolic
|
|
33
|
+
(`<DEL>`), `*`, `.`, breakend, IUPAC-ambiguity and `-` alleles are rejected, and REF must differ from ALT.
|
|
34
|
+
`N` is allowed in the key; reference validation decides whether a keyed variant can be scored.
|
|
35
|
+
|
|
36
|
+
The key does not left-align or trim indels, and it does not carry the genome build.
|
|
37
|
+
|
|
38
|
+
## Variant files
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from altar_identity import batched, read_variants
|
|
42
|
+
|
|
43
|
+
for batch in batched(read_variants("variants.tsv"), 1024):
|
|
44
|
+
...
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Altar writes the variants for a model container as a headerless, tab-separated UTF-8 file with the columns
|
|
48
|
+
`chr`, `pos`, `ref`, `alt`, and `variant_id`. `read_variants` also accepts four columns, or any label in the
|
|
49
|
+
fifth, and never uses the fifth column. Fields are never quoted: a `"` is an ordinary character, and a field
|
|
50
|
+
cannot hold a tab or line break. A leading UTF-8 byte-order mark is ignored. The reader yields canonical
|
|
51
|
+
`VariantKey` values, skips blank lines, and raises `VariantFileError` naming the file and line for a malformed
|
|
52
|
+
row. By default a repeated `variant_id` is an
|
|
53
|
+
error; pass `duplicates="skip"` to keep the first occurrence or `duplicates="allow"` to yield every row.
|
|
54
|
+
`read_variant_rows` yields `VariantRow(line, key)` records instead, for callers that apply their own policy
|
|
55
|
+
(for example, SNVs only) and need to report the offending line. An empty or blank-only file yields nothing.
|
|
56
|
+
|
|
57
|
+
## Content identity
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from altar_identity import sha256_file, verify_file
|
|
61
|
+
|
|
62
|
+
digest = sha256_file("weights.h5") # "sha256:<64 lowercase hex>"
|
|
63
|
+
verify_file("weights.h5", digest, label="weights")
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
A digest is `sha256:` followed by 64 lowercase hexadecimal digits and names the bytes of one regular file.
|
|
67
|
+
`verify_file` raises `DigestMismatchError` when the bytes differ, and also when the path is missing or is a
|
|
68
|
+
directory. Symlinks are followed. `parse_sha256_digest` and `is_sha256_digest` check the exact spelling, and
|
|
69
|
+
`SHA256_DIGEST_PATTERN` is the same rule as an unanchored regular expression for schemas that embed it.
|
|
70
|
+
|
|
71
|
+
A file that many tasks read, such as a reference genome on a shared volume, need not be hashed by every task.
|
|
72
|
+
`verify_file(..., trust_record=True, record=True)` accepts a file whose verification record is current and
|
|
73
|
+
writes a record after a successful hash. The record is a one-line file at `<path>.sha256-verified`:
|
|
74
|
+
|
|
75
|
+
```text
|
|
76
|
+
verified-file/1 sha256:<64 hex> size=<bytes> mtime=<seconds> ctime=<seconds> inode=<number>
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Rewriting, replacing, truncating, or appending to the file changes a recorded value and invalidates the record.
|
|
80
|
+
`has_verification_record`, `write_verification_record`, and `verification_record_line` are the primitives Altar
|
|
81
|
+
core's `verify_file_digest`, its staging backends, and the runtimes share. Anyone who can write the storage
|
|
82
|
+
can also forge a record, so do not trust records where bytes first arrive, such as a download.
|
|
83
|
+
|
|
84
|
+
## Development
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
uv run --isolated --no-project --python 3.9 --with pytest --with-editable identity \
|
|
88
|
+
python -m pytest identity/tests
|
|
89
|
+
```
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# Copyright 2026 Riya Sinha
|
|
2
|
+
"""Variant and content identity shared by Altar core and its model runtimes.
|
|
3
|
+
|
|
4
|
+
This package has no dependencies and supports Python 3.9, so a runtime pinned to an older scientific stack
|
|
5
|
+
builds variant IDs, reads Altar's variant files, and verifies staged files with the same code as Altar core.
|
|
6
|
+
|
|
7
|
+
- ``VariantKey``, ``canonical_chromosome``, ``canonical_variant_id``, ``parse_position``: the canonical
|
|
8
|
+
``chr:pos:REF:ALT`` key and its textual position rule.
|
|
9
|
+
- ``read_variants``, ``read_variant_rows``, ``batched``: the headerless variant file Altar hands to a model container.
|
|
10
|
+
- ``sha256_file``, ``verify_file``, ``parse_sha256_digest``: the ``sha256:<hex>`` identity of one regular file,
|
|
11
|
+
and the verification records that let many tasks share one hash of a staged file. ``SHA256_DIGEST_PATTERN``
|
|
12
|
+
is the one spelling of the digest rule, for schemas and patterns that embed it.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from altar_identity.digest import (
|
|
16
|
+
SHA256_DIGEST_PATTERN,
|
|
17
|
+
SHA256_PREFIX,
|
|
18
|
+
VERIFICATION_RECORD_FORMAT,
|
|
19
|
+
VERIFICATION_RECORD_SUFFIX,
|
|
20
|
+
DigestMismatchError,
|
|
21
|
+
has_verification_record,
|
|
22
|
+
is_sha256_digest,
|
|
23
|
+
parse_sha256_digest,
|
|
24
|
+
sha256_file,
|
|
25
|
+
verification_record_line,
|
|
26
|
+
verify_file,
|
|
27
|
+
write_verification_record,
|
|
28
|
+
)
|
|
29
|
+
from altar_identity.tsv import (
|
|
30
|
+
VARIANT_FIELD_COUNTS,
|
|
31
|
+
DuplicatePolicy,
|
|
32
|
+
VariantFileError,
|
|
33
|
+
VariantRow,
|
|
34
|
+
batched,
|
|
35
|
+
read_variant_rows,
|
|
36
|
+
read_variants,
|
|
37
|
+
)
|
|
38
|
+
from altar_identity.variant import (
|
|
39
|
+
VariantIdentityError,
|
|
40
|
+
VariantKey,
|
|
41
|
+
canonical_chromosome,
|
|
42
|
+
canonical_variant_id,
|
|
43
|
+
parse_position,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
__version__: str = "0.1.0"
|
|
48
|
+
|
|
49
|
+
__all__ = [
|
|
50
|
+
"SHA256_DIGEST_PATTERN",
|
|
51
|
+
"SHA256_PREFIX",
|
|
52
|
+
"VARIANT_FIELD_COUNTS",
|
|
53
|
+
"VERIFICATION_RECORD_FORMAT",
|
|
54
|
+
"VERIFICATION_RECORD_SUFFIX",
|
|
55
|
+
"DigestMismatchError",
|
|
56
|
+
"DuplicatePolicy",
|
|
57
|
+
"VariantFileError",
|
|
58
|
+
"VariantIdentityError",
|
|
59
|
+
"VariantKey",
|
|
60
|
+
"VariantRow",
|
|
61
|
+
"__version__",
|
|
62
|
+
"batched",
|
|
63
|
+
"canonical_chromosome",
|
|
64
|
+
"canonical_variant_id",
|
|
65
|
+
"has_verification_record",
|
|
66
|
+
"is_sha256_digest",
|
|
67
|
+
"parse_position",
|
|
68
|
+
"parse_sha256_digest",
|
|
69
|
+
"read_variant_rows",
|
|
70
|
+
"read_variants",
|
|
71
|
+
"sha256_file",
|
|
72
|
+
"verification_record_line",
|
|
73
|
+
"verify_file",
|
|
74
|
+
"write_verification_record",
|
|
75
|
+
]
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
# Copyright 2026 Riya Sinha
|
|
2
|
+
"""Content identity for staged files: one ``sha256:<64 lowercase hex>`` digest per regular file.
|
|
3
|
+
|
|
4
|
+
A digest names the bytes of exactly one regular file. Symlinks are followed. A path that is missing or is a
|
|
5
|
+
directory fails verification the same way as a digest mismatch, because neither has the bytes the digest
|
|
6
|
+
names; ship a multi-file resource as an archive and digest the archive.
|
|
7
|
+
|
|
8
|
+
Verification records
|
|
9
|
+
--------------------
|
|
10
|
+
A file read by many tasks, such as a reference genome, need not be hashed by every task. A verification record
|
|
11
|
+
is a one-line text file at the file's path plus ``VERIFICATION_RECORD_SUFFIX``::
|
|
12
|
+
|
|
13
|
+
verified-file/1 sha256:<64 hex> size=<bytes> mtime=<seconds> ctime=<seconds> inode=<number>
|
|
14
|
+
|
|
15
|
+
It states that the bytes matched the digest and pins the file's size, whole-second modification and
|
|
16
|
+
status-change times, and inode as they were then. Rewriting, replacing, truncating, or appending to the file
|
|
17
|
+
changes at least one of those values, which invalidates the record. Altar core's ``verify_file_digest``, its
|
|
18
|
+
shell verifier, and every runtime read and write this same line through this module.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
import contextlib
|
|
23
|
+
import hashlib
|
|
24
|
+
import os
|
|
25
|
+
import re
|
|
26
|
+
import stat
|
|
27
|
+
import time
|
|
28
|
+
import uuid
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
SHA256_PREFIX = "sha256:"
|
|
33
|
+
SHA256_DIGEST_PATTERN = f"{SHA256_PREFIX}[0-9a-f]{{64}}"
|
|
34
|
+
"""Unanchored regular expression for a canonical digest. Anchor it (``re.fullmatch``, or ``^...$`` in a schema)
|
|
35
|
+
or embed it in a larger pattern, such as an image reference, instead of spelling the digest rule again."""
|
|
36
|
+
VERIFICATION_RECORD_SUFFIX = ".sha256-verified"
|
|
37
|
+
VERIFICATION_RECORD_FORMAT = "verified-file/1"
|
|
38
|
+
_SHA256_DIGEST = re.compile(SHA256_DIGEST_PATTERN)
|
|
39
|
+
_BUFFER_SIZE = 1024 * 1024
|
|
40
|
+
_RECORD_MAX_BYTES = 512
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class DigestMismatchError(ValueError):
|
|
44
|
+
"""A path does not hold the single regular file whose bytes a digest names."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def is_sha256_digest(value: object) -> bool:
|
|
48
|
+
"""Return whether ``value`` is exactly ``sha256:`` followed by 64 lowercase hexadecimal digits."""
|
|
49
|
+
return isinstance(value, str) and _SHA256_DIGEST.fullmatch(value) is not None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def parse_sha256_digest(value: str) -> str:
|
|
53
|
+
"""Return ``value`` unchanged if it is a canonical SHA-256 digest; otherwise raise ``ValueError``.
|
|
54
|
+
|
|
55
|
+
The check is exact: no surrounding whitespace, no trailing newline, and no uppercase hexadecimal.
|
|
56
|
+
"""
|
|
57
|
+
if not is_sha256_digest(value):
|
|
58
|
+
msg = f"expected sha256:<64 lowercase hex>, got {value!r}"
|
|
59
|
+
raise ValueError(msg)
|
|
60
|
+
return value
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def sha256_file(path: str | os.PathLike[str]) -> str:
|
|
64
|
+
"""Return the ``sha256:<hex>`` digest of the bytes at ``path``, reading it in bounded chunks."""
|
|
65
|
+
digest = hashlib.sha256()
|
|
66
|
+
with Path(path).open("rb") as handle:
|
|
67
|
+
for chunk in iter(lambda: handle.read(_BUFFER_SIZE), b""):
|
|
68
|
+
digest.update(chunk)
|
|
69
|
+
return f"{SHA256_PREFIX}{digest.hexdigest()}"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def verification_record_line(digest: str, status: os.stat_result) -> str:
|
|
73
|
+
"""Return the verification-record line for ``digest`` and a file's ``os.stat`` result.
|
|
74
|
+
|
|
75
|
+
``int()`` of the float times matches the whole seconds ``stat -c %Y`` and ``%Z`` print for any file written
|
|
76
|
+
after 1970, so a shell verifier renders the same line.
|
|
77
|
+
"""
|
|
78
|
+
return (
|
|
79
|
+
f"{VERIFICATION_RECORD_FORMAT} {digest} size={status.st_size} mtime={int(status.st_mtime)} "
|
|
80
|
+
f"ctime={int(status.st_ctime)} inode={status.st_ino}"
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _record_path(path: str | os.PathLike[str]) -> Path:
|
|
85
|
+
return Path(f"{os.fspath(path)}{VERIFICATION_RECORD_SUFFIX}")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def has_verification_record(path: str | os.PathLike[str], digest: str) -> bool:
|
|
89
|
+
"""Return whether `path` is a regular file with a current verification record for `digest`.
|
|
90
|
+
|
|
91
|
+
The record must name `digest` and match the file's current size, modification time, status-change time,
|
|
92
|
+
and inode number. A missing, unreadable, or stale record returns `False`. Checking a record reads two
|
|
93
|
+
small pieces of metadata and never reads the file's bytes. Symlinks are followed.
|
|
94
|
+
"""
|
|
95
|
+
try:
|
|
96
|
+
status = Path(path).stat()
|
|
97
|
+
if not stat.S_ISREG(status.st_mode):
|
|
98
|
+
return False
|
|
99
|
+
with _record_path(path).open("rb") as handle:
|
|
100
|
+
content = handle.read(_RECORD_MAX_BYTES)
|
|
101
|
+
except OSError:
|
|
102
|
+
return False
|
|
103
|
+
return content.decode("ascii", "replace").rstrip("\n") == verification_record_line(digest, status)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def write_verification_record(path: str | os.PathLike[str], digest: str, before: os.stat_result) -> bool:
|
|
107
|
+
"""Write a record for bytes that just matched `digest`. Returns whether a record was written.
|
|
108
|
+
|
|
109
|
+
`before` is the file's status from before it was hashed. No record is written if the file changed while it
|
|
110
|
+
was being hashed. No record is written either while the file's status-change second is still the current
|
|
111
|
+
second: a later change within that same second would leave every recorded value unchanged. The next check
|
|
112
|
+
after that second writes the record instead. The write is atomic and best effort. A directory the caller
|
|
113
|
+
cannot write only means the next check hashes the file again.
|
|
114
|
+
"""
|
|
115
|
+
try:
|
|
116
|
+
after = Path(path).stat()
|
|
117
|
+
except OSError:
|
|
118
|
+
return False
|
|
119
|
+
line = verification_record_line(digest, after)
|
|
120
|
+
if line != verification_record_line(digest, before) or int(after.st_ctime) >= int(time.time()):
|
|
121
|
+
return False
|
|
122
|
+
record = _record_path(path)
|
|
123
|
+
temporary = record.with_name(f"{record.name}.{os.getpid()}.{uuid.uuid4().hex}.tmp")
|
|
124
|
+
try:
|
|
125
|
+
temporary.write_text(f"{line}\n", encoding="ascii")
|
|
126
|
+
temporary.replace(record)
|
|
127
|
+
except OSError:
|
|
128
|
+
with contextlib.suppress(OSError):
|
|
129
|
+
temporary.unlink()
|
|
130
|
+
return False
|
|
131
|
+
return True
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def verify_file(
|
|
135
|
+
path: str | os.PathLike[str],
|
|
136
|
+
expected: str,
|
|
137
|
+
*,
|
|
138
|
+
label: str = "resource",
|
|
139
|
+
trust_record: bool = False,
|
|
140
|
+
record: bool = False,
|
|
141
|
+
) -> bool:
|
|
142
|
+
"""Raise ``DigestMismatchError`` unless ``path`` is one regular file whose bytes have digest ``expected``.
|
|
143
|
+
|
|
144
|
+
``expected`` must be a canonical SHA-256 digest; a malformed one raises ``ValueError``. ``label`` names
|
|
145
|
+
the file in the error message.
|
|
146
|
+
|
|
147
|
+
By default the file is always hashed. With ``trust_record``, a current verification record accepts the file
|
|
148
|
+
without reading it, so a file staged once and read by many tasks is hashed once. With ``record``, a
|
|
149
|
+
successful hash writes a record for later checks. Anyone who can write the storage can also write a record,
|
|
150
|
+
so trust records only on storage whose writers you trust, and never where bytes first arrive (a download).
|
|
151
|
+
|
|
152
|
+
Returns ``True`` when the file was hashed and ``False`` when a record accepted it.
|
|
153
|
+
"""
|
|
154
|
+
parse_sha256_digest(expected)
|
|
155
|
+
if trust_record and has_verification_record(path, expected):
|
|
156
|
+
return False
|
|
157
|
+
if not Path(path).is_file():
|
|
158
|
+
msg = (
|
|
159
|
+
f"{label} at {str(path)!r} is not a regular file; a sha256 digest covers the bytes of one file, "
|
|
160
|
+
"so ship a directory as an archive"
|
|
161
|
+
)
|
|
162
|
+
raise DigestMismatchError(msg)
|
|
163
|
+
before = Path(path).stat()
|
|
164
|
+
actual = sha256_file(path)
|
|
165
|
+
if actual != expected:
|
|
166
|
+
msg = f"{label} at {str(path)!r} has digest {actual}; expected {expected}"
|
|
167
|
+
raise DigestMismatchError(msg)
|
|
168
|
+
if record:
|
|
169
|
+
write_verification_record(path, expected, before)
|
|
170
|
+
return True
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
__all__ = [
|
|
174
|
+
"SHA256_DIGEST_PATTERN",
|
|
175
|
+
"SHA256_PREFIX",
|
|
176
|
+
"VERIFICATION_RECORD_FORMAT",
|
|
177
|
+
"VERIFICATION_RECORD_SUFFIX",
|
|
178
|
+
"DigestMismatchError",
|
|
179
|
+
"has_verification_record",
|
|
180
|
+
"is_sha256_digest",
|
|
181
|
+
"parse_sha256_digest",
|
|
182
|
+
"sha256_file",
|
|
183
|
+
"verification_record_line",
|
|
184
|
+
"verify_file",
|
|
185
|
+
"write_verification_record",
|
|
186
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
# Copyright 2026 Riya Sinha
|
|
2
|
+
"""The canonical variant file that Altar hands to a model container.
|
|
3
|
+
|
|
4
|
+
Altar core's scoring preparation writes each batch as a headerless, tab-separated UTF-8 file with one variant
|
|
5
|
+
per line: ``chr``, ``pos`` (one-based), ``ref``, ``alt``, and the canonical ``variant_id``. The reader also
|
|
6
|
+
accepts rows without the fifth column, and rows whose fifth column is any other label, because callers may
|
|
7
|
+
hand a container a hand-written candidate file. The fifth column is never used: identity always comes from
|
|
8
|
+
the locus fields.
|
|
9
|
+
|
|
10
|
+
Core's scoring preparation reads its candidate files through this same reader, which applies ``VariantKey``'s
|
|
11
|
+
field rules (see ``altar_identity.variant``). It trims fields and uppercases alleles, normalizes chromosome aliases
|
|
12
|
+
(``1``, ``chr01`` and ``CHR1`` all become ``chr1``), and reads the position as ASCII digits. It rejects rows that
|
|
13
|
+
have no canonical spelling, such as a contig containing ``:``, a position below one or written as ``+5`` or
|
|
14
|
+
``1_000``, or an allele outside ``[ACGTN]``. Blank lines and a leading UTF-8 byte-order mark are skipped. Fields
|
|
15
|
+
are never quoted: a ``"`` is an ordinary character, and a field cannot contain a tab, carriage return, or newline.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
import csv
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import TYPE_CHECKING, Literal, NamedTuple, TypeVar
|
|
22
|
+
|
|
23
|
+
from altar_identity.variant import VariantIdentityError, VariantKey, parse_position
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
import os
|
|
28
|
+
from collections.abc import Iterable, Iterator
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
T = TypeVar("T")
|
|
32
|
+
VARIANT_FIELD_COUNTS = frozenset({4, 5})
|
|
33
|
+
DuplicatePolicy = Literal["error", "skip", "allow"]
|
|
34
|
+
_DUPLICATE_POLICIES = frozenset({"error", "skip", "allow"})
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class VariantFileError(ValueError):
|
|
38
|
+
"""A variant file row is malformed, has no canonical identity, or repeats a variant.
|
|
39
|
+
|
|
40
|
+
The message is ``path:line: reason``. ``line`` and ``reason`` are also attributes, so a caller that reports
|
|
41
|
+
errors in its own words (for example, without a temporary local path) need not parse the message.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
def __init__(self, message: str, *, line: int | None = None, reason: str | None = None) -> None:
|
|
45
|
+
super().__init__(message)
|
|
46
|
+
self.line = line
|
|
47
|
+
self.reason = message if reason is None else reason
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _row_error(path: str | os.PathLike[str], line: int, reason: str) -> VariantFileError:
|
|
51
|
+
return VariantFileError(f"{path}:{line}: {reason}", line=line, reason=reason)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class VariantRow(NamedTuple):
|
|
55
|
+
"""One canonical variant and the one-based file line it was read from."""
|
|
56
|
+
|
|
57
|
+
line: int
|
|
58
|
+
key: VariantKey
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _fields(handle: Iterable[str], path: str | os.PathLike[str]) -> Iterator[tuple[int, list[str]]]:
|
|
62
|
+
"""Yield ``(line, fields)`` for each non-blank row, split on tabs with no quoting."""
|
|
63
|
+
# No quoting: a `"` is an ordinary character, so one stray quote cannot swallow the lines after it.
|
|
64
|
+
reader = csv.reader(handle, delimiter="\t", quoting=csv.QUOTE_NONE)
|
|
65
|
+
while True:
|
|
66
|
+
try:
|
|
67
|
+
row = next(reader)
|
|
68
|
+
except StopIteration:
|
|
69
|
+
return
|
|
70
|
+
except csv.Error as error:
|
|
71
|
+
raise _row_error(path, reader.line_num, f"unreadable row: {error}") from error
|
|
72
|
+
if row:
|
|
73
|
+
yield reader.line_num, row
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def read_variants(path: str | os.PathLike[str], *, duplicates: DuplicatePolicy = "error") -> Iterator[VariantKey]:
|
|
77
|
+
"""Stream canonical ``VariantKey`` values from a headerless ``chr pos ref alt [label]`` file.
|
|
78
|
+
|
|
79
|
+
``duplicates`` sets what happens when two rows have the same canonical ``variant_id``: ``"error"``
|
|
80
|
+
(the default) raises ``VariantFileError``, ``"skip"`` yields only the first, and ``"allow"`` yields every
|
|
81
|
+
row. Core's scoring preparation never writes a variant twice. Errors name the file and line.
|
|
82
|
+
"""
|
|
83
|
+
for row in read_variant_rows(path, duplicates=duplicates):
|
|
84
|
+
yield row.key
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def read_variant_rows(path: str | os.PathLike[str], *, duplicates: DuplicatePolicy = "error") -> Iterator[VariantRow]:
|
|
88
|
+
"""Stream ``VariantRow(line, key)`` records, as ``read_variants`` does, for callers that report lines.
|
|
89
|
+
|
|
90
|
+
A runtime that applies its own policy to each variant (for example, SNVs only) uses the line to name the
|
|
91
|
+
offending row in its error, in the same ``path:line`` form as this reader's errors.
|
|
92
|
+
"""
|
|
93
|
+
if duplicates not in _DUPLICATE_POLICIES:
|
|
94
|
+
msg = f"duplicates must be one of {sorted(_DUPLICATE_POLICIES)}, got {duplicates!r}"
|
|
95
|
+
raise ValueError(msg)
|
|
96
|
+
seen: set[str] = set()
|
|
97
|
+
# `utf-8-sig` drops a leading byte-order mark, which would otherwise become part of the first contig name.
|
|
98
|
+
with Path(path).open(newline="", encoding="utf-8-sig") as handle:
|
|
99
|
+
for line, row in _fields(handle, path):
|
|
100
|
+
if len(row) not in VARIANT_FIELD_COUNTS:
|
|
101
|
+
reason = f"expected 4 or 5 tab-separated fields (chr, pos, ref, alt[, label]), got {len(row)}"
|
|
102
|
+
raise _row_error(path, line, reason)
|
|
103
|
+
chromosome, raw_position, reference, alternate = row[:4]
|
|
104
|
+
try:
|
|
105
|
+
position = parse_position(raw_position)
|
|
106
|
+
except VariantIdentityError as error:
|
|
107
|
+
raise _row_error(path, line, f"invalid position {raw_position!r}") from error
|
|
108
|
+
try:
|
|
109
|
+
key = VariantKey.from_fields(chromosome, position, reference, alternate)
|
|
110
|
+
except VariantIdentityError as error:
|
|
111
|
+
raise _row_error(path, line, str(error)) from error
|
|
112
|
+
if duplicates != "allow":
|
|
113
|
+
if key.variant_id in seen:
|
|
114
|
+
if duplicates == "skip":
|
|
115
|
+
continue
|
|
116
|
+
raise _row_error(path, line, f"contains duplicate variant_id {key.variant_id!r}")
|
|
117
|
+
seen.add(key.variant_id)
|
|
118
|
+
yield VariantRow(line, key)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def batched(values: Iterable[T], size: int) -> Iterator[tuple[T, ...]]:
|
|
122
|
+
"""Yield consecutive tuples of at most ``size`` items without reading ahead of the current batch."""
|
|
123
|
+
if isinstance(size, bool) or not isinstance(size, int) or size < 1:
|
|
124
|
+
msg = f"batch size must be a positive integer, got {size!r}"
|
|
125
|
+
raise ValueError(msg)
|
|
126
|
+
batch: list[T] = []
|
|
127
|
+
for value in values:
|
|
128
|
+
batch.append(value)
|
|
129
|
+
if len(batch) == size:
|
|
130
|
+
yield tuple(batch)
|
|
131
|
+
batch = []
|
|
132
|
+
if batch:
|
|
133
|
+
yield tuple(batch)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
__all__ = [
|
|
137
|
+
"VARIANT_FIELD_COUNTS",
|
|
138
|
+
"DuplicatePolicy",
|
|
139
|
+
"VariantFileError",
|
|
140
|
+
"VariantRow",
|
|
141
|
+
"batched",
|
|
142
|
+
"read_variant_rows",
|
|
143
|
+
"read_variants",
|
|
144
|
+
]
|