altar-identity 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,32 @@
1
+ *.pyc
2
+ *~
3
+ **/__pycache__/*
4
+ *.swp
5
+ .vscode/
6
+ .idea/
7
+ .DS_Store
8
+ .env*
9
+ .mypy_cache/
10
+ .ruff_cache/
11
+ .pytest_cache/
12
+ .coverage
13
+ htmlcov/
14
+ coverage.xml
15
+ pytest.xml
16
+ .hypothesis/
17
+ .python-version
18
+ .venv/
19
+ .venv-*/
20
+ *.egg-info/
21
+ build/
22
+ dist/
23
+ site/
24
+ *.sqlite3
25
+ *.log
26
+
27
+ # Generated variant indexes and datasets. The small canonical gene table is tracked.
28
+ altar/altar/variants/data/ccres.dnatree
29
+ altar/altar/variants/data/region_annotations.parquet
30
+ altar/altar/variants/data/variants.pkl.gz
31
+ altar/altar/variants/data/raw/
32
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Riya Sinha
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,114 @@
1
+ Metadata-Version: 2.5
2
+ Name: altar-identity
3
+ Version: 0.1.0
4
+ Summary: Dependency-free variant and content identity shared by Altar and its model runtimes
5
+ Project-URL: Documentation, https://kundajelab.github.io/altar/
6
+ Project-URL: Issues, https://github.com/kundajelab/altar/issues
7
+ Project-URL: Repository, https://github.com/kundajelab/altar
8
+ Author: Riya Sinha
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: bioinformatics,genomics,variant-identity
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.9
24
+ Description-Content-Type: text/markdown
25
+
26
+ # altar-identity
27
+
28
+ `altar-identity` holds the identity rules that Altar core and every Altar model runtime must agree on. It uses
29
+ only the Python standard library and supports Python 3.9 and later, so a runtime pinned to an older
30
+ TensorFlow or PyTorch stack runs the same code as Altar core instead of keeping a copy.
31
+
32
+ Most users do not install it directly: `altar` depends on it and re-exports the variant names from
33
+ `altar.variants`, `altar.models`, and `altar.sources`.
34
+
35
+ ## Variant identity
36
+
37
+ ```python
38
+ from altar_identity import VariantKey, canonical_chromosome, canonical_variant_id
39
+
40
+ canonical_chromosome("MT") # "chrM"
41
+ canonical_variant_id("1", 10, "a", "t") # "chr1:10:A:T"
42
+ VariantKey.require_canonical("chr1:10:A:T") # rejects aliases such as "1:10:A:T"
43
+ ```
44
+
45
+ A key is `chromosome:position:REF:ALT` with a one-based position, and every field is ASCII. Surrounding ASCII
46
+ whitespace is trimmed from each field; other whitespace and non-ASCII text raise `VariantIdentityError`, a
47
+ `ValueError`, rather than being folded (an Arabic-Indic or full-width digit never becomes `1`).
48
+
49
+ - **Chromosome.** The `chr` prefix is optional and case-insensitive. Primary chromosomes are normalized in any
50
+ case (`1`, `chr01`, `CHR1` → `chr1`; `M`, `MT` → `chrM`). Every other contig keeps its exact spelling after
51
+ the prefix (`CHRUn_KI270302v1` → `chrUn_KI270302v1`), because reference contig names are case-sensitive and
52
+ the key must match the name in the FASTA. The name uses the VCF contig-name characters: letters, digits, and
53
+ `!#$%&*+./;=?@^_|~-`, which exclude `:`.
54
+ - **Position.** A positive `int`. In text (`VariantKey.parse`, variant files, and `parse_position` for other
55
+ readers) it is ASCII digits `[0-9]+`: `+5`, `1_000`, `1e3`, and non-ASCII digits are rejected, although
56
+ Python's `int()` accepts some of them.
57
+ - **Alleles.** Uppercased, then `[ACGTN]+`, the VCF base alphabet for REF and a concrete ALT. Symbolic
58
+ (`<DEL>`), `*`, `.`, breakend, IUPAC-ambiguity and `-` alleles are rejected, and REF must differ from ALT.
59
+ `N` is allowed in the key; reference validation decides whether a keyed variant can be scored.
60
+
61
+ The key does not left-align or trim indels, and it does not carry the genome build.
62
+
63
+ ## Variant files
64
+
65
+ ```python
66
+ from altar_identity import batched, read_variants
67
+
68
+ for batch in batched(read_variants("variants.tsv"), 1024):
69
+ ...
70
+ ```
71
+
72
+ Altar writes the variants for a model container as a headerless, tab-separated UTF-8 file with the columns
73
+ `chr`, `pos`, `ref`, `alt`, and `variant_id`. `read_variants` also accepts four columns, or any label in the
74
+ fifth, and never uses the fifth column. Fields are never quoted: a `"` is an ordinary character, and a field
75
+ cannot hold a tab or line break. A leading UTF-8 byte-order mark is ignored. The reader yields canonical
76
+ `VariantKey` values, skips blank lines, and raises `VariantFileError` naming the file and line for a malformed
77
+ row. By default a repeated `variant_id` is an
78
+ error; pass `duplicates="skip"` to keep the first occurrence or `duplicates="allow"` to yield every row.
79
+ `read_variant_rows` yields `VariantRow(line, key)` records instead, for callers that apply their own policy
80
+ (for example, SNVs only) and need to report the offending line. An empty or blank-only file yields nothing.
81
+
82
+ ## Content identity
83
+
84
+ ```python
85
+ from altar_identity import sha256_file, verify_file
86
+
87
+ digest = sha256_file("weights.h5") # "sha256:<64 lowercase hex>"
88
+ verify_file("weights.h5", digest, label="weights")
89
+ ```
90
+
91
+ A digest is `sha256:` followed by 64 lowercase hexadecimal digits and names the bytes of one regular file.
92
+ `verify_file` raises `DigestMismatchError` when the bytes differ, and also when the path is missing or is a
93
+ directory. Symlinks are followed. `parse_sha256_digest` and `is_sha256_digest` check the exact spelling, and
94
+ `SHA256_DIGEST_PATTERN` is the same rule as an unanchored regular expression for schemas that embed it.
95
+
96
+ A file that many tasks read, such as a reference genome on a shared volume, need not be hashed by every task.
97
+ `verify_file(..., trust_record=True, record=True)` accepts a file whose verification record is current and
98
+ writes a record after a successful hash. The record is a one-line file at `<path>.sha256-verified`:
99
+
100
+ ```text
101
+ verified-file/1 sha256:<64 hex> size=<bytes> mtime=<seconds> ctime=<seconds> inode=<number>
102
+ ```
103
+
104
+ Rewriting, replacing, truncating, or appending to the file changes a recorded value and invalidates the record.
105
+ `has_verification_record`, `write_verification_record`, and `verification_record_line` are the primitives Altar
106
+ core's `verify_file_digest`, its staging backends, and the runtimes share. Anyone who can write the storage
107
+ can also forge a record, so do not trust records where bytes first arrive, such as a download.
108
+
109
+ ## Development
110
+
111
+ ```bash
112
+ uv run --isolated --no-project --python 3.9 --with pytest --with-editable identity \
113
+ python -m pytest identity/tests
114
+ ```
@@ -0,0 +1,89 @@
1
+ # altar-identity
2
+
3
+ `altar-identity` holds the identity rules that Altar core and every Altar model runtime must agree on. It uses
4
+ only the Python standard library and supports Python 3.9 and later, so a runtime pinned to an older
5
+ TensorFlow or PyTorch stack runs the same code as Altar core instead of keeping a copy.
6
+
7
+ Most users do not install it directly: `altar` depends on it and re-exports the variant names from
8
+ `altar.variants`, `altar.models`, and `altar.sources`.
9
+
10
+ ## Variant identity
11
+
12
+ ```python
13
+ from altar_identity import VariantKey, canonical_chromosome, canonical_variant_id
14
+
15
+ canonical_chromosome("MT") # "chrM"
16
+ canonical_variant_id("1", 10, "a", "t") # "chr1:10:A:T"
17
+ VariantKey.require_canonical("chr1:10:A:T") # rejects aliases such as "1:10:A:T"
18
+ ```
19
+
20
+ A key is `chromosome:position:REF:ALT` with a one-based position, and every field is ASCII. Surrounding ASCII
21
+ whitespace is trimmed from each field; other whitespace and non-ASCII text raise `VariantIdentityError`, a
22
+ `ValueError`, rather than being folded (an Arabic-Indic or full-width digit never becomes `1`).
23
+
24
+ - **Chromosome.** The `chr` prefix is optional and case-insensitive. Primary chromosomes are normalized in any
25
+ case (`1`, `chr01`, `CHR1` → `chr1`; `M`, `MT` → `chrM`). Every other contig keeps its exact spelling after
26
+ the prefix (`CHRUn_KI270302v1` → `chrUn_KI270302v1`), because reference contig names are case-sensitive and
27
+ the key must match the name in the FASTA. The name uses the VCF contig-name characters: letters, digits, and
28
+ `!#$%&*+./;=?@^_|~-`, which exclude `:`.
29
+ - **Position.** A positive `int`. In text (`VariantKey.parse`, variant files, and `parse_position` for other
30
+ readers) it is ASCII digits `[0-9]+`: `+5`, `1_000`, `1e3`, and non-ASCII digits are rejected, although
31
+ Python's `int()` accepts some of them.
32
+ - **Alleles.** Uppercased, then `[ACGTN]+`, the VCF base alphabet for REF and a concrete ALT. Symbolic
33
+ (`<DEL>`), `*`, `.`, breakend, IUPAC-ambiguity and `-` alleles are rejected, and REF must differ from ALT.
34
+ `N` is allowed in the key; reference validation decides whether a keyed variant can be scored.
35
+
36
+ The key does not left-align or trim indels, and it does not carry the genome build.
37
+
38
+ ## Variant files
39
+
40
+ ```python
41
+ from altar_identity import batched, read_variants
42
+
43
+ for batch in batched(read_variants("variants.tsv"), 1024):
44
+ ...
45
+ ```
46
+
47
+ Altar writes the variants for a model container as a headerless, tab-separated UTF-8 file with the columns
48
+ `chr`, `pos`, `ref`, `alt`, and `variant_id`. `read_variants` also accepts four columns, or any label in the
49
+ fifth, and never uses the fifth column. Fields are never quoted: a `"` is an ordinary character, and a field
50
+ cannot hold a tab or line break. A leading UTF-8 byte-order mark is ignored. The reader yields canonical
51
+ `VariantKey` values, skips blank lines, and raises `VariantFileError` naming the file and line for a malformed
52
+ row. By default a repeated `variant_id` is an
53
+ error; pass `duplicates="skip"` to keep the first occurrence or `duplicates="allow"` to yield every row.
54
+ `read_variant_rows` yields `VariantRow(line, key)` records instead, for callers that apply their own policy
55
+ (for example, SNVs only) and need to report the offending line. An empty or blank-only file yields nothing.
56
+
57
+ ## Content identity
58
+
59
+ ```python
60
+ from altar_identity import sha256_file, verify_file
61
+
62
+ digest = sha256_file("weights.h5") # "sha256:<64 lowercase hex>"
63
+ verify_file("weights.h5", digest, label="weights")
64
+ ```
65
+
66
+ A digest is `sha256:` followed by 64 lowercase hexadecimal digits and names the bytes of one regular file.
67
+ `verify_file` raises `DigestMismatchError` when the bytes differ, and also when the path is missing or is a
68
+ directory. Symlinks are followed. `parse_sha256_digest` and `is_sha256_digest` check the exact spelling, and
69
+ `SHA256_DIGEST_PATTERN` is the same rule as an unanchored regular expression for schemas that embed it.
70
+
71
+ A file that many tasks read, such as a reference genome on a shared volume, need not be hashed by every task.
72
+ `verify_file(..., trust_record=True, record=True)` accepts a file whose verification record is current and
73
+ writes a record after a successful hash. The record is a one-line file at `<path>.sha256-verified`:
74
+
75
+ ```text
76
+ verified-file/1 sha256:<64 hex> size=<bytes> mtime=<seconds> ctime=<seconds> inode=<number>
77
+ ```
78
+
79
+ Rewriting, replacing, truncating, or appending to the file changes a recorded value and invalidates the record.
80
+ `has_verification_record`, `write_verification_record`, and `verification_record_line` are the primitives Altar
81
+ core's `verify_file_digest`, its staging backends, and the runtimes share. Anyone who can write the storage
82
+ can also forge a record, so do not trust records where bytes first arrive, such as a download.
83
+
84
+ ## Development
85
+
86
+ ```bash
87
+ uv run --isolated --no-project --python 3.9 --with pytest --with-editable identity \
88
+ python -m pytest identity/tests
89
+ ```
@@ -0,0 +1,75 @@
1
+ # Copyright 2026 Riya Sinha
2
+ """Variant and content identity shared by Altar core and its model runtimes.
3
+
4
+ This package has no dependencies and supports Python 3.9, so a runtime pinned to an older scientific stack
5
+ builds variant IDs, reads Altar's variant files, and verifies staged files with the same code as Altar core.
6
+
7
+ - ``VariantKey``, ``canonical_chromosome``, ``canonical_variant_id``, ``parse_position``: the canonical
8
+ ``chr:pos:REF:ALT`` key and its textual position rule.
9
+ - ``read_variants``, ``read_variant_rows``, ``batched``: the headerless variant file Altar hands to a model container.
10
+ - ``sha256_file``, ``verify_file``, ``parse_sha256_digest``: the ``sha256:<hex>`` identity of one regular file,
11
+ and the verification records that let many tasks share one hash of a staged file. ``SHA256_DIGEST_PATTERN``
12
+ is the one spelling of the digest rule, for schemas and patterns that embed it.
13
+ """
14
+
15
+ from altar_identity.digest import (
16
+ SHA256_DIGEST_PATTERN,
17
+ SHA256_PREFIX,
18
+ VERIFICATION_RECORD_FORMAT,
19
+ VERIFICATION_RECORD_SUFFIX,
20
+ DigestMismatchError,
21
+ has_verification_record,
22
+ is_sha256_digest,
23
+ parse_sha256_digest,
24
+ sha256_file,
25
+ verification_record_line,
26
+ verify_file,
27
+ write_verification_record,
28
+ )
29
+ from altar_identity.tsv import (
30
+ VARIANT_FIELD_COUNTS,
31
+ DuplicatePolicy,
32
+ VariantFileError,
33
+ VariantRow,
34
+ batched,
35
+ read_variant_rows,
36
+ read_variants,
37
+ )
38
+ from altar_identity.variant import (
39
+ VariantIdentityError,
40
+ VariantKey,
41
+ canonical_chromosome,
42
+ canonical_variant_id,
43
+ parse_position,
44
+ )
45
+
46
+
47
+ __version__: str = "0.1.0"
48
+
49
+ __all__ = [
50
+ "SHA256_DIGEST_PATTERN",
51
+ "SHA256_PREFIX",
52
+ "VARIANT_FIELD_COUNTS",
53
+ "VERIFICATION_RECORD_FORMAT",
54
+ "VERIFICATION_RECORD_SUFFIX",
55
+ "DigestMismatchError",
56
+ "DuplicatePolicy",
57
+ "VariantFileError",
58
+ "VariantIdentityError",
59
+ "VariantKey",
60
+ "VariantRow",
61
+ "__version__",
62
+ "batched",
63
+ "canonical_chromosome",
64
+ "canonical_variant_id",
65
+ "has_verification_record",
66
+ "is_sha256_digest",
67
+ "parse_position",
68
+ "parse_sha256_digest",
69
+ "read_variant_rows",
70
+ "read_variants",
71
+ "sha256_file",
72
+ "verification_record_line",
73
+ "verify_file",
74
+ "write_verification_record",
75
+ ]
@@ -0,0 +1,186 @@
1
+ # Copyright 2026 Riya Sinha
2
+ """Content identity for staged files: one ``sha256:<64 lowercase hex>`` digest per regular file.
3
+
4
+ A digest names the bytes of exactly one regular file. Symlinks are followed. A path that is missing or is a
5
+ directory fails verification the same way as a digest mismatch, because neither has the bytes the digest
6
+ names; ship a multi-file resource as an archive and digest the archive.
7
+
8
+ Verification records
9
+ --------------------
10
+ A file read by many tasks, such as a reference genome, need not be hashed by every task. A verification record
11
+ is a one-line text file at the file's path plus ``VERIFICATION_RECORD_SUFFIX``::
12
+
13
+ verified-file/1 sha256:<64 hex> size=<bytes> mtime=<seconds> ctime=<seconds> inode=<number>
14
+
15
+ It states that the bytes matched the digest and pins the file's size, whole-second modification and
16
+ status-change times, and inode as they were then. Rewriting, replacing, truncating, or appending to the file
17
+ changes at least one of those values, which invalidates the record. Altar core's ``verify_file_digest``, its
18
+ shell verifier, and every runtime read and write this same line through this module.
19
+ """
20
+
21
+ from __future__ import annotations
22
+ import contextlib
23
+ import hashlib
24
+ import os
25
+ import re
26
+ import stat
27
+ import time
28
+ import uuid
29
+ from pathlib import Path
30
+
31
+
32
+ SHA256_PREFIX = "sha256:"
33
+ SHA256_DIGEST_PATTERN = f"{SHA256_PREFIX}[0-9a-f]{{64}}"
34
+ """Unanchored regular expression for a canonical digest. Anchor it (``re.fullmatch``, or ``^...$`` in a schema)
35
+ or embed it in a larger pattern, such as an image reference, instead of spelling the digest rule again."""
36
+ VERIFICATION_RECORD_SUFFIX = ".sha256-verified"
37
+ VERIFICATION_RECORD_FORMAT = "verified-file/1"
38
+ _SHA256_DIGEST = re.compile(SHA256_DIGEST_PATTERN)
39
+ _BUFFER_SIZE = 1024 * 1024
40
+ _RECORD_MAX_BYTES = 512
41
+
42
+
43
+ class DigestMismatchError(ValueError):
44
+ """A path does not hold the single regular file whose bytes a digest names."""
45
+
46
+
47
+ def is_sha256_digest(value: object) -> bool:
48
+ """Return whether ``value`` is exactly ``sha256:`` followed by 64 lowercase hexadecimal digits."""
49
+ return isinstance(value, str) and _SHA256_DIGEST.fullmatch(value) is not None
50
+
51
+
52
+ def parse_sha256_digest(value: str) -> str:
53
+ """Return ``value`` unchanged if it is a canonical SHA-256 digest; otherwise raise ``ValueError``.
54
+
55
+ The check is exact: no surrounding whitespace, no trailing newline, and no uppercase hexadecimal.
56
+ """
57
+ if not is_sha256_digest(value):
58
+ msg = f"expected sha256:<64 lowercase hex>, got {value!r}"
59
+ raise ValueError(msg)
60
+ return value
61
+
62
+
63
+ def sha256_file(path: str | os.PathLike[str]) -> str:
64
+ """Return the ``sha256:<hex>`` digest of the bytes at ``path``, reading it in bounded chunks."""
65
+ digest = hashlib.sha256()
66
+ with Path(path).open("rb") as handle:
67
+ for chunk in iter(lambda: handle.read(_BUFFER_SIZE), b""):
68
+ digest.update(chunk)
69
+ return f"{SHA256_PREFIX}{digest.hexdigest()}"
70
+
71
+
72
+ def verification_record_line(digest: str, status: os.stat_result) -> str:
73
+ """Return the verification-record line for ``digest`` and a file's ``os.stat`` result.
74
+
75
+ ``int()`` of the float times matches the whole seconds ``stat -c %Y`` and ``%Z`` print for any file written
76
+ after 1970, so a shell verifier renders the same line.
77
+ """
78
+ return (
79
+ f"{VERIFICATION_RECORD_FORMAT} {digest} size={status.st_size} mtime={int(status.st_mtime)} "
80
+ f"ctime={int(status.st_ctime)} inode={status.st_ino}"
81
+ )
82
+
83
+
84
+ def _record_path(path: str | os.PathLike[str]) -> Path:
85
+ return Path(f"{os.fspath(path)}{VERIFICATION_RECORD_SUFFIX}")
86
+
87
+
88
+ def has_verification_record(path: str | os.PathLike[str], digest: str) -> bool:
89
+ """Return whether `path` is a regular file with a current verification record for `digest`.
90
+
91
+ The record must name `digest` and match the file's current size, modification time, status-change time,
92
+ and inode number. A missing, unreadable, or stale record returns `False`. Checking a record reads two
93
+ small pieces of metadata and never reads the file's bytes. Symlinks are followed.
94
+ """
95
+ try:
96
+ status = Path(path).stat()
97
+ if not stat.S_ISREG(status.st_mode):
98
+ return False
99
+ with _record_path(path).open("rb") as handle:
100
+ content = handle.read(_RECORD_MAX_BYTES)
101
+ except OSError:
102
+ return False
103
+ return content.decode("ascii", "replace").rstrip("\n") == verification_record_line(digest, status)
104
+
105
+
106
+ def write_verification_record(path: str | os.PathLike[str], digest: str, before: os.stat_result) -> bool:
107
+ """Write a record for bytes that just matched `digest`. Returns whether a record was written.
108
+
109
+ `before` is the file's status from before it was hashed. No record is written if the file changed while it
110
+ was being hashed. No record is written either while the file's status-change second is still the current
111
+ second: a later change within that same second would leave every recorded value unchanged. The next check
112
+ after that second writes the record instead. The write is atomic and best effort. A directory the caller
113
+ cannot write only means the next check hashes the file again.
114
+ """
115
+ try:
116
+ after = Path(path).stat()
117
+ except OSError:
118
+ return False
119
+ line = verification_record_line(digest, after)
120
+ if line != verification_record_line(digest, before) or int(after.st_ctime) >= int(time.time()):
121
+ return False
122
+ record = _record_path(path)
123
+ temporary = record.with_name(f"{record.name}.{os.getpid()}.{uuid.uuid4().hex}.tmp")
124
+ try:
125
+ temporary.write_text(f"{line}\n", encoding="ascii")
126
+ temporary.replace(record)
127
+ except OSError:
128
+ with contextlib.suppress(OSError):
129
+ temporary.unlink()
130
+ return False
131
+ return True
132
+
133
+
134
+ def verify_file(
135
+ path: str | os.PathLike[str],
136
+ expected: str,
137
+ *,
138
+ label: str = "resource",
139
+ trust_record: bool = False,
140
+ record: bool = False,
141
+ ) -> bool:
142
+ """Raise ``DigestMismatchError`` unless ``path`` is one regular file whose bytes have digest ``expected``.
143
+
144
+ ``expected`` must be a canonical SHA-256 digest; a malformed one raises ``ValueError``. ``label`` names
145
+ the file in the error message.
146
+
147
+ By default the file is always hashed. With ``trust_record``, a current verification record accepts the file
148
+ without reading it, so a file staged once and read by many tasks is hashed once. With ``record``, a
149
+ successful hash writes a record for later checks. Anyone who can write the storage can also write a record,
150
+ so trust records only on storage whose writers you trust, and never where bytes first arrive (a download).
151
+
152
+ Returns ``True`` when the file was hashed and ``False`` when a record accepted it.
153
+ """
154
+ parse_sha256_digest(expected)
155
+ if trust_record and has_verification_record(path, expected):
156
+ return False
157
+ if not Path(path).is_file():
158
+ msg = (
159
+ f"{label} at {str(path)!r} is not a regular file; a sha256 digest covers the bytes of one file, "
160
+ "so ship a directory as an archive"
161
+ )
162
+ raise DigestMismatchError(msg)
163
+ before = Path(path).stat()
164
+ actual = sha256_file(path)
165
+ if actual != expected:
166
+ msg = f"{label} at {str(path)!r} has digest {actual}; expected {expected}"
167
+ raise DigestMismatchError(msg)
168
+ if record:
169
+ write_verification_record(path, expected, before)
170
+ return True
171
+
172
+
173
+ __all__ = [
174
+ "SHA256_DIGEST_PATTERN",
175
+ "SHA256_PREFIX",
176
+ "VERIFICATION_RECORD_FORMAT",
177
+ "VERIFICATION_RECORD_SUFFIX",
178
+ "DigestMismatchError",
179
+ "has_verification_record",
180
+ "is_sha256_digest",
181
+ "parse_sha256_digest",
182
+ "sha256_file",
183
+ "verification_record_line",
184
+ "verify_file",
185
+ "write_verification_record",
186
+ ]
File without changes
@@ -0,0 +1,144 @@
1
+ # Copyright 2026 Riya Sinha
2
+ """The canonical variant file that Altar hands to a model container.
3
+
4
+ Altar core's scoring preparation writes each batch as a headerless, tab-separated UTF-8 file with one variant
5
+ per line: ``chr``, ``pos`` (one-based), ``ref``, ``alt``, and the canonical ``variant_id``. The reader also
6
+ accepts rows without the fifth column, and rows whose fifth column is any other label, because callers may
7
+ hand a container a hand-written candidate file. The fifth column is never used: identity always comes from
8
+ the locus fields.
9
+
10
+ Core's scoring preparation reads its candidate files through this same reader, which applies ``VariantKey``'s
11
+ field rules (see ``altar_identity.variant``). It trims fields and uppercases alleles, normalizes chromosome aliases
12
+ (``1``, ``chr01`` and ``CHR1`` all become ``chr1``), and reads the position as ASCII digits. It rejects rows that
13
+ have no canonical spelling, such as a contig containing ``:``, a position below one or written as ``+5`` or
14
+ ``1_000``, or an allele outside ``[ACGTN]``. Blank lines and a leading UTF-8 byte-order mark are skipped. Fields
15
+ are never quoted: a ``"`` is an ordinary character, and a field cannot contain a tab, carriage return, or newline.
16
+ """
17
+
18
+ from __future__ import annotations
19
+ import csv
20
+ from pathlib import Path
21
+ from typing import TYPE_CHECKING, Literal, NamedTuple, TypeVar
22
+
23
+ from altar_identity.variant import VariantIdentityError, VariantKey, parse_position
24
+
25
+
26
+ if TYPE_CHECKING:
27
+ import os
28
+ from collections.abc import Iterable, Iterator
29
+
30
+
31
+ T = TypeVar("T")
32
+ VARIANT_FIELD_COUNTS = frozenset({4, 5})
33
+ DuplicatePolicy = Literal["error", "skip", "allow"]
34
+ _DUPLICATE_POLICIES = frozenset({"error", "skip", "allow"})
35
+
36
+
37
+ class VariantFileError(ValueError):
38
+ """A variant file row is malformed, has no canonical identity, or repeats a variant.
39
+
40
+ The message is ``path:line: reason``. ``line`` and ``reason`` are also attributes, so a caller that reports
41
+ errors in its own words (for example, without a temporary local path) need not parse the message.
42
+ """
43
+
44
+ def __init__(self, message: str, *, line: int | None = None, reason: str | None = None) -> None:
45
+ super().__init__(message)
46
+ self.line = line
47
+ self.reason = message if reason is None else reason
48
+
49
+
50
+ def _row_error(path: str | os.PathLike[str], line: int, reason: str) -> VariantFileError:
51
+ return VariantFileError(f"{path}:{line}: {reason}", line=line, reason=reason)
52
+
53
+
54
+ class VariantRow(NamedTuple):
55
+ """One canonical variant and the one-based file line it was read from."""
56
+
57
+ line: int
58
+ key: VariantKey
59
+
60
+
61
+ def _fields(handle: Iterable[str], path: str | os.PathLike[str]) -> Iterator[tuple[int, list[str]]]:
62
+ """Yield ``(line, fields)`` for each non-blank row, split on tabs with no quoting."""
63
+ # No quoting: a `"` is an ordinary character, so one stray quote cannot swallow the lines after it.
64
+ reader = csv.reader(handle, delimiter="\t", quoting=csv.QUOTE_NONE)
65
+ while True:
66
+ try:
67
+ row = next(reader)
68
+ except StopIteration:
69
+ return
70
+ except csv.Error as error:
71
+ raise _row_error(path, reader.line_num, f"unreadable row: {error}") from error
72
+ if row:
73
+ yield reader.line_num, row
74
+
75
+
76
+ def read_variants(path: str | os.PathLike[str], *, duplicates: DuplicatePolicy = "error") -> Iterator[VariantKey]:
77
+ """Stream canonical ``VariantKey`` values from a headerless ``chr pos ref alt [label]`` file.
78
+
79
+ ``duplicates`` sets what happens when two rows have the same canonical ``variant_id``: ``"error"``
80
+ (the default) raises ``VariantFileError``, ``"skip"`` yields only the first, and ``"allow"`` yields every
81
+ row. Core's scoring preparation never writes a variant twice. Errors name the file and line.
82
+ """
83
+ for row in read_variant_rows(path, duplicates=duplicates):
84
+ yield row.key
85
+
86
+
87
+ def read_variant_rows(path: str | os.PathLike[str], *, duplicates: DuplicatePolicy = "error") -> Iterator[VariantRow]:
88
+ """Stream ``VariantRow(line, key)`` records, as ``read_variants`` does, for callers that report lines.
89
+
90
+ A runtime that applies its own policy to each variant (for example, SNVs only) uses the line to name the
91
+ offending row in its error, in the same ``path:line`` form as this reader's errors.
92
+ """
93
+ if duplicates not in _DUPLICATE_POLICIES:
94
+ msg = f"duplicates must be one of {sorted(_DUPLICATE_POLICIES)}, got {duplicates!r}"
95
+ raise ValueError(msg)
96
+ seen: set[str] = set()
97
+ # `utf-8-sig` drops a leading byte-order mark, which would otherwise become part of the first contig name.
98
+ with Path(path).open(newline="", encoding="utf-8-sig") as handle:
99
+ for line, row in _fields(handle, path):
100
+ if len(row) not in VARIANT_FIELD_COUNTS:
101
+ reason = f"expected 4 or 5 tab-separated fields (chr, pos, ref, alt[, label]), got {len(row)}"
102
+ raise _row_error(path, line, reason)
103
+ chromosome, raw_position, reference, alternate = row[:4]
104
+ try:
105
+ position = parse_position(raw_position)
106
+ except VariantIdentityError as error:
107
+ raise _row_error(path, line, f"invalid position {raw_position!r}") from error
108
+ try:
109
+ key = VariantKey.from_fields(chromosome, position, reference, alternate)
110
+ except VariantIdentityError as error:
111
+ raise _row_error(path, line, str(error)) from error
112
+ if duplicates != "allow":
113
+ if key.variant_id in seen:
114
+ if duplicates == "skip":
115
+ continue
116
+ raise _row_error(path, line, f"contains duplicate variant_id {key.variant_id!r}")
117
+ seen.add(key.variant_id)
118
+ yield VariantRow(line, key)
119
+
120
+
121
+ def batched(values: Iterable[T], size: int) -> Iterator[tuple[T, ...]]:
122
+ """Yield consecutive tuples of at most ``size`` items without reading ahead of the current batch."""
123
+ if isinstance(size, bool) or not isinstance(size, int) or size < 1:
124
+ msg = f"batch size must be a positive integer, got {size!r}"
125
+ raise ValueError(msg)
126
+ batch: list[T] = []
127
+ for value in values:
128
+ batch.append(value)
129
+ if len(batch) == size:
130
+ yield tuple(batch)
131
+ batch = []
132
+ if batch:
133
+ yield tuple(batch)
134
+
135
+
136
+ __all__ = [
137
+ "VARIANT_FIELD_COUNTS",
138
+ "DuplicatePolicy",
139
+ "VariantFileError",
140
+ "VariantRow",
141
+ "batched",
142
+ "read_variant_rows",
143
+ "read_variants",
144
+ ]