osteosarc 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- osteosarc/__init__.py +34 -0
- osteosarc/__main__.py +3 -0
- osteosarc/cache.py +292 -0
- osteosarc/catalog.py +235 -0
- osteosarc/cli.py +180 -0
- osteosarc/curation.py +669 -0
- osteosarc/dataset.py +587 -0
- osteosarc/discovery.py +44 -0
- osteosarc/errors.py +21 -0
- osteosarc/explore.py +286 -0
- osteosarc/models.py +236 -0
- osteosarc/parsing.py +200 -0
- osteosarc/reads.py +402 -0
- osteosarc/timeline.py +369 -0
- osteosarc/urls.py +5 -0
- osteosarc-0.1.0.dist-info/METADATA +343 -0
- osteosarc-0.1.0.dist-info/RECORD +21 -0
- osteosarc-0.1.0.dist-info/WHEEL +5 -0
- osteosarc-0.1.0.dist-info/entry_points.txt +2 -0
- osteosarc-0.1.0.dist-info/licenses/LICENSE +201 -0
- osteosarc-0.1.0.dist-info/top_level.txt +1 -0
osteosarc/__init__.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Reproducible access to osteosarc.com. Importing performs no network I/O."""
|
|
2
|
+
|
|
3
|
+
from .cache import Cache, Receipt, digest
|
|
4
|
+
from .catalog import SNAPSHOT_SOURCES, TABLE_SOURCES, TIMELINE_SOURCES
|
|
5
|
+
from .curation import CORRECTIONS, Change, Correction, CurationWarning, glob
|
|
6
|
+
from .dataset import Dataset
|
|
7
|
+
from .discovery import list_bucket
|
|
8
|
+
from .errors import CoordinateError, IntegrityError, OfflineError, OsteosarcError, SchemaError
|
|
9
|
+
from .models import Asset, Assets, Region, SampleClaim, Variant, Variants
|
|
10
|
+
from .parsing import Table, parse_file, parse_table, parse_variant_index, parse_variants
|
|
11
|
+
from .reads import (
|
|
12
|
+
AlignmentInfo,
|
|
13
|
+
ReadFilter,
|
|
14
|
+
ReadSubset,
|
|
15
|
+
assembly_from_header,
|
|
16
|
+
extract_reads,
|
|
17
|
+
inspect_alignment,
|
|
18
|
+
normalize_assembly,
|
|
19
|
+
resolve_regions,
|
|
20
|
+
subset_templates,
|
|
21
|
+
)
|
|
22
|
+
from .timeline import Event, Timeline
|
|
23
|
+
|
|
24
|
+
__version__ = "0.1.0"
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"AlignmentInfo", "Asset", "Assets", "CORRECTIONS", "Cache", "Change", "CoordinateError", "Correction",
|
|
28
|
+
"CurationWarning", "Dataset", "Event", "IntegrityError", "TIMELINE_SOURCES", "Timeline", "glob",
|
|
29
|
+
"OfflineError", "OsteosarcError", "ReadFilter", "ReadSubset", "Receipt", "Region",
|
|
30
|
+
"SNAPSHOT_SOURCES", "SampleClaim", "SchemaError", "TABLE_SOURCES", "Table", "Variant",
|
|
31
|
+
"Variants", "assembly_from_header", "digest", "extract_reads", "inspect_alignment", "list_bucket",
|
|
32
|
+
"normalize_assembly", "parse_file", "parse_table", "parse_variant_index", "parse_variants",
|
|
33
|
+
"resolve_regions", "subset_templates",
|
|
34
|
+
]
|
osteosarc/__main__.py
ADDED
osteosarc/cache.py
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
"""Verified, shared, content-addressed downloads (curl; Linux and macOS).
|
|
2
|
+
|
|
3
|
+
Files live in the OpenVax shared cache, the same layout vaxrank and other
|
|
4
|
+
OpenVax tools use, so identical bytes are stored once:
|
|
5
|
+
|
|
6
|
+
<root>/objects/sha256/<sha256><original suffixes> shared content
|
|
7
|
+
<root>/osteosarc/... osteosarc's own records
|
|
8
|
+
|
|
9
|
+
<root> is OSTEOSARC_CACHE (an isolated cache), else OPENVAX_DATA_CACHE, else
|
|
10
|
+
the platform cache directory for "openvax" (as appdirs/datacache choose it).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import fcntl
|
|
16
|
+
import hashlib
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
import shutil
|
|
20
|
+
import subprocess
|
|
21
|
+
import sys
|
|
22
|
+
import tempfile
|
|
23
|
+
from contextlib import contextmanager
|
|
24
|
+
from dataclasses import asdict, dataclass
|
|
25
|
+
from datetime import datetime, timezone
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from urllib.parse import unquote, urlsplit
|
|
28
|
+
|
|
29
|
+
from .errors import IntegrityError, OfflineError
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def digest(path, algorithm="sha256"):
|
|
33
|
+
"""Hash a file incrementally, without loading large alignments into RAM."""
|
|
34
|
+
return digests(path, (algorithm,))[algorithm]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def digests(path, algorithms=("sha256",)):
|
|
38
|
+
"""Several hashes of a file in one read."""
|
|
39
|
+
results = {name: hashlib.new(name) for name in algorithms}
|
|
40
|
+
with Path(path).open("rb") as handle:
|
|
41
|
+
for block in iter(lambda: handle.read(1024 * 1024), b""):
|
|
42
|
+
for result in results.values():
|
|
43
|
+
result.update(block)
|
|
44
|
+
return {name: result.hexdigest() for name, result in results.items()}
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def file_identity(path):
|
|
48
|
+
"""Size, modification time and inode: changes whenever a file is rewritten."""
|
|
49
|
+
status = Path(path).stat()
|
|
50
|
+
return [str(Path(path).resolve()), status.st_size, status.st_mtime_ns, status.st_ino, status.st_dev]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
_UMASK = os.umask(0)
|
|
54
|
+
os.umask(_UMASK)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def share(path):
|
|
58
|
+
"""Give files the permissions the process umask allows (mkstemp/mkdtemp default to owner-only)."""
|
|
59
|
+
os.chmod(path, (0o777 if Path(path).is_dir() else 0o666) & ~_UMASK)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def default_root():
|
|
63
|
+
"""The shared OpenVax cache root, resolved exactly as vaxrank/datacache resolve it."""
|
|
64
|
+
if os.environ.get("OPENVAX_DATA_CACHE"):
|
|
65
|
+
return Path(os.environ["OPENVAX_DATA_CACHE"])
|
|
66
|
+
if sys.platform == "darwin":
|
|
67
|
+
return Path.home() / "Library" / "Caches" / "openvax"
|
|
68
|
+
return Path(os.environ.get("XDG_CACHE_HOME") or Path.home() / ".cache") / "openvax"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def object_name(sha256, filename):
|
|
72
|
+
"""OpenVax content name: the SHA256 followed by the original file's suffixes."""
|
|
73
|
+
return sha256 + "".join(Path(filename).suffixes)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def stable_id(value):
|
|
77
|
+
"""Return a deterministic SHA256 identity for a JSON-serializable value."""
|
|
78
|
+
return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def write_json(path, value):
|
|
82
|
+
"""Atomically replace a JSON file on the same filesystem."""
|
|
83
|
+
path = Path(path)
|
|
84
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
85
|
+
fd, name = tempfile.mkstemp(dir=path.parent, prefix=".json-")
|
|
86
|
+
try:
|
|
87
|
+
with os.fdopen(fd, "w") as handle:
|
|
88
|
+
json.dump(value, handle, indent=2, sort_keys=True)
|
|
89
|
+
handle.write("\n")
|
|
90
|
+
share(name)
|
|
91
|
+
os.replace(name, path)
|
|
92
|
+
finally:
|
|
93
|
+
Path(name).unlink(missing_ok=True)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@contextmanager
|
|
97
|
+
def file_lock(path):
|
|
98
|
+
"""Serialize writers; OS releases the advisory lock after interruption."""
|
|
99
|
+
path = Path(path)
|
|
100
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
101
|
+
with path.open("a") as handle:
|
|
102
|
+
fcntl.flock(handle, fcntl.LOCK_EX)
|
|
103
|
+
try:
|
|
104
|
+
yield
|
|
105
|
+
finally:
|
|
106
|
+
fcntl.flock(handle, fcntl.LOCK_UN)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass(frozen=True)
|
|
110
|
+
class Receipt:
|
|
111
|
+
url: str
|
|
112
|
+
sha256: str
|
|
113
|
+
size: int
|
|
114
|
+
filename: str
|
|
115
|
+
retrieved_at: str
|
|
116
|
+
etag: str | None = None
|
|
117
|
+
last_modified: str | None = None
|
|
118
|
+
md5: str | None = None
|
|
119
|
+
|
|
120
|
+
def to_dict(self):
|
|
121
|
+
return asdict(self)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class Cache:
|
|
125
|
+
"""A shared cache. Construction does no I/O; fetching is always explicit.
|
|
126
|
+
|
|
127
|
+
root defaults to OSTEOSARC_CACHE, else the OpenVax shared cache (see the
|
|
128
|
+
module docstring). Content is shared under objects/sha256; receipts,
|
|
129
|
+
snapshots and derived reads are kept under osteosarc/. Existing objects are
|
|
130
|
+
verified on every fetch. Refresh changes the URL's latest receipt but never
|
|
131
|
+
removes old snapshot objects.
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
def __init__(self, root=None, *, offline=False, timeout=600):
|
|
135
|
+
self.root = Path(root or os.environ.get("OSTEOSARC_CACHE") or default_root()).expanduser().resolve()
|
|
136
|
+
self.objects = self.root / "objects" / "sha256"
|
|
137
|
+
self.workspace = self.root / "osteosarc"
|
|
138
|
+
self.offline = offline
|
|
139
|
+
self.timeout = timeout
|
|
140
|
+
self._verified = {}
|
|
141
|
+
|
|
142
|
+
def path(self, receipt, *, verify=True):
|
|
143
|
+
"""Resolve a receipt, detecting missing or modified bytes.
|
|
144
|
+
|
|
145
|
+
Each object is hashed once per Cache instance; a rewritten file (new
|
|
146
|
+
size, mtime or inode) is hashed again.
|
|
147
|
+
"""
|
|
148
|
+
if not isinstance(receipt, Receipt):
|
|
149
|
+
receipt = Receipt(**receipt)
|
|
150
|
+
if (len(receipt.sha256) != 64 or any(c not in "0123456789abcdef" for c in receipt.sha256)
|
|
151
|
+
or Path(receipt.filename).name != receipt.filename or receipt.filename in ("", ".", "..")):
|
|
152
|
+
raise IntegrityError("Invalid cache receipt path")
|
|
153
|
+
path = self.objects / object_name(receipt.sha256, receipt.filename)
|
|
154
|
+
if not path.is_file():
|
|
155
|
+
raise OfflineError(f"Cached object is missing: {receipt.url}")
|
|
156
|
+
if verify:
|
|
157
|
+
identity = json.dumps(file_identity(path))
|
|
158
|
+
if self._verified.get(identity) != receipt.sha256:
|
|
159
|
+
if path.stat().st_size != receipt.size or digest(path) != receipt.sha256:
|
|
160
|
+
raise IntegrityError(f"Cached object was modified: {path}")
|
|
161
|
+
self._verified[identity] = receipt.sha256
|
|
162
|
+
return path
|
|
163
|
+
|
|
164
|
+
def file_digest(self, path):
|
|
165
|
+
"""SHA256 of a local file, remembered on disk by file identity.
|
|
166
|
+
|
|
167
|
+
Local BAMs are large; an unchanged file (same size, mtime and inode) is
|
|
168
|
+
not re-read. Any rewrite changes its identity and forces a new hash.
|
|
169
|
+
"""
|
|
170
|
+
identity = file_identity(path)
|
|
171
|
+
memo = self.workspace / "digests" / (stable_id(identity) + ".json")
|
|
172
|
+
if memo.is_file():
|
|
173
|
+
record = json.loads(memo.read_text())
|
|
174
|
+
if record.get("identity") == identity:
|
|
175
|
+
return record["sha256"]
|
|
176
|
+
value = digest(path)
|
|
177
|
+
if file_identity(path) == identity:
|
|
178
|
+
write_json(memo, dict(identity=identity, sha256=value))
|
|
179
|
+
return value
|
|
180
|
+
|
|
181
|
+
def fetch(self, url, *, refresh=False, sha256=None, md5=None, size=None, max_bytes=None):
|
|
182
|
+
"""Download or verify cached bytes, returning their immutable receipt.
|
|
183
|
+
|
|
184
|
+
Optional checksums are upstream claims checked against actual bytes;
|
|
185
|
+
ETags are retained as opaque HTTP identities, never treated as MD5s.
|
|
186
|
+
Interrupted downloads are discarded and retried on the next call.
|
|
187
|
+
"""
|
|
188
|
+
if urlsplit(url).scheme not in ("https", "http"):
|
|
189
|
+
raise ValueError("Downloads require an HTTP(S) URL; use Cache.import_file for local data")
|
|
190
|
+
if refresh and self.offline:
|
|
191
|
+
raise OfflineError("Cannot refresh in offline mode")
|
|
192
|
+
pointer = self.workspace / "urls" / (stable_id(url) + ".json")
|
|
193
|
+
with file_lock(self.workspace / "locks" / (stable_id(url) + ".lock")):
|
|
194
|
+
if pointer.exists() and not refresh:
|
|
195
|
+
receipt = Receipt(**json.loads(pointer.read_text()))
|
|
196
|
+
if receipt.url != url:
|
|
197
|
+
raise IntegrityError("Cache URL mismatch")
|
|
198
|
+
try:
|
|
199
|
+
path = self.path(receipt)
|
|
200
|
+
except OfflineError:
|
|
201
|
+
if self.offline:
|
|
202
|
+
raise
|
|
203
|
+
path = None # the object was pruned; download it again below
|
|
204
|
+
if path is not None:
|
|
205
|
+
# path() verified sha256; an md5 already checked when the object was stored is not re-read.
|
|
206
|
+
known_md5 = md5 and receipt.md5 and md5.lower() == receipt.md5.lower()
|
|
207
|
+
self._validate(path, sha256=None if sha256 == receipt.sha256 else sha256,
|
|
208
|
+
md5=None if known_md5 else md5, size=size, max_bytes=max_bytes)
|
|
209
|
+
return receipt
|
|
210
|
+
if self.offline:
|
|
211
|
+
raise OfflineError(f"Not cached: {url}")
|
|
212
|
+
staging = self.workspace / "staging"
|
|
213
|
+
staging.mkdir(parents=True, exist_ok=True)
|
|
214
|
+
with tempfile.TemporaryDirectory(dir=staging) as temporary:
|
|
215
|
+
path = Path(temporary) / "download"
|
|
216
|
+
headers = Path(temporary) / "headers"
|
|
217
|
+
command = ["curl", "--fail", "--location", "--silent", "--show-error",
|
|
218
|
+
"--proto", "=http,https", "--proto-redir", "=http,https",
|
|
219
|
+
"--connect-timeout", "30", "--max-time", str(self.timeout),
|
|
220
|
+
"--retry", "2", "--retry-all-errors",
|
|
221
|
+
"--dump-header", str(headers), "--output", str(path)]
|
|
222
|
+
if max_bytes is not None:
|
|
223
|
+
command += ["--max-filesize", str(max_bytes)]
|
|
224
|
+
subprocess.run(command + [url], check=True, capture_output=True,
|
|
225
|
+
timeout=self.timeout * 3 + 120)
|
|
226
|
+
hashes = digests(path, ("sha256", "md5") if md5 else ("sha256",))
|
|
227
|
+
self._validate(path, sha256=sha256, md5=md5, size=size, max_bytes=max_bytes, hashes=hashes)
|
|
228
|
+
# Keep the final HTTP response, rather than a redirect's headers.
|
|
229
|
+
final = headers.read_text().strip().split("\n\n")[-1]
|
|
230
|
+
fields = {k.lower(): v.strip() for line in final.splitlines()
|
|
231
|
+
if ":" in line for k, v in [line.split(":", 1)]}
|
|
232
|
+
receipt = self._store(path, url, fields, md5, move=True, checksum=hashes["sha256"])
|
|
233
|
+
write_json(pointer, receipt.to_dict())
|
|
234
|
+
return receipt
|
|
235
|
+
|
|
236
|
+
def import_file(self, path, url, *, sha256=None, md5=None, size=None):
|
|
237
|
+
"""Adopt an already downloaded file with its original source URL.
|
|
238
|
+
|
|
239
|
+
Does not assert when the upstream object was downloaded. retrieved_at
|
|
240
|
+
records the local import time. Useful for existing project caches.
|
|
241
|
+
"""
|
|
242
|
+
with file_lock(self.workspace / "locks" / (stable_id(url) + ".lock")):
|
|
243
|
+
path = Path(path)
|
|
244
|
+
self._validate(path, sha256=sha256, md5=md5, size=size)
|
|
245
|
+
receipt = self._store(path, url, {}, md5)
|
|
246
|
+
write_json(self.workspace / "urls" / (stable_id(url) + ".json"), receipt.to_dict())
|
|
247
|
+
return receipt
|
|
248
|
+
|
|
249
|
+
@staticmethod
|
|
250
|
+
def _validate(path, *, sha256=None, md5=None, size=None, max_bytes=None, hashes=None):
|
|
251
|
+
if size is not None and path.stat().st_size != size:
|
|
252
|
+
raise IntegrityError(f"Size mismatch: {path}")
|
|
253
|
+
if max_bytes is not None and path.stat().st_size > max_bytes:
|
|
254
|
+
raise IntegrityError(f"Download exceeds {max_bytes} bytes")
|
|
255
|
+
wanted = {name: value for name, value in (("sha256", sha256), ("md5", md5)) if value is not None}
|
|
256
|
+
if wanted:
|
|
257
|
+
hashes = dict(hashes or {})
|
|
258
|
+
missing = tuple(name for name in wanted if name not in hashes)
|
|
259
|
+
hashes.update(digests(path, missing) if missing else {})
|
|
260
|
+
for algorithm, expected in wanted.items():
|
|
261
|
+
if hashes[algorithm] != expected.lower():
|
|
262
|
+
raise IntegrityError(f"{algorithm} mismatch: {path}")
|
|
263
|
+
|
|
264
|
+
def _store(self, path, url, headers, md5, *, move=False, checksum=None):
|
|
265
|
+
checksum = checksum or digest(path)
|
|
266
|
+
filename = Path(unquote(urlsplit(url).path)).name or "index"
|
|
267
|
+
if filename in (".", ".."):
|
|
268
|
+
filename = "index"
|
|
269
|
+
receipt = Receipt(url, checksum, path.stat().st_size, filename,
|
|
270
|
+
datetime.now(timezone.utc).isoformat(), headers.get("etag"),
|
|
271
|
+
headers.get("last-modified"), md5)
|
|
272
|
+
output = self.objects / object_name(checksum, filename)
|
|
273
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
274
|
+
# The object may already be shared by another OpenVax tool; never rewrite valid bytes.
|
|
275
|
+
if output.is_file() and output.stat().st_size == receipt.size and digest(output) == checksum:
|
|
276
|
+
self._verified[json.dumps(file_identity(output))] = checksum
|
|
277
|
+
return receipt
|
|
278
|
+
if move:
|
|
279
|
+
share(path)
|
|
280
|
+
os.replace(path, output)
|
|
281
|
+
self._verified[json.dumps(file_identity(output))] = checksum
|
|
282
|
+
return receipt
|
|
283
|
+
fd, temporary = tempfile.mkstemp(dir=output.parent, prefix=".object-")
|
|
284
|
+
try:
|
|
285
|
+
with os.fdopen(fd, "wb") as handle, path.open("rb") as source:
|
|
286
|
+
shutil.copyfileobj(source, handle)
|
|
287
|
+
self._validate(Path(temporary), sha256=checksum, size=receipt.size)
|
|
288
|
+
share(temporary)
|
|
289
|
+
os.replace(temporary, output)
|
|
290
|
+
finally:
|
|
291
|
+
Path(temporary).unlink(missing_ok=True)
|
|
292
|
+
return receipt
|
osteosarc/catalog.py
ADDED
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
"""Reconcile all bucket objects with explicit, source-attributed metadata."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections import Counter, defaultdict
|
|
7
|
+
from pathlib import PurePosixPath
|
|
8
|
+
from urllib.parse import quote, unquote, urlsplit
|
|
9
|
+
|
|
10
|
+
from .cache import stable_id
|
|
11
|
+
from .curation import (
|
|
12
|
+
ASSAYS,
|
|
13
|
+
label_assay,
|
|
14
|
+
label_claims,
|
|
15
|
+
normalize_provider,
|
|
16
|
+
normalize_timepoint,
|
|
17
|
+
normalize_tissue,
|
|
18
|
+
)
|
|
19
|
+
from .errors import SchemaError
|
|
20
|
+
from .models import Asset, Assets, SampleClaim
|
|
21
|
+
from .urls import BUCKET, SITE, SOURCE_REPO
|
|
22
|
+
|
|
23
|
+
SNAPSHOT_SOURCES = {
|
|
24
|
+
"bams": SITE + "bams/bams.json",
|
|
25
|
+
"bucket": SITE + "bucket_listing.json",
|
|
26
|
+
"variant_index": SITE + "variants/",
|
|
27
|
+
"vafs": SITE + "variants/variant_vafs_long.tsv",
|
|
28
|
+
"vaf_columns": SITE + "variants/variant_vafs_long.columns.tsv",
|
|
29
|
+
"vaccine_overlap": SITE + "data/vaccine_overlap.json",
|
|
30
|
+
"source_variants": SOURCE_REPO + "src/data/variants.json",
|
|
31
|
+
"bam_metadata": SOURCE_REPO + "scripts/data/bam-metadata-consolidated.tsv",
|
|
32
|
+
"data_page": SITE + "data/",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
#: Dated sources behind the timeline and specimen views (about 1.7 MB).
|
|
36
|
+
#: Snapshots created before these existed still open; their timeline is unavailable.
|
|
37
|
+
TIMELINE_SOURCES = {
|
|
38
|
+
"events": SITE + "data/events.json",
|
|
39
|
+
"events_sheet": SOURCE_REPO + "scripts/timeline/timeline.csv",
|
|
40
|
+
"mrd": SITE + "data/mrd.json",
|
|
41
|
+
"specimens": SOURCE_REPO + "scripts/data/samples-consolidated.tsv",
|
|
42
|
+
"timepoint_summary": SOURCE_REPO + "src/data/samples.json",
|
|
43
|
+
"fastqs": SOURCE_REPO + "scripts/data/fastqs-consolidated.tsv",
|
|
44
|
+
"flow": SITE + "data/flow/manifest.json",
|
|
45
|
+
"imaging": SOURCE_REPO + "src/data/dicom-studies.json",
|
|
46
|
+
"pathology": SOURCE_REPO + "src/data/pathology-slides.json",
|
|
47
|
+
"labs": SITE + "data/lab_results.tsv",
|
|
48
|
+
"cytometry": SITE + "data/cytometry.tsv",
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
TABLE_SOURCES = {
|
|
52
|
+
"vafs": (SNAPSHOT_SOURCES["vafs"], "tsv"),
|
|
53
|
+
"vaf_columns": (SNAPSHOT_SOURCES["vaf_columns"], "tsv"),
|
|
54
|
+
"snv_top": (SITE + "oncoanalyser/tables/snv_top.tsv", "tsv"),
|
|
55
|
+
"dna_fusions": (SITE + "fusions/tables/prioritized.tsv", "tsv"),
|
|
56
|
+
"rna_fusions": (SITE + "ctat_lr_fusion/tables/lr_fusions_concordance.tsv", "tsv"),
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def asset_type(key):
|
|
61
|
+
"""Classify file format separately from scientific interpretation."""
|
|
62
|
+
lower = key.lower()
|
|
63
|
+
if lower.endswith((".bai", ".csi", ".crai", ".tbi", ".fai")):
|
|
64
|
+
return "index", lower.rsplit(".", 1)[-1]
|
|
65
|
+
base = lower.removesuffix(".gz").removesuffix(".bgz")
|
|
66
|
+
if base.endswith((".genes.results", ".isoforms.results")):
|
|
67
|
+
return "expression", "tsv"
|
|
68
|
+
suffix = base.rsplit(".", 1)[-1] if "." in base else ""
|
|
69
|
+
if suffix in ("bam", "cram", "sam"):
|
|
70
|
+
return "alignment", suffix
|
|
71
|
+
if suffix in ("vcf", "bcf"):
|
|
72
|
+
return "variants", suffix
|
|
73
|
+
if suffix in ("fastq", "fq"):
|
|
74
|
+
return "reads", "fastq"
|
|
75
|
+
if suffix in ("fasta", "fa", "fna", "faa"):
|
|
76
|
+
return "reference", "fasta"
|
|
77
|
+
if suffix in ("gtf", "gff", "gff3", "bed"):
|
|
78
|
+
return "annotation", suffix
|
|
79
|
+
if suffix in ("tsv", "csv", "json"):
|
|
80
|
+
return "table", suffix
|
|
81
|
+
if suffix in ("h5", "h5ad", "h5mu", "rds", "mtx", "cloupe"):
|
|
82
|
+
return "expression", suffix
|
|
83
|
+
return "other", suffix
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def bucket_url(key, base=BUCKET):
|
|
87
|
+
"""Encode an exact S3 object key once; '+' and spaces remain distinct."""
|
|
88
|
+
return base.rstrip("/") + "/" + quote(key, safe="/")
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def object_key(value, base=BUCKET):
|
|
92
|
+
if value.startswith(base):
|
|
93
|
+
return unquote(value[len(base):])
|
|
94
|
+
if urlsplit(value).scheme:
|
|
95
|
+
raise SchemaError(f"Object is outside the catalog bucket: {value}")
|
|
96
|
+
return value
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def build_assets(listing, bams, metadata, vafs, path_claims=(), *, tables=None):
|
|
100
|
+
"""Retain every listed object, enriching exact paths before basename matches.
|
|
101
|
+
|
|
102
|
+
Basename joins are used only when unique among alignment objects. Path
|
|
103
|
+
inferences are explicitly marked and excluded from default metadata filters.
|
|
104
|
+
The global viewer genome is retained as a claim, not assigned as assembly.
|
|
105
|
+
"""
|
|
106
|
+
base = listing.get("download_base", BUCKET)
|
|
107
|
+
objects = {}
|
|
108
|
+
for row in listing["files"]:
|
|
109
|
+
if len(row) < 3 or row[0] in objects:
|
|
110
|
+
raise SchemaError("Duplicate or malformed bucket object")
|
|
111
|
+
objects[row[0]] = dict(size=int(row[1]), modified=row[2])
|
|
112
|
+
catalog, claims, extra = {}, defaultdict(list), defaultdict(dict)
|
|
113
|
+
for category in bams["categories"]:
|
|
114
|
+
for row in category["bams"]:
|
|
115
|
+
key = object_key(row["url"], bams["baseUrl"])
|
|
116
|
+
if key in catalog:
|
|
117
|
+
raise SchemaError(f"Duplicate catalog alignment: {key}")
|
|
118
|
+
catalog[key] = row
|
|
119
|
+
objects.setdefault(key, dict(size=None, modified=None))
|
|
120
|
+
assay, platform = ASSAYS.get(category["name"], (None, None))
|
|
121
|
+
match = re.match(r"(T\d+)\b", row["name"])
|
|
122
|
+
claims[key].append(SampleClaim("bams", row["name"], match[1] if match else None,
|
|
123
|
+
assay=label_assay(row["name"], assay), platform=platform,
|
|
124
|
+
tissue=normalize_tissue(row.get("tissue")),
|
|
125
|
+
**label_claims(row["name"])))
|
|
126
|
+
extra[key].update(catalog=row, category=category["name"],
|
|
127
|
+
catalog_genome_assertion=bams.get("genome"))
|
|
128
|
+
for row in metadata:
|
|
129
|
+
if not row.get("s3_path"):
|
|
130
|
+
continue
|
|
131
|
+
key = object_key(row["s3_path"], base)
|
|
132
|
+
objects.setdefault(key, dict(size=None, modified=None))
|
|
133
|
+
assay, platform = ASSAYS.get(row.get("assay"), (None, None))
|
|
134
|
+
claims[key].append(SampleClaim("bam_metadata", row["display_name"], normalize_timepoint(row.get("timepoint")),
|
|
135
|
+
row.get("sample_date") or None, assay, platform,
|
|
136
|
+
normalize_tissue(row.get("tissue")), normalize_provider(row.get("provider"))))
|
|
137
|
+
extra[key].setdefault("metadata_rows", []).append(row)
|
|
138
|
+
counts = Counter(PurePosixPath(key).name for key in objects if asset_type(key)[0] == "alignment")
|
|
139
|
+
by_basename = defaultdict(set)
|
|
140
|
+
for row in vafs:
|
|
141
|
+
by_basename[row["bam_file"]].add(tuple(row.get(k, "") for k in
|
|
142
|
+
("sample_label", "timepoint", "sample_date", "assay_type", "tissue", "data_source")))
|
|
143
|
+
by_path = defaultdict(list)
|
|
144
|
+
for prefix, claim in path_claims:
|
|
145
|
+
by_path[prefix.rstrip("/")].append(claim)
|
|
146
|
+
assets = []
|
|
147
|
+
for key, object_metadata in objects.items():
|
|
148
|
+
kind, format = asset_type(key)
|
|
149
|
+
info = dict(extra.get(key, {}))
|
|
150
|
+
records = list(claims.get(key, ()))
|
|
151
|
+
# The most specific data-page path wins: a file's own row overrides the
|
|
152
|
+
# row for its directory, which can also hold other assays' files.
|
|
153
|
+
parts = key.split("/")
|
|
154
|
+
for depth in range(len(parts), 0, -1):
|
|
155
|
+
if "/".join(parts[:depth]) in by_path:
|
|
156
|
+
records.extend(by_path["/".join(parts[:depth])])
|
|
157
|
+
break
|
|
158
|
+
if kind == "alignment":
|
|
159
|
+
basename = PurePosixPath(key).name
|
|
160
|
+
if counts[basename] == 1:
|
|
161
|
+
for label, timepoint, date, assay_name, tissue, provider in sorted(by_basename[basename]):
|
|
162
|
+
assay, platform = ASSAYS.get(assay_name, (None, None))
|
|
163
|
+
records.append(SampleClaim("vafs", label, normalize_timepoint(timepoint), date or None,
|
|
164
|
+
assay, platform, normalize_tissue(tissue), normalize_provider(provider)))
|
|
165
|
+
elif by_basename[basename]:
|
|
166
|
+
info["ambiguous_vaf_basename"] = basename
|
|
167
|
+
# This inference is useful for discovery, but never establishes identity.
|
|
168
|
+
points = sorted(set(re.findall(r"(?:^|[/_ .-])(T[0-3])(?=[/_ .-]|$)", key)))
|
|
169
|
+
libraries = sorted(set(re.findall(r"\b(?:BG\d{6}|SARC\d{4}|TL-\d{2}-[A-Z0-9]+)\b", key)))
|
|
170
|
+
if kind != "other":
|
|
171
|
+
for point in points:
|
|
172
|
+
records.append(SampleClaim("bucket_path", key, timepoint=point, basis="inferred"))
|
|
173
|
+
for library in libraries:
|
|
174
|
+
records.append(SampleClaim("bucket_path", key, library=library, basis="inferred"))
|
|
175
|
+
suffixes = [key + ".bai", key[:-4] + ".bai", key + ".csi"] if format == "bam" else (
|
|
176
|
+
[key + ".crai", key[:-5] + ".crai", key + ".csi"] if format == "cram" else
|
|
177
|
+
[key + ".tbi", key + ".csi"] if format in ("vcf", "bcf") else [])
|
|
178
|
+
indexes = tuple(bucket_url(k, base) for k in suffixes if k in objects)
|
|
179
|
+
url = bucket_url(key, base)
|
|
180
|
+
assets.append(Asset(stable_id(url), key, url, kind, format, index_urls=indexes,
|
|
181
|
+
claims=tuple(records), metadata=info, **object_metadata))
|
|
182
|
+
for name, (url, format) in (tables or TABLE_SOURCES).items():
|
|
183
|
+
assets.append(Asset(stable_id(url), "site/" + name, url, "table", format,
|
|
184
|
+
metadata={"resource": name}))
|
|
185
|
+
return Assets(assets)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def data_page_rows(html):
|
|
189
|
+
"""Yield (section context, {column: text}) for data-page rows with a bucket path."""
|
|
190
|
+
from bs4 import BeautifulSoup
|
|
191
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
192
|
+
headings = {}
|
|
193
|
+
for element in soup.find_all(["h2", "h3", "h4", "table"]):
|
|
194
|
+
if element.name != "table":
|
|
195
|
+
level = int(element.name[1])
|
|
196
|
+
headings = {k: v for k, v in headings.items() if k < level}
|
|
197
|
+
headings[level] = element.get_text(" ", strip=True)
|
|
198
|
+
continue
|
|
199
|
+
context = " / ".join(headings.values())
|
|
200
|
+
headers = [c.get_text(" ", strip=True) for c in element.select("thead th")]
|
|
201
|
+
if not headers:
|
|
202
|
+
headers = [c.get_text(" ", strip=True) for c in element.select("tr:first-child th")]
|
|
203
|
+
for row in element.select("tr"):
|
|
204
|
+
cells = row.find_all("td", recursive=False)
|
|
205
|
+
if len(cells) != len(headers):
|
|
206
|
+
continue
|
|
207
|
+
values = dict(zip(headers, (c.get_text(" ", strip=True) for c in cells)))
|
|
208
|
+
if "Bucket Path" in values:
|
|
209
|
+
yield context, values
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def parse_data_paths(html):
|
|
213
|
+
"""Extract the data page's explicit path-to-sample claims, including FASTQs.
|
|
214
|
+
|
|
215
|
+
Assays/platforms come from section headings, tissues/providers from cells.
|
|
216
|
+
Composite tissues and timepoints remain unresolved rather than guessed.
|
|
217
|
+
"""
|
|
218
|
+
result = []
|
|
219
|
+
for context, values in data_page_rows(html):
|
|
220
|
+
title = context.lower()
|
|
221
|
+
assay = ("wgs" if "whole genome" in title or re.search(r"\bwgs\b", title) else
|
|
222
|
+
"wes" if "whole exome" in title or re.search(r"\bwes\b", title) else
|
|
223
|
+
"scrna-seq" if any(x in title for x in ("single cell", "single-cell")) else
|
|
224
|
+
"rna-seq" if "bulk rna" in title else None)
|
|
225
|
+
platform = ("ont" if "nanopore" in title else "pacbio" if "pacbio" in title else
|
|
226
|
+
"illumina" if "illumina" in title else None)
|
|
227
|
+
points = set(re.findall(r"\bT[0-3]\b", values.get("Timepoint", "")))
|
|
228
|
+
tissue = normalize_tissue(values.get("Tissue"))
|
|
229
|
+
label = " | ".join(v for k, v in values.items() if k not in ("Bucket Path", "Size", "Files"))
|
|
230
|
+
result.append((values["Bucket Path"].strip("`"), SampleClaim(
|
|
231
|
+
"data_page", context + " / " + label,
|
|
232
|
+
timepoint=next(iter(points)) if len(points) == 1 else None,
|
|
233
|
+
assay=assay, platform=platform, tissue=tissue,
|
|
234
|
+
provider=normalize_provider(values.get("Provider")))))
|
|
235
|
+
return tuple(result)
|