osteosarc 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
osteosarc/__init__.py ADDED
@@ -0,0 +1,34 @@
1
+ """Reproducible access to osteosarc.com. Importing performs no network I/O."""
2
+
3
+ from .cache import Cache, Receipt, digest
4
+ from .catalog import SNAPSHOT_SOURCES, TABLE_SOURCES, TIMELINE_SOURCES
5
+ from .curation import CORRECTIONS, Change, Correction, CurationWarning, glob
6
+ from .dataset import Dataset
7
+ from .discovery import list_bucket
8
+ from .errors import CoordinateError, IntegrityError, OfflineError, OsteosarcError, SchemaError
9
+ from .models import Asset, Assets, Region, SampleClaim, Variant, Variants
10
+ from .parsing import Table, parse_file, parse_table, parse_variant_index, parse_variants
11
+ from .reads import (
12
+ AlignmentInfo,
13
+ ReadFilter,
14
+ ReadSubset,
15
+ assembly_from_header,
16
+ extract_reads,
17
+ inspect_alignment,
18
+ normalize_assembly,
19
+ resolve_regions,
20
+ subset_templates,
21
+ )
22
+ from .timeline import Event, Timeline
23
+
24
+ __version__ = "0.1.0"
25
+
26
+ __all__ = [
27
+ "AlignmentInfo", "Asset", "Assets", "CORRECTIONS", "Cache", "Change", "CoordinateError", "Correction",
28
+ "CurationWarning", "Dataset", "Event", "IntegrityError", "TIMELINE_SOURCES", "Timeline", "glob",
29
+ "OfflineError", "OsteosarcError", "ReadFilter", "ReadSubset", "Receipt", "Region",
30
+ "SNAPSHOT_SOURCES", "SampleClaim", "SchemaError", "TABLE_SOURCES", "Table", "Variant",
31
+ "Variants", "assembly_from_header", "digest", "extract_reads", "inspect_alignment", "list_bucket",
32
+ "normalize_assembly", "parse_file", "parse_table", "parse_variant_index", "parse_variants",
33
+ "resolve_regions", "subset_templates",
34
+ ]
osteosarc/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
osteosarc/cache.py ADDED
@@ -0,0 +1,292 @@
1
+ """Verified, shared, content-addressed downloads (curl; Linux and macOS).
2
+
3
+ Files live in the OpenVax shared cache, the same layout vaxrank and other
4
+ OpenVax tools use, so identical bytes are stored once:
5
+
6
+ <root>/objects/sha256/<sha256><original suffixes> shared content
7
+ <root>/osteosarc/... osteosarc's own records
8
+
9
+ <root> is OSTEOSARC_CACHE (an isolated cache), else OPENVAX_DATA_CACHE, else
10
+ the platform cache directory for "openvax" (as appdirs/datacache choose it).
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import fcntl
16
+ import hashlib
17
+ import json
18
+ import os
19
+ import shutil
20
+ import subprocess
21
+ import sys
22
+ import tempfile
23
+ from contextlib import contextmanager
24
+ from dataclasses import asdict, dataclass
25
+ from datetime import datetime, timezone
26
+ from pathlib import Path
27
+ from urllib.parse import unquote, urlsplit
28
+
29
+ from .errors import IntegrityError, OfflineError
30
+
31
+
32
+ def digest(path, algorithm="sha256"):
33
+ """Hash a file incrementally, without loading large alignments into RAM."""
34
+ return digests(path, (algorithm,))[algorithm]
35
+
36
+
37
+ def digests(path, algorithms=("sha256",)):
38
+ """Several hashes of a file in one read."""
39
+ results = {name: hashlib.new(name) for name in algorithms}
40
+ with Path(path).open("rb") as handle:
41
+ for block in iter(lambda: handle.read(1024 * 1024), b""):
42
+ for result in results.values():
43
+ result.update(block)
44
+ return {name: result.hexdigest() for name, result in results.items()}
45
+
46
+
47
+ def file_identity(path):
48
+ """Size, modification time and inode: changes whenever a file is rewritten."""
49
+ status = Path(path).stat()
50
+ return [str(Path(path).resolve()), status.st_size, status.st_mtime_ns, status.st_ino, status.st_dev]
51
+
52
+
53
+ _UMASK = os.umask(0)
54
+ os.umask(_UMASK)
55
+
56
+
57
+ def share(path):
58
+ """Give files the permissions the process umask allows (mkstemp/mkdtemp default to owner-only)."""
59
+ os.chmod(path, (0o777 if Path(path).is_dir() else 0o666) & ~_UMASK)
60
+
61
+
62
+ def default_root():
63
+ """The shared OpenVax cache root, resolved exactly as vaxrank/datacache resolve it."""
64
+ if os.environ.get("OPENVAX_DATA_CACHE"):
65
+ return Path(os.environ["OPENVAX_DATA_CACHE"])
66
+ if sys.platform == "darwin":
67
+ return Path.home() / "Library" / "Caches" / "openvax"
68
+ return Path(os.environ.get("XDG_CACHE_HOME") or Path.home() / ".cache") / "openvax"
69
+
70
+
71
+ def object_name(sha256, filename):
72
+ """OpenVax content name: the SHA256 followed by the original file's suffixes."""
73
+ return sha256 + "".join(Path(filename).suffixes)
74
+
75
+
76
+ def stable_id(value):
77
+ """Return a deterministic SHA256 identity for a JSON-serializable value."""
78
+ return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
79
+
80
+
81
+ def write_json(path, value):
82
+ """Atomically replace a JSON file on the same filesystem."""
83
+ path = Path(path)
84
+ path.parent.mkdir(parents=True, exist_ok=True)
85
+ fd, name = tempfile.mkstemp(dir=path.parent, prefix=".json-")
86
+ try:
87
+ with os.fdopen(fd, "w") as handle:
88
+ json.dump(value, handle, indent=2, sort_keys=True)
89
+ handle.write("\n")
90
+ share(name)
91
+ os.replace(name, path)
92
+ finally:
93
+ Path(name).unlink(missing_ok=True)
94
+
95
+
96
+ @contextmanager
97
+ def file_lock(path):
98
+ """Serialize writers; OS releases the advisory lock after interruption."""
99
+ path = Path(path)
100
+ path.parent.mkdir(parents=True, exist_ok=True)
101
+ with path.open("a") as handle:
102
+ fcntl.flock(handle, fcntl.LOCK_EX)
103
+ try:
104
+ yield
105
+ finally:
106
+ fcntl.flock(handle, fcntl.LOCK_UN)
107
+
108
+
109
+ @dataclass(frozen=True)
110
+ class Receipt:
111
+ url: str
112
+ sha256: str
113
+ size: int
114
+ filename: str
115
+ retrieved_at: str
116
+ etag: str | None = None
117
+ last_modified: str | None = None
118
+ md5: str | None = None
119
+
120
+ def to_dict(self):
121
+ return asdict(self)
122
+
123
+
124
+ class Cache:
125
+ """A shared cache. Construction does no I/O; fetching is always explicit.
126
+
127
+ root defaults to OSTEOSARC_CACHE, else the OpenVax shared cache (see the
128
+ module docstring). Content is shared under objects/sha256; receipts,
129
+ snapshots and derived reads are kept under osteosarc/. Existing objects are
130
+ verified on every fetch. Refresh changes the URL's latest receipt but never
131
+ removes old snapshot objects.
132
+ """
133
+
134
+ def __init__(self, root=None, *, offline=False, timeout=600):
135
+ self.root = Path(root or os.environ.get("OSTEOSARC_CACHE") or default_root()).expanduser().resolve()
136
+ self.objects = self.root / "objects" / "sha256"
137
+ self.workspace = self.root / "osteosarc"
138
+ self.offline = offline
139
+ self.timeout = timeout
140
+ self._verified = {}
141
+
142
+ def path(self, receipt, *, verify=True):
143
+ """Resolve a receipt, detecting missing or modified bytes.
144
+
145
+ Each object is hashed once per Cache instance; a rewritten file (new
146
+ size, mtime or inode) is hashed again.
147
+ """
148
+ if not isinstance(receipt, Receipt):
149
+ receipt = Receipt(**receipt)
150
+ if (len(receipt.sha256) != 64 or any(c not in "0123456789abcdef" for c in receipt.sha256)
151
+ or Path(receipt.filename).name != receipt.filename or receipt.filename in ("", ".", "..")):
152
+ raise IntegrityError("Invalid cache receipt path")
153
+ path = self.objects / object_name(receipt.sha256, receipt.filename)
154
+ if not path.is_file():
155
+ raise OfflineError(f"Cached object is missing: {receipt.url}")
156
+ if verify:
157
+ identity = json.dumps(file_identity(path))
158
+ if self._verified.get(identity) != receipt.sha256:
159
+ if path.stat().st_size != receipt.size or digest(path) != receipt.sha256:
160
+ raise IntegrityError(f"Cached object was modified: {path}")
161
+ self._verified[identity] = receipt.sha256
162
+ return path
163
+
164
+ def file_digest(self, path):
165
+ """SHA256 of a local file, remembered on disk by file identity.
166
+
167
+ Local BAMs are large; an unchanged file (same size, mtime and inode) is
168
+ not re-read. Any rewrite changes its identity and forces a new hash.
169
+ """
170
+ identity = file_identity(path)
171
+ memo = self.workspace / "digests" / (stable_id(identity) + ".json")
172
+ if memo.is_file():
173
+ record = json.loads(memo.read_text())
174
+ if record.get("identity") == identity:
175
+ return record["sha256"]
176
+ value = digest(path)
177
+ if file_identity(path) == identity:
178
+ write_json(memo, dict(identity=identity, sha256=value))
179
+ return value
180
+
181
+ def fetch(self, url, *, refresh=False, sha256=None, md5=None, size=None, max_bytes=None):
182
+ """Download or verify cached bytes, returning their immutable receipt.
183
+
184
+ Optional checksums are upstream claims checked against actual bytes;
185
+ ETags are retained as opaque HTTP identities, never treated as MD5s.
186
+ Interrupted downloads are discarded and retried on the next call.
187
+ """
188
+ if urlsplit(url).scheme not in ("https", "http"):
189
+ raise ValueError("Downloads require an HTTP(S) URL; use Cache.import_file for local data")
190
+ if refresh and self.offline:
191
+ raise OfflineError("Cannot refresh in offline mode")
192
+ pointer = self.workspace / "urls" / (stable_id(url) + ".json")
193
+ with file_lock(self.workspace / "locks" / (stable_id(url) + ".lock")):
194
+ if pointer.exists() and not refresh:
195
+ receipt = Receipt(**json.loads(pointer.read_text()))
196
+ if receipt.url != url:
197
+ raise IntegrityError("Cache URL mismatch")
198
+ try:
199
+ path = self.path(receipt)
200
+ except OfflineError:
201
+ if self.offline:
202
+ raise
203
+ path = None # the object was pruned; download it again below
204
+ if path is not None:
205
+ # path() verified sha256; an md5 already checked when the object was stored is not re-read.
206
+ known_md5 = md5 and receipt.md5 and md5.lower() == receipt.md5.lower()
207
+ self._validate(path, sha256=None if sha256 == receipt.sha256 else sha256,
208
+ md5=None if known_md5 else md5, size=size, max_bytes=max_bytes)
209
+ return receipt
210
+ if self.offline:
211
+ raise OfflineError(f"Not cached: {url}")
212
+ staging = self.workspace / "staging"
213
+ staging.mkdir(parents=True, exist_ok=True)
214
+ with tempfile.TemporaryDirectory(dir=staging) as temporary:
215
+ path = Path(temporary) / "download"
216
+ headers = Path(temporary) / "headers"
217
+ command = ["curl", "--fail", "--location", "--silent", "--show-error",
218
+ "--proto", "=http,https", "--proto-redir", "=http,https",
219
+ "--connect-timeout", "30", "--max-time", str(self.timeout),
220
+ "--retry", "2", "--retry-all-errors",
221
+ "--dump-header", str(headers), "--output", str(path)]
222
+ if max_bytes is not None:
223
+ command += ["--max-filesize", str(max_bytes)]
224
+ subprocess.run(command + [url], check=True, capture_output=True,
225
+ timeout=self.timeout * 3 + 120)
226
+ hashes = digests(path, ("sha256", "md5") if md5 else ("sha256",))
227
+ self._validate(path, sha256=sha256, md5=md5, size=size, max_bytes=max_bytes, hashes=hashes)
228
+ # Keep the final HTTP response, rather than a redirect's headers.
229
+ final = headers.read_text().strip().split("\n\n")[-1]
230
+ fields = {k.lower(): v.strip() for line in final.splitlines()
231
+ if ":" in line for k, v in [line.split(":", 1)]}
232
+ receipt = self._store(path, url, fields, md5, move=True, checksum=hashes["sha256"])
233
+ write_json(pointer, receipt.to_dict())
234
+ return receipt
235
+
236
+ def import_file(self, path, url, *, sha256=None, md5=None, size=None):
237
+ """Adopt an already downloaded file with its original source URL.
238
+
239
+ Does not assert when the upstream object was downloaded. retrieved_at
240
+ records the local import time. Useful for existing project caches.
241
+ """
242
+ with file_lock(self.workspace / "locks" / (stable_id(url) + ".lock")):
243
+ path = Path(path)
244
+ self._validate(path, sha256=sha256, md5=md5, size=size)
245
+ receipt = self._store(path, url, {}, md5)
246
+ write_json(self.workspace / "urls" / (stable_id(url) + ".json"), receipt.to_dict())
247
+ return receipt
248
+
249
+ @staticmethod
250
+ def _validate(path, *, sha256=None, md5=None, size=None, max_bytes=None, hashes=None):
251
+ if size is not None and path.stat().st_size != size:
252
+ raise IntegrityError(f"Size mismatch: {path}")
253
+ if max_bytes is not None and path.stat().st_size > max_bytes:
254
+ raise IntegrityError(f"Download exceeds {max_bytes} bytes")
255
+ wanted = {name: value for name, value in (("sha256", sha256), ("md5", md5)) if value is not None}
256
+ if wanted:
257
+ hashes = dict(hashes or {})
258
+ missing = tuple(name for name in wanted if name not in hashes)
259
+ hashes.update(digests(path, missing) if missing else {})
260
+ for algorithm, expected in wanted.items():
261
+ if hashes[algorithm] != expected.lower():
262
+ raise IntegrityError(f"{algorithm} mismatch: {path}")
263
+
264
+ def _store(self, path, url, headers, md5, *, move=False, checksum=None):
265
+ checksum = checksum or digest(path)
266
+ filename = Path(unquote(urlsplit(url).path)).name or "index"
267
+ if filename in (".", ".."):
268
+ filename = "index"
269
+ receipt = Receipt(url, checksum, path.stat().st_size, filename,
270
+ datetime.now(timezone.utc).isoformat(), headers.get("etag"),
271
+ headers.get("last-modified"), md5)
272
+ output = self.objects / object_name(checksum, filename)
273
+ output.parent.mkdir(parents=True, exist_ok=True)
274
+ # The object may already be shared by another OpenVax tool; never rewrite valid bytes.
275
+ if output.is_file() and output.stat().st_size == receipt.size and digest(output) == checksum:
276
+ self._verified[json.dumps(file_identity(output))] = checksum
277
+ return receipt
278
+ if move:
279
+ share(path)
280
+ os.replace(path, output)
281
+ self._verified[json.dumps(file_identity(output))] = checksum
282
+ return receipt
283
+ fd, temporary = tempfile.mkstemp(dir=output.parent, prefix=".object-")
284
+ try:
285
+ with os.fdopen(fd, "wb") as handle, path.open("rb") as source:
286
+ shutil.copyfileobj(source, handle)
287
+ self._validate(Path(temporary), sha256=checksum, size=receipt.size)
288
+ share(temporary)
289
+ os.replace(temporary, output)
290
+ finally:
291
+ Path(temporary).unlink(missing_ok=True)
292
+ return receipt
osteosarc/catalog.py ADDED
@@ -0,0 +1,235 @@
1
+ """Reconcile all bucket objects with explicit, source-attributed metadata."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections import Counter, defaultdict
7
+ from pathlib import PurePosixPath
8
+ from urllib.parse import quote, unquote, urlsplit
9
+
10
+ from .cache import stable_id
11
+ from .curation import (
12
+ ASSAYS,
13
+ label_assay,
14
+ label_claims,
15
+ normalize_provider,
16
+ normalize_timepoint,
17
+ normalize_tissue,
18
+ )
19
+ from .errors import SchemaError
20
+ from .models import Asset, Assets, SampleClaim
21
+ from .urls import BUCKET, SITE, SOURCE_REPO
22
+
23
+ SNAPSHOT_SOURCES = {
24
+ "bams": SITE + "bams/bams.json",
25
+ "bucket": SITE + "bucket_listing.json",
26
+ "variant_index": SITE + "variants/",
27
+ "vafs": SITE + "variants/variant_vafs_long.tsv",
28
+ "vaf_columns": SITE + "variants/variant_vafs_long.columns.tsv",
29
+ "vaccine_overlap": SITE + "data/vaccine_overlap.json",
30
+ "source_variants": SOURCE_REPO + "src/data/variants.json",
31
+ "bam_metadata": SOURCE_REPO + "scripts/data/bam-metadata-consolidated.tsv",
32
+ "data_page": SITE + "data/",
33
+ }
34
+
35
+ #: Dated sources behind the timeline and specimen views (about 1.7 MB).
36
+ #: Snapshots created before these existed still open; their timeline is unavailable.
37
+ TIMELINE_SOURCES = {
38
+ "events": SITE + "data/events.json",
39
+ "events_sheet": SOURCE_REPO + "scripts/timeline/timeline.csv",
40
+ "mrd": SITE + "data/mrd.json",
41
+ "specimens": SOURCE_REPO + "scripts/data/samples-consolidated.tsv",
42
+ "timepoint_summary": SOURCE_REPO + "src/data/samples.json",
43
+ "fastqs": SOURCE_REPO + "scripts/data/fastqs-consolidated.tsv",
44
+ "flow": SITE + "data/flow/manifest.json",
45
+ "imaging": SOURCE_REPO + "src/data/dicom-studies.json",
46
+ "pathology": SOURCE_REPO + "src/data/pathology-slides.json",
47
+ "labs": SITE + "data/lab_results.tsv",
48
+ "cytometry": SITE + "data/cytometry.tsv",
49
+ }
50
+
51
+ TABLE_SOURCES = {
52
+ "vafs": (SNAPSHOT_SOURCES["vafs"], "tsv"),
53
+ "vaf_columns": (SNAPSHOT_SOURCES["vaf_columns"], "tsv"),
54
+ "snv_top": (SITE + "oncoanalyser/tables/snv_top.tsv", "tsv"),
55
+ "dna_fusions": (SITE + "fusions/tables/prioritized.tsv", "tsv"),
56
+ "rna_fusions": (SITE + "ctat_lr_fusion/tables/lr_fusions_concordance.tsv", "tsv"),
57
+ }
58
+
59
+
60
+ def asset_type(key):
61
+ """Classify file format separately from scientific interpretation."""
62
+ lower = key.lower()
63
+ if lower.endswith((".bai", ".csi", ".crai", ".tbi", ".fai")):
64
+ return "index", lower.rsplit(".", 1)[-1]
65
+ base = lower.removesuffix(".gz").removesuffix(".bgz")
66
+ if base.endswith((".genes.results", ".isoforms.results")):
67
+ return "expression", "tsv"
68
+ suffix = base.rsplit(".", 1)[-1] if "." in base else ""
69
+ if suffix in ("bam", "cram", "sam"):
70
+ return "alignment", suffix
71
+ if suffix in ("vcf", "bcf"):
72
+ return "variants", suffix
73
+ if suffix in ("fastq", "fq"):
74
+ return "reads", "fastq"
75
+ if suffix in ("fasta", "fa", "fna", "faa"):
76
+ return "reference", "fasta"
77
+ if suffix in ("gtf", "gff", "gff3", "bed"):
78
+ return "annotation", suffix
79
+ if suffix in ("tsv", "csv", "json"):
80
+ return "table", suffix
81
+ if suffix in ("h5", "h5ad", "h5mu", "rds", "mtx", "cloupe"):
82
+ return "expression", suffix
83
+ return "other", suffix
84
+
85
+
86
+ def bucket_url(key, base=BUCKET):
87
+ """Encode an exact S3 object key once; '+' and spaces remain distinct."""
88
+ return base.rstrip("/") + "/" + quote(key, safe="/")
89
+
90
+
91
+ def object_key(value, base=BUCKET):
92
+ if value.startswith(base):
93
+ return unquote(value[len(base):])
94
+ if urlsplit(value).scheme:
95
+ raise SchemaError(f"Object is outside the catalog bucket: {value}")
96
+ return value
97
+
98
+
99
+ def build_assets(listing, bams, metadata, vafs, path_claims=(), *, tables=None):
100
+ """Retain every listed object, enriching exact paths before basename matches.
101
+
102
+ Basename joins are used only when unique among alignment objects. Path
103
+ inferences are explicitly marked and excluded from default metadata filters.
104
+ The global viewer genome is retained as a claim, not assigned as assembly.
105
+ """
106
+ base = listing.get("download_base", BUCKET)
107
+ objects = {}
108
+ for row in listing["files"]:
109
+ if len(row) < 3 or row[0] in objects:
110
+ raise SchemaError("Duplicate or malformed bucket object")
111
+ objects[row[0]] = dict(size=int(row[1]), modified=row[2])
112
+ catalog, claims, extra = {}, defaultdict(list), defaultdict(dict)
113
+ for category in bams["categories"]:
114
+ for row in category["bams"]:
115
+ key = object_key(row["url"], bams["baseUrl"])
116
+ if key in catalog:
117
+ raise SchemaError(f"Duplicate catalog alignment: {key}")
118
+ catalog[key] = row
119
+ objects.setdefault(key, dict(size=None, modified=None))
120
+ assay, platform = ASSAYS.get(category["name"], (None, None))
121
+ match = re.match(r"(T\d+)\b", row["name"])
122
+ claims[key].append(SampleClaim("bams", row["name"], match[1] if match else None,
123
+ assay=label_assay(row["name"], assay), platform=platform,
124
+ tissue=normalize_tissue(row.get("tissue")),
125
+ **label_claims(row["name"])))
126
+ extra[key].update(catalog=row, category=category["name"],
127
+ catalog_genome_assertion=bams.get("genome"))
128
+ for row in metadata:
129
+ if not row.get("s3_path"):
130
+ continue
131
+ key = object_key(row["s3_path"], base)
132
+ objects.setdefault(key, dict(size=None, modified=None))
133
+ assay, platform = ASSAYS.get(row.get("assay"), (None, None))
134
+ claims[key].append(SampleClaim("bam_metadata", row["display_name"], normalize_timepoint(row.get("timepoint")),
135
+ row.get("sample_date") or None, assay, platform,
136
+ normalize_tissue(row.get("tissue")), normalize_provider(row.get("provider"))))
137
+ extra[key].setdefault("metadata_rows", []).append(row)
138
+ counts = Counter(PurePosixPath(key).name for key in objects if asset_type(key)[0] == "alignment")
139
+ by_basename = defaultdict(set)
140
+ for row in vafs:
141
+ by_basename[row["bam_file"]].add(tuple(row.get(k, "") for k in
142
+ ("sample_label", "timepoint", "sample_date", "assay_type", "tissue", "data_source")))
143
+ by_path = defaultdict(list)
144
+ for prefix, claim in path_claims:
145
+ by_path[prefix.rstrip("/")].append(claim)
146
+ assets = []
147
+ for key, object_metadata in objects.items():
148
+ kind, format = asset_type(key)
149
+ info = dict(extra.get(key, {}))
150
+ records = list(claims.get(key, ()))
151
+ # The most specific data-page path wins: a file's own row overrides the
152
+ # row for its directory, which can also hold other assays' files.
153
+ parts = key.split("/")
154
+ for depth in range(len(parts), 0, -1):
155
+ if "/".join(parts[:depth]) in by_path:
156
+ records.extend(by_path["/".join(parts[:depth])])
157
+ break
158
+ if kind == "alignment":
159
+ basename = PurePosixPath(key).name
160
+ if counts[basename] == 1:
161
+ for label, timepoint, date, assay_name, tissue, provider in sorted(by_basename[basename]):
162
+ assay, platform = ASSAYS.get(assay_name, (None, None))
163
+ records.append(SampleClaim("vafs", label, normalize_timepoint(timepoint), date or None,
164
+ assay, platform, normalize_tissue(tissue), normalize_provider(provider)))
165
+ elif by_basename[basename]:
166
+ info["ambiguous_vaf_basename"] = basename
167
+ # This inference is useful for discovery, but never establishes identity.
168
+ points = sorted(set(re.findall(r"(?:^|[/_ .-])(T[0-3])(?=[/_ .-]|$)", key)))
169
+ libraries = sorted(set(re.findall(r"\b(?:BG\d{6}|SARC\d{4}|TL-\d{2}-[A-Z0-9]+)\b", key)))
170
+ if kind != "other":
171
+ for point in points:
172
+ records.append(SampleClaim("bucket_path", key, timepoint=point, basis="inferred"))
173
+ for library in libraries:
174
+ records.append(SampleClaim("bucket_path", key, library=library, basis="inferred"))
175
+ suffixes = [key + ".bai", key[:-4] + ".bai", key + ".csi"] if format == "bam" else (
176
+ [key + ".crai", key[:-5] + ".crai", key + ".csi"] if format == "cram" else
177
+ [key + ".tbi", key + ".csi"] if format in ("vcf", "bcf") else [])
178
+ indexes = tuple(bucket_url(k, base) for k in suffixes if k in objects)
179
+ url = bucket_url(key, base)
180
+ assets.append(Asset(stable_id(url), key, url, kind, format, index_urls=indexes,
181
+ claims=tuple(records), metadata=info, **object_metadata))
182
+ for name, (url, format) in (tables or TABLE_SOURCES).items():
183
+ assets.append(Asset(stable_id(url), "site/" + name, url, "table", format,
184
+ metadata={"resource": name}))
185
+ return Assets(assets)
186
+
187
+
188
+ def data_page_rows(html):
189
+ """Yield (section context, {column: text}) for data-page rows with a bucket path."""
190
+ from bs4 import BeautifulSoup
191
+ soup = BeautifulSoup(html, "html.parser")
192
+ headings = {}
193
+ for element in soup.find_all(["h2", "h3", "h4", "table"]):
194
+ if element.name != "table":
195
+ level = int(element.name[1])
196
+ headings = {k: v for k, v in headings.items() if k < level}
197
+ headings[level] = element.get_text(" ", strip=True)
198
+ continue
199
+ context = " / ".join(headings.values())
200
+ headers = [c.get_text(" ", strip=True) for c in element.select("thead th")]
201
+ if not headers:
202
+ headers = [c.get_text(" ", strip=True) for c in element.select("tr:first-child th")]
203
+ for row in element.select("tr"):
204
+ cells = row.find_all("td", recursive=False)
205
+ if len(cells) != len(headers):
206
+ continue
207
+ values = dict(zip(headers, (c.get_text(" ", strip=True) for c in cells)))
208
+ if "Bucket Path" in values:
209
+ yield context, values
210
+
211
+
212
+ def parse_data_paths(html):
213
+ """Extract the data page's explicit path-to-sample claims, including FASTQs.
214
+
215
+ Assays/platforms come from section headings, tissues/providers from cells.
216
+ Composite tissues and timepoints remain unresolved rather than guessed.
217
+ """
218
+ result = []
219
+ for context, values in data_page_rows(html):
220
+ title = context.lower()
221
+ assay = ("wgs" if "whole genome" in title or re.search(r"\bwgs\b", title) else
222
+ "wes" if "whole exome" in title or re.search(r"\bwes\b", title) else
223
+ "scrna-seq" if any(x in title for x in ("single cell", "single-cell")) else
224
+ "rna-seq" if "bulk rna" in title else None)
225
+ platform = ("ont" if "nanopore" in title else "pacbio" if "pacbio" in title else
226
+ "illumina" if "illumina" in title else None)
227
+ points = set(re.findall(r"\bT[0-3]\b", values.get("Timepoint", "")))
228
+ tissue = normalize_tissue(values.get("Tissue"))
229
+ label = " | ".join(v for k, v in values.items() if k not in ("Bucket Path", "Size", "Files"))
230
+ result.append((values["Bucket Path"].strip("`"), SampleClaim(
231
+ "data_page", context + " / " + label,
232
+ timepoint=next(iter(points)) if len(points) == 1 else None,
233
+ assay=assay, platform=platform, tissue=tissue,
234
+ provider=normalize_provider(values.get("Provider")))))
235
+ return tuple(result)