survscope 0.4.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- survscope/__init__.py +34 -0
- survscope/analysis.py +152 -0
- survscope/builder.py +657 -0
- survscope/cli.py +189 -0
- survscope/constants.py +125 -0
- survscope/cptac.py +620 -0
- survscope/data.py +236 -0
- survscope/grouping.py +163 -0
- survscope/models.py +128 -0
- survscope/plotting.py +163 -0
- survscope/quality.py +91 -0
- survscope/statistics.py +250 -0
- survscope/validation.py +192 -0
- survscope-0.4.3.dist-info/METADATA +113 -0
- survscope-0.4.3.dist-info/RECORD +19 -0
- survscope-0.4.3.dist-info/WHEEL +5 -0
- survscope-0.4.3.dist-info/entry_points.txt +3 -0
- survscope-0.4.3.dist-info/licenses/LICENSE +21 -0
- survscope-0.4.3.dist-info/top_level.txt +1 -0
survscope/builder.py
ADDED
|
@@ -0,0 +1,657 @@
|
|
|
1
|
+
"""Streaming builder for compact, versioned SurvScope static data assets."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import contextlib
|
|
7
|
+
import copy
|
|
8
|
+
import csv
|
|
9
|
+
import gzip
|
|
10
|
+
import hashlib
|
|
11
|
+
import io
|
|
12
|
+
import json
|
|
13
|
+
import math
|
|
14
|
+
import shutil
|
|
15
|
+
import tempfile
|
|
16
|
+
import urllib.request
|
|
17
|
+
import zipfile
|
|
18
|
+
from collections import Counter
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from datetime import datetime, timezone
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any, BinaryIO
|
|
23
|
+
|
|
24
|
+
import numpy as np
|
|
25
|
+
|
|
26
|
+
from .constants import (
|
|
27
|
+
BUCKET_COUNT,
|
|
28
|
+
COHORT_LABELS,
|
|
29
|
+
COHORT_PRIMARY_SAMPLE_CODES,
|
|
30
|
+
COHORTS,
|
|
31
|
+
DEFAULT_DATA_VERSION,
|
|
32
|
+
DEFAULT_PRIMARY_SAMPLE_CODES,
|
|
33
|
+
ENDPOINT_COLUMNS,
|
|
34
|
+
ENDPOINTS,
|
|
35
|
+
EXPRESSION_SCALE,
|
|
36
|
+
GDC_EXPRESSION_URL,
|
|
37
|
+
GDC_PIPELINE_URL,
|
|
38
|
+
GDC_PROBEMAP_URL,
|
|
39
|
+
GDC_SAMPLE_TYPE_CODES_URL,
|
|
40
|
+
MISSING_EXPRESSION,
|
|
41
|
+
SAMPLE_TYPE_LABELS,
|
|
42
|
+
SCHEMA_VERSION,
|
|
43
|
+
TCGA_CDR_CITATION_URL,
|
|
44
|
+
TCGA_CDR_URL,
|
|
45
|
+
)
|
|
46
|
+
from .cptac import CPTAC_COHORTS, build_cptac_cohort
|
|
47
|
+
from .quality import endpoint_quality
|
|
48
|
+
from .validation import validate_release
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _json_bytes(value: Any) -> bytes:
|
|
52
|
+
return (json.dumps(value, separators=(",", ":"), allow_nan=False) + "\n").encode()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _parse_number(value: str) -> float | None:
|
|
56
|
+
try:
|
|
57
|
+
number = float(value)
|
|
58
|
+
except (TypeError, ValueError):
|
|
59
|
+
return None
|
|
60
|
+
return number if math.isfinite(number) else None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _sample_case(sample: str) -> str:
|
|
64
|
+
return sample[:15]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _sample_type_code(sample: str) -> str | None:
|
|
68
|
+
parts = sample.split("-")
|
|
69
|
+
return parts[3][:2] if len(parts) >= 4 and len(parts[3]) >= 2 else None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _primary_sample_codes(cohort: str) -> tuple[str, ...]:
|
|
73
|
+
return COHORT_PRIMARY_SAMPLE_CODES.get(cohort.upper(), DEFAULT_PRIMARY_SAMPLE_CODES)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _is_primary_cancer_sample(sample: str, cohort: str) -> bool:
|
|
77
|
+
return _sample_type_code(sample) in _primary_sample_codes(cohort)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _bucket_for(symbol: str) -> str:
|
|
81
|
+
value = hashlib.sha256(symbol.upper().encode()).digest()[0] % BUCKET_COUNT
|
|
82
|
+
return f"{value:02x}"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class HashingReader:
|
|
86
|
+
"""Binary reader that hashes source bytes as they stream past."""
|
|
87
|
+
|
|
88
|
+
def __init__(self, source: BinaryIO) -> None:
|
|
89
|
+
self.source = source
|
|
90
|
+
self.sha256 = hashlib.sha256()
|
|
91
|
+
self.byte_count = 0
|
|
92
|
+
|
|
93
|
+
def read(self, size: int = -1) -> bytes:
|
|
94
|
+
payload = self.source.read(size)
|
|
95
|
+
self.sha256.update(payload)
|
|
96
|
+
self.byte_count += len(payload)
|
|
97
|
+
return payload
|
|
98
|
+
|
|
99
|
+
def readinto(self, target) -> int:
|
|
100
|
+
payload = self.read(len(target))
|
|
101
|
+
target[: len(payload)] = payload
|
|
102
|
+
return len(payload)
|
|
103
|
+
|
|
104
|
+
def readable(self) -> bool:
|
|
105
|
+
return True
|
|
106
|
+
|
|
107
|
+
def close(self) -> None:
|
|
108
|
+
self.source.close()
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@dataclass(frozen=True)
|
|
112
|
+
class SourceText:
|
|
113
|
+
text: str
|
|
114
|
+
url: str
|
|
115
|
+
sha256: str
|
|
116
|
+
byte_count: int
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _read_small_source(url_or_path: str, public_url: str | None = None) -> SourceText:
|
|
120
|
+
path = Path(url_or_path)
|
|
121
|
+
if path.is_file():
|
|
122
|
+
payload = path.read_bytes()
|
|
123
|
+
url = public_url or str(path.resolve())
|
|
124
|
+
else:
|
|
125
|
+
request = urllib.request.Request(url_or_path, headers={"User-Agent": "survscope/0.1"})
|
|
126
|
+
with urllib.request.urlopen(request, timeout=120) as response:
|
|
127
|
+
payload = response.read()
|
|
128
|
+
url = public_url or url_or_path
|
|
129
|
+
return SourceText(
|
|
130
|
+
text=payload.decode(),
|
|
131
|
+
url=url,
|
|
132
|
+
sha256=hashlib.sha256(payload).hexdigest(),
|
|
133
|
+
byte_count=len(payload),
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def load_probemap(source: SourceText) -> tuple[dict[str, tuple[str, str]], int]:
|
|
138
|
+
"""Return mappings only for symbols associated with exactly one Ensembl gene."""
|
|
139
|
+
reader = csv.DictReader(io.StringIO(source.text), delimiter="\t")
|
|
140
|
+
records = []
|
|
141
|
+
symbols = Counter()
|
|
142
|
+
for row in reader:
|
|
143
|
+
ensembl_version = row["id"].strip()
|
|
144
|
+
symbol = row["gene"].strip()
|
|
145
|
+
if not ensembl_version or not symbol:
|
|
146
|
+
continue
|
|
147
|
+
ensembl = ensembl_version.split(".", 1)[0]
|
|
148
|
+
symbol_key = symbol.upper()
|
|
149
|
+
records.append((ensembl_version, ensembl, symbol, symbol_key))
|
|
150
|
+
symbols[symbol_key] += 1
|
|
151
|
+
ambiguous = {symbol for symbol, count in symbols.items() if count != 1}
|
|
152
|
+
mapping: dict[str, tuple[str, str]] = {}
|
|
153
|
+
for ensembl_version, ensembl, symbol, symbol_key in records:
|
|
154
|
+
if symbol_key in ambiguous:
|
|
155
|
+
continue
|
|
156
|
+
mapping[ensembl_version] = (symbol, ensembl)
|
|
157
|
+
mapping.setdefault(ensembl, (symbol, ensembl))
|
|
158
|
+
return mapping, len(ambiguous)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def load_survival(source: SourceText) -> dict[str, dict[str, dict[str, float | None]]]:
|
|
162
|
+
result: dict[str, dict[str, dict[str, float | None]]] = {}
|
|
163
|
+
reader = csv.DictReader(io.StringIO(source.text), delimiter="\t")
|
|
164
|
+
for row in reader:
|
|
165
|
+
cohort = row["cancer type abbreviation"].strip().upper()
|
|
166
|
+
sample = row["sample"].strip()
|
|
167
|
+
if not cohort or not sample:
|
|
168
|
+
continue
|
|
169
|
+
endpoint_values = {}
|
|
170
|
+
for endpoint, (time_column, event_column) in ENDPOINT_COLUMNS.items():
|
|
171
|
+
endpoint_values[endpoint] = {
|
|
172
|
+
"time": _parse_number(row.get(time_column, "")),
|
|
173
|
+
"event": _parse_number(row.get(event_column, "")),
|
|
174
|
+
}
|
|
175
|
+
result.setdefault(cohort, {})[sample] = endpoint_values
|
|
176
|
+
return result
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@contextlib.contextmanager
|
|
180
|
+
def _expression_stream(source: str) -> Any:
|
|
181
|
+
"""Yield a decompressed text stream and its source-byte hasher."""
|
|
182
|
+
path = Path(source)
|
|
183
|
+
if path.is_file():
|
|
184
|
+
raw: BinaryIO = path.open("rb")
|
|
185
|
+
label = str(path.resolve())
|
|
186
|
+
else:
|
|
187
|
+
request = urllib.request.Request(source, headers={"User-Agent": "survscope/0.1"})
|
|
188
|
+
raw = urllib.request.urlopen(request, timeout=180)
|
|
189
|
+
label = source
|
|
190
|
+
hashing = HashingReader(raw)
|
|
191
|
+
compressed = gzip.GzipFile(fileobj=hashing, mode="rb")
|
|
192
|
+
text = io.TextIOWrapper(compressed, encoding="utf-8", newline="")
|
|
193
|
+
try:
|
|
194
|
+
yield text, hashing, label
|
|
195
|
+
finally:
|
|
196
|
+
text.close()
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _clinical_asset(
|
|
200
|
+
cohort: str,
|
|
201
|
+
sample_cases: list[str],
|
|
202
|
+
survival_rows: dict[str, dict[str, dict[str, float | None]]],
|
|
203
|
+
) -> dict[str, Any]:
|
|
204
|
+
sample_codes = _primary_sample_codes(cohort)
|
|
205
|
+
sample_labels = [SAMPLE_TYPE_LABELS[code] for code in sample_codes]
|
|
206
|
+
endpoints = {}
|
|
207
|
+
for endpoint in ENDPOINTS:
|
|
208
|
+
quality, note = endpoint_quality(cohort, endpoint)
|
|
209
|
+
endpoints[endpoint] = {
|
|
210
|
+
"time": [survival_rows[sample][endpoint]["time"] for sample in sample_cases],
|
|
211
|
+
"event": [survival_rows[sample][endpoint]["event"] for sample in sample_cases],
|
|
212
|
+
"quality": quality,
|
|
213
|
+
"quality_note": note,
|
|
214
|
+
}
|
|
215
|
+
return {
|
|
216
|
+
"schema_version": SCHEMA_VERSION,
|
|
217
|
+
"cohort": cohort,
|
|
218
|
+
"sample_count": len(sample_cases),
|
|
219
|
+
"sample_type": " / ".join(sample_labels),
|
|
220
|
+
"sample_type_codes": list(sample_codes),
|
|
221
|
+
"identifiers_included": False,
|
|
222
|
+
"endpoints": endpoints,
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _median_metadata(
|
|
227
|
+
expression_log2: np.ndarray,
|
|
228
|
+
encoded: np.ndarray,
|
|
229
|
+
clinical: dict[str, Any],
|
|
230
|
+
) -> dict[str, dict[str, Any]]:
|
|
231
|
+
exact_tpm = np.exp2(expression_log2) - 1.0
|
|
232
|
+
decoded_log2 = encoded.astype(float) / EXPRESSION_SCALE
|
|
233
|
+
decoded_log2[encoded == MISSING_EXPRESSION] = np.nan
|
|
234
|
+
decoded_tpm = np.exp2(decoded_log2) - 1.0
|
|
235
|
+
metadata = {}
|
|
236
|
+
for endpoint in ENDPOINTS:
|
|
237
|
+
time = np.asarray(
|
|
238
|
+
[
|
|
239
|
+
np.nan if value is None else value
|
|
240
|
+
for value in clinical["endpoints"][endpoint]["time"]
|
|
241
|
+
],
|
|
242
|
+
dtype=float,
|
|
243
|
+
)
|
|
244
|
+
event = np.asarray(
|
|
245
|
+
[
|
|
246
|
+
np.nan if value is None else value
|
|
247
|
+
for value in clinical["endpoints"][endpoint]["event"]
|
|
248
|
+
],
|
|
249
|
+
dtype=float,
|
|
250
|
+
)
|
|
251
|
+
valid = (
|
|
252
|
+
np.isfinite(exact_tpm) & np.isfinite(time) & np.isfinite(event) & (time > 0)
|
|
253
|
+
)
|
|
254
|
+
if not np.any(valid):
|
|
255
|
+
metadata[endpoint] = {"cutoff_tpm": None, "flips": []}
|
|
256
|
+
continue
|
|
257
|
+
cutoff = float(np.median(exact_tpm[valid]))
|
|
258
|
+
exact_high = exact_tpm > cutoff
|
|
259
|
+
encoded_high = decoded_tpm > cutoff
|
|
260
|
+
flips = np.flatnonzero(valid & (exact_high != encoded_high)).astype(int).tolist()
|
|
261
|
+
metadata[endpoint] = {"cutoff_tpm": cutoff, "flips": flips}
|
|
262
|
+
return metadata
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def build_cohort(
|
|
266
|
+
cohort: str,
|
|
267
|
+
*,
|
|
268
|
+
expression_source: str,
|
|
269
|
+
survival: dict[str, dict[str, dict[str, float | None]]],
|
|
270
|
+
probemap: dict[str, tuple[str, str]],
|
|
271
|
+
outdir: Path,
|
|
272
|
+
wanted_genes: set[str] | None = None,
|
|
273
|
+
expression_public_url: str | None = None,
|
|
274
|
+
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
|
|
275
|
+
"""Stream one cohort and emit only compact plot data assets."""
|
|
276
|
+
cohort = cohort.upper()
|
|
277
|
+
survival_rows = survival.get(cohort)
|
|
278
|
+
if not survival_rows:
|
|
279
|
+
raise RuntimeError(f"No TCGA-CDR rows found for {cohort}")
|
|
280
|
+
outdir.mkdir(parents=True, exist_ok=True)
|
|
281
|
+
bucket_ids = [f"{index:02x}" for index in range(BUCKET_COUNT)]
|
|
282
|
+
bucket_metadata: dict[str, list[dict[str, Any]]] = {key: [] for key in bucket_ids}
|
|
283
|
+
genes: list[dict[str, Any]] = []
|
|
284
|
+
|
|
285
|
+
with tempfile.TemporaryDirectory(prefix=f"survscope-{cohort.lower()}-") as temp_name:
|
|
286
|
+
temporary = Path(temp_name)
|
|
287
|
+
raw_paths = {key: temporary / f"{cohort}-bucket-{key}.u16le" for key in bucket_ids}
|
|
288
|
+
handles = {key: path.open("wb") for key, path in raw_paths.items()}
|
|
289
|
+
try:
|
|
290
|
+
with _expression_stream(expression_source) as (text, source_hash, source_label):
|
|
291
|
+
reader = csv.reader(text, delimiter="\t")
|
|
292
|
+
header = next(reader)
|
|
293
|
+
selected: list[tuple[int, str]] = []
|
|
294
|
+
seen_cases = set()
|
|
295
|
+
for index, sample in enumerate(header[1:]):
|
|
296
|
+
case = _sample_case(sample)
|
|
297
|
+
if (
|
|
298
|
+
_is_primary_cancer_sample(sample, cohort)
|
|
299
|
+
and case in survival_rows
|
|
300
|
+
and case not in seen_cases
|
|
301
|
+
):
|
|
302
|
+
selected.append((index, case))
|
|
303
|
+
seen_cases.add(case)
|
|
304
|
+
if not selected:
|
|
305
|
+
codes = ", ".join(_primary_sample_codes(cohort))
|
|
306
|
+
raise RuntimeError(
|
|
307
|
+
f"No matched primary cancer samples found for {cohort} "
|
|
308
|
+
f"(TCGA sample codes: {codes})"
|
|
309
|
+
)
|
|
310
|
+
selected_indices = [index for index, _ in selected]
|
|
311
|
+
sample_cases = [case for _, case in selected]
|
|
312
|
+
clinical = _clinical_asset(cohort, sample_cases, survival_rows)
|
|
313
|
+
processed_rows = 0
|
|
314
|
+
included_rows = 0
|
|
315
|
+
for row in reader:
|
|
316
|
+
processed_rows += 1
|
|
317
|
+
if not row:
|
|
318
|
+
continue
|
|
319
|
+
identifier = row[0].strip()
|
|
320
|
+
mapped = probemap.get(identifier) or probemap.get(
|
|
321
|
+
identifier.split(".", 1)[0]
|
|
322
|
+
)
|
|
323
|
+
if mapped is None:
|
|
324
|
+
continue
|
|
325
|
+
symbol, ensembl = mapped
|
|
326
|
+
if wanted_genes and symbol.upper() not in wanted_genes:
|
|
327
|
+
continue
|
|
328
|
+
values = np.asarray(
|
|
329
|
+
[
|
|
330
|
+
float(row[index + 1]) if row[index + 1] else np.nan
|
|
331
|
+
for index in selected_indices
|
|
332
|
+
],
|
|
333
|
+
dtype=float,
|
|
334
|
+
)
|
|
335
|
+
encoded = np.full(len(values), MISSING_EXPRESSION, dtype="<u2")
|
|
336
|
+
finite = np.isfinite(values)
|
|
337
|
+
quantized = np.rint(values[finite] * EXPRESSION_SCALE)
|
|
338
|
+
quantized = np.clip(quantized, 0, MISSING_EXPRESSION - 1)
|
|
339
|
+
encoded[finite] = quantized.astype("<u2")
|
|
340
|
+
bucket = _bucket_for(symbol)
|
|
341
|
+
row_index = len(bucket_metadata[bucket])
|
|
342
|
+
handles[bucket].write(encoded.tobytes())
|
|
343
|
+
record = {
|
|
344
|
+
"symbol": symbol,
|
|
345
|
+
"ensembl": ensembl,
|
|
346
|
+
"row": row_index,
|
|
347
|
+
"medians": _median_metadata(values, encoded, clinical),
|
|
348
|
+
}
|
|
349
|
+
bucket_metadata[bucket].append(record)
|
|
350
|
+
genes.append(
|
|
351
|
+
{
|
|
352
|
+
"symbol": symbol,
|
|
353
|
+
"ensembl": ensembl,
|
|
354
|
+
"bucket": bucket,
|
|
355
|
+
}
|
|
356
|
+
)
|
|
357
|
+
included_rows += 1
|
|
358
|
+
if processed_rows % 5000 == 0:
|
|
359
|
+
print(
|
|
360
|
+
f"{cohort}: streamed {processed_rows:,} source rows; "
|
|
361
|
+
f"kept {included_rows:,}",
|
|
362
|
+
flush=True,
|
|
363
|
+
)
|
|
364
|
+
source_details = {
|
|
365
|
+
"url": expression_public_url or source_label,
|
|
366
|
+
"sha256": source_hash.sha256.hexdigest(),
|
|
367
|
+
"compressed_bytes_streamed": source_hash.byte_count,
|
|
368
|
+
}
|
|
369
|
+
finally:
|
|
370
|
+
for handle in handles.values():
|
|
371
|
+
handle.close()
|
|
372
|
+
|
|
373
|
+
bucket_assets = {}
|
|
374
|
+
for bucket in bucket_ids:
|
|
375
|
+
records = bucket_metadata[bucket]
|
|
376
|
+
if not records:
|
|
377
|
+
continue
|
|
378
|
+
asset_name = f"{cohort}-bucket-{bucket}.zip"
|
|
379
|
+
asset_path = outdir / asset_name
|
|
380
|
+
meta = {
|
|
381
|
+
"schema_version": SCHEMA_VERSION,
|
|
382
|
+
"cohort": cohort,
|
|
383
|
+
"sample_count": len(sample_cases),
|
|
384
|
+
"scale": EXPRESSION_SCALE,
|
|
385
|
+
"missing": MISSING_EXPRESSION,
|
|
386
|
+
"transform": "log2(TPM+1)",
|
|
387
|
+
"genes": records,
|
|
388
|
+
}
|
|
389
|
+
with zipfile.ZipFile(
|
|
390
|
+
asset_path,
|
|
391
|
+
mode="w",
|
|
392
|
+
compression=zipfile.ZIP_DEFLATED,
|
|
393
|
+
compresslevel=9,
|
|
394
|
+
) as archive:
|
|
395
|
+
archive.writestr("meta.json", _json_bytes(meta))
|
|
396
|
+
archive.write(raw_paths[bucket], arcname="expression.u16le")
|
|
397
|
+
bucket_assets[bucket] = asset_name
|
|
398
|
+
|
|
399
|
+
clinical_name = f"{cohort}-clinical.json"
|
|
400
|
+
(outdir / clinical_name).write_bytes(_json_bytes(clinical))
|
|
401
|
+
cohort_manifest = {
|
|
402
|
+
"program": "TCGA",
|
|
403
|
+
"label": COHORT_LABELS[cohort],
|
|
404
|
+
"sample_count": len(sample_cases),
|
|
405
|
+
"gene_count": len(genes),
|
|
406
|
+
"sample_type_codes": list(_primary_sample_codes(cohort)),
|
|
407
|
+
"clinical_asset": clinical_name,
|
|
408
|
+
"bucket_assets": bucket_assets,
|
|
409
|
+
"source": source_details,
|
|
410
|
+
}
|
|
411
|
+
print(
|
|
412
|
+
f"{cohort}: completed {len(genes):,} genes across {len(bucket_assets)} buckets "
|
|
413
|
+
f"for {len(sample_cases):,} matched primary cancer samples",
|
|
414
|
+
flush=True,
|
|
415
|
+
)
|
|
416
|
+
return cohort_manifest, genes
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _asset_checksums(outdir: Path) -> dict[str, dict[str, Any]]:
|
|
420
|
+
checksums = {}
|
|
421
|
+
for path in sorted(outdir.iterdir()):
|
|
422
|
+
if path.is_file() and not path.name.startswith(("manifest-", "SHA256SUMS")):
|
|
423
|
+
payload = path.read_bytes()
|
|
424
|
+
checksums[path.name] = {
|
|
425
|
+
"sha256": hashlib.sha256(payload).hexdigest(),
|
|
426
|
+
"bytes": len(payload),
|
|
427
|
+
}
|
|
428
|
+
return checksums
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def build_release(
|
|
432
|
+
*,
|
|
433
|
+
cohorts: list[str],
|
|
434
|
+
outdir: Path,
|
|
435
|
+
data_version: str,
|
|
436
|
+
survival_source: str,
|
|
437
|
+
probemap_source: str,
|
|
438
|
+
expression_file: str | None = None,
|
|
439
|
+
expression_url_template: str = GDC_EXPRESSION_URL,
|
|
440
|
+
wanted_genes: set[str] | None = None,
|
|
441
|
+
survival_public_url: str | None = None,
|
|
442
|
+
expression_public_url: str | None = None,
|
|
443
|
+
tcga_release: Path | None = None,
|
|
444
|
+
) -> Path:
|
|
445
|
+
"""Build one immutable data-release directory."""
|
|
446
|
+
data_version = data_version.removeprefix("data-v")
|
|
447
|
+
if datetime.strptime(data_version, "%Y.%m.%d").strftime("%Y.%m.%d") != data_version:
|
|
448
|
+
raise ValueError("Data version must use YYYY.MM.DD")
|
|
449
|
+
cohorts = list(dict.fromkeys(cohort.upper() for cohort in cohorts))
|
|
450
|
+
unknown = set(cohorts) - set(COHORTS) - set(CPTAC_COHORTS)
|
|
451
|
+
if not cohorts or unknown:
|
|
452
|
+
raise ValueError(f"Unsupported or empty cohort selection: {sorted(unknown)}")
|
|
453
|
+
reused_manifest = None
|
|
454
|
+
reused_provenance = None
|
|
455
|
+
if tcga_release is not None:
|
|
456
|
+
if wanted_genes or expression_file:
|
|
457
|
+
raise ValueError("--tcga-release cannot be combined with --genes or --expression-file")
|
|
458
|
+
reused_manifest = validate_release(tcga_release)
|
|
459
|
+
missing = (set(cohorts) & set(COHORTS)) - set(reused_manifest["cohorts"])
|
|
460
|
+
if missing:
|
|
461
|
+
raise ValueError(f"TCGA release is missing requested cohorts: {sorted(missing)}")
|
|
462
|
+
reused_provenance = {
|
|
463
|
+
"data_version": reused_manifest["data_version"],
|
|
464
|
+
"manifest_sha256": hashlib.sha256(
|
|
465
|
+
next(tcga_release.glob("manifest-*.json")).read_bytes()
|
|
466
|
+
).hexdigest(),
|
|
467
|
+
}
|
|
468
|
+
if outdir.exists() and any(outdir.iterdir()):
|
|
469
|
+
raise ValueError("Release directory must be empty; existing data releases are immutable")
|
|
470
|
+
outdir.mkdir(parents=True, exist_ok=True)
|
|
471
|
+
tcga_sources = {}
|
|
472
|
+
survival = {}
|
|
473
|
+
probemap = {}
|
|
474
|
+
if reused_manifest is not None:
|
|
475
|
+
tcga_sources = reused_manifest["sources"]
|
|
476
|
+
elif any(cohort in COHORTS for cohort in cohorts):
|
|
477
|
+
survival_text = _read_small_source(survival_source, public_url=survival_public_url)
|
|
478
|
+
probemap_text = _read_small_source(probemap_source)
|
|
479
|
+
survival = load_survival(survival_text)
|
|
480
|
+
probemap, ambiguous_count = load_probemap(probemap_text)
|
|
481
|
+
tcga_sources = {
|
|
482
|
+
"expression": {
|
|
483
|
+
"label": "GDC STAR TPM", "pipeline": GDC_PIPELINE_URL,
|
|
484
|
+
"dataset_template": expression_url_template,
|
|
485
|
+
"wrangling": "Xena GDC ETL; log2(TPM+1)",
|
|
486
|
+
},
|
|
487
|
+
"survival": {
|
|
488
|
+
"label": "PanCanAtlas TCGA-CDR", "url": survival_text.url,
|
|
489
|
+
"sha256": survival_text.sha256, "bytes": survival_text.byte_count,
|
|
490
|
+
"citation": TCGA_CDR_CITATION_URL,
|
|
491
|
+
},
|
|
492
|
+
"gene_map": {
|
|
493
|
+
"label": "GENCODE v36 gene probemap", "url": probemap_text.url,
|
|
494
|
+
"sha256": probemap_text.sha256, "bytes": probemap_text.byte_count,
|
|
495
|
+
"ambiguous_symbols_excluded": ambiguous_count,
|
|
496
|
+
},
|
|
497
|
+
}
|
|
498
|
+
catalog: dict[tuple[str, str], dict[str, Any]] = {}
|
|
499
|
+
cohort_details = {}
|
|
500
|
+
for cohort in cohorts:
|
|
501
|
+
if cohort in CPTAC_COHORTS:
|
|
502
|
+
if expression_file:
|
|
503
|
+
raise ValueError("--expression-file is only supported for TCGA builds")
|
|
504
|
+
details, genes = build_cptac_cohort(cohort, outdir=outdir, wanted_genes=wanted_genes)
|
|
505
|
+
elif reused_manifest is not None:
|
|
506
|
+
details = copy.deepcopy(reused_manifest["cohorts"][cohort])
|
|
507
|
+
if details.get("program", "TCGA") != "TCGA":
|
|
508
|
+
raise ValueError(f"{cohort} is not TCGA in the supplied compact release")
|
|
509
|
+
details["program"] = "TCGA"
|
|
510
|
+
details.setdefault("sources", tcga_sources)
|
|
511
|
+
details["reused_from"] = reused_provenance
|
|
512
|
+
for name in [details["clinical_asset"], *details["bucket_assets"].values()]:
|
|
513
|
+
shutil.copyfile(tcga_release / name, outdir / name)
|
|
514
|
+
genes = [
|
|
515
|
+
{key: value for key, value in gene.items() if key != "cohorts"}
|
|
516
|
+
for gene in reused_manifest["genes"] if cohort in gene["cohorts"]
|
|
517
|
+
]
|
|
518
|
+
else:
|
|
519
|
+
expression_source = expression_file or expression_url_template.format(cohort=cohort)
|
|
520
|
+
details, genes = build_cohort(
|
|
521
|
+
cohort, expression_source=expression_source, survival=survival,
|
|
522
|
+
probemap=probemap, outdir=outdir, wanted_genes=wanted_genes,
|
|
523
|
+
expression_public_url=(
|
|
524
|
+
expression_public_url or expression_url_template.format(cohort=cohort)
|
|
525
|
+
),
|
|
526
|
+
)
|
|
527
|
+
details["sources"] = tcga_sources
|
|
528
|
+
cohort_details[cohort] = details
|
|
529
|
+
print(f"Built {cohort}: {details['sample_count']} patients, "
|
|
530
|
+
f"{details['gene_count']} genes", flush=True)
|
|
531
|
+
for gene in genes:
|
|
532
|
+
key = (gene["symbol"].upper(), gene["ensembl"])
|
|
533
|
+
entry = catalog.setdefault(key, {**gene, "cohorts": []})
|
|
534
|
+
entry["cohorts"].append(cohort)
|
|
535
|
+
|
|
536
|
+
checksums = _asset_checksums(outdir)
|
|
537
|
+
manifest = {
|
|
538
|
+
# Older readers hard-code TCGA names and release-level provenance.
|
|
539
|
+
"schema_version": 2 if any(code in CPTAC_COHORTS for code in cohorts) else SCHEMA_VERSION,
|
|
540
|
+
"data_version": data_version.removeprefix("data-v"),
|
|
541
|
+
"created_at": datetime.now(timezone.utc).isoformat(),
|
|
542
|
+
"expression_encoding": {
|
|
543
|
+
"transform": "log2(TPM+1)",
|
|
544
|
+
"storage": "uint16 little-endian",
|
|
545
|
+
"scale": EXPRESSION_SCALE,
|
|
546
|
+
"missing": MISSING_EXPRESSION,
|
|
547
|
+
"maximum_quantization_error_log2": 0.0005,
|
|
548
|
+
"exact_median_membership_corrections": True,
|
|
549
|
+
},
|
|
550
|
+
"sample_policy": {
|
|
551
|
+
"sample_type": "Cohort-specific primary cancer specimen",
|
|
552
|
+
"default_tcga_sample_codes": list(DEFAULT_PRIMARY_SAMPLE_CODES),
|
|
553
|
+
"cohort_overrides": {
|
|
554
|
+
cohort: list(codes)
|
|
555
|
+
for cohort, codes in sorted(COHORT_PRIMARY_SAMPLE_CODES.items())
|
|
556
|
+
},
|
|
557
|
+
"code_definitions": SAMPLE_TYPE_LABELS,
|
|
558
|
+
"code_table": GDC_SAMPLE_TYPE_CODES_URL,
|
|
559
|
+
"one_sample_per_case": True,
|
|
560
|
+
"case_identifiers_published": False,
|
|
561
|
+
"cptac": "Explicit project/site/histology filters and primary sample types; "
|
|
562
|
+
"see per-cohort selection, coverage, and provenance",
|
|
563
|
+
},
|
|
564
|
+
"sources": tcga_sources or next(iter(cohort_details.values()))["sources"],
|
|
565
|
+
"requested_cohorts": cohorts,
|
|
566
|
+
"cohorts": dict(sorted(cohort_details.items())),
|
|
567
|
+
"genes": sorted(
|
|
568
|
+
catalog.values(),
|
|
569
|
+
key=lambda item: (item["symbol"].upper(), item["ensembl"]),
|
|
570
|
+
),
|
|
571
|
+
"assets": checksums,
|
|
572
|
+
}
|
|
573
|
+
manifest_name = f"manifest-{manifest['data_version']}.json"
|
|
574
|
+
manifest_path = outdir / manifest_name
|
|
575
|
+
manifest_path.write_bytes(_json_bytes(manifest))
|
|
576
|
+
checksum_lines = [
|
|
577
|
+
f"{details['sha256']} {name}" for name, details in sorted(checksums.items())
|
|
578
|
+
]
|
|
579
|
+
checksum_lines.append(
|
|
580
|
+
f"{hashlib.sha256(manifest_path.read_bytes()).hexdigest()} {manifest_name}"
|
|
581
|
+
)
|
|
582
|
+
(outdir / "SHA256SUMS").write_text("\n".join(checksum_lines) + "\n")
|
|
583
|
+
return manifest_path
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
587
|
+
parser = argparse.ArgumentParser(
|
|
588
|
+
prog="survscope-build-data",
|
|
589
|
+
description="Stream TCGA and CPTAC sources into compact SurvScope release assets.",
|
|
590
|
+
)
|
|
591
|
+
parser.add_argument(
|
|
592
|
+
"--cohorts",
|
|
593
|
+
nargs="+",
|
|
594
|
+
default=list(COHORTS),
|
|
595
|
+
help="TCGA abbreviations or CPTAC-3 codes; defaults to all 33 TCGA cohorts.",
|
|
596
|
+
)
|
|
597
|
+
parser.add_argument("--include-cptac", action="store_true",
|
|
598
|
+
help="Also build all supported CPTAC-3 RNA/OS cohorts.")
|
|
599
|
+
parser.add_argument("--outdir", required=True)
|
|
600
|
+
parser.add_argument("--data-version", default=DEFAULT_DATA_VERSION)
|
|
601
|
+
parser.add_argument("--survival-source", default=TCGA_CDR_URL)
|
|
602
|
+
parser.add_argument("--probemap-source", default=GDC_PROBEMAP_URL)
|
|
603
|
+
parser.add_argument(
|
|
604
|
+
"--tcga-release", type=Path,
|
|
605
|
+
help="Reuse a verified compact TCGA release directory without downloading raw matrices.",
|
|
606
|
+
)
|
|
607
|
+
parser.add_argument(
|
|
608
|
+
"--expression-file",
|
|
609
|
+
help="Local .tsv.gz for a one-cohort validation build.",
|
|
610
|
+
)
|
|
611
|
+
parser.add_argument("--expression-url-template", default=GDC_EXPRESSION_URL)
|
|
612
|
+
parser.add_argument(
|
|
613
|
+
"--expression-public-url",
|
|
614
|
+
help="Canonical URL recorded when --expression-file is used.",
|
|
615
|
+
)
|
|
616
|
+
parser.add_argument(
|
|
617
|
+
"--survival-public-url",
|
|
618
|
+
help="Canonical URL recorded when --survival-source is a local fixture.",
|
|
619
|
+
)
|
|
620
|
+
parser.add_argument(
|
|
621
|
+
"--genes",
|
|
622
|
+
nargs="+",
|
|
623
|
+
help="Optional gene-symbol subset for fixtures or smoke builds.",
|
|
624
|
+
)
|
|
625
|
+
return parser
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def main(argv: list[str] | None = None) -> int:
|
|
629
|
+
args = build_parser().parse_args(argv)
|
|
630
|
+
cohorts = [cohort.upper() for cohort in args.cohorts]
|
|
631
|
+
if args.include_cptac:
|
|
632
|
+
cohorts = list(dict.fromkeys([*cohorts, *CPTAC_COHORTS]))
|
|
633
|
+
unknown = sorted(set(cohorts) - set(COHORTS) - set(CPTAC_COHORTS))
|
|
634
|
+
if unknown:
|
|
635
|
+
raise SystemExit(f"Unsupported cohorts: {', '.join(unknown)}. "
|
|
636
|
+
"CPTAC-2 currently lacks usable GDC overall-survival records.")
|
|
637
|
+
if args.expression_file and len(cohorts) != 1:
|
|
638
|
+
raise SystemExit("--expression-file requires exactly one --cohorts value")
|
|
639
|
+
manifest = build_release(
|
|
640
|
+
cohorts=cohorts,
|
|
641
|
+
outdir=Path(args.outdir),
|
|
642
|
+
data_version=args.data_version,
|
|
643
|
+
survival_source=args.survival_source,
|
|
644
|
+
probemap_source=args.probemap_source,
|
|
645
|
+
expression_file=args.expression_file,
|
|
646
|
+
expression_url_template=args.expression_url_template,
|
|
647
|
+
wanted_genes={gene.upper() for gene in args.genes} if args.genes else None,
|
|
648
|
+
survival_public_url=args.survival_public_url,
|
|
649
|
+
expression_public_url=args.expression_public_url,
|
|
650
|
+
tcga_release=args.tcga_release,
|
|
651
|
+
)
|
|
652
|
+
print(manifest)
|
|
653
|
+
return 0
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
if __name__ == "__main__":
|
|
657
|
+
raise SystemExit(main())
|