survscope 0.4.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
survscope/builder.py ADDED
@@ -0,0 +1,657 @@
1
+ """Streaming builder for compact, versioned SurvScope static data assets."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import contextlib
7
+ import copy
8
+ import csv
9
+ import gzip
10
+ import hashlib
11
+ import io
12
+ import json
13
+ import math
14
+ import shutil
15
+ import tempfile
16
+ import urllib.request
17
+ import zipfile
18
+ from collections import Counter
19
+ from dataclasses import dataclass
20
+ from datetime import datetime, timezone
21
+ from pathlib import Path
22
+ from typing import Any, BinaryIO
23
+
24
+ import numpy as np
25
+
26
+ from .constants import (
27
+ BUCKET_COUNT,
28
+ COHORT_LABELS,
29
+ COHORT_PRIMARY_SAMPLE_CODES,
30
+ COHORTS,
31
+ DEFAULT_DATA_VERSION,
32
+ DEFAULT_PRIMARY_SAMPLE_CODES,
33
+ ENDPOINT_COLUMNS,
34
+ ENDPOINTS,
35
+ EXPRESSION_SCALE,
36
+ GDC_EXPRESSION_URL,
37
+ GDC_PIPELINE_URL,
38
+ GDC_PROBEMAP_URL,
39
+ GDC_SAMPLE_TYPE_CODES_URL,
40
+ MISSING_EXPRESSION,
41
+ SAMPLE_TYPE_LABELS,
42
+ SCHEMA_VERSION,
43
+ TCGA_CDR_CITATION_URL,
44
+ TCGA_CDR_URL,
45
+ )
46
+ from .cptac import CPTAC_COHORTS, build_cptac_cohort
47
+ from .quality import endpoint_quality
48
+ from .validation import validate_release
49
+
50
+
51
+ def _json_bytes(value: Any) -> bytes:
52
+ return (json.dumps(value, separators=(",", ":"), allow_nan=False) + "\n").encode()
53
+
54
+
55
+ def _parse_number(value: str) -> float | None:
56
+ try:
57
+ number = float(value)
58
+ except (TypeError, ValueError):
59
+ return None
60
+ return number if math.isfinite(number) else None
61
+
62
+
63
+ def _sample_case(sample: str) -> str:
64
+ return sample[:15]
65
+
66
+
67
+ def _sample_type_code(sample: str) -> str | None:
68
+ parts = sample.split("-")
69
+ return parts[3][:2] if len(parts) >= 4 and len(parts[3]) >= 2 else None
70
+
71
+
72
+ def _primary_sample_codes(cohort: str) -> tuple[str, ...]:
73
+ return COHORT_PRIMARY_SAMPLE_CODES.get(cohort.upper(), DEFAULT_PRIMARY_SAMPLE_CODES)
74
+
75
+
76
+ def _is_primary_cancer_sample(sample: str, cohort: str) -> bool:
77
+ return _sample_type_code(sample) in _primary_sample_codes(cohort)
78
+
79
+
80
+ def _bucket_for(symbol: str) -> str:
81
+ value = hashlib.sha256(symbol.upper().encode()).digest()[0] % BUCKET_COUNT
82
+ return f"{value:02x}"
83
+
84
+
85
+ class HashingReader:
86
+ """Binary reader that hashes source bytes as they stream past."""
87
+
88
+ def __init__(self, source: BinaryIO) -> None:
89
+ self.source = source
90
+ self.sha256 = hashlib.sha256()
91
+ self.byte_count = 0
92
+
93
+ def read(self, size: int = -1) -> bytes:
94
+ payload = self.source.read(size)
95
+ self.sha256.update(payload)
96
+ self.byte_count += len(payload)
97
+ return payload
98
+
99
+ def readinto(self, target) -> int:
100
+ payload = self.read(len(target))
101
+ target[: len(payload)] = payload
102
+ return len(payload)
103
+
104
+ def readable(self) -> bool:
105
+ return True
106
+
107
+ def close(self) -> None:
108
+ self.source.close()
109
+
110
+
111
+ @dataclass(frozen=True)
112
+ class SourceText:
113
+ text: str
114
+ url: str
115
+ sha256: str
116
+ byte_count: int
117
+
118
+
119
+ def _read_small_source(url_or_path: str, public_url: str | None = None) -> SourceText:
120
+ path = Path(url_or_path)
121
+ if path.is_file():
122
+ payload = path.read_bytes()
123
+ url = public_url or str(path.resolve())
124
+ else:
125
+ request = urllib.request.Request(url_or_path, headers={"User-Agent": "survscope/0.1"})
126
+ with urllib.request.urlopen(request, timeout=120) as response:
127
+ payload = response.read()
128
+ url = public_url or url_or_path
129
+ return SourceText(
130
+ text=payload.decode(),
131
+ url=url,
132
+ sha256=hashlib.sha256(payload).hexdigest(),
133
+ byte_count=len(payload),
134
+ )
135
+
136
+
137
+ def load_probemap(source: SourceText) -> tuple[dict[str, tuple[str, str]], int]:
138
+ """Return mappings only for symbols associated with exactly one Ensembl gene."""
139
+ reader = csv.DictReader(io.StringIO(source.text), delimiter="\t")
140
+ records = []
141
+ symbols = Counter()
142
+ for row in reader:
143
+ ensembl_version = row["id"].strip()
144
+ symbol = row["gene"].strip()
145
+ if not ensembl_version or not symbol:
146
+ continue
147
+ ensembl = ensembl_version.split(".", 1)[0]
148
+ symbol_key = symbol.upper()
149
+ records.append((ensembl_version, ensembl, symbol, symbol_key))
150
+ symbols[symbol_key] += 1
151
+ ambiguous = {symbol for symbol, count in symbols.items() if count != 1}
152
+ mapping: dict[str, tuple[str, str]] = {}
153
+ for ensembl_version, ensembl, symbol, symbol_key in records:
154
+ if symbol_key in ambiguous:
155
+ continue
156
+ mapping[ensembl_version] = (symbol, ensembl)
157
+ mapping.setdefault(ensembl, (symbol, ensembl))
158
+ return mapping, len(ambiguous)
159
+
160
+
161
+ def load_survival(source: SourceText) -> dict[str, dict[str, dict[str, float | None]]]:
162
+ result: dict[str, dict[str, dict[str, float | None]]] = {}
163
+ reader = csv.DictReader(io.StringIO(source.text), delimiter="\t")
164
+ for row in reader:
165
+ cohort = row["cancer type abbreviation"].strip().upper()
166
+ sample = row["sample"].strip()
167
+ if not cohort or not sample:
168
+ continue
169
+ endpoint_values = {}
170
+ for endpoint, (time_column, event_column) in ENDPOINT_COLUMNS.items():
171
+ endpoint_values[endpoint] = {
172
+ "time": _parse_number(row.get(time_column, "")),
173
+ "event": _parse_number(row.get(event_column, "")),
174
+ }
175
+ result.setdefault(cohort, {})[sample] = endpoint_values
176
+ return result
177
+
178
+
179
+ @contextlib.contextmanager
180
+ def _expression_stream(source: str) -> Any:
181
+ """Yield a decompressed text stream and its source-byte hasher."""
182
+ path = Path(source)
183
+ if path.is_file():
184
+ raw: BinaryIO = path.open("rb")
185
+ label = str(path.resolve())
186
+ else:
187
+ request = urllib.request.Request(source, headers={"User-Agent": "survscope/0.1"})
188
+ raw = urllib.request.urlopen(request, timeout=180)
189
+ label = source
190
+ hashing = HashingReader(raw)
191
+ compressed = gzip.GzipFile(fileobj=hashing, mode="rb")
192
+ text = io.TextIOWrapper(compressed, encoding="utf-8", newline="")
193
+ try:
194
+ yield text, hashing, label
195
+ finally:
196
+ text.close()
197
+
198
+
199
+ def _clinical_asset(
200
+ cohort: str,
201
+ sample_cases: list[str],
202
+ survival_rows: dict[str, dict[str, dict[str, float | None]]],
203
+ ) -> dict[str, Any]:
204
+ sample_codes = _primary_sample_codes(cohort)
205
+ sample_labels = [SAMPLE_TYPE_LABELS[code] for code in sample_codes]
206
+ endpoints = {}
207
+ for endpoint in ENDPOINTS:
208
+ quality, note = endpoint_quality(cohort, endpoint)
209
+ endpoints[endpoint] = {
210
+ "time": [survival_rows[sample][endpoint]["time"] for sample in sample_cases],
211
+ "event": [survival_rows[sample][endpoint]["event"] for sample in sample_cases],
212
+ "quality": quality,
213
+ "quality_note": note,
214
+ }
215
+ return {
216
+ "schema_version": SCHEMA_VERSION,
217
+ "cohort": cohort,
218
+ "sample_count": len(sample_cases),
219
+ "sample_type": " / ".join(sample_labels),
220
+ "sample_type_codes": list(sample_codes),
221
+ "identifiers_included": False,
222
+ "endpoints": endpoints,
223
+ }
224
+
225
+
226
+ def _median_metadata(
227
+ expression_log2: np.ndarray,
228
+ encoded: np.ndarray,
229
+ clinical: dict[str, Any],
230
+ ) -> dict[str, dict[str, Any]]:
231
+ exact_tpm = np.exp2(expression_log2) - 1.0
232
+ decoded_log2 = encoded.astype(float) / EXPRESSION_SCALE
233
+ decoded_log2[encoded == MISSING_EXPRESSION] = np.nan
234
+ decoded_tpm = np.exp2(decoded_log2) - 1.0
235
+ metadata = {}
236
+ for endpoint in ENDPOINTS:
237
+ time = np.asarray(
238
+ [
239
+ np.nan if value is None else value
240
+ for value in clinical["endpoints"][endpoint]["time"]
241
+ ],
242
+ dtype=float,
243
+ )
244
+ event = np.asarray(
245
+ [
246
+ np.nan if value is None else value
247
+ for value in clinical["endpoints"][endpoint]["event"]
248
+ ],
249
+ dtype=float,
250
+ )
251
+ valid = (
252
+ np.isfinite(exact_tpm) & np.isfinite(time) & np.isfinite(event) & (time > 0)
253
+ )
254
+ if not np.any(valid):
255
+ metadata[endpoint] = {"cutoff_tpm": None, "flips": []}
256
+ continue
257
+ cutoff = float(np.median(exact_tpm[valid]))
258
+ exact_high = exact_tpm > cutoff
259
+ encoded_high = decoded_tpm > cutoff
260
+ flips = np.flatnonzero(valid & (exact_high != encoded_high)).astype(int).tolist()
261
+ metadata[endpoint] = {"cutoff_tpm": cutoff, "flips": flips}
262
+ return metadata
263
+
264
+
265
+ def build_cohort(
266
+ cohort: str,
267
+ *,
268
+ expression_source: str,
269
+ survival: dict[str, dict[str, dict[str, float | None]]],
270
+ probemap: dict[str, tuple[str, str]],
271
+ outdir: Path,
272
+ wanted_genes: set[str] | None = None,
273
+ expression_public_url: str | None = None,
274
+ ) -> tuple[dict[str, Any], list[dict[str, Any]]]:
275
+ """Stream one cohort and emit only compact plot data assets."""
276
+ cohort = cohort.upper()
277
+ survival_rows = survival.get(cohort)
278
+ if not survival_rows:
279
+ raise RuntimeError(f"No TCGA-CDR rows found for {cohort}")
280
+ outdir.mkdir(parents=True, exist_ok=True)
281
+ bucket_ids = [f"{index:02x}" for index in range(BUCKET_COUNT)]
282
+ bucket_metadata: dict[str, list[dict[str, Any]]] = {key: [] for key in bucket_ids}
283
+ genes: list[dict[str, Any]] = []
284
+
285
+ with tempfile.TemporaryDirectory(prefix=f"survscope-{cohort.lower()}-") as temp_name:
286
+ temporary = Path(temp_name)
287
+ raw_paths = {key: temporary / f"{cohort}-bucket-{key}.u16le" for key in bucket_ids}
288
+ handles = {key: path.open("wb") for key, path in raw_paths.items()}
289
+ try:
290
+ with _expression_stream(expression_source) as (text, source_hash, source_label):
291
+ reader = csv.reader(text, delimiter="\t")
292
+ header = next(reader)
293
+ selected: list[tuple[int, str]] = []
294
+ seen_cases = set()
295
+ for index, sample in enumerate(header[1:]):
296
+ case = _sample_case(sample)
297
+ if (
298
+ _is_primary_cancer_sample(sample, cohort)
299
+ and case in survival_rows
300
+ and case not in seen_cases
301
+ ):
302
+ selected.append((index, case))
303
+ seen_cases.add(case)
304
+ if not selected:
305
+ codes = ", ".join(_primary_sample_codes(cohort))
306
+ raise RuntimeError(
307
+ f"No matched primary cancer samples found for {cohort} "
308
+ f"(TCGA sample codes: {codes})"
309
+ )
310
+ selected_indices = [index for index, _ in selected]
311
+ sample_cases = [case for _, case in selected]
312
+ clinical = _clinical_asset(cohort, sample_cases, survival_rows)
313
+ processed_rows = 0
314
+ included_rows = 0
315
+ for row in reader:
316
+ processed_rows += 1
317
+ if not row:
318
+ continue
319
+ identifier = row[0].strip()
320
+ mapped = probemap.get(identifier) or probemap.get(
321
+ identifier.split(".", 1)[0]
322
+ )
323
+ if mapped is None:
324
+ continue
325
+ symbol, ensembl = mapped
326
+ if wanted_genes and symbol.upper() not in wanted_genes:
327
+ continue
328
+ values = np.asarray(
329
+ [
330
+ float(row[index + 1]) if row[index + 1] else np.nan
331
+ for index in selected_indices
332
+ ],
333
+ dtype=float,
334
+ )
335
+ encoded = np.full(len(values), MISSING_EXPRESSION, dtype="<u2")
336
+ finite = np.isfinite(values)
337
+ quantized = np.rint(values[finite] * EXPRESSION_SCALE)
338
+ quantized = np.clip(quantized, 0, MISSING_EXPRESSION - 1)
339
+ encoded[finite] = quantized.astype("<u2")
340
+ bucket = _bucket_for(symbol)
341
+ row_index = len(bucket_metadata[bucket])
342
+ handles[bucket].write(encoded.tobytes())
343
+ record = {
344
+ "symbol": symbol,
345
+ "ensembl": ensembl,
346
+ "row": row_index,
347
+ "medians": _median_metadata(values, encoded, clinical),
348
+ }
349
+ bucket_metadata[bucket].append(record)
350
+ genes.append(
351
+ {
352
+ "symbol": symbol,
353
+ "ensembl": ensembl,
354
+ "bucket": bucket,
355
+ }
356
+ )
357
+ included_rows += 1
358
+ if processed_rows % 5000 == 0:
359
+ print(
360
+ f"{cohort}: streamed {processed_rows:,} source rows; "
361
+ f"kept {included_rows:,}",
362
+ flush=True,
363
+ )
364
+ source_details = {
365
+ "url": expression_public_url or source_label,
366
+ "sha256": source_hash.sha256.hexdigest(),
367
+ "compressed_bytes_streamed": source_hash.byte_count,
368
+ }
369
+ finally:
370
+ for handle in handles.values():
371
+ handle.close()
372
+
373
+ bucket_assets = {}
374
+ for bucket in bucket_ids:
375
+ records = bucket_metadata[bucket]
376
+ if not records:
377
+ continue
378
+ asset_name = f"{cohort}-bucket-{bucket}.zip"
379
+ asset_path = outdir / asset_name
380
+ meta = {
381
+ "schema_version": SCHEMA_VERSION,
382
+ "cohort": cohort,
383
+ "sample_count": len(sample_cases),
384
+ "scale": EXPRESSION_SCALE,
385
+ "missing": MISSING_EXPRESSION,
386
+ "transform": "log2(TPM+1)",
387
+ "genes": records,
388
+ }
389
+ with zipfile.ZipFile(
390
+ asset_path,
391
+ mode="w",
392
+ compression=zipfile.ZIP_DEFLATED,
393
+ compresslevel=9,
394
+ ) as archive:
395
+ archive.writestr("meta.json", _json_bytes(meta))
396
+ archive.write(raw_paths[bucket], arcname="expression.u16le")
397
+ bucket_assets[bucket] = asset_name
398
+
399
+ clinical_name = f"{cohort}-clinical.json"
400
+ (outdir / clinical_name).write_bytes(_json_bytes(clinical))
401
+ cohort_manifest = {
402
+ "program": "TCGA",
403
+ "label": COHORT_LABELS[cohort],
404
+ "sample_count": len(sample_cases),
405
+ "gene_count": len(genes),
406
+ "sample_type_codes": list(_primary_sample_codes(cohort)),
407
+ "clinical_asset": clinical_name,
408
+ "bucket_assets": bucket_assets,
409
+ "source": source_details,
410
+ }
411
+ print(
412
+ f"{cohort}: completed {len(genes):,} genes across {len(bucket_assets)} buckets "
413
+ f"for {len(sample_cases):,} matched primary cancer samples",
414
+ flush=True,
415
+ )
416
+ return cohort_manifest, genes
417
+
418
+
419
+ def _asset_checksums(outdir: Path) -> dict[str, dict[str, Any]]:
420
+ checksums = {}
421
+ for path in sorted(outdir.iterdir()):
422
+ if path.is_file() and not path.name.startswith(("manifest-", "SHA256SUMS")):
423
+ payload = path.read_bytes()
424
+ checksums[path.name] = {
425
+ "sha256": hashlib.sha256(payload).hexdigest(),
426
+ "bytes": len(payload),
427
+ }
428
+ return checksums
429
+
430
+
431
+ def build_release(
432
+ *,
433
+ cohorts: list[str],
434
+ outdir: Path,
435
+ data_version: str,
436
+ survival_source: str,
437
+ probemap_source: str,
438
+ expression_file: str | None = None,
439
+ expression_url_template: str = GDC_EXPRESSION_URL,
440
+ wanted_genes: set[str] | None = None,
441
+ survival_public_url: str | None = None,
442
+ expression_public_url: str | None = None,
443
+ tcga_release: Path | None = None,
444
+ ) -> Path:
445
+ """Build one immutable data-release directory."""
446
+ data_version = data_version.removeprefix("data-v")
447
+ if datetime.strptime(data_version, "%Y.%m.%d").strftime("%Y.%m.%d") != data_version:
448
+ raise ValueError("Data version must use YYYY.MM.DD")
449
+ cohorts = list(dict.fromkeys(cohort.upper() for cohort in cohorts))
450
+ unknown = set(cohorts) - set(COHORTS) - set(CPTAC_COHORTS)
451
+ if not cohorts or unknown:
452
+ raise ValueError(f"Unsupported or empty cohort selection: {sorted(unknown)}")
453
+ reused_manifest = None
454
+ reused_provenance = None
455
+ if tcga_release is not None:
456
+ if wanted_genes or expression_file:
457
+ raise ValueError("--tcga-release cannot be combined with --genes or --expression-file")
458
+ reused_manifest = validate_release(tcga_release)
459
+ missing = (set(cohorts) & set(COHORTS)) - set(reused_manifest["cohorts"])
460
+ if missing:
461
+ raise ValueError(f"TCGA release is missing requested cohorts: {sorted(missing)}")
462
+ reused_provenance = {
463
+ "data_version": reused_manifest["data_version"],
464
+ "manifest_sha256": hashlib.sha256(
465
+ next(tcga_release.glob("manifest-*.json")).read_bytes()
466
+ ).hexdigest(),
467
+ }
468
+ if outdir.exists() and any(outdir.iterdir()):
469
+ raise ValueError("Release directory must be empty; existing data releases are immutable")
470
+ outdir.mkdir(parents=True, exist_ok=True)
471
+ tcga_sources = {}
472
+ survival = {}
473
+ probemap = {}
474
+ if reused_manifest is not None:
475
+ tcga_sources = reused_manifest["sources"]
476
+ elif any(cohort in COHORTS for cohort in cohorts):
477
+ survival_text = _read_small_source(survival_source, public_url=survival_public_url)
478
+ probemap_text = _read_small_source(probemap_source)
479
+ survival = load_survival(survival_text)
480
+ probemap, ambiguous_count = load_probemap(probemap_text)
481
+ tcga_sources = {
482
+ "expression": {
483
+ "label": "GDC STAR TPM", "pipeline": GDC_PIPELINE_URL,
484
+ "dataset_template": expression_url_template,
485
+ "wrangling": "Xena GDC ETL; log2(TPM+1)",
486
+ },
487
+ "survival": {
488
+ "label": "PanCanAtlas TCGA-CDR", "url": survival_text.url,
489
+ "sha256": survival_text.sha256, "bytes": survival_text.byte_count,
490
+ "citation": TCGA_CDR_CITATION_URL,
491
+ },
492
+ "gene_map": {
493
+ "label": "GENCODE v36 gene probemap", "url": probemap_text.url,
494
+ "sha256": probemap_text.sha256, "bytes": probemap_text.byte_count,
495
+ "ambiguous_symbols_excluded": ambiguous_count,
496
+ },
497
+ }
498
+ catalog: dict[tuple[str, str], dict[str, Any]] = {}
499
+ cohort_details = {}
500
+ for cohort in cohorts:
501
+ if cohort in CPTAC_COHORTS:
502
+ if expression_file:
503
+ raise ValueError("--expression-file is only supported for TCGA builds")
504
+ details, genes = build_cptac_cohort(cohort, outdir=outdir, wanted_genes=wanted_genes)
505
+ elif reused_manifest is not None:
506
+ details = copy.deepcopy(reused_manifest["cohorts"][cohort])
507
+ if details.get("program", "TCGA") != "TCGA":
508
+ raise ValueError(f"{cohort} is not TCGA in the supplied compact release")
509
+ details["program"] = "TCGA"
510
+ details.setdefault("sources", tcga_sources)
511
+ details["reused_from"] = reused_provenance
512
+ for name in [details["clinical_asset"], *details["bucket_assets"].values()]:
513
+ shutil.copyfile(tcga_release / name, outdir / name)
514
+ genes = [
515
+ {key: value for key, value in gene.items() if key != "cohorts"}
516
+ for gene in reused_manifest["genes"] if cohort in gene["cohorts"]
517
+ ]
518
+ else:
519
+ expression_source = expression_file or expression_url_template.format(cohort=cohort)
520
+ details, genes = build_cohort(
521
+ cohort, expression_source=expression_source, survival=survival,
522
+ probemap=probemap, outdir=outdir, wanted_genes=wanted_genes,
523
+ expression_public_url=(
524
+ expression_public_url or expression_url_template.format(cohort=cohort)
525
+ ),
526
+ )
527
+ details["sources"] = tcga_sources
528
+ cohort_details[cohort] = details
529
+ print(f"Built {cohort}: {details['sample_count']} patients, "
530
+ f"{details['gene_count']} genes", flush=True)
531
+ for gene in genes:
532
+ key = (gene["symbol"].upper(), gene["ensembl"])
533
+ entry = catalog.setdefault(key, {**gene, "cohorts": []})
534
+ entry["cohorts"].append(cohort)
535
+
536
+ checksums = _asset_checksums(outdir)
537
+ manifest = {
538
+ # Older readers hard-code TCGA names and release-level provenance.
539
+ "schema_version": 2 if any(code in CPTAC_COHORTS for code in cohorts) else SCHEMA_VERSION,
540
+ "data_version": data_version.removeprefix("data-v"),
541
+ "created_at": datetime.now(timezone.utc).isoformat(),
542
+ "expression_encoding": {
543
+ "transform": "log2(TPM+1)",
544
+ "storage": "uint16 little-endian",
545
+ "scale": EXPRESSION_SCALE,
546
+ "missing": MISSING_EXPRESSION,
547
+ "maximum_quantization_error_log2": 0.0005,
548
+ "exact_median_membership_corrections": True,
549
+ },
550
+ "sample_policy": {
551
+ "sample_type": "Cohort-specific primary cancer specimen",
552
+ "default_tcga_sample_codes": list(DEFAULT_PRIMARY_SAMPLE_CODES),
553
+ "cohort_overrides": {
554
+ cohort: list(codes)
555
+ for cohort, codes in sorted(COHORT_PRIMARY_SAMPLE_CODES.items())
556
+ },
557
+ "code_definitions": SAMPLE_TYPE_LABELS,
558
+ "code_table": GDC_SAMPLE_TYPE_CODES_URL,
559
+ "one_sample_per_case": True,
560
+ "case_identifiers_published": False,
561
+ "cptac": "Explicit project/site/histology filters and primary sample types; "
562
+ "see per-cohort selection, coverage, and provenance",
563
+ },
564
+ "sources": tcga_sources or next(iter(cohort_details.values()))["sources"],
565
+ "requested_cohorts": cohorts,
566
+ "cohorts": dict(sorted(cohort_details.items())),
567
+ "genes": sorted(
568
+ catalog.values(),
569
+ key=lambda item: (item["symbol"].upper(), item["ensembl"]),
570
+ ),
571
+ "assets": checksums,
572
+ }
573
+ manifest_name = f"manifest-{manifest['data_version']}.json"
574
+ manifest_path = outdir / manifest_name
575
+ manifest_path.write_bytes(_json_bytes(manifest))
576
+ checksum_lines = [
577
+ f"{details['sha256']} {name}" for name, details in sorted(checksums.items())
578
+ ]
579
+ checksum_lines.append(
580
+ f"{hashlib.sha256(manifest_path.read_bytes()).hexdigest()} {manifest_name}"
581
+ )
582
+ (outdir / "SHA256SUMS").write_text("\n".join(checksum_lines) + "\n")
583
+ return manifest_path
584
+
585
+
586
+ def build_parser() -> argparse.ArgumentParser:
587
+ parser = argparse.ArgumentParser(
588
+ prog="survscope-build-data",
589
+ description="Stream TCGA and CPTAC sources into compact SurvScope release assets.",
590
+ )
591
+ parser.add_argument(
592
+ "--cohorts",
593
+ nargs="+",
594
+ default=list(COHORTS),
595
+ help="TCGA abbreviations or CPTAC-3 codes; defaults to all 33 TCGA cohorts.",
596
+ )
597
+ parser.add_argument("--include-cptac", action="store_true",
598
+ help="Also build all supported CPTAC-3 RNA/OS cohorts.")
599
+ parser.add_argument("--outdir", required=True)
600
+ parser.add_argument("--data-version", default=DEFAULT_DATA_VERSION)
601
+ parser.add_argument("--survival-source", default=TCGA_CDR_URL)
602
+ parser.add_argument("--probemap-source", default=GDC_PROBEMAP_URL)
603
+ parser.add_argument(
604
+ "--tcga-release", type=Path,
605
+ help="Reuse a verified compact TCGA release directory without downloading raw matrices.",
606
+ )
607
+ parser.add_argument(
608
+ "--expression-file",
609
+ help="Local .tsv.gz for a one-cohort validation build.",
610
+ )
611
+ parser.add_argument("--expression-url-template", default=GDC_EXPRESSION_URL)
612
+ parser.add_argument(
613
+ "--expression-public-url",
614
+ help="Canonical URL recorded when --expression-file is used.",
615
+ )
616
+ parser.add_argument(
617
+ "--survival-public-url",
618
+ help="Canonical URL recorded when --survival-source is a local fixture.",
619
+ )
620
+ parser.add_argument(
621
+ "--genes",
622
+ nargs="+",
623
+ help="Optional gene-symbol subset for fixtures or smoke builds.",
624
+ )
625
+ return parser
626
+
627
+
628
+ def main(argv: list[str] | None = None) -> int:
629
+ args = build_parser().parse_args(argv)
630
+ cohorts = [cohort.upper() for cohort in args.cohorts]
631
+ if args.include_cptac:
632
+ cohorts = list(dict.fromkeys([*cohorts, *CPTAC_COHORTS]))
633
+ unknown = sorted(set(cohorts) - set(COHORTS) - set(CPTAC_COHORTS))
634
+ if unknown:
635
+ raise SystemExit(f"Unsupported cohorts: {', '.join(unknown)}. "
636
+ "CPTAC-2 currently lacks usable GDC overall-survival records.")
637
+ if args.expression_file and len(cohorts) != 1:
638
+ raise SystemExit("--expression-file requires exactly one --cohorts value")
639
+ manifest = build_release(
640
+ cohorts=cohorts,
641
+ outdir=Path(args.outdir),
642
+ data_version=args.data_version,
643
+ survival_source=args.survival_source,
644
+ probemap_source=args.probemap_source,
645
+ expression_file=args.expression_file,
646
+ expression_url_template=args.expression_url_template,
647
+ wanted_genes={gene.upper() for gene in args.genes} if args.genes else None,
648
+ survival_public_url=args.survival_public_url,
649
+ expression_public_url=args.expression_public_url,
650
+ tcga_release=args.tcga_release,
651
+ )
652
+ print(manifest)
653
+ return 0
654
+
655
+
656
+ if __name__ == "__main__":
657
+ raise SystemExit(main())