cellme 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cellme/__init__.py +3 -0
- cellme/builds.py +128 -0
- cellme/cbioportal.py +331 -0
- cellme/main.py +34 -0
- cellme/py.typed +0 -0
- cellme/tools/__init__.py +0 -0
- cellme/tools/truth_track.py +77 -0
- cellme/vcf.py +472 -0
- cellme-1.0.0.dist-info/METADATA +119 -0
- cellme-1.0.0.dist-info/RECORD +13 -0
- cellme-1.0.0.dist-info/WHEEL +4 -0
- cellme-1.0.0.dist-info/entry_points.txt +2 -0
- cellme-1.0.0.dist-info/licenses/LICENSE +21 -0
cellme/__init__.py
ADDED
cellme/builds.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Human genome build definitions and their reference contig tables."""
|
|
2
|
+
|
|
3
|
+
from enum import StrEnum
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class GenomeBuild(StrEnum):
|
|
7
|
+
"""
|
|
8
|
+
A human reference genome build supported by cellme.
|
|
9
|
+
|
|
10
|
+
Only the two builds relevant to CCLE truth tracks are modeled: GRCh37, the
|
|
11
|
+
build that CCLE reports its coordinates against, and GRCh38, the current
|
|
12
|
+
human reference. The ``hg19`` and ``hg38`` aliases resolve to GRCh37 and
|
|
13
|
+
GRCh38 respectively when the enum is constructed from a string.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
GRCh38 = "GRCh38"
|
|
17
|
+
GRCh37 = "GRCh37"
|
|
18
|
+
|
|
19
|
+
@classmethod
|
|
20
|
+
def _missing_(cls, value: object) -> "GenomeBuild | None":
|
|
21
|
+
"""
|
|
22
|
+
Resolve the ``hg19`` and ``hg38`` aliases case-insensitively.
|
|
23
|
+
|
|
24
|
+
Args:
|
|
25
|
+
value: The value that did not match a canonical member name.
|
|
26
|
+
|
|
27
|
+
Returns:
|
|
28
|
+
The matching build for a known alias, otherwise None.
|
|
29
|
+
"""
|
|
30
|
+
aliases = {"hg38": cls.GRCh38, "hg19": cls.GRCh37}
|
|
31
|
+
if isinstance(value, str):
|
|
32
|
+
return aliases.get(value.lower())
|
|
33
|
+
return None
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def ucsc_name(self) -> str:
|
|
37
|
+
"""The UCSC-style assembly name used by liftover chain files."""
|
|
38
|
+
return {GenomeBuild.GRCh38: "hg38", GenomeBuild.GRCh37: "hg19"}[self]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
GRCH37_CONTIGS: tuple[tuple[str, int], ...] = (
|
|
42
|
+
("1", 249250621),
|
|
43
|
+
("2", 243199373),
|
|
44
|
+
("3", 198022430),
|
|
45
|
+
("4", 191154276),
|
|
46
|
+
("5", 180915260),
|
|
47
|
+
("6", 171115067),
|
|
48
|
+
("7", 159138663),
|
|
49
|
+
("8", 146364022),
|
|
50
|
+
("9", 141213431),
|
|
51
|
+
("10", 135534747),
|
|
52
|
+
("11", 135006516),
|
|
53
|
+
("12", 133851895),
|
|
54
|
+
("13", 115169878),
|
|
55
|
+
("14", 107349540),
|
|
56
|
+
("15", 102531392),
|
|
57
|
+
("16", 90354753),
|
|
58
|
+
("17", 81195210),
|
|
59
|
+
("18", 78077248),
|
|
60
|
+
("19", 59128983),
|
|
61
|
+
("20", 63025520),
|
|
62
|
+
("21", 48129895),
|
|
63
|
+
("22", 51304566),
|
|
64
|
+
("X", 155270560),
|
|
65
|
+
("Y", 59373566),
|
|
66
|
+
("MT", 16569),
|
|
67
|
+
)
|
|
68
|
+
"""Primary-assembly contig names and lengths for GRCh37 (Ensembl naming)."""
|
|
69
|
+
|
|
70
|
+
GRCH38_CONTIGS: tuple[tuple[str, int], ...] = (
|
|
71
|
+
("1", 248956422),
|
|
72
|
+
("2", 242193529),
|
|
73
|
+
("3", 198295559),
|
|
74
|
+
("4", 190214555),
|
|
75
|
+
("5", 181538259),
|
|
76
|
+
("6", 170805979),
|
|
77
|
+
("7", 159345973),
|
|
78
|
+
("8", 145138636),
|
|
79
|
+
("9", 138394717),
|
|
80
|
+
("10", 133797422),
|
|
81
|
+
("11", 135086622),
|
|
82
|
+
("12", 133275309),
|
|
83
|
+
("13", 114364328),
|
|
84
|
+
("14", 107043718),
|
|
85
|
+
("15", 101991189),
|
|
86
|
+
("16", 90338345),
|
|
87
|
+
("17", 83257441),
|
|
88
|
+
("18", 80373285),
|
|
89
|
+
("19", 58617616),
|
|
90
|
+
("20", 64444167),
|
|
91
|
+
("21", 46709983),
|
|
92
|
+
("22", 50818468),
|
|
93
|
+
("X", 156040895),
|
|
94
|
+
("Y", 57227415),
|
|
95
|
+
("MT", 16569),
|
|
96
|
+
)
|
|
97
|
+
"""Primary-assembly contig names and lengths for GRCh38 (Ensembl naming)."""
|
|
98
|
+
|
|
99
|
+
_CONTIGS_BY_BUILD: dict[GenomeBuild, tuple[tuple[str, int], ...]] = {
|
|
100
|
+
GenomeBuild.GRCh37: GRCH37_CONTIGS,
|
|
101
|
+
GenomeBuild.GRCh38: GRCH38_CONTIGS,
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def contigs_for(build: GenomeBuild) -> tuple[tuple[str, int], ...]:
|
|
106
|
+
"""
|
|
107
|
+
Return the ordered primary-assembly contigs for a build.
|
|
108
|
+
|
|
109
|
+
Args:
|
|
110
|
+
build: The genome build to look up.
|
|
111
|
+
|
|
112
|
+
Returns:
|
|
113
|
+
A tuple of (contig name, length) pairs in canonical karyotype order.
|
|
114
|
+
"""
|
|
115
|
+
return _CONTIGS_BY_BUILD[build]
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def contig_order(build: GenomeBuild) -> dict[str, int]:
|
|
119
|
+
"""
|
|
120
|
+
Return a mapping from contig name to its sort index for a build.
|
|
121
|
+
|
|
122
|
+
Args:
|
|
123
|
+
build: The genome build to look up.
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
A mapping from contig name to a zero-based rank in karyotype order.
|
|
127
|
+
"""
|
|
128
|
+
return {name: index for index, (name, _length) in enumerate(contigs_for(build))}
|
cellme/cbioportal.py
ADDED
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
"""Client for the cBioPortal REST API and its CCLE mutation study."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import requests
|
|
9
|
+
from rapidfuzz import fuzz
|
|
10
|
+
from rapidfuzz import process
|
|
11
|
+
|
|
12
|
+
CBIOPORTAL_API: str = "https://www.cbioportal.org/api"
|
|
13
|
+
"""Base URL of the public cBioPortal REST API."""
|
|
14
|
+
|
|
15
|
+
DEFAULT_STUDY: str = "ccle_broad_2019"
|
|
16
|
+
"""The Cancer Cell Line Encyclopedia study used as the mutation source."""
|
|
17
|
+
|
|
18
|
+
_REQUEST_TIMEOUT: float = 60.0
|
|
19
|
+
"""Seconds to wait on any single cBioPortal request before giving up."""
|
|
20
|
+
|
|
21
|
+
_MISSING_VALUES: frozenset[str] = frozenset({"", ".", "NA", "N/A", "NULL"})
|
|
22
|
+
"""String placeholders that cBioPortal uses to mean a value is absent."""
|
|
23
|
+
|
|
24
|
+
_SUGGESTION_LIMIT: int = 5
|
|
25
|
+
"""How many close cell-line names to offer when a query does not resolve."""
|
|
26
|
+
|
|
27
|
+
_SUGGESTION_SCORE_CUTOFF: float = 70.0
|
|
28
|
+
"""The minimum rapidfuzz similarity, out of 100, for a name to be suggested."""
|
|
29
|
+
|
|
30
|
+
_CHROMOSOME_ALIASES: dict[str, str] = {"23": "X", "24": "Y", "25": "MT", "M": "MT"}
|
|
31
|
+
"""CCLE numeric chromosome codes mapped to their standard contig names."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class CellLineError(ValueError):
|
|
35
|
+
"""Base class for failures to resolve a cell line query to a sample."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class CellLineNotFoundError(CellLineError):
|
|
39
|
+
"""Raised when a query matches no cell line in the study."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class AmbiguousCellLineError(CellLineError):
|
|
43
|
+
"""Raised when a query matches more than one cell line in the study."""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class Mutation:
|
|
48
|
+
"""
|
|
49
|
+
A single somatic mutation call for one cell line in the CCLE study.
|
|
50
|
+
|
|
51
|
+
The fields mirror the subset of the cBioPortal mutation record needed to
|
|
52
|
+
build a VCF record. Insertions and deletions follow the MAF convention in
|
|
53
|
+
which the absent allele is encoded as a single dash.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
gene: str
|
|
57
|
+
entrez_gene_id: int | None
|
|
58
|
+
chromosome: str
|
|
59
|
+
start_position: int
|
|
60
|
+
end_position: int
|
|
61
|
+
reference_allele: str
|
|
62
|
+
variant_allele: str
|
|
63
|
+
protein_change: str | None
|
|
64
|
+
variant_class: str
|
|
65
|
+
variant_type: str
|
|
66
|
+
ncbi_build: str
|
|
67
|
+
refseq_mrna_id: str | None
|
|
68
|
+
protein_pos_start: int | None
|
|
69
|
+
protein_pos_end: int | None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _clean(value: object) -> str | None:
|
|
73
|
+
"""
|
|
74
|
+
Normalize a cBioPortal string field, mapping placeholders to None.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
value: The raw field value from the API response.
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
The stripped string, or None when the value is missing or a placeholder.
|
|
81
|
+
"""
|
|
82
|
+
if value is None:
|
|
83
|
+
return None
|
|
84
|
+
text = str(value).strip()
|
|
85
|
+
if text.upper() in _MISSING_VALUES:
|
|
86
|
+
return None
|
|
87
|
+
return text
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def normalize_cell_line(name: str) -> str:
|
|
91
|
+
"""
|
|
92
|
+
Reduce a cell line name to an uppercase alphanumeric matching key.
|
|
93
|
+
|
|
94
|
+
This lets a user pass ``MOLT-4``, ``MOLT4``, or ``molt 4`` and have them all
|
|
95
|
+
compare equal to the leading token of a CCLE sample identifier.
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
name: A cell line name or CCLE sample identifier token.
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
The name uppercased with every non-alphanumeric character removed.
|
|
102
|
+
"""
|
|
103
|
+
return re.sub(r"[^0-9A-Za-z]", "", name).upper()
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def cell_line_name(sample_id: str) -> str:
|
|
107
|
+
"""
|
|
108
|
+
Return the leading cell line token of a CCLE sample identifier.
|
|
109
|
+
|
|
110
|
+
CCLE sample identifiers join a cell line name to its tissue of origin with
|
|
111
|
+
underscores, for example ``MOLT4_HAEMATOPOIETIC_AND_LYMPHOID_TISSUE``.
|
|
112
|
+
|
|
113
|
+
Args:
|
|
114
|
+
sample_id: A CCLE sample identifier.
|
|
115
|
+
|
|
116
|
+
Returns:
|
|
117
|
+
The substring before the first underscore.
|
|
118
|
+
"""
|
|
119
|
+
return sample_id.split("_", 1)[0]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def humanize_cell_line(token: str) -> str:
|
|
123
|
+
"""
|
|
124
|
+
Render a CCLE cell-line token in its conventional hyphenated form.
|
|
125
|
+
|
|
126
|
+
A token of a letter prefix followed by a trailing digit run is split with a
|
|
127
|
+
hyphen, so ``MOLT4`` becomes ``MOLT-4``. This is a light heuristic for
|
|
128
|
+
readability; tokens that do not fit that simple shape are returned unchanged.
|
|
129
|
+
|
|
130
|
+
Args:
|
|
131
|
+
token: The leading cell line token of a CCLE sample identifier.
|
|
132
|
+
|
|
133
|
+
Returns:
|
|
134
|
+
The token with a hyphen inserted before a trailing digit run when one
|
|
135
|
+
applies, otherwise the token unchanged.
|
|
136
|
+
"""
|
|
137
|
+
match = re.fullmatch(r"([A-Za-z]+)(\d+)", token)
|
|
138
|
+
if match is None:
|
|
139
|
+
return token
|
|
140
|
+
return f"{match.group(1)}-{match.group(2)}"
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def suggest_cell_lines(
|
|
144
|
+
query: str,
|
|
145
|
+
sample_ids: Iterable[str],
|
|
146
|
+
*,
|
|
147
|
+
limit: int = _SUGGESTION_LIMIT,
|
|
148
|
+
score_cutoff: float = _SUGGESTION_SCORE_CUTOFF,
|
|
149
|
+
) -> list[str]:
|
|
150
|
+
"""
|
|
151
|
+
Return the closest CCLE cell-line names to a query by fuzzy similarity.
|
|
152
|
+
|
|
153
|
+
The query is compared against the unique leading cell-line token of every
|
|
154
|
+
sample, on their normalized alphanumeric form, using a Levenshtein-based
|
|
155
|
+
similarity. Only names scoring at or above the cutoff are kept, so a query
|
|
156
|
+
with no near neighbors yields an empty list.
|
|
157
|
+
|
|
158
|
+
Args:
|
|
159
|
+
query: The unresolved cell line identifier supplied by the user.
|
|
160
|
+
sample_ids: All sample identifiers available in the study.
|
|
161
|
+
limit: The maximum number of names to return.
|
|
162
|
+
score_cutoff: The minimum similarity, out of 100, to include a name.
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
Up to ``limit`` human-friendly cell-line names, best match first.
|
|
166
|
+
"""
|
|
167
|
+
display_names: list[str] = []
|
|
168
|
+
normalized_names: list[str] = []
|
|
169
|
+
seen: set[str] = set()
|
|
170
|
+
for sample_id in sample_ids:
|
|
171
|
+
token = cell_line_name(sample_id)
|
|
172
|
+
key = normalize_cell_line(token)
|
|
173
|
+
if key in seen:
|
|
174
|
+
continue
|
|
175
|
+
seen.add(key)
|
|
176
|
+
display_names.append(token)
|
|
177
|
+
normalized_names.append(key)
|
|
178
|
+
|
|
179
|
+
matches = process.extract(
|
|
180
|
+
normalize_cell_line(query),
|
|
181
|
+
normalized_names,
|
|
182
|
+
scorer=fuzz.ratio,
|
|
183
|
+
limit=limit,
|
|
184
|
+
score_cutoff=score_cutoff,
|
|
185
|
+
)
|
|
186
|
+
return [humanize_cell_line(display_names[index]) for _name, _score, index in matches]
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def resolve_sample(query: str, sample_ids: Iterable[str]) -> str:
|
|
190
|
+
"""
|
|
191
|
+
Resolve a cell line query to a single CCLE sample identifier.
|
|
192
|
+
|
|
193
|
+
A query first tries to match a whole sample identifier, then falls back to
|
|
194
|
+
matching the leading cell line token, both compared on their normalized
|
|
195
|
+
alphanumeric form.
|
|
196
|
+
|
|
197
|
+
Args:
|
|
198
|
+
query: The cell line identifier supplied by the user.
|
|
199
|
+
sample_ids: All sample identifiers available in the study.
|
|
200
|
+
|
|
201
|
+
Returns:
|
|
202
|
+
The single matching sample identifier.
|
|
203
|
+
|
|
204
|
+
Raises:
|
|
205
|
+
CellLineNotFoundError: If no sample matches the query.
|
|
206
|
+
AmbiguousCellLineError: If more than one sample matches the query.
|
|
207
|
+
"""
|
|
208
|
+
wanted = normalize_cell_line(query)
|
|
209
|
+
identifiers = list(sample_ids)
|
|
210
|
+
|
|
211
|
+
whole = [s for s in identifiers if normalize_cell_line(s) == wanted]
|
|
212
|
+
if len(whole) == 1:
|
|
213
|
+
return whole[0]
|
|
214
|
+
|
|
215
|
+
by_token = [s for s in identifiers if normalize_cell_line(cell_line_name(s)) == wanted]
|
|
216
|
+
if len(by_token) == 1:
|
|
217
|
+
return by_token[0]
|
|
218
|
+
if not by_token:
|
|
219
|
+
suggestions = suggest_cell_lines(query, identifiers)
|
|
220
|
+
if suggestions:
|
|
221
|
+
hint = "Did you mean: " + ", ".join(suggestions) + "?"
|
|
222
|
+
else:
|
|
223
|
+
hint = "No close matches."
|
|
224
|
+
raise CellLineNotFoundError(f"no CCLE cell line matched {query!r}. {hint}")
|
|
225
|
+
raise AmbiguousCellLineError(
|
|
226
|
+
f"Query {query!r} matched multiple cell lines: {', '.join(sorted(by_token))}."
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _normalize_chromosome(chromosome: str) -> str:
|
|
231
|
+
"""
|
|
232
|
+
Map a CCLE chromosome code to its standard contig name.
|
|
233
|
+
|
|
234
|
+
CCLE reports the sex and mitochondrial chromosomes with numeric codes, where
|
|
235
|
+
23 is X, 24 is Y, and 25 is the mitochondrion.
|
|
236
|
+
|
|
237
|
+
Args:
|
|
238
|
+
chromosome: The raw ``chr`` value from a cBioPortal record.
|
|
239
|
+
|
|
240
|
+
Returns:
|
|
241
|
+
The standard contig name, unchanged when no alias applies.
|
|
242
|
+
"""
|
|
243
|
+
return _CHROMOSOME_ALIASES.get(chromosome, chromosome)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _mutation_from_record(record: dict[str, Any]) -> Mutation:
|
|
247
|
+
"""
|
|
248
|
+
Build a Mutation from one cBioPortal mutation record.
|
|
249
|
+
|
|
250
|
+
Args:
|
|
251
|
+
record: A decoded JSON object from the mutations fetch endpoint.
|
|
252
|
+
|
|
253
|
+
Returns:
|
|
254
|
+
The parsed Mutation.
|
|
255
|
+
"""
|
|
256
|
+
gene = record.get("gene") or {}
|
|
257
|
+
symbol = gene.get("hugoGeneSymbol") or record.get("hugoGeneSymbol") or ""
|
|
258
|
+
entrez = record.get("entrezGeneId")
|
|
259
|
+
if entrez is None:
|
|
260
|
+
entrez = gene.get("entrezGeneId")
|
|
261
|
+
return Mutation(
|
|
262
|
+
gene=symbol,
|
|
263
|
+
entrez_gene_id=int(entrez) if entrez is not None else None,
|
|
264
|
+
chromosome=_normalize_chromosome(str(record["chr"])),
|
|
265
|
+
start_position=int(record["startPosition"]),
|
|
266
|
+
end_position=int(record["endPosition"]),
|
|
267
|
+
reference_allele=str(record.get("referenceAllele") or ""),
|
|
268
|
+
variant_allele=str(record.get("variantAllele") or ""),
|
|
269
|
+
protein_change=_clean(record.get("proteinChange")),
|
|
270
|
+
variant_class=str(record.get("mutationType") or ""),
|
|
271
|
+
variant_type=str(record.get("variantType") or ""),
|
|
272
|
+
ncbi_build=str(record.get("ncbiBuild") or ""),
|
|
273
|
+
refseq_mrna_id=_clean(record.get("refseqMrnaId")),
|
|
274
|
+
protein_pos_start=record.get("proteinPosStart"),
|
|
275
|
+
protein_pos_end=record.get("proteinPosEnd"),
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def fetch_sample_ids(
|
|
280
|
+
study: str = DEFAULT_STUDY,
|
|
281
|
+
*,
|
|
282
|
+
session: requests.Session | None = None,
|
|
283
|
+
base_url: str = CBIOPORTAL_API,
|
|
284
|
+
) -> list[str]:
|
|
285
|
+
"""
|
|
286
|
+
Fetch every sample identifier in a cBioPortal study.
|
|
287
|
+
|
|
288
|
+
Args:
|
|
289
|
+
study: The cBioPortal study identifier.
|
|
290
|
+
session: An optional requests session to reuse a connection.
|
|
291
|
+
base_url: The base URL of the cBioPortal API.
|
|
292
|
+
|
|
293
|
+
Returns:
|
|
294
|
+
The sample identifiers reported by the study.
|
|
295
|
+
"""
|
|
296
|
+
http = session or requests
|
|
297
|
+
response = http.get(f"{base_url}/studies/{study}/samples", timeout=_REQUEST_TIMEOUT)
|
|
298
|
+
response.raise_for_status()
|
|
299
|
+
return [str(sample["sampleId"]) for sample in response.json()]
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def fetch_mutations(
|
|
303
|
+
study: str,
|
|
304
|
+
sample_id: str,
|
|
305
|
+
*,
|
|
306
|
+
session: requests.Session | None = None,
|
|
307
|
+
base_url: str = CBIOPORTAL_API,
|
|
308
|
+
) -> list[Mutation]:
|
|
309
|
+
"""
|
|
310
|
+
Fetch all mutations for a single sample from a study's mutation profile.
|
|
311
|
+
|
|
312
|
+
Args:
|
|
313
|
+
study: The cBioPortal study identifier.
|
|
314
|
+
sample_id: The sample identifier to fetch mutations for.
|
|
315
|
+
session: An optional requests session to reuse a connection.
|
|
316
|
+
base_url: The base URL of the cBioPortal API.
|
|
317
|
+
|
|
318
|
+
Returns:
|
|
319
|
+
The parsed mutations for the sample.
|
|
320
|
+
"""
|
|
321
|
+
http = session or requests
|
|
322
|
+
profile = f"{study}_mutations"
|
|
323
|
+
url = f"{base_url}/molecular-profiles/{profile}/mutations/fetch"
|
|
324
|
+
response = http.post(
|
|
325
|
+
url,
|
|
326
|
+
params={"projection": "DETAILED"},
|
|
327
|
+
json={"sampleIds": [sample_id]},
|
|
328
|
+
timeout=_REQUEST_TIMEOUT,
|
|
329
|
+
)
|
|
330
|
+
response.raise_for_status()
|
|
331
|
+
return [_mutation_from_record(record) for record in response.json()]
|
cellme/main.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import sys
|
|
3
|
+
from collections.abc import Callable
|
|
4
|
+
|
|
5
|
+
import defopt
|
|
6
|
+
|
|
7
|
+
from cellme.cbioportal import CellLineError
|
|
8
|
+
from cellme.tools.truth_track import truth_track
|
|
9
|
+
|
|
10
|
+
_tools: list[Callable[..., None]] = [
|
|
11
|
+
truth_track,
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def setup_logging(level: str = "INFO") -> None:
|
|
16
|
+
"""Set up basic logging to print to the console."""
|
|
17
|
+
logging.basicConfig(
|
|
18
|
+
level=level,
|
|
19
|
+
format="%(asctime)s %(name)s:%(funcName)s:%(lineno)s [%(levelname)s]: %(message)s",
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def run() -> None:
|
|
24
|
+
"""Set up logging, then hand over to defopt for running the command line tool."""
|
|
25
|
+
setup_logging()
|
|
26
|
+
logger = logging.getLogger("cellme")
|
|
27
|
+
logger.info("Executing: " + " ".join(sys.argv))
|
|
28
|
+
(command,) = _tools
|
|
29
|
+
try:
|
|
30
|
+
defopt.run(command, argv=sys.argv[1:], version=True)
|
|
31
|
+
except CellLineError as error:
|
|
32
|
+
print(f"Error: {error}", file=sys.stderr)
|
|
33
|
+
raise SystemExit(1) from error
|
|
34
|
+
logger.info("Finished executing successfully.")
|
cellme/py.typed
ADDED
|
File without changes
|
cellme/tools/__init__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""The cellme command: resolve a cell line and write its truth-track VCF."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
from cellme import __version__
|
|
9
|
+
from cellme.builds import GenomeBuild
|
|
10
|
+
from cellme.cbioportal import DEFAULT_STUDY
|
|
11
|
+
from cellme.cbioportal import cell_line_name
|
|
12
|
+
from cellme.cbioportal import fetch_mutations
|
|
13
|
+
from cellme.cbioportal import fetch_sample_ids
|
|
14
|
+
from cellme.cbioportal import resolve_sample
|
|
15
|
+
from cellme.vcf import TrackContext
|
|
16
|
+
from cellme.vcf import build_header
|
|
17
|
+
from cellme.vcf import build_records
|
|
18
|
+
from cellme.vcf import make_anchor_base
|
|
19
|
+
from cellme.vcf import make_lifter
|
|
20
|
+
from cellme.vcf import write_vcf
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger("cellme")
|
|
23
|
+
|
|
24
|
+
CCLE_BUILD: GenomeBuild = GenomeBuild.GRCh37
|
|
25
|
+
"""CCLE reports its coordinates against GRCh37, the source build for lifting."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def truth_track(
|
|
29
|
+
query: str,
|
|
30
|
+
*,
|
|
31
|
+
build: GenomeBuild = GenomeBuild.GRCh38,
|
|
32
|
+
reference: Path | None = None,
|
|
33
|
+
output: Path | None = None,
|
|
34
|
+
study: str = DEFAULT_STUDY,
|
|
35
|
+
) -> None:
|
|
36
|
+
"""
|
|
37
|
+
Write a truth-track VCF of a human cell line's known CCLE mutations.
|
|
38
|
+
|
|
39
|
+
The query is resolved to a Cancer Cell Line Encyclopedia sample, its somatic
|
|
40
|
+
mutations are fetched from cBioPortal, and each is written as a VCF record on
|
|
41
|
+
the requested genome build. CCLE coordinates are GRCh37; when the target
|
|
42
|
+
build is GRCh38 every coordinate is lifted with a UCSC chain file.
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
query: Cell line identifier, e.g. MOLT-4, MOLT4, or a full CCLE sample id.
|
|
46
|
+
build: Target genome build for the emitted VCF.
|
|
47
|
+
reference: Reference FASTA for the target build, used only to place
|
|
48
|
+
spec-compliant anchor bases on insertions and deletions. Without it,
|
|
49
|
+
indel anchors use a placeholder N and are marked ANCHOR=placeholder.
|
|
50
|
+
output: Output VCF path. Writes to standard output when omitted.
|
|
51
|
+
study: cBioPortal study identifier to query.
|
|
52
|
+
"""
|
|
53
|
+
with requests.Session() as session:
|
|
54
|
+
session.headers["User-Agent"] = f"cellme/{__version__}"
|
|
55
|
+
sample_ids = fetch_sample_ids(study, session=session)
|
|
56
|
+
sample_id = resolve_sample(query, sample_ids)
|
|
57
|
+
logger.info(f"Resolved {query!r} to CCLE sample {sample_id}")
|
|
58
|
+
mutations = fetch_mutations(study, sample_id, session=session)
|
|
59
|
+
logger.info(f"Fetched {len(mutations)} mutations for {sample_id}")
|
|
60
|
+
|
|
61
|
+
context = TrackContext(
|
|
62
|
+
cell_line=cell_line_name(sample_id),
|
|
63
|
+
sample_id=sample_id,
|
|
64
|
+
study=study,
|
|
65
|
+
source_build=CCLE_BUILD,
|
|
66
|
+
target_build=build,
|
|
67
|
+
)
|
|
68
|
+
lift_position = make_lifter(CCLE_BUILD, build)
|
|
69
|
+
anchor_base = make_anchor_base(reference)
|
|
70
|
+
records, dropped = build_records(
|
|
71
|
+
mutations, context, lift_position=lift_position, anchor_base=anchor_base
|
|
72
|
+
)
|
|
73
|
+
if dropped:
|
|
74
|
+
logger.warning(f"Dropped {dropped} mutations that could not be lifted to {build}")
|
|
75
|
+
header = build_header(context, __version__)
|
|
76
|
+
write_vcf(records, header, output)
|
|
77
|
+
logger.info(f"Wrote {len(records)} records for {context.cell_line} on {build}")
|
cellme/vcf.py
ADDED
|
@@ -0,0 +1,472 @@
|
|
|
1
|
+
"""Conversion of CCLE mutations into a sorted, well-described truth-track VCF."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Callable
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import pysam
|
|
9
|
+
from liftover import get_lifter
|
|
10
|
+
|
|
11
|
+
from cellme.builds import GenomeBuild
|
|
12
|
+
from cellme.builds import contig_order
|
|
13
|
+
from cellme.builds import contigs_for
|
|
14
|
+
from cellme.cbioportal import Mutation
|
|
15
|
+
|
|
16
|
+
LiftPosition = Callable[[str, int], int | None]
|
|
17
|
+
"""A callable mapping a source (contig, 1-based position) to a lifted position."""
|
|
18
|
+
|
|
19
|
+
AnchorBase = Callable[[str, int], str]
|
|
20
|
+
"""A callable returning the single reference base at a (contig, 1-based position)."""
|
|
21
|
+
|
|
22
|
+
_DASH: str = "-"
|
|
23
|
+
"""The MAF sentinel for the absent allele of an insertion or deletion."""
|
|
24
|
+
|
|
25
|
+
InfoValue = str | int | bool
|
|
26
|
+
"""The value types cellme writes into a VCF INFO field."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class InfoField:
|
|
31
|
+
"""A single INFO field definition used for both the header and records."""
|
|
32
|
+
|
|
33
|
+
key: str
|
|
34
|
+
number: str
|
|
35
|
+
type: str
|
|
36
|
+
description: str
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
INFO_FIELDS: tuple[InfoField, ...] = (
|
|
40
|
+
InfoField("GENE", "1", "String", "HUGO gene symbol"),
|
|
41
|
+
InfoField(
|
|
42
|
+
"PROTEIN_CHANGE",
|
|
43
|
+
"1",
|
|
44
|
+
"String",
|
|
45
|
+
"Protein-level change in CCLE short form (HGVS p.-like), e.g. R306*",
|
|
46
|
+
),
|
|
47
|
+
InfoField(
|
|
48
|
+
"VARIANT_CLASS",
|
|
49
|
+
"1",
|
|
50
|
+
"String",
|
|
51
|
+
"Variant classification (MAF-style), e.g. Missense_Mutation",
|
|
52
|
+
),
|
|
53
|
+
InfoField(
|
|
54
|
+
"VARIANT_TYPE",
|
|
55
|
+
"1",
|
|
56
|
+
"String",
|
|
57
|
+
"Sequence alteration type reported by CCLE: SNP, DNP, INS, or DEL",
|
|
58
|
+
),
|
|
59
|
+
InfoField("CELL_LINE", "1", "String", "Cell line resolved from the query"),
|
|
60
|
+
InfoField("SAMPLE_ID", "1", "String", "CCLE sample identifier for the cell line"),
|
|
61
|
+
InfoField("ENTREZ", "1", "Integer", "NCBI Entrez gene identifier"),
|
|
62
|
+
InfoField("REFSEQ", "1", "String", "RefSeq mRNA accession for the annotated transcript"),
|
|
63
|
+
InfoField("PROTEIN_POS", "1", "String", "Affected protein position or range, 1-based"),
|
|
64
|
+
InfoField(
|
|
65
|
+
"SOURCE",
|
|
66
|
+
"1",
|
|
67
|
+
"String",
|
|
68
|
+
"Source database and study, e.g. cBioPortal CCLE ccle_broad_2019",
|
|
69
|
+
),
|
|
70
|
+
InfoField(
|
|
71
|
+
"ORIGINAL_BUILD",
|
|
72
|
+
"1",
|
|
73
|
+
"String",
|
|
74
|
+
"Genome build of the source coordinates before any liftover",
|
|
75
|
+
),
|
|
76
|
+
InfoField(
|
|
77
|
+
"ORIGINAL_LOCUS",
|
|
78
|
+
"1",
|
|
79
|
+
"String",
|
|
80
|
+
"Original locus before liftover in UCSC position format (chrom:pos) on ORIGINAL_BUILD",
|
|
81
|
+
),
|
|
82
|
+
InfoField(
|
|
83
|
+
"LIFTED",
|
|
84
|
+
"0",
|
|
85
|
+
"Flag",
|
|
86
|
+
"Coordinate was lifted from ORIGINAL_BUILD to the output reference build",
|
|
87
|
+
),
|
|
88
|
+
InfoField(
|
|
89
|
+
"ANCHOR",
|
|
90
|
+
"1",
|
|
91
|
+
"String",
|
|
92
|
+
"Provenance of the indel anchor base: reference (from --reference) or placeholder (N)",
|
|
93
|
+
),
|
|
94
|
+
)
|
|
95
|
+
"""The complete, ordered INFO schema emitted by cellme."""
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@dataclass(frozen=True)
|
|
99
|
+
class TrackContext:
|
|
100
|
+
"""The per-run context shared by every record in a truth-track VCF."""
|
|
101
|
+
|
|
102
|
+
cell_line: str
|
|
103
|
+
sample_id: str
|
|
104
|
+
study: str
|
|
105
|
+
source_build: GenomeBuild
|
|
106
|
+
target_build: GenomeBuild
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass(frozen=True)
|
|
110
|
+
class VcfRecord:
|
|
111
|
+
"""A single VCF record with its 1-based position and resolved alleles."""
|
|
112
|
+
|
|
113
|
+
contig: str
|
|
114
|
+
position: int
|
|
115
|
+
identifier: str | None
|
|
116
|
+
reference_allele: str
|
|
117
|
+
alternate_allele: str
|
|
118
|
+
info: dict[str, InfoValue]
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _identifier(mutation: Mutation) -> str | None:
|
|
122
|
+
"""
|
|
123
|
+
Build a stable, meaningful VCF identifier for a mutation.
|
|
124
|
+
|
|
125
|
+
Args:
|
|
126
|
+
mutation: The source mutation.
|
|
127
|
+
|
|
128
|
+
Returns:
|
|
129
|
+
A ``gene:proteinChange`` token when available, a ``gene:chrom:pos`` token
|
|
130
|
+
when only the gene is known, otherwise None.
|
|
131
|
+
"""
|
|
132
|
+
if mutation.gene and mutation.protein_change:
|
|
133
|
+
return f"{mutation.gene}:{mutation.protein_change}"
|
|
134
|
+
if mutation.gene:
|
|
135
|
+
return f"{mutation.gene}:{mutation.chromosome}:{mutation.start_position}"
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _format_protein_position(mutation: Mutation) -> str | None:
|
|
140
|
+
"""
|
|
141
|
+
Format the affected protein position or range for the PROTEIN_POS field.
|
|
142
|
+
|
|
143
|
+
Args:
|
|
144
|
+
mutation: The source mutation.
|
|
145
|
+
|
|
146
|
+
Returns:
|
|
147
|
+
A single position, a hyphenated range, or None when unknown.
|
|
148
|
+
"""
|
|
149
|
+
start = mutation.protein_pos_start
|
|
150
|
+
end = mutation.protein_pos_end
|
|
151
|
+
if start is None:
|
|
152
|
+
return None
|
|
153
|
+
if end is None or end == start:
|
|
154
|
+
return str(start)
|
|
155
|
+
return f"{start}-{end}"
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _original_locus(mutation: Mutation) -> str:
|
|
159
|
+
"""
|
|
160
|
+
Format the pre-liftover locus in UCSC position syntax.
|
|
161
|
+
|
|
162
|
+
CCLE contigs arrive without a ``chr`` prefix, so one is added. A variant
|
|
163
|
+
spanning a single base is rendered as ``chrom:pos``; a wider variant is
|
|
164
|
+
rendered as the range ``chrom:start-end`` using the original coordinates.
|
|
165
|
+
|
|
166
|
+
Args:
|
|
167
|
+
mutation: The source mutation, on the source build.
|
|
168
|
+
|
|
169
|
+
Returns:
|
|
170
|
+
The UCSC-style locus string, for example ``chr17:7577022``.
|
|
171
|
+
"""
|
|
172
|
+
if mutation.start_position == mutation.end_position:
|
|
173
|
+
return f"chr{mutation.chromosome}:{mutation.start_position}"
|
|
174
|
+
return f"chr{mutation.chromosome}:{mutation.start_position}-{mutation.end_position}"
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _build_info(
|
|
178
|
+
mutation: Mutation,
|
|
179
|
+
context: TrackContext,
|
|
180
|
+
*,
|
|
181
|
+
lifted: bool,
|
|
182
|
+
anchor_source: str | None,
|
|
183
|
+
) -> dict[str, InfoValue]:
|
|
184
|
+
"""
|
|
185
|
+
Assemble the INFO mapping for one record from its mutation and context.
|
|
186
|
+
|
|
187
|
+
Args:
|
|
188
|
+
mutation: The source mutation.
|
|
189
|
+
context: The shared per-run context.
|
|
190
|
+
lifted: Whether the coordinate was lifted to the target build.
|
|
191
|
+
anchor_source: The indel anchor provenance, or None for substitutions.
|
|
192
|
+
|
|
193
|
+
Returns:
|
|
194
|
+
A mapping of INFO keys to values, omitting fields with no value.
|
|
195
|
+
"""
|
|
196
|
+
info: dict[str, InfoValue] = {}
|
|
197
|
+
if mutation.gene:
|
|
198
|
+
info["GENE"] = mutation.gene
|
|
199
|
+
if mutation.protein_change:
|
|
200
|
+
info["PROTEIN_CHANGE"] = mutation.protein_change
|
|
201
|
+
if mutation.variant_class:
|
|
202
|
+
info["VARIANT_CLASS"] = mutation.variant_class
|
|
203
|
+
if mutation.variant_type:
|
|
204
|
+
info["VARIANT_TYPE"] = mutation.variant_type
|
|
205
|
+
info["CELL_LINE"] = context.cell_line
|
|
206
|
+
info["SAMPLE_ID"] = context.sample_id
|
|
207
|
+
if mutation.entrez_gene_id is not None:
|
|
208
|
+
info["ENTREZ"] = mutation.entrez_gene_id
|
|
209
|
+
if mutation.refseq_mrna_id:
|
|
210
|
+
info["REFSEQ"] = mutation.refseq_mrna_id
|
|
211
|
+
protein_position = _format_protein_position(mutation)
|
|
212
|
+
if protein_position is not None:
|
|
213
|
+
info["PROTEIN_POS"] = protein_position
|
|
214
|
+
info["SOURCE"] = f"cBioPortal CCLE {context.study}"
|
|
215
|
+
info["ORIGINAL_BUILD"] = str(context.source_build)
|
|
216
|
+
if lifted:
|
|
217
|
+
info["ORIGINAL_LOCUS"] = _original_locus(mutation)
|
|
218
|
+
info["LIFTED"] = True
|
|
219
|
+
if anchor_source is not None:
|
|
220
|
+
info["ANCHOR"] = anchor_source
|
|
221
|
+
return info
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def build_record(
|
|
225
|
+
mutation: Mutation,
|
|
226
|
+
context: TrackContext,
|
|
227
|
+
*,
|
|
228
|
+
lift_position: LiftPosition | None,
|
|
229
|
+
anchor_base: AnchorBase | None,
|
|
230
|
+
) -> VcfRecord | None:
|
|
231
|
+
"""
|
|
232
|
+
Convert one CCLE mutation into a VCF record on the target build.
|
|
233
|
+
|
|
234
|
+
Substitutions carry their CCLE alleles through unchanged. Insertions and
|
|
235
|
+
deletions are left-anchored: a deletion sits one base to the left of the
|
|
236
|
+
deleted sequence and an insertion at the base preceding the insertion point.
|
|
237
|
+
The anchoring position is computed in the source build, then lifted, so the
|
|
238
|
+
anchor base is read from the target reference when one is supplied.
|
|
239
|
+
|
|
240
|
+
Args:
|
|
241
|
+
mutation: The source mutation to convert.
|
|
242
|
+
context: The shared per-run context.
|
|
243
|
+
lift_position: A callable that lifts a source position to the target
|
|
244
|
+
build, or None when the source and target builds are the same.
|
|
245
|
+
anchor_base: A callable returning the target reference base for indel
|
|
246
|
+
anchoring, or None to use a placeholder ``N`` anchor.
|
|
247
|
+
|
|
248
|
+
Returns:
|
|
249
|
+
The converted record, or None when the mutation is malformed or its
|
|
250
|
+
coordinate cannot be lifted to the target build.
|
|
251
|
+
"""
|
|
252
|
+
is_deletion = mutation.variant_allele == _DASH
|
|
253
|
+
is_insertion = mutation.reference_allele == _DASH
|
|
254
|
+
if is_deletion and is_insertion:
|
|
255
|
+
return None
|
|
256
|
+
|
|
257
|
+
if is_deletion:
|
|
258
|
+
source_position = mutation.start_position - 1
|
|
259
|
+
if source_position < 1:
|
|
260
|
+
return None
|
|
261
|
+
else:
|
|
262
|
+
source_position = mutation.start_position
|
|
263
|
+
|
|
264
|
+
if lift_position is None:
|
|
265
|
+
position = source_position
|
|
266
|
+
else:
|
|
267
|
+
lifted_position = lift_position(mutation.chromosome, source_position)
|
|
268
|
+
if lifted_position is None:
|
|
269
|
+
return None
|
|
270
|
+
position = lifted_position
|
|
271
|
+
|
|
272
|
+
anchor_source: str | None = None
|
|
273
|
+
if is_deletion or is_insertion:
|
|
274
|
+
if anchor_base is None:
|
|
275
|
+
anchor = "N"
|
|
276
|
+
anchor_source = "placeholder"
|
|
277
|
+
else:
|
|
278
|
+
anchor = anchor_base(mutation.chromosome, position)
|
|
279
|
+
anchor_source = "reference"
|
|
280
|
+
if is_deletion:
|
|
281
|
+
reference_allele = anchor + mutation.reference_allele
|
|
282
|
+
alternate_allele = anchor
|
|
283
|
+
else:
|
|
284
|
+
reference_allele = anchor
|
|
285
|
+
alternate_allele = anchor + mutation.variant_allele
|
|
286
|
+
else:
|
|
287
|
+
reference_allele = mutation.reference_allele
|
|
288
|
+
alternate_allele = mutation.variant_allele
|
|
289
|
+
if not reference_allele or not alternate_allele:
|
|
290
|
+
return None
|
|
291
|
+
|
|
292
|
+
info = _build_info(
|
|
293
|
+
mutation,
|
|
294
|
+
context,
|
|
295
|
+
lifted=lift_position is not None,
|
|
296
|
+
anchor_source=anchor_source,
|
|
297
|
+
)
|
|
298
|
+
return VcfRecord(
|
|
299
|
+
contig=mutation.chromosome,
|
|
300
|
+
position=position,
|
|
301
|
+
identifier=_identifier(mutation),
|
|
302
|
+
reference_allele=reference_allele.upper(),
|
|
303
|
+
alternate_allele=alternate_allele.upper(),
|
|
304
|
+
info=info,
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def build_records(
|
|
309
|
+
mutations: Iterable[Mutation],
|
|
310
|
+
context: TrackContext,
|
|
311
|
+
*,
|
|
312
|
+
lift_position: LiftPosition | None,
|
|
313
|
+
anchor_base: AnchorBase | None,
|
|
314
|
+
) -> tuple[list[VcfRecord], int]:
|
|
315
|
+
"""
|
|
316
|
+
Convert many mutations, sorted into the target build's contig order.
|
|
317
|
+
|
|
318
|
+
Args:
|
|
319
|
+
mutations: The source mutations.
|
|
320
|
+
context: The shared per-run context.
|
|
321
|
+
lift_position: A callable that lifts a source position to the target
|
|
322
|
+
build, or None when the source and target builds are the same.
|
|
323
|
+
anchor_base: A callable returning the target reference base for indel
|
|
324
|
+
anchoring, or None to use a placeholder ``N`` anchor.
|
|
325
|
+
|
|
326
|
+
Returns:
|
|
327
|
+
A tuple of the sorted records and the count of mutations that were
|
|
328
|
+
dropped because they were malformed, could not be lifted, or sit on a
|
|
329
|
+
contig that is not part of the target build.
|
|
330
|
+
"""
|
|
331
|
+
order = contig_order(context.target_build)
|
|
332
|
+
records: list[VcfRecord] = []
|
|
333
|
+
dropped = 0
|
|
334
|
+
for mutation in mutations:
|
|
335
|
+
record = build_record(
|
|
336
|
+
mutation, context, lift_position=lift_position, anchor_base=anchor_base
|
|
337
|
+
)
|
|
338
|
+
if record is None or record.contig not in order:
|
|
339
|
+
dropped += 1
|
|
340
|
+
continue
|
|
341
|
+
records.append(record)
|
|
342
|
+
records.sort(key=lambda record: (order[record.contig], record.position))
|
|
343
|
+
return records, dropped
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def build_header(context: TrackContext, version: str) -> "pysam.VariantHeader":
|
|
347
|
+
"""
|
|
348
|
+
Build a VCF header for the target build with the full INFO schema.
|
|
349
|
+
|
|
350
|
+
Args:
|
|
351
|
+
context: The shared per-run context.
|
|
352
|
+
version: The cellme version string, written to the ``source`` line.
|
|
353
|
+
|
|
354
|
+
Returns:
|
|
355
|
+
A pysam variant header populated with meta, contig, and INFO lines.
|
|
356
|
+
"""
|
|
357
|
+
header = pysam.VariantHeader()
|
|
358
|
+
header.add_line(f"##source=cellme {version}")
|
|
359
|
+
header.add_line(f"##reference={context.target_build}")
|
|
360
|
+
header.add_line(f"##cellme_cellLine={context.cell_line}")
|
|
361
|
+
header.add_line(f"##cellme_sampleId={context.sample_id}")
|
|
362
|
+
header.add_line(f"##cellme_sourceStudy=cBioPortal CCLE {context.study}")
|
|
363
|
+
header.add_line(f"##cellme_sourceBuild={context.source_build}")
|
|
364
|
+
for name, length in contigs_for(context.target_build):
|
|
365
|
+
header.add_line(f"##contig=<ID={name},length={length}>")
|
|
366
|
+
for field in INFO_FIELDS:
|
|
367
|
+
header.add_line(
|
|
368
|
+
f"##INFO=<ID={field.key},Number={field.number},"
|
|
369
|
+
f'Type={field.type},Description="{field.description}">'
|
|
370
|
+
)
|
|
371
|
+
return header
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def _to_pysam_record(header: "pysam.VariantHeader", record: VcfRecord) -> "pysam.VariantRecord":
|
|
375
|
+
"""
|
|
376
|
+
Materialize a VcfRecord as a pysam record bound to a header.
|
|
377
|
+
|
|
378
|
+
Args:
|
|
379
|
+
header: The header the record will be written under.
|
|
380
|
+
record: The record to materialize.
|
|
381
|
+
|
|
382
|
+
Returns:
|
|
383
|
+
A pysam variant record ready to be written.
|
|
384
|
+
"""
|
|
385
|
+
start = record.position - 1
|
|
386
|
+
stop = start + len(record.reference_allele)
|
|
387
|
+
return header.new_record(
|
|
388
|
+
contig=record.contig,
|
|
389
|
+
start=start,
|
|
390
|
+
stop=stop,
|
|
391
|
+
alleles=(record.reference_allele, record.alternate_allele),
|
|
392
|
+
id=record.identifier,
|
|
393
|
+
info=dict(record.info),
|
|
394
|
+
)
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def write_vcf(
|
|
398
|
+
records: Iterable[VcfRecord],
|
|
399
|
+
header: "pysam.VariantHeader",
|
|
400
|
+
output: Path | None,
|
|
401
|
+
) -> None:
|
|
402
|
+
"""
|
|
403
|
+
Write records to a VCF at a path, or to standard output when no path given.
|
|
404
|
+
|
|
405
|
+
Args:
|
|
406
|
+
records: The records to write, already in sorted order.
|
|
407
|
+
header: The header to write them under.
|
|
408
|
+
output: The destination path, or None to write to standard output.
|
|
409
|
+
"""
|
|
410
|
+
destination = str(output) if output is not None else "-"
|
|
411
|
+
with pysam.VariantFile(destination, "w", header=header) as out:
|
|
412
|
+
for record in records:
|
|
413
|
+
out.write(_to_pysam_record(header, record))
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def make_lifter(source: GenomeBuild, target: GenomeBuild) -> LiftPosition | None:
|
|
417
|
+
"""
|
|
418
|
+
Create a position lifter between two builds, or None when they match.
|
|
419
|
+
|
|
420
|
+
The returned callable drops coordinates that fail to lift or that lift to a
|
|
421
|
+
different contig, so that only confidently mapped positions are emitted.
|
|
422
|
+
|
|
423
|
+
Args:
|
|
424
|
+
source: The build the source coordinates are on.
|
|
425
|
+
target: The build to lift coordinates onto.
|
|
426
|
+
|
|
427
|
+
Returns:
|
|
428
|
+
A lifter callable, or None when ``source`` and ``target`` are equal.
|
|
429
|
+
"""
|
|
430
|
+
if source == target:
|
|
431
|
+
return None
|
|
432
|
+
lifter = get_lifter(source.ucsc_name, target.ucsc_name)
|
|
433
|
+
|
|
434
|
+
def lift(chromosome: str, position: int) -> int | None:
|
|
435
|
+
result = lifter.query(chromosome, position)
|
|
436
|
+
if not result:
|
|
437
|
+
return None
|
|
438
|
+
lifted_chromosome, lifted_position, _ = result[0]
|
|
439
|
+
if lifted_chromosome.removeprefix("chr") != chromosome.removeprefix("chr"):
|
|
440
|
+
return None
|
|
441
|
+
return int(lifted_position)
|
|
442
|
+
|
|
443
|
+
return lift
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def make_anchor_base(reference: Path | None) -> AnchorBase | None:
|
|
447
|
+
"""
|
|
448
|
+
Open a reference FASTA and return a base lookup for indel anchoring.
|
|
449
|
+
|
|
450
|
+
The FASTA is matched by contig name with and without a ``chr`` prefix, so a
|
|
451
|
+
reference using either naming convention works.
|
|
452
|
+
|
|
453
|
+
Args:
|
|
454
|
+
reference: The path to a FASTA for the target build, or None.
|
|
455
|
+
|
|
456
|
+
Returns:
|
|
457
|
+
A callable returning the reference base at a 1-based position, or None
|
|
458
|
+
when no reference was supplied.
|
|
459
|
+
"""
|
|
460
|
+
if reference is None:
|
|
461
|
+
return None
|
|
462
|
+
fasta = pysam.FastaFile(str(reference))
|
|
463
|
+
contigs = set(fasta.references)
|
|
464
|
+
|
|
465
|
+
def anchor(chromosome: str, position: int) -> str:
|
|
466
|
+
for name in (chromosome, f"chr{chromosome}"):
|
|
467
|
+
if name in contigs:
|
|
468
|
+
base = fasta.fetch(name, position - 1, position)
|
|
469
|
+
return base.upper() if base else "N"
|
|
470
|
+
return "N"
|
|
471
|
+
|
|
472
|
+
return anchor
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: cellme
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Convert a human cell line identifier into a truth-track VCF of its known mutations.
|
|
5
|
+
Project-URL: homepage, https://github.com/clintval/cellme
|
|
6
|
+
Project-URL: repository, https://github.com/clintval/cellme
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/clintval/cellme/issues
|
|
8
|
+
Author-email: Clint Valentine <valentine.clint@gmail.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Natural Language :: English
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Python: >=3.12
|
|
23
|
+
Requires-Dist: defopt~=6.4
|
|
24
|
+
Requires-Dist: liftover>=1.3
|
|
25
|
+
Requires-Dist: pysam>=0.23
|
|
26
|
+
Requires-Dist: rapidfuzz>=3.14.5
|
|
27
|
+
Requires-Dist: requests>=2.32
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# cellme
|
|
31
|
+
|
|
32
|
+
[](https://pypi.org/project/cellme/)
|
|
33
|
+
[](https://github.com/clintval/cellme/actions/workflows/python_package.yml?query=branch%3Amain)
|
|
34
|
+
[](https://github.com/clintval/cellme)
|
|
35
|
+
[](LICENSE)
|
|
36
|
+
[](http://mypy-lang.org/)
|
|
37
|
+
[](https://github.com/astral-sh/uv)
|
|
38
|
+
[](https://docs.astral.sh/ruff/)
|
|
39
|
+
|
|
40
|
+
Convert a human cell line identifier into a truth-track VCF of its known mutations.
|
|
41
|
+
|
|
42
|
+
## Introduction
|
|
43
|
+
|
|
44
|
+
cellme builds a truth/known VCF for a cell line by resolving a name such as `MOLT-4` to a [Cancer Cell Line Encyclopedia](https://sites.broadinstitute.org/ccle/) sample, fetching that sample's somatic mutations from the [cBioPortal](https://www.cbioportal.org/) REST API, and writing them as a sorted VCF.
|
|
45
|
+
The mutation source is the `ccle_broad_2019` study.
|
|
46
|
+
Records are against the GRCh37 assembly but a liftover can be performed if you need GRCh38 output.
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
Install the release from PyPI with `pip`:
|
|
51
|
+
|
|
52
|
+
```console
|
|
53
|
+
pip install cellme
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Add cellme to a [`uv`](https://docs.astral.sh/uv/) project, or run it without installing, with:
|
|
57
|
+
|
|
58
|
+
```console
|
|
59
|
+
uv add cellme
|
|
60
|
+
uvx cellme --help
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Development Installation
|
|
64
|
+
|
|
65
|
+
Install the Python package and dependency management tool [`uv`](https://docs.astral.sh/uv/getting-started/installation/) using the official documentation.
|
|
66
|
+
|
|
67
|
+
Install the dependencies of the project with:
|
|
68
|
+
|
|
69
|
+
```console
|
|
70
|
+
uv sync --locked
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
To check successful installation, run:
|
|
74
|
+
|
|
75
|
+
```console
|
|
76
|
+
uv run cellme --help
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
## Usage
|
|
80
|
+
|
|
81
|
+
Write a GRCh38 truth track for the cell line MOLT-4 to a file:
|
|
82
|
+
|
|
83
|
+
```console
|
|
84
|
+
uv run cellme "MOLT-4" --build GRCh38 --reference /ref/hg38.fa --output MOLT-4.GRCh38.vcf
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
> [!IMPORTANT]
|
|
88
|
+
> Always pass `--reference` with a FASTA for the build you target, as shown above.
|
|
89
|
+
> It ensure the correct anchor bases are set for insertions and deletions.
|
|
90
|
+
> Without it, indel anchor bases fall back to a placeholder `N` and those records are marked `ANCHOR=placeholder`.
|
|
91
|
+
|
|
92
|
+
The query is matched case-insensitively on the leading cell line token, so `MOLT-4`, `MOLT4`, and the full `MOLT4_HAEMATOPOIETIC_AND_LYMPHOID_TISSUE` sample ID all resolve to the same sample.
|
|
93
|
+
|
|
94
|
+
## VCF INFO Fields
|
|
95
|
+
|
|
96
|
+
Every record is annotated with compact, self-describing INFO fields.
|
|
97
|
+
|
|
98
|
+
| Key | Type | Description |
|
|
99
|
+
| ---------------- | ------- | --------------------------------------------------------------------------- |
|
|
100
|
+
| `GENE` | String | HUGO gene symbol. |
|
|
101
|
+
| `PROTEIN_CHANGE` | String | Protein-level change in CCLE short form (HGVS p.-like), e.g. `R306*`. |
|
|
102
|
+
| `VARIANT_CLASS` | String | Variant classification (MAF-style), e.g. `Missense_Mutation`. |
|
|
103
|
+
| `VARIANT_TYPE` | String | Sequence alteration type reported by CCLE: `SNP`, `DNP`, `INS`, or `DEL`. |
|
|
104
|
+
| `CELL_LINE` | String | Cell line resolved from the query. |
|
|
105
|
+
| `SAMPLE_ID` | String | CCLE sample identifier for the cell line. |
|
|
106
|
+
| `ENTREZ` | Integer | NCBI Entrez gene identifier. |
|
|
107
|
+
| `REFSEQ` | String | RefSeq mRNA accession for the annotated transcript. |
|
|
108
|
+
| `PROTEIN_POS` | String | Affected protein position or range, 1-based. |
|
|
109
|
+
| `SOURCE` | String | Source database and study, e.g. `cBioPortal CCLE ccle_broad_2019`. |
|
|
110
|
+
| `ORIGINAL_BUILD` | String | Genome build of the source coordinates before any liftover. |
|
|
111
|
+
| `ORIGINAL_LOCUS` | String | Original locus before liftover in UCSC position format, e.g. `chr17:7577022`. |
|
|
112
|
+
| `LIFTED` | Flag | Coordinate was lifted from `ORIGINAL_BUILD` to the output reference build. |
|
|
113
|
+
| `ANCHOR` | String | Provenance of the indel anchor base: `reference` or `placeholder`. |
|
|
114
|
+
|
|
115
|
+
The `ID` column is set to a stable `gene:proteinChange` token where one is available, for example `TP53:R306*`, and falls back to `gene:chrom:pos` otherwise.
|
|
116
|
+
|
|
117
|
+
## Development and Testing
|
|
118
|
+
|
|
119
|
+
See the [contributing guide](./CONTRIBUTING.md) for more information.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
cellme/__init__.py,sha256=5D2ExwAMD0Ey6XJ0aSfWq2Q3OEfhi6eqewCKZvvoMzg,72
|
|
2
|
+
cellme/builds.py,sha256=27EA18OJmnk4NS_-64ieyKGZLc1bTwg3Z6XjJUsDnh8,3468
|
|
3
|
+
cellme/cbioportal.py,sha256=6898btBLTw4ixw7wPLCVWxF6p-Iw59USWWvrD3YaFkc,10746
|
|
4
|
+
cellme/main.py,sha256=SI0Lapl65-2xyVIBhfXZsv1Hv6W6BcsqzP-y4C6Z5s0,970
|
|
5
|
+
cellme/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
cellme/vcf.py,sha256=zMmkiqd5uYLbFZnUXXIoSoiWt4MqM_Xc1Hpb46uNvcA,15522
|
|
7
|
+
cellme/tools/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
8
|
+
cellme/tools/truth_track.py,sha256=UuChceqm9hmINFnYqeUrQmRGzPJqJyKneEm5kBG9USk,3049
|
|
9
|
+
cellme-1.0.0.dist-info/METADATA,sha256=Vezrw1HQuDohglwTLKWPAmVZph2n5Kut2Hb_3JIt-Eg,5941
|
|
10
|
+
cellme-1.0.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
11
|
+
cellme-1.0.0.dist-info/entry_points.txt,sha256=JJFK4NDS42GjJ2YNePs0SvsyfihMwx1O8BkInnDLG-E,43
|
|
12
|
+
cellme-1.0.0.dist-info/licenses/LICENSE,sha256=ANRqU0eXlG-kIJQDac3GBzNMxSLphMGs1QieZ_Bc0Ww,1076
|
|
13
|
+
cellme-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright © 2026 Fulcrum Genomics LLC
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|