telometer 1.0__tar.gz → 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- telometer-2.0.0/PKG-INFO +16 -0
- telometer-2.0.0/pyproject.toml +31 -0
- {telometer-1.0 → telometer-2.0.0}/telometer/__init__.py +1 -1
- telometer-2.0.0/telometer/telometer.py +503 -0
- telometer-2.0.0/telometer.egg-info/PKG-INFO +16 -0
- {telometer-1.0 → telometer-2.0.0}/telometer.egg-info/SOURCES.txt +2 -3
- telometer-2.0.0/telometer.egg-info/requires.txt +2 -0
- telometer-1.0/LICENSE.txt +0 -19
- telometer-1.0/PKG-INFO +0 -16
- telometer-1.0/setup.py +0 -23
- telometer-1.0/telometer/telometer.py +0 -292
- telometer-1.0/telometer.egg-info/PKG-INFO +0 -16
- telometer-1.0/telometer.egg-info/entry_points.txt +0 -2
- {telometer-1.0 → telometer-2.0.0}/README.md +0 -0
- {telometer-1.0 → telometer-2.0.0}/setup.cfg +0 -0
- {telometer-1.0 → telometer-2.0.0}/telometer.egg-info/dependency_links.txt +0 -0
- {telometer-1.0 → telometer-2.0.0}/telometer.egg-info/top_level.txt +0 -0
telometer-2.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: telometer
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Quantitative digital telomere measurement from nanopore long-reads
|
|
5
|
+
Author-email: "Santiago E. Sanchez" <santy.esanchez@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Requires-Python: >=3.7
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
Requires-Dist: requests
|
|
12
|
+
Requires-Dist: numpy
|
|
13
|
+
|
|
14
|
+
A simple tool for measuring chromosome-specific telomeres from long-read alignments.
|
|
15
|
+
|
|
16
|
+
[Telometer Github](https://github.com/santiago-es/Telometer)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77.0.3", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "telometer"
|
|
7
|
+
version = "2.0.0"
|
|
8
|
+
description = "Quantitative digital telomere measurement from nanopore long-reads"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.7"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
|
|
14
|
+
authors = [
|
|
15
|
+
{ name = "Santiago E. Sanchez", email = "santy.esanchez@gmail.com" }
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
classifiers = [
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Operating System :: OS Independent"
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
dependencies = [
|
|
24
|
+
"requests",
|
|
25
|
+
"numpy"
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[tool.setuptools.packages.find]
|
|
29
|
+
where = ["."]
|
|
30
|
+
include = ["telometer*"]
|
|
31
|
+
exclude = ["build*", "dist*"]
|
|
@@ -0,0 +1,503 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# Telometer v2.0: eight-feature baseline with optional variant detection and enrichment
|
|
3
|
+
# Baseline changes retained: 01, 03, 07, 11, 12, 14, 15, 16
|
|
4
|
+
# Created by: Santiago E Sanchez
|
|
5
|
+
# Artandi Lab, Stanford University, 2024
|
|
6
|
+
# Measures telomeres from ONT or PacBio long reads aligned to a T2T genome assembly
|
|
7
|
+
# Simple Usage: telometer -b sorted_t2t.bam -o output.tsv
|
|
8
|
+
# Optional enrichment: add -e to your existing Telometer command.
|
|
9
|
+
# -e adds the sample-level telomeres_per_gb column and console totals. Each QNAME
|
|
10
|
+
# contributes one consistent recorded full length, including soft/hard clips,
|
|
11
|
+
# before any Telometer filter. Missing/conflicting lengths yield NA, not a
|
|
12
|
+
# partial-denominator rate. This measures yield represented by the input BAM;
|
|
13
|
+
# reads/bases removed upstream without retained metadata cannot be recovered.
|
|
14
|
+
# Memory for -e scales with unique read IDs; -l is not a total-RSS limit.
|
|
15
|
+
|
|
16
|
+
__version__ = "2.0"
|
|
17
|
+
|
|
18
|
+
import pysam
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from variant_support_v2 import (BASE_COLUMNS, VARIANT_COLUMNS, TRACT_COLUMNS,
|
|
21
|
+
validate_variant_options, strict_runs, coverage_prefix, combined_window_density, summarize_variant)
|
|
22
|
+
import regex as re
|
|
23
|
+
import pandas as pd
|
|
24
|
+
import time
|
|
25
|
+
import argparse
|
|
26
|
+
import sys
|
|
27
|
+
from multiprocessing import Pool, cpu_count
|
|
28
|
+
import numpy as np
|
|
29
|
+
from scipy.signal import savgol_filter
|
|
30
|
+
from telometer_variant_support import alignment_fields, cached_findall, note, reverse_complement_iupac, stable_record_key
|
|
31
|
+
|
|
32
|
+
def reverse_complement(seq):
|
|
33
|
+
complement = {'A': 'T', 'C': 'G', 'G': 'C', 'T': 'A', 'N': 'N'}
|
|
34
|
+
return "".join(complement[base] for base in reversed(seq))
|
|
35
|
+
|
|
36
|
+
def get_telomere_repeats():
|
|
37
|
+
telomere_repeats = ['CCCTAA', 'TTAGGG']
|
|
38
|
+
telomere_repeats_rc = [reverse_complement(repeat) for repeat in telomere_repeats]
|
|
39
|
+
return telomere_repeats + telomere_repeats_rc
|
|
40
|
+
|
|
41
|
+
def identify_telomere_regions(seq, base_qualities, telomere_motifs, window_size=120, step_size=12,
|
|
42
|
+
density_threshold=0.1, max_gap_length=20, quality_threshold=15, variant_motif=None, min_variant_units=2):
|
|
43
|
+
variant_prefix = coverage_prefix(len(seq), strict_runs(seq, variant_motif, min_variant_units)) if variant_motif is not None else None
|
|
44
|
+
telomere_regions = []
|
|
45
|
+
current_start = None
|
|
46
|
+
gap_length = 0
|
|
47
|
+
has_large_gap = False
|
|
48
|
+
densities = []
|
|
49
|
+
|
|
50
|
+
combined_motif_pattern = '|'.join(f'({motif})' for motif in telomere_motifs)
|
|
51
|
+
|
|
52
|
+
for i in range(0, len(seq) - window_size + 1, step_size):
|
|
53
|
+
window_seq = seq[i:i + window_size]
|
|
54
|
+
motif_count = len(cached_findall(combined_motif_pattern, window_seq))
|
|
55
|
+
density = motif_count / (window_size / len(telomere_motifs[0]))
|
|
56
|
+
if variant_prefix is not None:
|
|
57
|
+
density = combined_window_density(window_seq, i, combined_motif_pattern, motif_count, variant_prefix)
|
|
58
|
+
densities.append(density)
|
|
59
|
+
|
|
60
|
+
if density >= density_threshold:
|
|
61
|
+
if current_start is None:
|
|
62
|
+
current_start = i
|
|
63
|
+
gap_length = 0
|
|
64
|
+
else:
|
|
65
|
+
if current_start is not None:
|
|
66
|
+
gap_length += step_size
|
|
67
|
+
|
|
68
|
+
gap_start = i - gap_length
|
|
69
|
+
gap_end = i
|
|
70
|
+
gap_quality = base_qualities[gap_start:gap_end]
|
|
71
|
+
|
|
72
|
+
if len(gap_quality) > 0:
|
|
73
|
+
avg_gap_quality = sum(gap_quality) / len(gap_quality)
|
|
74
|
+
else:
|
|
75
|
+
avg_gap_quality = 0
|
|
76
|
+
|
|
77
|
+
if avg_gap_quality < quality_threshold:
|
|
78
|
+
gap_length = 0
|
|
79
|
+
continue
|
|
80
|
+
|
|
81
|
+
if gap_length > max_gap_length:
|
|
82
|
+
mismatch_motif_count = len(cached_findall(f"({combined_motif_pattern}){{e<=1}}", window_seq))
|
|
83
|
+
mismatch_density = mismatch_motif_count / (window_size / len(telomere_motifs[0]))
|
|
84
|
+
|
|
85
|
+
if mismatch_density >= density_threshold:
|
|
86
|
+
gap_length = 0
|
|
87
|
+
else:
|
|
88
|
+
has_large_gap = True
|
|
89
|
+
telomere_regions.append((current_start, i - gap_length + step_size))
|
|
90
|
+
current_start = None
|
|
91
|
+
gap_length = 0
|
|
92
|
+
|
|
93
|
+
if current_start is not None:
|
|
94
|
+
telomere_regions.append((current_start, len(seq)))
|
|
95
|
+
|
|
96
|
+
return telomere_regions, densities, has_large_gap, combined_motif_pattern
|
|
97
|
+
|
|
98
|
+
def detect_discontinuities(base_qualities, window_length=11, polyorder=2, gradient_threshold=-1):
|
|
99
|
+
if len(base_qualities) < window_length:
|
|
100
|
+
window_length = len(base_qualities) if len(base_qualities) % 2 == 1 else len(base_qualities) - 1
|
|
101
|
+
|
|
102
|
+
if len(base_qualities) > polyorder:
|
|
103
|
+
smoothed_qualities = savgol_filter(base_qualities, window_length=window_length, polyorder=polyorder)
|
|
104
|
+
else:
|
|
105
|
+
smoothed_qualities = base_qualities
|
|
106
|
+
|
|
107
|
+
gradient = np.diff(smoothed_qualities)
|
|
108
|
+
discontinuities = np.where(gradient < gradient_threshold)[0]
|
|
109
|
+
|
|
110
|
+
return discontinuities, smoothed_qualities
|
|
111
|
+
|
|
112
|
+
def check_gap(seq, base_qualities, telomere_regions, gradient_threshold=-1, quality_threshold=15):
|
|
113
|
+
if not telomere_regions:
|
|
114
|
+
return []
|
|
115
|
+
merged_regions = []
|
|
116
|
+
current_start, current_end = telomere_regions[0]
|
|
117
|
+
for start, end in telomere_regions[1:]:
|
|
118
|
+
gap_quality = base_qualities[current_end:start]
|
|
119
|
+
discontinuities, smoothed_qualities = detect_discontinuities(gap_quality, gradient_threshold=gradient_threshold)
|
|
120
|
+
avg_gap_quality = sum(gap_quality) / len(gap_quality) if len(gap_quality) > 0 else 0
|
|
121
|
+
if avg_gap_quality < quality_threshold or len(discontinuities) == 0:
|
|
122
|
+
current_end = max(current_end, end)
|
|
123
|
+
else:
|
|
124
|
+
merged_regions.append((current_start, current_end))
|
|
125
|
+
current_start, current_end = start, end
|
|
126
|
+
merged_regions.append((current_start, current_end))
|
|
127
|
+
return merged_regions
|
|
128
|
+
|
|
129
|
+
def measure_telomere_length(seq, telomere_motifs, base_qualities, window_size=120, step_size=12,
|
|
130
|
+
density_threshold=0.1, max_gap_length=20, quality_threshold=15,
|
|
131
|
+
variant_motif=None, min_variant_units=2):
|
|
132
|
+
|
|
133
|
+
telomere_regions, densities, has_large_gap, combined_motif_pattern = identify_telomere_regions(
|
|
134
|
+
seq, base_qualities, telomere_motifs, window_size, step_size, density_threshold, max_gap_length, quality_threshold,
|
|
135
|
+
variant_motif=variant_motif, min_variant_units=min_variant_units)
|
|
136
|
+
|
|
137
|
+
if not telomere_regions:
|
|
138
|
+
return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
if not any(density >= 0.75 for density in densities):
|
|
142
|
+
return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
merged_telomere_regions = check_gap(seq, base_qualities, telomere_regions, gradient_threshold=-1, quality_threshold=quality_threshold)
|
|
146
|
+
|
|
147
|
+
terminus_regions = []
|
|
148
|
+
for start, end in merged_telomere_regions:
|
|
149
|
+
if start < 100 or end > len(seq) - 100:
|
|
150
|
+
terminus_regions.append((start, end))
|
|
151
|
+
|
|
152
|
+
if not terminus_regions:
|
|
153
|
+
return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
telomere_start, telomere_end = terminus_regions[0]
|
|
157
|
+
if not any(density >= 0.75 and telomere_start <= j * step_size and
|
|
158
|
+
j * step_size + window_size <= telomere_end
|
|
159
|
+
for j, density in enumerate(densities)):
|
|
160
|
+
note('candidate_local_dense_failed')
|
|
161
|
+
return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
|
|
162
|
+
telomere_length = telomere_end - telomere_start
|
|
163
|
+
|
|
164
|
+
return telomere_start, telomere_end, telomere_length, densities, has_large_gap, combined_motif_pattern
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def process_read(read_data, telomere_motifs, max_gap_length, min_read_len, variant_motif=None, min_variant_units=2):
|
|
168
|
+
if read_data['is_unmapped'] or read_data['reference_name'] == 'chrM':
|
|
169
|
+
return None
|
|
170
|
+
|
|
171
|
+
if read_data['reference_start'] > 30000 and read_data['reference_end'] < read_data['reference_length'] - 30000:
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
seq = read_data['query_sequence']
|
|
175
|
+
base_qualities = read_data['query_qualities']
|
|
176
|
+
if seq is None or len(seq) < min_read_len:
|
|
177
|
+
return None
|
|
178
|
+
if read_data['is_reverse']:
|
|
179
|
+
seq = reverse_complement_iupac(seq)
|
|
180
|
+
base_qualities = base_qualities[::-1]
|
|
181
|
+
|
|
182
|
+
alignment_start = read_data['reference_start']
|
|
183
|
+
alignment_end = read_data['reference_end']
|
|
184
|
+
reference_genome_length = read_data['reference_length']
|
|
185
|
+
|
|
186
|
+
near_p = alignment_start <= 30000
|
|
187
|
+
near_q = alignment_end >= reference_genome_length - 30000
|
|
188
|
+
if near_p and not near_q:
|
|
189
|
+
arm = "p"
|
|
190
|
+
elif near_q and not near_p:
|
|
191
|
+
arm = "q"
|
|
192
|
+
elif near_p and near_q:
|
|
193
|
+
arm = "ambiguous"
|
|
194
|
+
else:
|
|
195
|
+
arm = "unassigned"
|
|
196
|
+
|
|
197
|
+
direction = "rev" if read_data['is_reverse'] else "fwd"
|
|
198
|
+
|
|
199
|
+
telomere_start, telomere_end, telomere_length, densities, has_large_gap, combined_motif_pattern = measure_telomere_length(
|
|
200
|
+
seq, telomere_motifs, base_qualities, max_gap_length=max_gap_length,
|
|
201
|
+
variant_motif=variant_motif, min_variant_units=min_variant_units)
|
|
202
|
+
|
|
203
|
+
if telomere_length < 1:
|
|
204
|
+
return None
|
|
205
|
+
|
|
206
|
+
if (telomere_length + 50) > len(seq):
|
|
207
|
+
return None
|
|
208
|
+
|
|
209
|
+
result = {
|
|
210
|
+
'chromosome': read_data['reference_name'],
|
|
211
|
+
'reference_start': alignment_start,
|
|
212
|
+
'reference_end': alignment_end,
|
|
213
|
+
'telomere_length': telomere_length,
|
|
214
|
+
'read_id': read_data['query_name'],
|
|
215
|
+
'mapping_quality': read_data['mapping_quality'],
|
|
216
|
+
'read_length': len(seq),
|
|
217
|
+
'arm': arm,
|
|
218
|
+
'direction': direction,
|
|
219
|
+
'_selection_key': stable_record_key(read_data['_ranking_fields'])
|
|
220
|
+
}
|
|
221
|
+
if variant_motif is not None:
|
|
222
|
+
fields, tracts, details = summarize_variant(seq, base_qualities, telomere_start, telomere_end,
|
|
223
|
+
variant_motif, min_variant_units, read_data)
|
|
224
|
+
result.update(fields)
|
|
225
|
+
result['_variant_tracts'] = tracts
|
|
226
|
+
result['_variant_details'] = details
|
|
227
|
+
return result
|
|
228
|
+
|
|
229
|
+
def process_read_wrapper(args):
|
|
230
|
+
return process_read(*args)
|
|
231
|
+
|
|
232
|
+
def estimate_memory_usage(reads_list):
|
|
233
|
+
"""
|
|
234
|
+
Estimate the memory usage of the reads in bytes.
|
|
235
|
+
Approximate size based on the size of sequences, base qualities, and metadata.
|
|
236
|
+
"""
|
|
237
|
+
memory_usage = 0
|
|
238
|
+
for read in reads_list:
|
|
239
|
+
seq_len = len(read['query_sequence']) if read['query_sequence'] else 0
|
|
240
|
+
quality_len = len(read['query_qualities']) if read['query_qualities'] else 0
|
|
241
|
+
memory_usage += sys.getsizeof(read['query_name']) + seq_len + quality_len + sys.getsizeof(read)
|
|
242
|
+
return memory_usage
|
|
243
|
+
|
|
244
|
+
def recorded_full_read_length(read):
|
|
245
|
+
"""Return the positive full length recorded by SEQ/CIGAR, or None if unknown.
|
|
246
|
+
|
|
247
|
+
Includes soft clips and reported hard clips, but never reconstructs bases.
|
|
248
|
+
Malformed or mutually inconsistent length metadata raises ValueError.
|
|
249
|
+
"""
|
|
250
|
+
sequence_length = read.query_length
|
|
251
|
+
if sequence_length is None:
|
|
252
|
+
sequence_length = 0
|
|
253
|
+
if (isinstance(sequence_length, bool) or not isinstance(sequence_length, int)
|
|
254
|
+
or sequence_length < 0):
|
|
255
|
+
raise ValueError('Invalid stored query length')
|
|
256
|
+
cigar = read.cigartuples or []
|
|
257
|
+
if not cigar:
|
|
258
|
+
return sequence_length or None
|
|
259
|
+
|
|
260
|
+
if not isinstance(cigar, (list, tuple)) or any(
|
|
261
|
+
not isinstance(item, (tuple, list)) or len(item) != 2 for item in cigar):
|
|
262
|
+
raise ValueError('Malformed CIGAR operations')
|
|
263
|
+
query_span, hard_clips = 0, 0
|
|
264
|
+
first_inner = 1 if cigar[0][0] == 5 else 0
|
|
265
|
+
last_inner = len(cigar)-2 if cigar[-1][0] == 5 else len(cigar)-1
|
|
266
|
+
for i, item in enumerate(cigar):
|
|
267
|
+
if not isinstance(item, (tuple, list)) or len(item) != 2:
|
|
268
|
+
raise ValueError('Malformed CIGAR operation')
|
|
269
|
+
op, length = item
|
|
270
|
+
if (isinstance(op, bool) or not isinstance(op, int) or op not in range(9)
|
|
271
|
+
or isinstance(length, bool) or not isinstance(length, int) or length <= 0):
|
|
272
|
+
raise ValueError('Invalid CIGAR operation or length')
|
|
273
|
+
if op == 5:
|
|
274
|
+
if i not in (0, len(cigar)-1):
|
|
275
|
+
raise ValueError('Internal hard clip')
|
|
276
|
+
hard_clips += length
|
|
277
|
+
elif op == 4 and i not in (first_inner, last_inner):
|
|
278
|
+
raise ValueError('Internal soft clip')
|
|
279
|
+
if op in (0, 1, 4, 7, 8):
|
|
280
|
+
query_span += length
|
|
281
|
+
if sequence_length and sequence_length != query_span:
|
|
282
|
+
raise ValueError('SEQ length disagrees with query-consuming CIGAR span')
|
|
283
|
+
return query_span + hard_clips or None
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
class EnrichmentCounter:
|
|
287
|
+
"""One length/status per QNAME; no sequences or worker-local counters.
|
|
288
|
+
|
|
289
|
+
Memory scales with distinct read IDs, independently of the -l batch target.
|
|
290
|
+
This estimates only yield represented by the BAM, not reads/bases removed
|
|
291
|
+
upstream. Distinct IDs count separately even for duplicate-marked reads.
|
|
292
|
+
"""
|
|
293
|
+
def __init__(self):
|
|
294
|
+
self._lengths = {} # None: unresolved; -1: conflicting/invalid; >0: known.
|
|
295
|
+
self._missing_id_records = 0
|
|
296
|
+
self._paired_records = 0
|
|
297
|
+
|
|
298
|
+
def add(self, read):
|
|
299
|
+
if read.is_paired:
|
|
300
|
+
self._paired_records += 1
|
|
301
|
+
read_id = read.query_name
|
|
302
|
+
if not isinstance(read_id, str) or not read_id.strip() or read_id == '*':
|
|
303
|
+
self._missing_id_records += 1
|
|
304
|
+
return
|
|
305
|
+
previous = self._lengths.setdefault(read_id, None)
|
|
306
|
+
if previous == -1:
|
|
307
|
+
return
|
|
308
|
+
try:
|
|
309
|
+
length = recorded_full_read_length(read)
|
|
310
|
+
except (ValueError, TypeError, IndexError):
|
|
311
|
+
self._lengths[read_id] = -1
|
|
312
|
+
return
|
|
313
|
+
if length is None:
|
|
314
|
+
return # Another alignment can still establish this read's length.
|
|
315
|
+
if previous is not None and previous != length:
|
|
316
|
+
self._lengths[read_id] = -1
|
|
317
|
+
else:
|
|
318
|
+
self._lengths[read_id] = length
|
|
319
|
+
|
|
320
|
+
def summarize(self, measured_read_ids):
|
|
321
|
+
unknown = sum(length is None for length in self._lengths.values())
|
|
322
|
+
invalid = sum(length == -1 for length in self._lengths.values())
|
|
323
|
+
reasons = []
|
|
324
|
+
if unknown:
|
|
325
|
+
reasons.append(f'{unknown} read IDs without a usable recorded full length')
|
|
326
|
+
if invalid:
|
|
327
|
+
reasons.append(f'{invalid} read IDs with conflicting or invalid length metadata')
|
|
328
|
+
if self._missing_id_records:
|
|
329
|
+
reasons.append(f'{self._missing_id_records} records without a usable read ID')
|
|
330
|
+
if self._paired_records:
|
|
331
|
+
reasons.append(f'{self._paired_records} paired-end records (unpaired long reads required)')
|
|
332
|
+
total_bases = None if reasons else sum(self._lengths.values())
|
|
333
|
+
total_gb = None if total_bases is None else total_bases / 1_000_000_000
|
|
334
|
+
rate = measured_read_ids * 1_000_000_000 / total_bases if total_bases else None
|
|
335
|
+
if total_bases == 0:
|
|
336
|
+
reasons.append('zero total recorded bases in the input BAM')
|
|
337
|
+
return {'unique_read_ids': len(self._lengths), 'total_bases': total_bases,
|
|
338
|
+
'total_gb': total_gb, 'telomeres_per_gb': rate,
|
|
339
|
+
'warning': 'Enrichment unavailable: ' + '; '.join(reasons) if reasons else None}
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def process_bam_file(bam_file_path, output_file_path, max_gap_length=20, min_read_len=1000, num_processes=8, memory_limit_gb=8,
|
|
343
|
+
variant_motif=None, min_variant_units=None, enrichment=False):
|
|
344
|
+
variant_motif, min_variant_units = validate_variant_options(variant_motif, min_variant_units)
|
|
345
|
+
enrichment_counter = EnrichmentCounter() if enrichment else None
|
|
346
|
+
start_time = time.time()
|
|
347
|
+
bam_file = pysam.AlignmentFile(bam_file_path, "rb")
|
|
348
|
+
telomere_motifs = get_telomere_repeats()
|
|
349
|
+
|
|
350
|
+
results = []
|
|
351
|
+
read_results = {}
|
|
352
|
+
total_reads = 0
|
|
353
|
+
memory_limit_bytes = memory_limit_gb * (1024 ** 3)
|
|
354
|
+
|
|
355
|
+
batch = []
|
|
356
|
+
batch_memory = 0
|
|
357
|
+
pool = Pool(processes=num_processes)
|
|
358
|
+
|
|
359
|
+
def process_batch(batch_data):
|
|
360
|
+
nonlocal total_reads, read_results
|
|
361
|
+
|
|
362
|
+
for result in pool.imap_unordered(process_read_wrapper, [(rd, telomere_motifs, max_gap_length, min_read_len, variant_motif, min_variant_units) for rd in batch_data]):
|
|
363
|
+
if result:
|
|
364
|
+
total_reads += 1
|
|
365
|
+
existing_result = read_results.get(result['read_id'])
|
|
366
|
+
if existing_result:
|
|
367
|
+
if (result['mapping_quality'] > existing_result['mapping_quality'] or
|
|
368
|
+
(result['mapping_quality'] == existing_result['mapping_quality'] and
|
|
369
|
+
result['telomere_length'] > existing_result['telomere_length']) or
|
|
370
|
+
(result['mapping_quality'] == existing_result['mapping_quality'] and
|
|
371
|
+
result['telomere_length'] == existing_result['telomere_length'] and
|
|
372
|
+
result['_selection_key'] < existing_result['_selection_key'])):
|
|
373
|
+
read_results[result['read_id']] = result
|
|
374
|
+
else:
|
|
375
|
+
read_results[result['read_id']] = result
|
|
376
|
+
|
|
377
|
+
batch_data.clear()
|
|
378
|
+
|
|
379
|
+
for read in bam_file:
|
|
380
|
+
if enrichment_counter is not None:
|
|
381
|
+
enrichment_counter.add(read)
|
|
382
|
+
if read.reference_name is not None and read.reference_name != 'chrM':
|
|
383
|
+
if read.is_unmapped:
|
|
384
|
+
continue
|
|
385
|
+
if read.reference_start > 30000 and read.reference_end < bam_file.get_reference_length(read.reference_name) - 30000:
|
|
386
|
+
continue
|
|
387
|
+
if read.query_sequence is None or len(read.query_sequence) < min_read_len:
|
|
388
|
+
continue
|
|
389
|
+
base_qualities = read.query_qualities
|
|
390
|
+
|
|
391
|
+
if base_qualities is not None and len(base_qualities) > 0:
|
|
392
|
+
avg_phred_score = sum(base_qualities) / len(base_qualities)
|
|
393
|
+
else:
|
|
394
|
+
avg_phred_score = 0
|
|
395
|
+
#print("no phred score")
|
|
396
|
+
|
|
397
|
+
if avg_phred_score > 9:
|
|
398
|
+
read_data = {
|
|
399
|
+
'query_name': read.query_name,
|
|
400
|
+
'is_unmapped': read.is_unmapped,
|
|
401
|
+
'is_reverse': read.is_reverse,
|
|
402
|
+
'reference_start': read.reference_start,
|
|
403
|
+
'reference_end': read.reference_end,
|
|
404
|
+
'reference_name': read.reference_name,
|
|
405
|
+
'mapping_quality': read.mapping_quality,
|
|
406
|
+
'query_sequence': read.query_sequence,
|
|
407
|
+
'query_qualities': base_qualities,
|
|
408
|
+
'_ranking_fields': alignment_fields(read),
|
|
409
|
+
'reference_length': bam_file.get_reference_length(read.reference_name) if read.reference_name is not None else None
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
if variant_motif is not None:
|
|
413
|
+
read_data['_variant_cigar'] = read.cigartuples or []
|
|
414
|
+
read_memory = estimate_memory_usage([read_data])
|
|
415
|
+
batch_memory += read_memory
|
|
416
|
+
batch.append(read_data)
|
|
417
|
+
|
|
418
|
+
if batch_memory >= memory_limit_bytes:
|
|
419
|
+
print(f"Processing batch of {len(batch)} reads, approx memory used: {batch_memory / (1024 ** 3):.2f} GB")
|
|
420
|
+
process_batch(batch)
|
|
421
|
+
batch_memory = 0
|
|
422
|
+
batch = []
|
|
423
|
+
#else:
|
|
424
|
+
#print("low phred")
|
|
425
|
+
|
|
426
|
+
if batch:
|
|
427
|
+
print(f"Processing final batch of {len(batch)} reads")
|
|
428
|
+
process_batch(batch)
|
|
429
|
+
|
|
430
|
+
pool.close()
|
|
431
|
+
pool.join()
|
|
432
|
+
bam_file.close()
|
|
433
|
+
|
|
434
|
+
enrichment_summary = enrichment_counter.summarize(len(read_results)) if enrichment_counter is not None else None
|
|
435
|
+
results_df = pd.DataFrame(list(read_results.values()), columns=BASE_COLUMNS + (VARIANT_COLUMNS if variant_motif is not None else []))
|
|
436
|
+
if enrichment_summary is not None:
|
|
437
|
+
rate = enrichment_summary['telomeres_per_gb']
|
|
438
|
+
results_df['telomeres_per_gb'] = rate if rate is not None else np.nan
|
|
439
|
+
results_df.to_csv(output_file_path, sep='\t', index=False)
|
|
440
|
+
if variant_motif is not None:
|
|
441
|
+
tract_rows = [tract for result in read_results.values() for tract in result['_variant_tracts']]
|
|
442
|
+
tract_path = Path(output_file_path).with_name(Path(output_file_path).stem + '_variant_tracts.tsv')
|
|
443
|
+
pd.DataFrame(tract_rows, columns=TRACT_COLUMNS).to_csv(tract_path, sep='\t', index=False)
|
|
444
|
+
|
|
445
|
+
print(f"Telometer completed successfully. Total telomeres measured: {len(read_results)}")
|
|
446
|
+
print(f"Total processing time: {time.time() - start_time:.2f} seconds")
|
|
447
|
+
if enrichment_summary is not None:
|
|
448
|
+
for key, label in [('total_bases', 'Nonredundant recorded bases in BAM'),
|
|
449
|
+
('total_gb', 'Recorded sequencing yield in BAM (Gb)'),
|
|
450
|
+
('telomeres_per_gb', 'Telomeres measured per Gb')]:
|
|
451
|
+
value = enrichment_summary[key]
|
|
452
|
+
print(f"{label}: {value if value is not None else 'NA'}")
|
|
453
|
+
if enrichment_summary['warning']:
|
|
454
|
+
print(f"Warning: {enrichment_summary['warning']}", file=sys.stderr)
|
|
455
|
+
# After the final print statements
|
|
456
|
+
if len(read_results) > 0:
|
|
457
|
+
# Convert read results to a DataFrame for easy statistical computation
|
|
458
|
+
results_df = pd.DataFrame(list(read_results.values()))
|
|
459
|
+
|
|
460
|
+
# Calculate summary statistics for the 'telomere_length' column
|
|
461
|
+
telomere_lengths = results_df['telomere_length']
|
|
462
|
+
|
|
463
|
+
min_length = telomere_lengths.min()
|
|
464
|
+
percentile_25 = telomere_lengths.quantile(0.25)
|
|
465
|
+
median_length = telomere_lengths.median()
|
|
466
|
+
mean_length = telomere_lengths.mean()
|
|
467
|
+
percentile_75 = telomere_lengths.quantile(0.75)
|
|
468
|
+
max_length = telomere_lengths.max()
|
|
469
|
+
|
|
470
|
+
# Print summary statistics
|
|
471
|
+
print("\nTelomere Length Summary Statistics:")
|
|
472
|
+
print(f"Min: {min_length:.2f}")
|
|
473
|
+
print(f"25th Percentile: {percentile_25:.2f}")
|
|
474
|
+
print(f"Median: {median_length:.2f}")
|
|
475
|
+
print(f"Mean: {mean_length:.2f}")
|
|
476
|
+
print(f"75th Percentile: {percentile_75:.2f}")
|
|
477
|
+
print(f"Max: {max_length:.2f}")
|
|
478
|
+
else:
|
|
479
|
+
print("No telomere measurements to summarize.")
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def run_telometer():
|
|
483
|
+
parser = argparse.ArgumentParser(description=f'Telometer v{__version__}: telomere length with optional exact 6-bp variant-repeat tracts and enrichment.')
|
|
484
|
+
parser.add_argument('--version', action='version', version=f'Telometer v{__version__}')
|
|
485
|
+
parser.add_argument('-b', '--bam', help='The path to the sorted BAM file.', required=True)
|
|
486
|
+
parser.add_argument('-o', '--output', help='The path to the output file.', required=True)
|
|
487
|
+
parser.add_argument('-m', '--minreadlen', default=1000, type=int, help='Minimum read length to consider (Default: 1000 for telomere capture, use 4000 for WGS). Optional', required=False)
|
|
488
|
+
parser.add_argument('-g', '--maxgaplen', default=20, type=int, help='Maximum allowed gap length between telomere regions. Optional', required=False)
|
|
489
|
+
parser.add_argument('-t', '--threads', default=cpu_count(), type=int, help='Number of processing threads to use. Optional', required=False)
|
|
490
|
+
parser.add_argument('-l', '--memlimit', default=8, type=int, help="Maximum amount of memory to commit per batch of reads while processing. Optional, default = 8 Gb", required=False)
|
|
491
|
+
parser.add_argument('-e', '--enrichment', action='store_true', help='Append telomeres_per_gb and print sample totals using one recorded full length per read ID across the entire BAM. Optional; off by default.')
|
|
492
|
+
parser.add_argument('--variant-motif', default=None, help='Optional noncanonical 6-bp DNA motif; rotations and reverse complements are included.')
|
|
493
|
+
parser.add_argument('--min-variant-units', default=None, type=int, help='Minimum non-overlapping exact 6-bp units (effective default: 2; requires --variant-motif).')
|
|
494
|
+
args = parser.parse_args()
|
|
495
|
+
try:
|
|
496
|
+
validate_variant_options(args.variant_motif, args.min_variant_units, emit_warnings=False)
|
|
497
|
+
except ValueError as error:
|
|
498
|
+
parser.error(str(error))
|
|
499
|
+
process_bam_file(args.bam, args.output, max_gap_length=args.maxgaplen, min_read_len=args.minreadlen, num_processes=args.threads, memory_limit_gb=args.memlimit,
|
|
500
|
+
variant_motif=args.variant_motif, min_variant_units=args.min_variant_units, enrichment=args.enrichment)
|
|
501
|
+
|
|
502
|
+
if __name__ == "__main__":
|
|
503
|
+
run_telometer()
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: telometer
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Quantitative digital telomere measurement from nanopore long-reads
|
|
5
|
+
Author-email: "Santiago E. Sanchez" <santy.esanchez@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Requires-Python: >=3.7
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
Requires-Dist: requests
|
|
12
|
+
Requires-Dist: numpy
|
|
13
|
+
|
|
14
|
+
A simple tool for measuring chromosome-specific telomeres from long-read alignments.
|
|
15
|
+
|
|
16
|
+
[Telometer Github](https://github.com/santiago-es/Telometer)
|
|
@@ -1,10 +1,9 @@
|
|
|
1
|
-
LICENSE.txt
|
|
2
1
|
README.md
|
|
3
|
-
|
|
2
|
+
pyproject.toml
|
|
4
3
|
telometer/__init__.py
|
|
5
4
|
telometer/telometer.py
|
|
6
5
|
telometer.egg-info/PKG-INFO
|
|
7
6
|
telometer.egg-info/SOURCES.txt
|
|
8
7
|
telometer.egg-info/dependency_links.txt
|
|
9
|
-
telometer.egg-info/
|
|
8
|
+
telometer.egg-info/requires.txt
|
|
10
9
|
telometer.egg-info/top_level.txt
|
telometer-1.0/LICENSE.txt
DELETED
|
@@ -1,19 +0,0 @@
|
|
|
1
|
-
Copyright (c) 2024 The Python Packaging Authority
|
|
2
|
-
|
|
3
|
-
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
4
|
-
of this software and associated documentation files (the "Software"), to deal
|
|
5
|
-
in the Software without restriction, including without limitation the rights
|
|
6
|
-
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
7
|
-
copies of the Software, and to permit persons to whom the Software is
|
|
8
|
-
furnished to do so, subject to the following conditions:
|
|
9
|
-
|
|
10
|
-
The above copyright notice and this permission notice shall be included in all
|
|
11
|
-
copies or substantial portions of the Software.
|
|
12
|
-
|
|
13
|
-
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
14
|
-
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
15
|
-
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
16
|
-
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
17
|
-
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
18
|
-
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
19
|
-
SOFTWARE.
|
telometer-1.0/PKG-INFO
DELETED
|
@@ -1,16 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.1
|
|
2
|
-
Name: telometer
|
|
3
|
-
Version: 1.0
|
|
4
|
-
Summary: a simple regular expression based method for measuring individual, chromosome-specific telomere lengths from long-read sequencing data
|
|
5
|
-
Author: Santiago E Sanchez
|
|
6
|
-
Author-email: ses94@stanford.edu
|
|
7
|
-
License: MIT
|
|
8
|
-
Classifier: Programming Language :: Python :: 3
|
|
9
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
-
Classifier: Operating System :: OS Independent
|
|
11
|
-
Requires-Python: >=3.7
|
|
12
|
-
License-File: LICENSE.txt
|
|
13
|
-
|
|
14
|
-
A simple tool for measuring chromosome-specific telomeres from long-read alignments.
|
|
15
|
-
|
|
16
|
-
[Telometer Github](https://github.com/santiago-es/Telometer)
|
telometer-1.0/setup.py
DELETED
|
@@ -1,23 +0,0 @@
|
|
|
1
|
-
import setuptools
|
|
2
|
-
|
|
3
|
-
setuptools.setup(
|
|
4
|
-
name="telometer",
|
|
5
|
-
version="1.0",
|
|
6
|
-
author="Santiago E Sanchez",
|
|
7
|
-
author_email="ses94@stanford.edu",
|
|
8
|
-
description="a simple regular expression based method for measuring individual, chromosome-specific telomere lengths from long-read sequencing data",
|
|
9
|
-
packages=setuptools.find_packages(),
|
|
10
|
-
license='MIT',
|
|
11
|
-
long_description=open('README.md').read(),
|
|
12
|
-
classifiers=[
|
|
13
|
-
"Programming Language :: Python :: 3",
|
|
14
|
-
"License :: OSI Approved :: MIT License",
|
|
15
|
-
"Operating System :: OS Independent",
|
|
16
|
-
],
|
|
17
|
-
python_requires='>=3.7',
|
|
18
|
-
entry_points={
|
|
19
|
-
'console_scripts': [
|
|
20
|
-
'telometer=telometer:run_telometer', # 'telometer' is the command, 'telometer:main' means the main function in telometer.py
|
|
21
|
-
],
|
|
22
|
-
},
|
|
23
|
-
)
|
|
@@ -1,292 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
# Telometer v1.0
|
|
3
|
-
# Created by: Santiago E Sanchez
|
|
4
|
-
# Artandi Lab, Stanford University, 2024
|
|
5
|
-
# Measures telomeres from ONT or PacBio long reads aligned to a T2T genome assembly
|
|
6
|
-
# Simple Usage: telometer -b sorted_t2t.bam -o output.tsv
|
|
7
|
-
|
|
8
|
-
import pysam
|
|
9
|
-
import regex as re
|
|
10
|
-
import pandas as pd
|
|
11
|
-
import time
|
|
12
|
-
import argparse
|
|
13
|
-
import sys
|
|
14
|
-
from multiprocessing import Pool, cpu_count
|
|
15
|
-
import numpy as np
|
|
16
|
-
from scipy.signal import savgol_filter
|
|
17
|
-
|
|
18
|
-
def reverse_complement(seq):
|
|
19
|
-
complement = {'A': 'T', 'C': 'G', 'G': 'C', 'T': 'A', 'N': 'N'}
|
|
20
|
-
return "".join(complement[base] for base in reversed(seq))
|
|
21
|
-
|
|
22
|
-
def get_telomere_repeats():
|
|
23
|
-
telomere_repeats = ['CCCTAA', 'TTAGGG']
|
|
24
|
-
telomere_repeats_rc = [reverse_complement(repeat) for repeat in telomere_repeats]
|
|
25
|
-
return telomere_repeats + telomere_repeats_rc
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
def identify_telomere_regions(seq, base_qualities, telomere_motifs, window_size=120, step_size=12,
|
|
29
|
-
density_threshold=0.1, max_gap_length=250, quality_threshold=15):
|
|
30
|
-
telomere_regions = []
|
|
31
|
-
current_start = None
|
|
32
|
-
gap_length = 0
|
|
33
|
-
has_large_gap = False
|
|
34
|
-
densities = []
|
|
35
|
-
|
|
36
|
-
combined_motif_pattern = '|'.join(f'({motif})' for motif in telomere_motifs)
|
|
37
|
-
|
|
38
|
-
for i in range(0, len(seq) - window_size + 1, step_size):
|
|
39
|
-
window_seq = seq[i:i + window_size]
|
|
40
|
-
motif_count = len(re.findall(combined_motif_pattern, window_seq))
|
|
41
|
-
density = motif_count / (window_size / len(telomere_motifs[0])) # Assuming all motifs have similar lengths
|
|
42
|
-
densities.append(density)
|
|
43
|
-
|
|
44
|
-
if density >= density_threshold:
|
|
45
|
-
if current_start is None:
|
|
46
|
-
current_start = i
|
|
47
|
-
gap_length = 0
|
|
48
|
-
else:
|
|
49
|
-
if current_start is not None:
|
|
50
|
-
gap_length += step_size
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
gap_start = i - gap_length
|
|
54
|
-
gap_end = i
|
|
55
|
-
gap_quality = base_qualities[gap_start:gap_end]
|
|
56
|
-
|
|
57
|
-
if len(gap_quality) > 0:
|
|
58
|
-
avg_gap_quality = sum(gap_quality) / len(gap_quality)
|
|
59
|
-
else:
|
|
60
|
-
avg_gap_quality = 0
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
if avg_gap_quality < quality_threshold:
|
|
64
|
-
gap_length = 0
|
|
65
|
-
continue
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
if gap_length > max_gap_length:
|
|
69
|
-
|
|
70
|
-
mismatch_motif_count = len(re.findall(f"({combined_motif_pattern}){{e<=1}}", window_seq))
|
|
71
|
-
mismatch_density = mismatch_motif_count / (window_size / len(telomere_motifs[0]))
|
|
72
|
-
|
|
73
|
-
if mismatch_density >= density_threshold:
|
|
74
|
-
|
|
75
|
-
gap_length = 0
|
|
76
|
-
else:
|
|
77
|
-
has_large_gap = True
|
|
78
|
-
telomere_regions.append((current_start, i - gap_length + step_size))
|
|
79
|
-
current_start = None
|
|
80
|
-
gap_length = 0
|
|
81
|
-
|
|
82
|
-
if current_start is not None:
|
|
83
|
-
telomere_regions.append((current_start, len(seq)))
|
|
84
|
-
|
|
85
|
-
return telomere_regions, densities, has_large_gap, combined_motif_pattern
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
def detect_discontinuities(base_qualities, window_length=11, polyorder=2, gradient_threshold=-1):
|
|
89
|
-
|
|
90
|
-
if len(base_qualities) < window_length:
|
|
91
|
-
window_length = len(base_qualities) if len(base_qualities) % 2 == 1 else len(base_qualities) - 1 # Window length must be odd
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
if len(base_qualities) > polyorder:
|
|
95
|
-
smoothed_qualities = savgol_filter(base_qualities, window_length=window_length, polyorder=polyorder)
|
|
96
|
-
else:
|
|
97
|
-
|
|
98
|
-
smoothed_qualities = base_qualities
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
gradient = np.diff(smoothed_qualities)
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
discontinuities = np.where(gradient < gradient_threshold)[0]
|
|
105
|
-
|
|
106
|
-
return discontinuities, smoothed_qualities
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
def check_gap(seq, base_qualities, telomere_regions, gradient_threshold=-1, quality_threshold=15):
|
|
110
|
-
merged_regions = []
|
|
111
|
-
prev_end = None
|
|
112
|
-
current_start = None
|
|
113
|
-
|
|
114
|
-
for i, (start, end) in enumerate(telomere_regions):
|
|
115
|
-
if current_start is None:
|
|
116
|
-
current_start = start
|
|
117
|
-
|
|
118
|
-
if prev_end is not None:
|
|
119
|
-
gap_start = prev_end
|
|
120
|
-
gap_end = start
|
|
121
|
-
gap_quality = base_qualities[gap_start:gap_end]
|
|
122
|
-
discontinuities, smoothed_qualities = detect_discontinuities(gap_quality, gradient_threshold=gradient_threshold)
|
|
123
|
-
avg_gap_quality = sum(gap_quality) / len(gap_quality) if len(gap_quality) > 0 else 0
|
|
124
|
-
if avg_gap_quality < quality_threshold or len(discontinuities) == 0:
|
|
125
|
-
continue
|
|
126
|
-
merged_regions.append((current_start, end))
|
|
127
|
-
current_start = None
|
|
128
|
-
prev_end = end
|
|
129
|
-
|
|
130
|
-
return merged_regions
|
|
131
|
-
|
|
132
|
-
def measure_telomere_length(seq, telomere_motifs, base_qualities, window_size=120, step_size=12, density_threshold=0.1, max_gap_length=150, quality_threshold=15):
|
|
133
|
-
telomere_regions, densities, has_large_gap, combined_motif_pattern = identify_telomere_regions(
|
|
134
|
-
seq, base_qualities, telomere_motifs, window_size, step_size, density_threshold, max_gap_length, quality_threshold)
|
|
135
|
-
|
|
136
|
-
if not telomere_regions:
|
|
137
|
-
return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
|
|
138
|
-
|
|
139
|
-
merged_telomere_regions = check_gap(seq, base_qualities, telomere_regions, gradient_threshold=-1, quality_threshold=quality_threshold)
|
|
140
|
-
terminus_regions = [region for region in merged_telomere_regions if region[0] < 100 or region[1] > len(seq) - 100]
|
|
141
|
-
|
|
142
|
-
if not terminus_regions:
|
|
143
|
-
return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
|
|
144
|
-
|
|
145
|
-
telomere_start, telomere_end = terminus_regions[0]
|
|
146
|
-
telomere_length = telomere_end - telomere_start
|
|
147
|
-
return telomere_start, telomere_end, telomere_length, densities, has_large_gap, combined_motif_pattern
|
|
148
|
-
|
|
149
|
-
# New function to process a single read
|
|
150
|
-
def process_read(read_data, telomere_motifs, max_gap_length, min_read_len):
|
|
151
|
-
if read_data['is_unmapped'] or read_data['reference_name'] == 'chrM':
|
|
152
|
-
return None
|
|
153
|
-
|
|
154
|
-
if read_data['reference_start'] > 30000 and read_data['reference_end'] < read_data['reference_length'] - 30000:
|
|
155
|
-
return None
|
|
156
|
-
|
|
157
|
-
seq = read_data['query_sequence']
|
|
158
|
-
base_qualities = read_data['query_qualities'] # Extract base qualities here
|
|
159
|
-
if seq is None or len(seq) < min_read_len:
|
|
160
|
-
return None
|
|
161
|
-
|
|
162
|
-
alignment_start = read_data['reference_start']
|
|
163
|
-
alignment_end = read_data['reference_end']
|
|
164
|
-
reference_genome_length = read_data['reference_length']
|
|
165
|
-
|
|
166
|
-
if alignment_start < 15000 and alignment_end <= reference_genome_length - 30000:
|
|
167
|
-
arm = "p"
|
|
168
|
-
else:
|
|
169
|
-
arm = "q"
|
|
170
|
-
|
|
171
|
-
direction = "rev" if read_data['is_reverse'] else "fwd"
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
telomere_start, telomere_end, telomere_length, densities, has_large_gap, combined_motif_pattern = measure_telomere_length(
|
|
175
|
-
seq, telomere_motifs, base_qualities, max_gap_length=max_gap_length)
|
|
176
|
-
|
|
177
|
-
if telomere_length < 1:
|
|
178
|
-
return None
|
|
179
|
-
|
|
180
|
-
return {
|
|
181
|
-
'chromosome': read_data['reference_name'],
|
|
182
|
-
'reference_start': alignment_start,
|
|
183
|
-
'reference_end': alignment_end,
|
|
184
|
-
'telomere_length': telomere_length,
|
|
185
|
-
'read_id': read_data['query_name'],
|
|
186
|
-
'mapping_quality': read_data['mapping_quality'],
|
|
187
|
-
'read_length': len(seq),
|
|
188
|
-
'arm': arm,
|
|
189
|
-
'direction': direction
|
|
190
|
-
}
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
def process_read_wrapper(args):
|
|
194
|
-
return process_read(*args)
|
|
195
|
-
|
|
196
|
-
def estimate_memory_usage(reads_list):
|
|
197
|
-
"""
|
|
198
|
-
Estimate the memory usage of the reads in bytes.
|
|
199
|
-
Approximate size based on the size of sequences, base qualities, and metadata.
|
|
200
|
-
"""
|
|
201
|
-
memory_usage = 0
|
|
202
|
-
for read in reads_list:
|
|
203
|
-
seq_len = len(read['query_sequence']) if read['query_sequence'] else 0
|
|
204
|
-
quality_len = len(read['query_qualities']) if read['query_qualities'] else 0
|
|
205
|
-
memory_usage += sys.getsizeof(read['query_name']) + seq_len + quality_len + sys.getsizeof(read)
|
|
206
|
-
return memory_usage
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
def process_bam_file(bam_file_path, output_file_path, max_gap_length=150, min_read_len=1000, num_processes=8, memory_limit_gb=8):
|
|
210
|
-
start_time = time.time()
|
|
211
|
-
bam_file = pysam.AlignmentFile(bam_file_path, "rb")
|
|
212
|
-
telomere_motifs = get_telomere_repeats()
|
|
213
|
-
|
|
214
|
-
results = []
|
|
215
|
-
read_results = {}
|
|
216
|
-
total_reads = 0
|
|
217
|
-
memory_limit_bytes = memory_limit_gb * (1024 ** 3)
|
|
218
|
-
|
|
219
|
-
batch = []
|
|
220
|
-
batch_memory = 0
|
|
221
|
-
|
|
222
|
-
def process_batch(batch_data):
|
|
223
|
-
nonlocal total_reads, read_results
|
|
224
|
-
|
|
225
|
-
with Pool(processes=num_processes) as pool:
|
|
226
|
-
for result in pool.imap_unordered(process_read_wrapper, [(rd, telomere_motifs, max_gap_length, min_read_len) for rd in batch_data]):
|
|
227
|
-
if result:
|
|
228
|
-
total_reads += 1
|
|
229
|
-
existing_result = read_results.get(result['read_id'])
|
|
230
|
-
if existing_result:
|
|
231
|
-
if (result['mapping_quality'] > existing_result['mapping_quality'] or
|
|
232
|
-
(result['mapping_quality'] == existing_result['mapping_quality'] and
|
|
233
|
-
result['telomere_length'] > existing_result['telomere_length'])):
|
|
234
|
-
read_results[result['read_id']] = result
|
|
235
|
-
else:
|
|
236
|
-
read_results[result['read_id']] = result
|
|
237
|
-
|
|
238
|
-
batch_data.clear()
|
|
239
|
-
|
|
240
|
-
for read in bam_file:
|
|
241
|
-
if read.reference_name is not None and read.reference_name != 'chrM':
|
|
242
|
-
read_data = {
|
|
243
|
-
'query_name': read.query_name,
|
|
244
|
-
'is_unmapped': read.is_unmapped,
|
|
245
|
-
'is_reverse': read.is_reverse,
|
|
246
|
-
'reference_start': read.reference_start,
|
|
247
|
-
'reference_end': read.reference_end,
|
|
248
|
-
'reference_name': read.reference_name,
|
|
249
|
-
'mapping_quality': read.mapping_quality,
|
|
250
|
-
'query_sequence': read.query_sequence,
|
|
251
|
-
'query_qualities': read.query_qualities,
|
|
252
|
-
'reference_length': bam_file.get_reference_length(read.reference_name) if read.reference_name is not None else None
|
|
253
|
-
}
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
read_memory = estimate_memory_usage([read_data])
|
|
257
|
-
batch_memory += read_memory
|
|
258
|
-
batch.append(read_data)
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
if batch_memory >= memory_limit_bytes:
|
|
262
|
-
print(f"Processing batch of {len(batch)} reads, approx memory used: {batch_memory / (1024 ** 3):.2f} GB")
|
|
263
|
-
process_batch(batch)
|
|
264
|
-
batch_memory = 0
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
if batch:
|
|
268
|
-
print(f"Processing final batch of {len(batch)} reads")
|
|
269
|
-
process_batch(batch)
|
|
270
|
-
|
|
271
|
-
bam_file.close()
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
results_df = pd.DataFrame(list(read_results.values()))
|
|
275
|
-
results_df.to_csv(output_file_path, sep='\t', index=False)
|
|
276
|
-
|
|
277
|
-
print(f"Telometer completed successfully. Total telomeres measured: {len(read_results)}")
|
|
278
|
-
print(f"Total processing time: {time.time() - start_time:.2f} seconds")
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
def run_telometer():
|
|
282
|
-
parser = argparse.ArgumentParser(description='Calculate telomere length from a BAM file.')
|
|
283
|
-
parser.add_argument('-b', '--bam', help='The path to the sorted BAM file.', required=True)
|
|
284
|
-
parser.add_argument('-o', '--output', help='The path to the output file.', required=True)
|
|
285
|
-
parser.add_argument('-m', '--minreadlen', default=1000, type=int, help='Minimum read length to consider (Default: 1000 for telomere capture, use 4000 for WGS). Optional', required=False)
|
|
286
|
-
parser.add_argument('-g', '--maxgaplen', default=250, type=int, help='Maximum allowed gap length between telomere regions. Optional', required=False)
|
|
287
|
-
parser.add_argument('-t', '--threads', default=cpu_count(), type=int, help='Number of processing threads to use. Optional', required=False)
|
|
288
|
-
args = parser.parse_args()
|
|
289
|
-
process_bam_file(args.bam, args.output, max_gap_length=args.maxgaplen, min_read_len=args.minreadlen, num_processes=args.threads)
|
|
290
|
-
|
|
291
|
-
if __name__ == "__main__":
|
|
292
|
-
run_telometer()
|
|
@@ -1,16 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.1
|
|
2
|
-
Name: telometer
|
|
3
|
-
Version: 1.0
|
|
4
|
-
Summary: a simple regular expression based method for measuring individual, chromosome-specific telomere lengths from long-read sequencing data
|
|
5
|
-
Author: Santiago E Sanchez
|
|
6
|
-
Author-email: ses94@stanford.edu
|
|
7
|
-
License: MIT
|
|
8
|
-
Classifier: Programming Language :: Python :: 3
|
|
9
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
-
Classifier: Operating System :: OS Independent
|
|
11
|
-
Requires-Python: >=3.7
|
|
12
|
-
License-File: LICENSE.txt
|
|
13
|
-
|
|
14
|
-
A simple tool for measuring chromosome-specific telomeres from long-read alignments.
|
|
15
|
-
|
|
16
|
-
[Telometer Github](https://github.com/santiago-es/Telometer)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|