telometer 1.1__tar.gz → 2.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,16 @@
1
+ Metadata-Version: 2.4
2
+ Name: telometer
3
+ Version: 2.0.1
4
+ Summary: Quantitative digital telomere measurement from nanopore long-reads
5
+ Author-email: "Santiago E. Sanchez" <santy.esanchez@gmail.com>
6
+ License-Expression: MIT
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: Operating System :: OS Independent
9
+ Requires-Python: >=3.7
10
+ Description-Content-Type: text/markdown
11
+ Requires-Dist: requests
12
+ Requires-Dist: numpy
13
+
14
+ A simple tool for measuring chromosome-specific telomeres from long-read alignments.
15
+
16
+ [Telometer Github](https://github.com/santiago-es/Telometer)
@@ -0,0 +1,31 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77.0.3", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "telometer"
7
+ version = "2.0.1"
8
+ description = "Quantitative digital telomere measurement from nanopore long-reads"
9
+ readme = "README.md"
10
+ requires-python = ">=3.7"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+
14
+ authors = [
15
+ { name = "Santiago E. Sanchez", email = "santy.esanchez@gmail.com" }
16
+ ]
17
+
18
+ classifiers = [
19
+ "Programming Language :: Python :: 3",
20
+ "Operating System :: OS Independent"
21
+ ]
22
+
23
+ dependencies = [
24
+ "requests",
25
+ "numpy"
26
+ ]
27
+
28
+ [tool.setuptools.packages.find]
29
+ where = ["."]
30
+ include = ["telometer*"]
31
+ exclude = ["build*", "dist*"]
@@ -0,0 +1,503 @@
1
+ #!/usr/bin/env python3
2
+ # Telometer v2.0: eight-feature baseline with optional variant detection and enrichment
3
+ # Baseline changes retained: 01, 03, 07, 11, 12, 14, 15, 16
4
+ # Created by: Santiago E Sanchez
5
+ # Artandi Lab, Stanford University, 2024
6
+ # Measures telomeres from ONT or PacBio long reads aligned to a T2T genome assembly
7
+ # Simple Usage: telometer -b sorted_t2t.bam -o output.tsv
8
+ # Optional enrichment: add -e to your existing Telometer command.
9
+ # -e adds the sample-level telomeres_per_gb column and console totals. Each QNAME
10
+ # contributes one consistent recorded full length, including soft/hard clips,
11
+ # before any Telometer filter. Missing/conflicting lengths yield NA, not a
12
+ # partial-denominator rate. This measures yield represented by the input BAM;
13
+ # reads/bases removed upstream without retained metadata cannot be recovered.
14
+ # Memory for -e scales with unique read IDs; -l is not a total-RSS limit.
15
+
16
+ __version__ = "2.0"
17
+
18
+ import pysam
19
+ from pathlib import Path
20
+ from variant_support_v2 import (BASE_COLUMNS, VARIANT_COLUMNS, TRACT_COLUMNS,
21
+ validate_variant_options, strict_runs, coverage_prefix, combined_window_density, summarize_variant)
22
+ import regex as re
23
+ import pandas as pd
24
+ import time
25
+ import argparse
26
+ import sys
27
+ from multiprocessing import Pool, cpu_count
28
+ import numpy as np
29
+ from scipy.signal import savgol_filter
30
+ from telometer_variant_support import alignment_fields, cached_findall, note, reverse_complement_iupac, stable_record_key
31
+
32
+ def reverse_complement(seq):
33
+ complement = {'A': 'T', 'C': 'G', 'G': 'C', 'T': 'A', 'N': 'N'}
34
+ return "".join(complement[base] for base in reversed(seq))
35
+
36
+ def get_telomere_repeats():
37
+ telomere_repeats = ['CCCTAA', 'TTAGGG']
38
+ telomere_repeats_rc = [reverse_complement(repeat) for repeat in telomere_repeats]
39
+ return telomere_repeats + telomere_repeats_rc
40
+
41
+ def identify_telomere_regions(seq, base_qualities, telomere_motifs, window_size=120, step_size=12,
42
+ density_threshold=0.1, max_gap_length=20, quality_threshold=15, variant_motif=None, min_variant_units=2):
43
+ variant_prefix = coverage_prefix(len(seq), strict_runs(seq, variant_motif, min_variant_units)) if variant_motif is not None else None
44
+ telomere_regions = []
45
+ current_start = None
46
+ gap_length = 0
47
+ has_large_gap = False
48
+ densities = []
49
+
50
+ combined_motif_pattern = '|'.join(f'({motif})' for motif in telomere_motifs)
51
+
52
+ for i in range(0, len(seq) - window_size + 1, step_size):
53
+ window_seq = seq[i:i + window_size]
54
+ motif_count = len(cached_findall(combined_motif_pattern, window_seq))
55
+ density = motif_count / (window_size / len(telomere_motifs[0]))
56
+ if variant_prefix is not None:
57
+ density = combined_window_density(window_seq, i, combined_motif_pattern, motif_count, variant_prefix)
58
+ densities.append(density)
59
+
60
+ if density >= density_threshold:
61
+ if current_start is None:
62
+ current_start = i
63
+ gap_length = 0
64
+ else:
65
+ if current_start is not None:
66
+ gap_length += step_size
67
+
68
+ gap_start = i - gap_length
69
+ gap_end = i
70
+ gap_quality = base_qualities[gap_start:gap_end]
71
+
72
+ if len(gap_quality) > 0:
73
+ avg_gap_quality = sum(gap_quality) / len(gap_quality)
74
+ else:
75
+ avg_gap_quality = 0
76
+
77
+ if avg_gap_quality < quality_threshold:
78
+ gap_length = 0
79
+ continue
80
+
81
+ if gap_length > max_gap_length:
82
+ mismatch_motif_count = len(cached_findall(f"({combined_motif_pattern}){{e<=1}}", window_seq))
83
+ mismatch_density = mismatch_motif_count / (window_size / len(telomere_motifs[0]))
84
+
85
+ if mismatch_density >= density_threshold:
86
+ gap_length = 0
87
+ else:
88
+ has_large_gap = True
89
+ telomere_regions.append((current_start, i - gap_length + step_size))
90
+ current_start = None
91
+ gap_length = 0
92
+
93
+ if current_start is not None:
94
+ telomere_regions.append((current_start, len(seq)))
95
+
96
+ return telomere_regions, densities, has_large_gap, combined_motif_pattern
97
+
98
+ def detect_discontinuities(base_qualities, window_length=11, polyorder=2, gradient_threshold=-1):
99
+ if len(base_qualities) < window_length:
100
+ window_length = len(base_qualities) if len(base_qualities) % 2 == 1 else len(base_qualities) - 1
101
+
102
+ if len(base_qualities) > polyorder:
103
+ smoothed_qualities = savgol_filter(base_qualities, window_length=window_length, polyorder=polyorder)
104
+ else:
105
+ smoothed_qualities = base_qualities
106
+
107
+ gradient = np.diff(smoothed_qualities)
108
+ discontinuities = np.where(gradient < gradient_threshold)[0]
109
+
110
+ return discontinuities, smoothed_qualities
111
+
112
+ def check_gap(seq, base_qualities, telomere_regions, gradient_threshold=-1, quality_threshold=15):
113
+ if not telomere_regions:
114
+ return []
115
+ merged_regions = []
116
+ current_start, current_end = telomere_regions[0]
117
+ for start, end in telomere_regions[1:]:
118
+ gap_quality = base_qualities[current_end:start]
119
+ discontinuities, smoothed_qualities = detect_discontinuities(gap_quality, gradient_threshold=gradient_threshold)
120
+ avg_gap_quality = sum(gap_quality) / len(gap_quality) if len(gap_quality) > 0 else 0
121
+ if avg_gap_quality < quality_threshold or len(discontinuities) == 0:
122
+ current_end = max(current_end, end)
123
+ else:
124
+ merged_regions.append((current_start, current_end))
125
+ current_start, current_end = start, end
126
+ merged_regions.append((current_start, current_end))
127
+ return merged_regions
128
+
129
+ def measure_telomere_length(seq, telomere_motifs, base_qualities, window_size=120, step_size=12,
130
+ density_threshold=0.1, max_gap_length=20, quality_threshold=15,
131
+ variant_motif=None, min_variant_units=2):
132
+
133
+ telomere_regions, densities, has_large_gap, combined_motif_pattern = identify_telomere_regions(
134
+ seq, base_qualities, telomere_motifs, window_size, step_size, density_threshold, max_gap_length, quality_threshold,
135
+ variant_motif=variant_motif, min_variant_units=min_variant_units)
136
+
137
+ if not telomere_regions:
138
+ return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
139
+
140
+
141
+ if not any(density >= 0.75 for density in densities):
142
+ return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
143
+
144
+
145
+ merged_telomere_regions = check_gap(seq, base_qualities, telomere_regions, gradient_threshold=-1, quality_threshold=quality_threshold)
146
+
147
+ terminus_regions = []
148
+ for start, end in merged_telomere_regions:
149
+ if start < 100 or end > len(seq) - 100:
150
+ terminus_regions.append((start, end))
151
+
152
+ if not terminus_regions:
153
+ return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
154
+
155
+
156
+ telomere_start, telomere_end = terminus_regions[0]
157
+ if not any(density >= 0.75 and telomere_start <= j * step_size and
158
+ j * step_size + window_size <= telomere_end
159
+ for j, density in enumerate(densities)):
160
+ note('candidate_local_dense_failed')
161
+ return 0, 0, 0, densities, has_large_gap, combined_motif_pattern
162
+ telomere_length = telomere_end - telomere_start
163
+
164
+ return telomere_start, telomere_end, telomere_length, densities, has_large_gap, combined_motif_pattern
165
+
166
+
167
+ def process_read(read_data, telomere_motifs, max_gap_length, min_read_len, variant_motif=None, min_variant_units=2):
168
+ if read_data['is_unmapped'] or read_data['reference_name'] == 'chrM':
169
+ return None
170
+
171
+ if read_data['reference_start'] > 30000 and read_data['reference_end'] < read_data['reference_length'] - 30000:
172
+ return None
173
+
174
+ seq = read_data['query_sequence']
175
+ base_qualities = read_data['query_qualities']
176
+ if seq is None or len(seq) < min_read_len:
177
+ return None
178
+ if read_data['is_reverse']:
179
+ seq = reverse_complement_iupac(seq)
180
+ base_qualities = base_qualities[::-1]
181
+
182
+ alignment_start = read_data['reference_start']
183
+ alignment_end = read_data['reference_end']
184
+ reference_genome_length = read_data['reference_length']
185
+
186
+ near_p = alignment_start <= 30000
187
+ near_q = alignment_end >= reference_genome_length - 30000
188
+ if near_p and not near_q:
189
+ arm = "p"
190
+ elif near_q and not near_p:
191
+ arm = "q"
192
+ elif near_p and near_q:
193
+ arm = "ambiguous"
194
+ else:
195
+ arm = "unassigned"
196
+
197
+ direction = "rev" if read_data['is_reverse'] else "fwd"
198
+
199
+ telomere_start, telomere_end, telomere_length, densities, has_large_gap, combined_motif_pattern = measure_telomere_length(
200
+ seq, telomere_motifs, base_qualities, max_gap_length=max_gap_length,
201
+ variant_motif=variant_motif, min_variant_units=min_variant_units)
202
+
203
+ if telomere_length < 1:
204
+ return None
205
+
206
+ if (telomere_length + 50) > len(seq):
207
+ return None
208
+
209
+ result = {
210
+ 'chromosome': read_data['reference_name'],
211
+ 'reference_start': alignment_start,
212
+ 'reference_end': alignment_end,
213
+ 'telomere_length': telomere_length,
214
+ 'read_id': read_data['query_name'],
215
+ 'mapping_quality': read_data['mapping_quality'],
216
+ 'read_length': len(seq),
217
+ 'arm': arm,
218
+ 'direction': direction,
219
+ '_selection_key': stable_record_key(read_data['_ranking_fields'])
220
+ }
221
+ if variant_motif is not None:
222
+ fields, tracts, details = summarize_variant(seq, base_qualities, telomere_start, telomere_end,
223
+ variant_motif, min_variant_units, read_data)
224
+ result.update(fields)
225
+ result['_variant_tracts'] = tracts
226
+ result['_variant_details'] = details
227
+ return result
228
+
229
+ def process_read_wrapper(args):
230
+ return process_read(*args)
231
+
232
+ def estimate_memory_usage(reads_list):
233
+ """
234
+ Estimate the memory usage of the reads in bytes.
235
+ Approximate size based on the size of sequences, base qualities, and metadata.
236
+ """
237
+ memory_usage = 0
238
+ for read in reads_list:
239
+ seq_len = len(read['query_sequence']) if read['query_sequence'] else 0
240
+ quality_len = len(read['query_qualities']) if read['query_qualities'] else 0
241
+ memory_usage += sys.getsizeof(read['query_name']) + seq_len + quality_len + sys.getsizeof(read)
242
+ return memory_usage
243
+
244
+ def recorded_full_read_length(read):
245
+ """Return the positive full length recorded by SEQ/CIGAR, or None if unknown.
246
+
247
+ Includes soft clips and reported hard clips, but never reconstructs bases.
248
+ Malformed or mutually inconsistent length metadata raises ValueError.
249
+ """
250
+ sequence_length = read.query_length
251
+ if sequence_length is None:
252
+ sequence_length = 0
253
+ if (isinstance(sequence_length, bool) or not isinstance(sequence_length, int)
254
+ or sequence_length < 0):
255
+ raise ValueError('Invalid stored query length')
256
+ cigar = read.cigartuples or []
257
+ if not cigar:
258
+ return sequence_length or None
259
+
260
+ if not isinstance(cigar, (list, tuple)) or any(
261
+ not isinstance(item, (tuple, list)) or len(item) != 2 for item in cigar):
262
+ raise ValueError('Malformed CIGAR operations')
263
+ query_span, hard_clips = 0, 0
264
+ first_inner = 1 if cigar[0][0] == 5 else 0
265
+ last_inner = len(cigar)-2 if cigar[-1][0] == 5 else len(cigar)-1
266
+ for i, item in enumerate(cigar):
267
+ if not isinstance(item, (tuple, list)) or len(item) != 2:
268
+ raise ValueError('Malformed CIGAR operation')
269
+ op, length = item
270
+ if (isinstance(op, bool) or not isinstance(op, int) or op not in range(9)
271
+ or isinstance(length, bool) or not isinstance(length, int) or length <= 0):
272
+ raise ValueError('Invalid CIGAR operation or length')
273
+ if op == 5:
274
+ if i not in (0, len(cigar)-1):
275
+ raise ValueError('Internal hard clip')
276
+ hard_clips += length
277
+ elif op == 4 and i not in (first_inner, last_inner):
278
+ raise ValueError('Internal soft clip')
279
+ if op in (0, 1, 4, 7, 8):
280
+ query_span += length
281
+ if sequence_length and sequence_length != query_span:
282
+ raise ValueError('SEQ length disagrees with query-consuming CIGAR span')
283
+ return query_span + hard_clips or None
284
+
285
+
286
+ class EnrichmentCounter:
287
+ """One length/status per QNAME; no sequences or worker-local counters.
288
+
289
+ Memory scales with distinct read IDs, independently of the -l batch target.
290
+ This estimates only yield represented by the BAM, not reads/bases removed
291
+ upstream. Distinct IDs count separately even for duplicate-marked reads.
292
+ """
293
+ def __init__(self):
294
+ self._lengths = {} # None: unresolved; -1: conflicting/invalid; >0: known.
295
+ self._missing_id_records = 0
296
+ self._paired_records = 0
297
+
298
+ def add(self, read):
299
+ if read.is_paired:
300
+ self._paired_records += 1
301
+ read_id = read.query_name
302
+ if not isinstance(read_id, str) or not read_id.strip() or read_id == '*':
303
+ self._missing_id_records += 1
304
+ return
305
+ previous = self._lengths.setdefault(read_id, None)
306
+ if previous == -1:
307
+ return
308
+ try:
309
+ length = recorded_full_read_length(read)
310
+ except (ValueError, TypeError, IndexError):
311
+ self._lengths[read_id] = -1
312
+ return
313
+ if length is None:
314
+ return # Another alignment can still establish this read's length.
315
+ if previous is not None and previous != length:
316
+ self._lengths[read_id] = -1
317
+ else:
318
+ self._lengths[read_id] = length
319
+
320
+ def summarize(self, measured_read_ids):
321
+ unknown = sum(length is None for length in self._lengths.values())
322
+ invalid = sum(length == -1 for length in self._lengths.values())
323
+ reasons = []
324
+ if unknown:
325
+ reasons.append(f'{unknown} read IDs without a usable recorded full length')
326
+ if invalid:
327
+ reasons.append(f'{invalid} read IDs with conflicting or invalid length metadata')
328
+ if self._missing_id_records:
329
+ reasons.append(f'{self._missing_id_records} records without a usable read ID')
330
+ if self._paired_records:
331
+ reasons.append(f'{self._paired_records} paired-end records (unpaired long reads required)')
332
+ total_bases = None if reasons else sum(self._lengths.values())
333
+ total_gb = None if total_bases is None else total_bases / 1_000_000_000
334
+ rate = measured_read_ids * 1_000_000_000 / total_bases if total_bases else None
335
+ if total_bases == 0:
336
+ reasons.append('zero total recorded bases in the input BAM')
337
+ return {'unique_read_ids': len(self._lengths), 'total_bases': total_bases,
338
+ 'total_gb': total_gb, 'telomeres_per_gb': rate,
339
+ 'warning': 'Enrichment unavailable: ' + '; '.join(reasons) if reasons else None}
340
+
341
+
342
+ def process_bam_file(bam_file_path, output_file_path, max_gap_length=20, min_read_len=1000, num_processes=8, memory_limit_gb=8,
343
+ variant_motif=None, min_variant_units=None, enrichment=False):
344
+ variant_motif, min_variant_units = validate_variant_options(variant_motif, min_variant_units)
345
+ enrichment_counter = EnrichmentCounter() if enrichment else None
346
+ start_time = time.time()
347
+ bam_file = pysam.AlignmentFile(bam_file_path, "rb")
348
+ telomere_motifs = get_telomere_repeats()
349
+
350
+ results = []
351
+ read_results = {}
352
+ total_reads = 0
353
+ memory_limit_bytes = memory_limit_gb * (1024 ** 3)
354
+
355
+ batch = []
356
+ batch_memory = 0
357
+ pool = Pool(processes=num_processes)
358
+
359
+ def process_batch(batch_data):
360
+ nonlocal total_reads, read_results
361
+
362
+ for result in pool.imap_unordered(process_read_wrapper, [(rd, telomere_motifs, max_gap_length, min_read_len, variant_motif, min_variant_units) for rd in batch_data]):
363
+ if result:
364
+ total_reads += 1
365
+ existing_result = read_results.get(result['read_id'])
366
+ if existing_result:
367
+ if (result['mapping_quality'] > existing_result['mapping_quality'] or
368
+ (result['mapping_quality'] == existing_result['mapping_quality'] and
369
+ result['telomere_length'] > existing_result['telomere_length']) or
370
+ (result['mapping_quality'] == existing_result['mapping_quality'] and
371
+ result['telomere_length'] == existing_result['telomere_length'] and
372
+ result['_selection_key'] < existing_result['_selection_key'])):
373
+ read_results[result['read_id']] = result
374
+ else:
375
+ read_results[result['read_id']] = result
376
+
377
+ batch_data.clear()
378
+
379
+ for read in bam_file:
380
+ if enrichment_counter is not None:
381
+ enrichment_counter.add(read)
382
+ if read.reference_name is not None and read.reference_name != 'chrM':
383
+ if read.is_unmapped:
384
+ continue
385
+ if read.reference_start > 30000 and read.reference_end < bam_file.get_reference_length(read.reference_name) - 30000:
386
+ continue
387
+ if read.query_sequence is None or len(read.query_sequence) < min_read_len:
388
+ continue
389
+ base_qualities = read.query_qualities
390
+
391
+ if base_qualities is not None and len(base_qualities) > 0:
392
+ avg_phred_score = sum(base_qualities) / len(base_qualities)
393
+ else:
394
+ avg_phred_score = 0
395
+ #print("no phred score")
396
+
397
+ if avg_phred_score > 9:
398
+ read_data = {
399
+ 'query_name': read.query_name,
400
+ 'is_unmapped': read.is_unmapped,
401
+ 'is_reverse': read.is_reverse,
402
+ 'reference_start': read.reference_start,
403
+ 'reference_end': read.reference_end,
404
+ 'reference_name': read.reference_name,
405
+ 'mapping_quality': read.mapping_quality,
406
+ 'query_sequence': read.query_sequence,
407
+ 'query_qualities': base_qualities,
408
+ '_ranking_fields': alignment_fields(read),
409
+ 'reference_length': bam_file.get_reference_length(read.reference_name) if read.reference_name is not None else None
410
+ }
411
+
412
+ if variant_motif is not None:
413
+ read_data['_variant_cigar'] = read.cigartuples or []
414
+ read_memory = estimate_memory_usage([read_data])
415
+ batch_memory += read_memory
416
+ batch.append(read_data)
417
+
418
+ if batch_memory >= memory_limit_bytes:
419
+ print(f"Processing batch of {len(batch)} reads, approx memory used: {batch_memory / (1024 ** 3):.2f} GB")
420
+ process_batch(batch)
421
+ batch_memory = 0
422
+ batch = []
423
+ #else:
424
+ #print("low phred")
425
+
426
+ if batch:
427
+ print(f"Processing final batch of {len(batch)} reads")
428
+ process_batch(batch)
429
+
430
+ pool.close()
431
+ pool.join()
432
+ bam_file.close()
433
+
434
+ enrichment_summary = enrichment_counter.summarize(len(read_results)) if enrichment_counter is not None else None
435
+ results_df = pd.DataFrame(list(read_results.values()), columns=BASE_COLUMNS + (VARIANT_COLUMNS if variant_motif is not None else []))
436
+ if enrichment_summary is not None:
437
+ rate = enrichment_summary['telomeres_per_gb']
438
+ results_df['telomeres_per_gb'] = rate if rate is not None else np.nan
439
+ results_df.to_csv(output_file_path, sep='\t', index=False)
440
+ if variant_motif is not None:
441
+ tract_rows = [tract for result in read_results.values() for tract in result['_variant_tracts']]
442
+ tract_path = Path(output_file_path).with_name(Path(output_file_path).stem + '_variant_tracts.tsv')
443
+ pd.DataFrame(tract_rows, columns=TRACT_COLUMNS).to_csv(tract_path, sep='\t', index=False)
444
+
445
+ print(f"Telometer completed successfully. Total telomeres measured: {len(read_results)}")
446
+ print(f"Total processing time: {time.time() - start_time:.2f} seconds")
447
+ if enrichment_summary is not None:
448
+ for key, label in [('total_bases', 'Nonredundant recorded bases in BAM'),
449
+ ('total_gb', 'Recorded sequencing yield in BAM (Gb)'),
450
+ ('telomeres_per_gb', 'Telomeres measured per Gb')]:
451
+ value = enrichment_summary[key]
452
+ print(f"{label}: {value if value is not None else 'NA'}")
453
+ if enrichment_summary['warning']:
454
+ print(f"Warning: {enrichment_summary['warning']}", file=sys.stderr)
455
+ # After the final print statements
456
+ if len(read_results) > 0:
457
+ # Convert read results to a DataFrame for easy statistical computation
458
+ results_df = pd.DataFrame(list(read_results.values()))
459
+
460
+ # Calculate summary statistics for the 'telomere_length' column
461
+ telomere_lengths = results_df['telomere_length']
462
+
463
+ min_length = telomere_lengths.min()
464
+ percentile_25 = telomere_lengths.quantile(0.25)
465
+ median_length = telomere_lengths.median()
466
+ mean_length = telomere_lengths.mean()
467
+ percentile_75 = telomere_lengths.quantile(0.75)
468
+ max_length = telomere_lengths.max()
469
+
470
+ # Print summary statistics
471
+ print("\nTelomere Length Summary Statistics:")
472
+ print(f"Min: {min_length:.2f}")
473
+ print(f"25th Percentile: {percentile_25:.2f}")
474
+ print(f"Median: {median_length:.2f}")
475
+ print(f"Mean: {mean_length:.2f}")
476
+ print(f"75th Percentile: {percentile_75:.2f}")
477
+ print(f"Max: {max_length:.2f}")
478
+ else:
479
+ print("No telomere measurements to summarize.")
480
+
481
+
482
+ def run_telometer():
483
+ parser = argparse.ArgumentParser(description=f'Telometer v{__version__}: telomere length with optional exact 6-bp variant-repeat tracts and enrichment.')
484
+ parser.add_argument('--version', action='version', version=f'Telometer v{__version__}')
485
+ parser.add_argument('-b', '--bam', help='The path to the sorted BAM file.', required=True)
486
+ parser.add_argument('-o', '--output', help='The path to the output file.', required=True)
487
+ parser.add_argument('-m', '--minreadlen', default=1000, type=int, help='Minimum read length to consider (Default: 1000 for telomere capture, use 4000 for WGS). Optional', required=False)
488
+ parser.add_argument('-g', '--maxgaplen', default=20, type=int, help='Maximum allowed gap length between telomere regions. Optional', required=False)
489
+ parser.add_argument('-t', '--threads', default=cpu_count(), type=int, help='Number of processing threads to use. Optional', required=False)
490
+ parser.add_argument('-l', '--memlimit', default=8, type=int, help="Maximum amount of memory to commit per batch of reads while processing. Optional, default = 8 Gb", required=False)
491
+ parser.add_argument('-e', '--enrichment', action='store_true', help='Append telomeres_per_gb and print sample totals using one recorded full length per read ID across the entire BAM. Optional; off by default.')
492
+ parser.add_argument('--variant-motif', default=None, help='Optional noncanonical 6-bp DNA motif; rotations and reverse complements are included.')
493
+ parser.add_argument('--min-variant-units', default=None, type=int, help='Minimum non-overlapping exact 6-bp units (effective default: 2; requires --variant-motif).')
494
+ args = parser.parse_args()
495
+ try:
496
+ validate_variant_options(args.variant_motif, args.min_variant_units, emit_warnings=False)
497
+ except ValueError as error:
498
+ parser.error(str(error))
499
+ process_bam_file(args.bam, args.output, max_gap_length=args.maxgaplen, min_read_len=args.minreadlen, num_processes=args.threads, memory_limit_gb=args.memlimit,
500
+ variant_motif=args.variant_motif, min_variant_units=args.min_variant_units, enrichment=args.enrichment)
501
+
502
+ if __name__ == "__main__":
503
+ run_telometer()