nanotimesort 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nanotimesort-1.0.0/LICENSE +21 -0
- nanotimesort-1.0.0/PKG-INFO +94 -0
- nanotimesort-1.0.0/README.md +47 -0
- nanotimesort-1.0.0/nanotimesort/__init__.py +4 -0
- nanotimesort-1.0.0/nanotimesort/binner.py +315 -0
- nanotimesort-1.0.0/nanotimesort/cli.py +78 -0
- nanotimesort-1.0.0/nanotimesort/timestamps.py +69 -0
- nanotimesort-1.0.0/nanotimesort.egg-info/PKG-INFO +94 -0
- nanotimesort-1.0.0/nanotimesort.egg-info/SOURCES.txt +16 -0
- nanotimesort-1.0.0/nanotimesort.egg-info/dependency_links.txt +1 -0
- nanotimesort-1.0.0/nanotimesort.egg-info/entry_points.txt +2 -0
- nanotimesort-1.0.0/nanotimesort.egg-info/requires.txt +5 -0
- nanotimesort-1.0.0/nanotimesort.egg-info/top_level.txt +1 -0
- nanotimesort-1.0.0/pyproject.toml +47 -0
- nanotimesort-1.0.0/setup.cfg +4 -0
- nanotimesort-1.0.0/tests/test_binner.py +125 -0
- nanotimesort-1.0.0/tests/test_cli.py +55 -0
- nanotimesort-1.0.0/tests/test_timestamps.py +48 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2019 Marc-Olivier Duceppe
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: nanotimesort
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Bin Oxford Nanopore reads by cumulative sequencing time intervals (Guppy and Dorado compatible)
|
|
5
|
+
Author-email: Marc-Olivier Duceppe <duceppemo@gmail.com>
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2019 Marc-Olivier Duceppe
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Project-URL: Homepage, https://github.com/duceppemo/nanoTimeSort
|
|
29
|
+
Project-URL: Issues, https://github.com/duceppemo/nanoTimeSort/issues
|
|
30
|
+
Project-URL: Wiki, https://github.com/duceppemo/nanoTimeSort/wiki
|
|
31
|
+
Keywords: nanopore,fastq,bioinformatics,minion,dorado,guppy,binning
|
|
32
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
33
|
+
Classifier: Environment :: Console
|
|
34
|
+
Classifier: Intended Audience :: Science/Research
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Operating System :: OS Independent
|
|
37
|
+
Classifier: Programming Language :: Python :: 3
|
|
38
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
39
|
+
Requires-Python: >=3.8
|
|
40
|
+
Description-Content-Type: text/markdown
|
|
41
|
+
License-File: LICENSE
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: pytest; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
45
|
+
Requires-Dist: ruff; extra == "dev"
|
|
46
|
+
Dynamic: license-file
|
|
47
|
+
|
|
48
|
+
<p align="center">
|
|
49
|
+
<img src="docs/images/logo.svg" alt="nanoTimeSort" width="520">
|
|
50
|
+
</p>
|
|
51
|
+
|
|
52
|
+
<p align="center">
|
|
53
|
+
<a href="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
54
|
+
<a href="https://codecov.io/gh/duceppemo/nanoTimeSort"><img src="https://codecov.io/gh/duceppemo/nanoTimeSort/branch/master/graph/badge.svg" alt="codecov"></a>
|
|
55
|
+
<a href="LICENSE"><img src="https://img.shields.io/badge/License-MIT-yellow.svg" alt="License: MIT"></a>
|
|
56
|
+
<a href="https://www.python.org/downloads/"><img src="https://img.shields.io/badge/python-3.8%2B-blue.svg" alt="Python 3.8+"></a>
|
|
57
|
+
<a href="pyproject.toml"><img src="https://img.shields.io/badge/dependencies-none-brightgreen.svg" alt="No dependencies"></a>
|
|
58
|
+
</p>
|
|
59
|
+
|
|
60
|
+
Bin Oxford Nanopore reads by **cumulative sequencing time intervals**, using the read start time
|
|
61
|
+
that MinKNOW, Guppy and Dorado embed in every FASTQ header. Useful to answer *"how long did I
|
|
62
|
+
actually need to sequence?"* — for time-to-detection studies, assembly saturation curves, or
|
|
63
|
+
benchmarking real-time pipelines.
|
|
64
|
+
|
|
65
|
+
```text
|
|
66
|
+
fastq_pass/ ──► sample_0-1h_125437reads_1103093674bp.fastq.gz
|
|
67
|
+
sample_0-2h_225891reads_2087456221bp.fastq.gz (bins are cumulative)
|
|
68
|
+
sample_0-3h_301255reads_2812345678bp.fastq.gz
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Compatible with **Guppy/MinKNOW** (`start_time=`) and **Dorado** (`st:Z:`) headers. Fast
|
|
72
|
+
(each read compressed once, ~100× faster than v0.1), low-memory (streaming), parallel, and
|
|
73
|
+
pure standard library.
|
|
74
|
+
|
|
75
|
+
## Install
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pip install git+https://github.com/duceppemo/nanoTimeSort.git
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Requires Python ≥ 3.8. No other dependencies.
|
|
82
|
+
|
|
83
|
+
## Usage
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
nanotimesort -f /path/to/fastq_pass/ -o /path/to/output/ -i 1h -p my_sample
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
📖 **Full documentation** — CLI reference, tutorial with example data, header compatibility,
|
|
90
|
+
design notes and FAQ — is in the [**wiki**](https://github.com/duceppemo/nanoTimeSort/wiki).
|
|
91
|
+
|
|
92
|
+
## License
|
|
93
|
+
|
|
94
|
+
[MIT](LICENSE) © Marc-Olivier Duceppe
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="docs/images/logo.svg" alt="nanoTimeSort" width="520">
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
<p align="center">
|
|
6
|
+
<a href="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
7
|
+
<a href="https://codecov.io/gh/duceppemo/nanoTimeSort"><img src="https://codecov.io/gh/duceppemo/nanoTimeSort/branch/master/graph/badge.svg" alt="codecov"></a>
|
|
8
|
+
<a href="LICENSE"><img src="https://img.shields.io/badge/License-MIT-yellow.svg" alt="License: MIT"></a>
|
|
9
|
+
<a href="https://www.python.org/downloads/"><img src="https://img.shields.io/badge/python-3.8%2B-blue.svg" alt="Python 3.8+"></a>
|
|
10
|
+
<a href="pyproject.toml"><img src="https://img.shields.io/badge/dependencies-none-brightgreen.svg" alt="No dependencies"></a>
|
|
11
|
+
</p>
|
|
12
|
+
|
|
13
|
+
Bin Oxford Nanopore reads by **cumulative sequencing time intervals**, using the read start time
|
|
14
|
+
that MinKNOW, Guppy and Dorado embed in every FASTQ header. Useful to answer *"how long did I
|
|
15
|
+
actually need to sequence?"* — for time-to-detection studies, assembly saturation curves, or
|
|
16
|
+
benchmarking real-time pipelines.
|
|
17
|
+
|
|
18
|
+
```text
|
|
19
|
+
fastq_pass/ ──► sample_0-1h_125437reads_1103093674bp.fastq.gz
|
|
20
|
+
sample_0-2h_225891reads_2087456221bp.fastq.gz (bins are cumulative)
|
|
21
|
+
sample_0-3h_301255reads_2812345678bp.fastq.gz
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Compatible with **Guppy/MinKNOW** (`start_time=`) and **Dorado** (`st:Z:`) headers. Fast
|
|
25
|
+
(each read compressed once, ~100× faster than v0.1), low-memory (streaming), parallel, and
|
|
26
|
+
pure standard library.
|
|
27
|
+
|
|
28
|
+
## Install
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install git+https://github.com/duceppemo/nanoTimeSort.git
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Requires Python ≥ 3.8. No other dependencies.
|
|
35
|
+
|
|
36
|
+
## Usage
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
nanotimesort -f /path/to/fastq_pass/ -o /path/to/output/ -i 1h -p my_sample
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
📖 **Full documentation** — CLI reference, tutorial with example data, header compatibility,
|
|
43
|
+
design notes and FAQ — is in the [**wiki**](https://github.com/duceppemo/nanoTimeSort/wiki).
|
|
44
|
+
|
|
45
|
+
## License
|
|
46
|
+
|
|
47
|
+
[MIT](LICENSE) © Marc-Olivier Duceppe
|
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
"""Core binning engine.
|
|
2
|
+
|
|
3
|
+
The pipeline runs in three stages, none of which holds reads in memory:
|
|
4
|
+
|
|
5
|
+
1. **Scan** (parallel): stream every FASTQ once to find the earliest and
|
|
6
|
+
latest read start time and the total read count.
|
|
7
|
+
2. **Chunk** (parallel): stream every FASTQ again; each read is gzip-
|
|
8
|
+
compressed exactly once, into the chunk file of the single interval it
|
|
9
|
+
belongs to.
|
|
10
|
+
3. **Assemble** (sequential): cumulative output *k* is built by raw byte
|
|
11
|
+
concatenation of output *k-1* and the chunks of interval *k*. Gzip members
|
|
12
|
+
concatenate into a valid gzip stream, so no data is ever recompressed.
|
|
13
|
+
|
|
14
|
+
Compared to the original implementation (which recompressed every read into
|
|
15
|
+
every cumulative bin at gzip level 9), this is typically one to two orders of
|
|
16
|
+
magnitude faster.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import gzip
|
|
22
|
+
import os
|
|
23
|
+
import shutil
|
|
24
|
+
import sys
|
|
25
|
+
import tempfile
|
|
26
|
+
from concurrent.futures import ProcessPoolExecutor
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
from datetime import datetime
|
|
29
|
+
from math import floor
|
|
30
|
+
from time import time
|
|
31
|
+
from typing import IO, Dict, Iterator, List, Optional, Tuple
|
|
32
|
+
|
|
33
|
+
from .timestamps import extract_start_time
|
|
34
|
+
|
|
35
|
+
UNIT_SECONDS = {"h": 3600.0, "m": 60.0, "s": 1.0}
|
|
36
|
+
UNIT_NAMES = {"h": "hour", "m": "minute", "s": "second"}
|
|
37
|
+
|
|
38
|
+
_COPY_BUFFER = 4 * 1024 * 1024
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class ScanResult:
|
|
43
|
+
"""Summary of one scan pass over a FASTQ file."""
|
|
44
|
+
|
|
45
|
+
reads: int = 0
|
|
46
|
+
missing: int = 0 # reads without a recognizable start time
|
|
47
|
+
t_min: Optional[datetime] = None
|
|
48
|
+
t_max: Optional[datetime] = None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def open_fastq(path: str) -> IO[bytes]:
|
|
52
|
+
"""Open a plain or gzipped FASTQ for binary reading."""
|
|
53
|
+
if path.endswith(".gz"):
|
|
54
|
+
return gzip.open(path, "rb")
|
|
55
|
+
return open(path, "rb", buffering=1024 * 1024)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def fastq_records(handle: IO[bytes]) -> Iterator[Tuple[bytes, bytes, bytes]]:
|
|
59
|
+
"""Yield (header, sequence, quality) byte lines, newline-stripped."""
|
|
60
|
+
while True:
|
|
61
|
+
header = handle.readline()
|
|
62
|
+
if not header:
|
|
63
|
+
return
|
|
64
|
+
seq = handle.readline()
|
|
65
|
+
handle.readline() # '+' separator line
|
|
66
|
+
qual = handle.readline()
|
|
67
|
+
if not qual:
|
|
68
|
+
return # truncated final record: skip it
|
|
69
|
+
yield header.rstrip(), seq.rstrip(), qual.rstrip()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def find_fastq_files(input_path: str) -> List[str]:
|
|
73
|
+
"""Collect FASTQ files from a folder (recursively) or a single file path."""
|
|
74
|
+
extensions = (".fastq", ".fastq.gz", ".fq", ".fq.gz")
|
|
75
|
+
if os.path.isfile(input_path):
|
|
76
|
+
return [input_path] if input_path.endswith(extensions) else []
|
|
77
|
+
fastq_list = []
|
|
78
|
+
for root, _dirs, filenames in os.walk(input_path):
|
|
79
|
+
for filename in filenames:
|
|
80
|
+
if filename.endswith(extensions):
|
|
81
|
+
fastq_list.append(os.path.join(root, filename))
|
|
82
|
+
return sorted(fastq_list)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def scan_file(path: str) -> ScanResult:
|
|
86
|
+
"""Pass 1: stream one file and record read count and time extremes."""
|
|
87
|
+
result = ScanResult()
|
|
88
|
+
if os.path.getsize(path) == 0:
|
|
89
|
+
return result
|
|
90
|
+
with open_fastq(path) as handle:
|
|
91
|
+
for header, _seq, _qual in fastq_records(handle):
|
|
92
|
+
start_time = extract_start_time(header)
|
|
93
|
+
if start_time is None:
|
|
94
|
+
result.missing += 1
|
|
95
|
+
continue
|
|
96
|
+
result.reads += 1
|
|
97
|
+
if result.t_min is None or start_time < result.t_min:
|
|
98
|
+
result.t_min = start_time
|
|
99
|
+
if result.t_max is None or start_time > result.t_max:
|
|
100
|
+
result.t_max = start_time
|
|
101
|
+
return result
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def chunk_file(
|
|
105
|
+
path: str,
|
|
106
|
+
file_index: int,
|
|
107
|
+
t_min: datetime,
|
|
108
|
+
bin_seconds: float,
|
|
109
|
+
num_bins: int,
|
|
110
|
+
chunk_dir: str,
|
|
111
|
+
compresslevel: int,
|
|
112
|
+
) -> Tuple[List[int], List[int]]:
|
|
113
|
+
"""Pass 2: write each read of one file into its interval chunk.
|
|
114
|
+
|
|
115
|
+
Chunk files are opened lazily, so intervals with no reads in this file
|
|
116
|
+
cost nothing. Returns per-interval read and base-pair counts.
|
|
117
|
+
"""
|
|
118
|
+
reads_per_bin = [0] * num_bins
|
|
119
|
+
bp_per_bin = [0] * num_bins
|
|
120
|
+
handles: Dict[int, IO[bytes]] = {}
|
|
121
|
+
|
|
122
|
+
if os.path.getsize(path) == 0:
|
|
123
|
+
return reads_per_bin, bp_per_bin
|
|
124
|
+
|
|
125
|
+
try:
|
|
126
|
+
with open_fastq(path) as handle:
|
|
127
|
+
for header, seq, qual in fastq_records(handle):
|
|
128
|
+
start_time = extract_start_time(header)
|
|
129
|
+
if start_time is None:
|
|
130
|
+
continue
|
|
131
|
+
elapsed = (start_time - t_min).total_seconds()
|
|
132
|
+
bin_index = min(floor(elapsed / bin_seconds), num_bins - 1)
|
|
133
|
+
out = handles.get(bin_index)
|
|
134
|
+
if out is None:
|
|
135
|
+
chunk_path = os.path.join(
|
|
136
|
+
chunk_dir, "chunk_b{:06d}_f{:06d}.fastq.gz".format(bin_index, file_index)
|
|
137
|
+
)
|
|
138
|
+
out = gzip.open(chunk_path, "wb", compresslevel=compresslevel)
|
|
139
|
+
handles[bin_index] = out
|
|
140
|
+
out.write(header + b"\n" + seq + b"\n+\n" + qual + b"\n")
|
|
141
|
+
reads_per_bin[bin_index] += 1
|
|
142
|
+
bp_per_bin[bin_index] += len(seq)
|
|
143
|
+
finally:
|
|
144
|
+
for out in handles.values():
|
|
145
|
+
out.close()
|
|
146
|
+
|
|
147
|
+
return reads_per_bin, bp_per_bin
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def format_number(value: float) -> str:
|
|
151
|
+
"""Render 2.0 as '2' and 0.5 as '0.5' for use in file names."""
|
|
152
|
+
return str(int(value)) if float(value).is_integer() else str(value)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class NanoTimeSort:
|
|
156
|
+
"""Bin Nanopore reads into cumulative sequencing-time interval files."""
|
|
157
|
+
|
|
158
|
+
def __init__(
|
|
159
|
+
self,
|
|
160
|
+
input_path: str,
|
|
161
|
+
output_folder: str,
|
|
162
|
+
interval: str,
|
|
163
|
+
prefix: str = "interval",
|
|
164
|
+
threads: int = 1,
|
|
165
|
+
compresslevel: int = 4,
|
|
166
|
+
):
|
|
167
|
+
self.input_path = input_path
|
|
168
|
+
self.output_folder = output_folder
|
|
169
|
+
self.prefix = prefix
|
|
170
|
+
self.threads = max(1, threads)
|
|
171
|
+
self.compresslevel = compresslevel
|
|
172
|
+
self.bin_size, self.units = self._parse_interval(interval)
|
|
173
|
+
self.bin_seconds = self.bin_size * UNIT_SECONDS[self.units]
|
|
174
|
+
|
|
175
|
+
@staticmethod
|
|
176
|
+
def _parse_interval(interval: str) -> Tuple[float, str]:
|
|
177
|
+
units = interval[-1].lower()
|
|
178
|
+
if units not in UNIT_SECONDS:
|
|
179
|
+
raise ValueError(
|
|
180
|
+
"Invalid interval unit '{}'. Use one of: {}".format(
|
|
181
|
+
units, ", ".join(sorted(UNIT_SECONDS))
|
|
182
|
+
)
|
|
183
|
+
)
|
|
184
|
+
try:
|
|
185
|
+
bin_size = float(interval[:-1])
|
|
186
|
+
except ValueError:
|
|
187
|
+
raise ValueError("Invalid interval value: '{}'".format(interval)) from None
|
|
188
|
+
if bin_size <= 0:
|
|
189
|
+
raise ValueError("Interval must be greater than zero.")
|
|
190
|
+
return bin_size, units
|
|
191
|
+
|
|
192
|
+
def run(self) -> List[str]:
|
|
193
|
+
"""Execute the full pipeline. Returns the list of output file paths."""
|
|
194
|
+
overall_start = time()
|
|
195
|
+
os.makedirs(self.output_folder, exist_ok=True)
|
|
196
|
+
|
|
197
|
+
fastq_list = find_fastq_files(self.input_path)
|
|
198
|
+
if not fastq_list:
|
|
199
|
+
raise FileNotFoundError("No FASTQ files found in {}".format(self.input_path))
|
|
200
|
+
workers = min(self.threads, len(fastq_list))
|
|
201
|
+
|
|
202
|
+
# Pass 1: find the time range of the run.
|
|
203
|
+
print("Scanning {} FASTQ file(s)...".format(len(fastq_list)), end="", flush=True)
|
|
204
|
+
start = time()
|
|
205
|
+
scans = self._map(scan_file, fastq_list, workers)
|
|
206
|
+
total_reads = sum(s.reads for s in scans)
|
|
207
|
+
total_missing = sum(s.missing for s in scans)
|
|
208
|
+
t_mins = [s.t_min for s in scans if s.t_min is not None]
|
|
209
|
+
t_maxs = [s.t_max for s in scans if s.t_max is not None]
|
|
210
|
+
print(" {} reads in {}".format(total_reads, elapsed_time(time() - start)))
|
|
211
|
+
if total_missing:
|
|
212
|
+
print(
|
|
213
|
+
"Warning: {} read(s) without a 'start_time=' or 'st:Z:' header field "
|
|
214
|
+
"were skipped.".format(total_missing),
|
|
215
|
+
file=sys.stderr,
|
|
216
|
+
)
|
|
217
|
+
if not t_mins:
|
|
218
|
+
raise ValueError(
|
|
219
|
+
"No read start times found. Headers must contain a Guppy/MinKNOW "
|
|
220
|
+
"'start_time=' field or a Dorado 'st:Z:' tag."
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
t_min, t_max = min(t_mins), max(t_maxs)
|
|
224
|
+
run_seconds = (t_max - t_min).total_seconds()
|
|
225
|
+
num_bins = int(run_seconds // self.bin_seconds) + 1
|
|
226
|
+
print(
|
|
227
|
+
"Run spans {} -> {} intervals of {}{}".format(
|
|
228
|
+
elapsed_time(run_seconds), num_bins, format_number(self.bin_size), self.units
|
|
229
|
+
)
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
# Pass 2: compress each read once into its interval chunk.
|
|
233
|
+
print("Binning reads...", end="", flush=True)
|
|
234
|
+
start = time()
|
|
235
|
+
with tempfile.TemporaryDirectory(prefix="nanotimesort_", dir=self.output_folder) as chunk_dir:
|
|
236
|
+
jobs = [
|
|
237
|
+
(path, i, t_min, self.bin_seconds, num_bins, chunk_dir, self.compresslevel)
|
|
238
|
+
for i, path in enumerate(fastq_list)
|
|
239
|
+
]
|
|
240
|
+
results = self._starmap(chunk_file, jobs, workers)
|
|
241
|
+
reads_per_bin = [0] * num_bins
|
|
242
|
+
bp_per_bin = [0] * num_bins
|
|
243
|
+
for file_reads, file_bp in results:
|
|
244
|
+
for b in range(num_bins):
|
|
245
|
+
reads_per_bin[b] += file_reads[b]
|
|
246
|
+
bp_per_bin[b] += file_bp[b]
|
|
247
|
+
print(" done in {}".format(elapsed_time(time() - start)))
|
|
248
|
+
|
|
249
|
+
# Pass 3: assemble cumulative outputs by gzip member concatenation.
|
|
250
|
+
print("Writing cumulative interval files...", end="", flush=True)
|
|
251
|
+
start = time()
|
|
252
|
+
outputs = self._assemble(chunk_dir, num_bins, reads_per_bin, bp_per_bin)
|
|
253
|
+
print(" done in {}".format(elapsed_time(time() - start)))
|
|
254
|
+
|
|
255
|
+
print("Total run time: {}".format(elapsed_time(time() - overall_start)))
|
|
256
|
+
return outputs
|
|
257
|
+
|
|
258
|
+
def _assemble(
|
|
259
|
+
self,
|
|
260
|
+
chunk_dir: str,
|
|
261
|
+
num_bins: int,
|
|
262
|
+
reads_per_bin: List[int],
|
|
263
|
+
bp_per_bin: List[int],
|
|
264
|
+
) -> List[str]:
|
|
265
|
+
chunk_names = sorted(os.listdir(chunk_dir))
|
|
266
|
+
outputs: List[str] = []
|
|
267
|
+
previous_path: Optional[str] = None
|
|
268
|
+
cumulative_reads = 0
|
|
269
|
+
cumulative_bp = 0
|
|
270
|
+
|
|
271
|
+
for b in range(num_bins):
|
|
272
|
+
cumulative_reads += reads_per_bin[b]
|
|
273
|
+
cumulative_bp += bp_per_bin[b]
|
|
274
|
+
label = format_number((b + 1) * self.bin_size)
|
|
275
|
+
out_name = "{}_0-{}{}_{}reads_{}bp.fastq.gz".format(
|
|
276
|
+
self.prefix, label, self.units, cumulative_reads, cumulative_bp
|
|
277
|
+
)
|
|
278
|
+
out_path = os.path.join(self.output_folder, out_name)
|
|
279
|
+
prefix = "chunk_b{:06d}_".format(b)
|
|
280
|
+
with open(out_path, "wb") as out:
|
|
281
|
+
if previous_path is not None:
|
|
282
|
+
with open(previous_path, "rb") as prev:
|
|
283
|
+
shutil.copyfileobj(prev, out, _COPY_BUFFER)
|
|
284
|
+
for name in chunk_names:
|
|
285
|
+
if name.startswith(prefix):
|
|
286
|
+
with open(os.path.join(chunk_dir, name), "rb") as chunk:
|
|
287
|
+
shutil.copyfileobj(chunk, out, _COPY_BUFFER)
|
|
288
|
+
outputs.append(out_path)
|
|
289
|
+
previous_path = out_path
|
|
290
|
+
|
|
291
|
+
return outputs
|
|
292
|
+
|
|
293
|
+
def _map(self, func, items, workers):
|
|
294
|
+
if workers <= 1 or len(items) <= 1:
|
|
295
|
+
return [func(item) for item in items]
|
|
296
|
+
with ProcessPoolExecutor(max_workers=workers) as pool:
|
|
297
|
+
return list(pool.map(func, items))
|
|
298
|
+
|
|
299
|
+
def _starmap(self, func, jobs, workers):
|
|
300
|
+
if workers <= 1 or len(jobs) <= 1:
|
|
301
|
+
return [func(*job) for job in jobs]
|
|
302
|
+
with ProcessPoolExecutor(max_workers=workers) as pool:
|
|
303
|
+
futures = [pool.submit(func, *job) for job in jobs]
|
|
304
|
+
return [f.result() for f in futures]
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def elapsed_time(seconds: float) -> str:
|
|
308
|
+
"""Format a duration in seconds as a compact '1d2h3m4s' style string."""
|
|
309
|
+
if seconds < 1:
|
|
310
|
+
return "{:.2f}s".format(seconds)
|
|
311
|
+
minutes, secs = divmod(int(round(seconds)), 60)
|
|
312
|
+
hours, minutes = divmod(minutes, 60)
|
|
313
|
+
days, hours = divmod(hours, 24)
|
|
314
|
+
parts = [("d", days), ("h", hours), ("m", minutes), ("s", secs)]
|
|
315
|
+
return "".join("{}{}".format(value, name) for name, value in parts if value) or "0s"
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Command-line interface for nanoTimeSort."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from argparse import ArgumentParser, RawTextHelpFormatter
|
|
7
|
+
from multiprocessing import cpu_count
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .binner import NanoTimeSort
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def build_parser() -> ArgumentParser:
|
|
14
|
+
cpu = cpu_count()
|
|
15
|
+
parser = ArgumentParser(
|
|
16
|
+
prog="nanotimesort",
|
|
17
|
+
description="Bin Oxford Nanopore reads by cumulative sequencing time intervals.\n"
|
|
18
|
+
"Supports Guppy/MinKNOW ('start_time=') and Dorado ('st:Z:') FASTQ headers.",
|
|
19
|
+
formatter_class=RawTextHelpFormatter,
|
|
20
|
+
)
|
|
21
|
+
parser.add_argument(
|
|
22
|
+
"-f", "--fastq", metavar="/basecalled/folder/", required=True,
|
|
23
|
+
help="Input folder (searched recursively) or single FASTQ file.\n"
|
|
24
|
+
"Accepts .fastq, .fq, .fastq.gz and .fq.gz.",
|
|
25
|
+
)
|
|
26
|
+
parser.add_argument(
|
|
27
|
+
"-o", "--output", metavar="/output/folder/", required=True,
|
|
28
|
+
help="Output folder. Created if it does not exist.",
|
|
29
|
+
)
|
|
30
|
+
parser.add_argument(
|
|
31
|
+
"-i", "--interval", metavar="1h", required=True,
|
|
32
|
+
help="Time interval for the bins, e.g. '1h', '30m' or '90s'.\n"
|
|
33
|
+
"Bins are cumulative: with '-i 1h', the second file also\n"
|
|
34
|
+
"contains the reads of the first hour.",
|
|
35
|
+
)
|
|
36
|
+
parser.add_argument(
|
|
37
|
+
"-p", "--prefix", metavar="my_sample", default="interval",
|
|
38
|
+
help="Output file prefix. Files are named like\n"
|
|
39
|
+
"'my_sample_0-1h_123reads_456789bp.fastq.gz'.\n"
|
|
40
|
+
"Default: interval",
|
|
41
|
+
)
|
|
42
|
+
parser.add_argument(
|
|
43
|
+
"-t", "--threads", metavar=str(cpu), type=int, default=cpu,
|
|
44
|
+
help="Number of FASTQ files to process in parallel.\n"
|
|
45
|
+
"Default: {}".format(cpu),
|
|
46
|
+
)
|
|
47
|
+
parser.add_argument(
|
|
48
|
+
"-c", "--compression-level", metavar="4", type=int, default=4,
|
|
49
|
+
choices=range(1, 10),
|
|
50
|
+
help="Gzip compression level for output files (1=fastest, 9=smallest).\n"
|
|
51
|
+
"Default: 4",
|
|
52
|
+
)
|
|
53
|
+
parser.add_argument(
|
|
54
|
+
"-v", "--version", action="version", version="nanoTimeSort v{}".format(__version__),
|
|
55
|
+
)
|
|
56
|
+
return parser
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def main(argv=None) -> int:
|
|
60
|
+
args = build_parser().parse_args(argv)
|
|
61
|
+
try:
|
|
62
|
+
binner = NanoTimeSort(
|
|
63
|
+
input_path=args.fastq,
|
|
64
|
+
output_folder=args.output,
|
|
65
|
+
interval=args.interval,
|
|
66
|
+
prefix=args.prefix,
|
|
67
|
+
threads=args.threads,
|
|
68
|
+
compresslevel=args.compression_level,
|
|
69
|
+
)
|
|
70
|
+
binner.run()
|
|
71
|
+
except (ValueError, FileNotFoundError) as err:
|
|
72
|
+
print("Error: {}".format(err), file=sys.stderr)
|
|
73
|
+
return 1
|
|
74
|
+
return 0
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
if __name__ == "__main__":
|
|
78
|
+
sys.exit(main())
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Extraction and parsing of read start times from Nanopore FASTQ headers.
|
|
2
|
+
|
|
3
|
+
Two header dialects are supported:
|
|
4
|
+
|
|
5
|
+
* Guppy / MinKNOW (key=value fields)::
|
|
6
|
+
|
|
7
|
+
@<read_id> runid=... read=... ch=... start_time=2019-07-16T19:51:22Z
|
|
8
|
+
@<read_id> ... start_time=2025-01-13T10:45:28.681306+00:00 ...
|
|
9
|
+
|
|
10
|
+
* Dorado (SAM-style tags)::
|
|
11
|
+
|
|
12
|
+
@<read_id> qs:f:21.3 du:f:12.44 ch:i:942 st:Z:2023-09-01T11:13:45.731+00:00 ...
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
from typing import Optional, Union
|
|
20
|
+
|
|
21
|
+
# Guppy/MinKNOW style field.
|
|
22
|
+
_GUPPY_PREFIX = b"start_time="
|
|
23
|
+
# Dorado SAM-tag style field (st:Z:<ISO 8601>).
|
|
24
|
+
_DORADO_PREFIX = b"st:Z:"
|
|
25
|
+
|
|
26
|
+
_FRACTION_RE = re.compile(r"\.(\d+)")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def extract_start_time(header: bytes) -> Optional[datetime]:
|
|
30
|
+
"""Return the read start time from a FASTQ header line, or None if absent.
|
|
31
|
+
|
|
32
|
+
:param header: raw FASTQ header line (bytes, with or without trailing newline)
|
|
33
|
+
:return: timezone-aware datetime (UTC assumed when the timestamp is naive)
|
|
34
|
+
"""
|
|
35
|
+
for item in header.split():
|
|
36
|
+
if item.startswith(_GUPPY_PREFIX):
|
|
37
|
+
return parse_timestamp(item[len(_GUPPY_PREFIX):])
|
|
38
|
+
if item.startswith(_DORADO_PREFIX):
|
|
39
|
+
return parse_timestamp(item[len(_DORADO_PREFIX):])
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def parse_timestamp(raw: Union[bytes, str]) -> datetime:
|
|
44
|
+
"""Parse an ISO 8601 / RFC 3339 timestamp into a timezone-aware datetime.
|
|
45
|
+
|
|
46
|
+
Handles 'Z' suffixes and unusual fractional-second precision on Python
|
|
47
|
+
versions where datetime.fromisoformat() is strict (< 3.11).
|
|
48
|
+
"""
|
|
49
|
+
if isinstance(raw, bytes):
|
|
50
|
+
raw = raw.decode("ascii")
|
|
51
|
+
try:
|
|
52
|
+
dt = datetime.fromisoformat(raw)
|
|
53
|
+
except ValueError:
|
|
54
|
+
dt = datetime.fromisoformat(_normalize(raw))
|
|
55
|
+
if dt.tzinfo is None:
|
|
56
|
+
dt = dt.replace(tzinfo=timezone.utc)
|
|
57
|
+
return dt
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _normalize(s: str) -> str:
|
|
61
|
+
"""Rewrite a timestamp so strict fromisoformat() implementations accept it."""
|
|
62
|
+
if s.endswith(("Z", "z")):
|
|
63
|
+
s = s[:-1] + "+00:00"
|
|
64
|
+
|
|
65
|
+
def _pad(match: "re.Match[str]") -> str:
|
|
66
|
+
digits = match.group(1)[:6]
|
|
67
|
+
return "." + digits.ljust(6, "0")
|
|
68
|
+
|
|
69
|
+
return _FRACTION_RE.sub(_pad, s, count=1)
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: nanotimesort
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Bin Oxford Nanopore reads by cumulative sequencing time intervals (Guppy and Dorado compatible)
|
|
5
|
+
Author-email: Marc-Olivier Duceppe <duceppemo@gmail.com>
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2019 Marc-Olivier Duceppe
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Project-URL: Homepage, https://github.com/duceppemo/nanoTimeSort
|
|
29
|
+
Project-URL: Issues, https://github.com/duceppemo/nanoTimeSort/issues
|
|
30
|
+
Project-URL: Wiki, https://github.com/duceppemo/nanoTimeSort/wiki
|
|
31
|
+
Keywords: nanopore,fastq,bioinformatics,minion,dorado,guppy,binning
|
|
32
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
33
|
+
Classifier: Environment :: Console
|
|
34
|
+
Classifier: Intended Audience :: Science/Research
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Operating System :: OS Independent
|
|
37
|
+
Classifier: Programming Language :: Python :: 3
|
|
38
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
39
|
+
Requires-Python: >=3.8
|
|
40
|
+
Description-Content-Type: text/markdown
|
|
41
|
+
License-File: LICENSE
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: pytest; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
45
|
+
Requires-Dist: ruff; extra == "dev"
|
|
46
|
+
Dynamic: license-file
|
|
47
|
+
|
|
48
|
+
<p align="center">
|
|
49
|
+
<img src="docs/images/logo.svg" alt="nanoTimeSort" width="520">
|
|
50
|
+
</p>
|
|
51
|
+
|
|
52
|
+
<p align="center">
|
|
53
|
+
<a href="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
54
|
+
<a href="https://codecov.io/gh/duceppemo/nanoTimeSort"><img src="https://codecov.io/gh/duceppemo/nanoTimeSort/branch/master/graph/badge.svg" alt="codecov"></a>
|
|
55
|
+
<a href="LICENSE"><img src="https://img.shields.io/badge/License-MIT-yellow.svg" alt="License: MIT"></a>
|
|
56
|
+
<a href="https://www.python.org/downloads/"><img src="https://img.shields.io/badge/python-3.8%2B-blue.svg" alt="Python 3.8+"></a>
|
|
57
|
+
<a href="pyproject.toml"><img src="https://img.shields.io/badge/dependencies-none-brightgreen.svg" alt="No dependencies"></a>
|
|
58
|
+
</p>
|
|
59
|
+
|
|
60
|
+
Bin Oxford Nanopore reads by **cumulative sequencing time intervals**, using the read start time
|
|
61
|
+
that MinKNOW, Guppy and Dorado embed in every FASTQ header. Useful to answer *"how long did I
|
|
62
|
+
actually need to sequence?"* — for time-to-detection studies, assembly saturation curves, or
|
|
63
|
+
benchmarking real-time pipelines.
|
|
64
|
+
|
|
65
|
+
```text
|
|
66
|
+
fastq_pass/ ──► sample_0-1h_125437reads_1103093674bp.fastq.gz
|
|
67
|
+
sample_0-2h_225891reads_2087456221bp.fastq.gz (bins are cumulative)
|
|
68
|
+
sample_0-3h_301255reads_2812345678bp.fastq.gz
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Compatible with **Guppy/MinKNOW** (`start_time=`) and **Dorado** (`st:Z:`) headers. Fast
|
|
72
|
+
(each read compressed once, ~100× faster than v0.1), low-memory (streaming), parallel, and
|
|
73
|
+
pure standard library.
|
|
74
|
+
|
|
75
|
+
## Install
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pip install git+https://github.com/duceppemo/nanoTimeSort.git
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Requires Python ≥ 3.8. No other dependencies.
|
|
82
|
+
|
|
83
|
+
## Usage
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
nanotimesort -f /path/to/fastq_pass/ -o /path/to/output/ -i 1h -p my_sample
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
📖 **Full documentation** — CLI reference, tutorial with example data, header compatibility,
|
|
90
|
+
design notes and FAQ — is in the [**wiki**](https://github.com/duceppemo/nanoTimeSort/wiki).
|
|
91
|
+
|
|
92
|
+
## License
|
|
93
|
+
|
|
94
|
+
[MIT](LICENSE) © Marc-Olivier Duceppe
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
nanotimesort/__init__.py
|
|
5
|
+
nanotimesort/binner.py
|
|
6
|
+
nanotimesort/cli.py
|
|
7
|
+
nanotimesort/timestamps.py
|
|
8
|
+
nanotimesort.egg-info/PKG-INFO
|
|
9
|
+
nanotimesort.egg-info/SOURCES.txt
|
|
10
|
+
nanotimesort.egg-info/dependency_links.txt
|
|
11
|
+
nanotimesort.egg-info/entry_points.txt
|
|
12
|
+
nanotimesort.egg-info/requires.txt
|
|
13
|
+
nanotimesort.egg-info/top_level.txt
|
|
14
|
+
tests/test_binner.py
|
|
15
|
+
tests/test_cli.py
|
|
16
|
+
tests/test_timestamps.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
nanotimesort
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=64"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "nanotimesort"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "Bin Oxford Nanopore reads by cumulative sequencing time intervals (Guppy and Dorado compatible)"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { file = "LICENSE" }
|
|
11
|
+
authors = [{ name = "Marc-Olivier Duceppe", email = "duceppemo@gmail.com" }]
|
|
12
|
+
requires-python = ">=3.8"
|
|
13
|
+
keywords = ["nanopore", "fastq", "bioinformatics", "minion", "dorado", "guppy", "binning"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 5 - Production/Stable",
|
|
16
|
+
"Environment :: Console",
|
|
17
|
+
"Intended Audience :: Science/Research",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
22
|
+
]
|
|
23
|
+
dependencies = []
|
|
24
|
+
|
|
25
|
+
[project.optional-dependencies]
|
|
26
|
+
dev = ["pytest", "pytest-cov", "ruff"]
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Homepage = "https://github.com/duceppemo/nanoTimeSort"
|
|
30
|
+
Issues = "https://github.com/duceppemo/nanoTimeSort/issues"
|
|
31
|
+
Wiki = "https://github.com/duceppemo/nanoTimeSort/wiki"
|
|
32
|
+
|
|
33
|
+
[project.scripts]
|
|
34
|
+
nanotimesort = "nanotimesort.cli:main"
|
|
35
|
+
|
|
36
|
+
[tool.setuptools]
|
|
37
|
+
packages = ["nanotimesort"]
|
|
38
|
+
|
|
39
|
+
[tool.ruff]
|
|
40
|
+
line-length = 110
|
|
41
|
+
target-version = "py38"
|
|
42
|
+
|
|
43
|
+
[tool.ruff.lint]
|
|
44
|
+
select = ["E4", "E7", "E9", "F", "B"]
|
|
45
|
+
|
|
46
|
+
[tool.pytest.ini_options]
|
|
47
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
import gzip
|
|
2
|
+
import os
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from nanotimesort.binner import NanoTimeSort, find_fastq_files
|
|
7
|
+
|
|
8
|
+
BASE = "2024-05-01T{:02d}:{:02d}:00Z"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def make_read(read_id, minutes, style="guppy", seq="ACGT" * 10):
|
|
12
|
+
"""Build one FASTQ record with a start time `minutes` after 10:00 UTC."""
|
|
13
|
+
hour, minute = divmod(600 + minutes, 60)
|
|
14
|
+
stamp = BASE.format(hour, minute)
|
|
15
|
+
if style == "guppy":
|
|
16
|
+
header = "@{} runid=abc ch=1 start_time={}".format(read_id, stamp)
|
|
17
|
+
else: # dorado
|
|
18
|
+
stamp = stamp.replace("Z", ".000+00:00")
|
|
19
|
+
header = "@{} qs:f:20.0 ch:i:1 st:Z:{}".format(read_id, stamp)
|
|
20
|
+
return "{}\n{}\n+\n{}\n".format(header, seq, "I" * len(seq))
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@pytest.fixture
|
|
24
|
+
def run_folder(tmp_path):
|
|
25
|
+
"""Two FASTQ files (one gzipped, one plain) spanning ~2.5 hours."""
|
|
26
|
+
fastq_dir = tmp_path / "fastq"
|
|
27
|
+
fastq_dir.mkdir()
|
|
28
|
+
# File 1 (gzipped, Guppy headers): reads at 0, 30 and 70 minutes.
|
|
29
|
+
content1 = (
|
|
30
|
+
make_read("read1", 0)
|
|
31
|
+
+ make_read("read2", 30)
|
|
32
|
+
+ make_read("read3", 70)
|
|
33
|
+
)
|
|
34
|
+
with gzip.open(fastq_dir / "part1.fastq.gz", "wt") as handle:
|
|
35
|
+
handle.write(content1)
|
|
36
|
+
# File 2 (plain, Dorado headers): reads at 90 and 150 minutes.
|
|
37
|
+
content2 = make_read("read4", 90, style="dorado") + make_read("read5", 150, style="dorado")
|
|
38
|
+
(fastq_dir / "part2.fastq").write_text(content2)
|
|
39
|
+
return fastq_dir
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def read_ids(path):
|
|
43
|
+
with gzip.open(path, "rt") as handle:
|
|
44
|
+
return [line.split()[0][1:] for i, line in enumerate(handle) if i % 4 == 0]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def test_find_fastq_files(run_folder):
|
|
48
|
+
files = find_fastq_files(str(run_folder))
|
|
49
|
+
assert [os.path.basename(f) for f in files] == ["part1.fastq.gz", "part2.fastq"]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_cumulative_binning(run_folder, tmp_path):
|
|
53
|
+
out_dir = tmp_path / "out"
|
|
54
|
+
binner = NanoTimeSort(
|
|
55
|
+
input_path=str(run_folder),
|
|
56
|
+
output_folder=str(out_dir),
|
|
57
|
+
interval="1h",
|
|
58
|
+
prefix="test",
|
|
59
|
+
threads=1,
|
|
60
|
+
)
|
|
61
|
+
outputs = binner.run()
|
|
62
|
+
|
|
63
|
+
# Run spans 150 minutes -> three cumulative 1 h bins.
|
|
64
|
+
names = [os.path.basename(p) for p in outputs]
|
|
65
|
+
bp = 40 # each read is 40 bp
|
|
66
|
+
assert names == [
|
|
67
|
+
"test_0-1h_2reads_{}bp.fastq.gz".format(2 * bp),
|
|
68
|
+
"test_0-2h_4reads_{}bp.fastq.gz".format(4 * bp),
|
|
69
|
+
"test_0-3h_5reads_{}bp.fastq.gz".format(5 * bp),
|
|
70
|
+
]
|
|
71
|
+
|
|
72
|
+
# Bin contents are cumulative and mix Guppy + Dorado reads.
|
|
73
|
+
assert sorted(read_ids(outputs[0])) == ["read1", "read2"]
|
|
74
|
+
assert sorted(read_ids(outputs[1])) == ["read1", "read2", "read3", "read4"]
|
|
75
|
+
assert sorted(read_ids(outputs[2])) == ["read1", "read2", "read3", "read4", "read5"]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_parallel_matches_serial(run_folder, tmp_path):
|
|
79
|
+
results = {}
|
|
80
|
+
for threads in (1, 2):
|
|
81
|
+
out_dir = tmp_path / "out_t{}".format(threads)
|
|
82
|
+
binner = NanoTimeSort(
|
|
83
|
+
input_path=str(run_folder),
|
|
84
|
+
output_folder=str(out_dir),
|
|
85
|
+
interval="30m",
|
|
86
|
+
prefix="par",
|
|
87
|
+
threads=threads,
|
|
88
|
+
)
|
|
89
|
+
outputs = binner.run()
|
|
90
|
+
results[threads] = {
|
|
91
|
+
os.path.basename(p): sorted(read_ids(p)) for p in outputs
|
|
92
|
+
}
|
|
93
|
+
assert results[1] == results[2]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def test_single_file_input(run_folder, tmp_path):
|
|
97
|
+
out_dir = tmp_path / "out_single"
|
|
98
|
+
binner = NanoTimeSort(
|
|
99
|
+
input_path=str(run_folder / "part2.fastq"),
|
|
100
|
+
output_folder=str(out_dir),
|
|
101
|
+
interval="2h",
|
|
102
|
+
prefix="single",
|
|
103
|
+
threads=1,
|
|
104
|
+
)
|
|
105
|
+
outputs = binner.run()
|
|
106
|
+
assert len(outputs) == 1
|
|
107
|
+
assert sorted(read_ids(outputs[0])) == ["read4", "read5"]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def test_invalid_interval():
|
|
111
|
+
with pytest.raises(ValueError):
|
|
112
|
+
NanoTimeSort("in", "out", interval="1x")
|
|
113
|
+
with pytest.raises(ValueError):
|
|
114
|
+
NanoTimeSort("in", "out", interval="-5m")
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def test_reads_without_timestamp_are_skipped(tmp_path, capsys):
|
|
118
|
+
fastq_dir = tmp_path / "fastq"
|
|
119
|
+
fastq_dir.mkdir()
|
|
120
|
+
content = make_read("good1", 0) + "@orphan length=4\nACGT\n+\nIIII\n" + make_read("good2", 10)
|
|
121
|
+
(fastq_dir / "mixed.fastq").write_text(content)
|
|
122
|
+
out_dir = tmp_path / "out"
|
|
123
|
+
binner = NanoTimeSort(str(fastq_dir), str(out_dir), interval="1h", threads=1)
|
|
124
|
+
outputs = binner.run()
|
|
125
|
+
assert sorted(read_ids(outputs[-1])) == ["good1", "good2"]
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import gzip
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from nanotimesort import __version__
|
|
6
|
+
from nanotimesort.cli import main
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@pytest.fixture
|
|
10
|
+
def fastq_folder(tmp_path):
|
|
11
|
+
folder = tmp_path / "fastq"
|
|
12
|
+
folder.mkdir()
|
|
13
|
+
reads = "".join(
|
|
14
|
+
"@read{} runid=abc start_time=2024-05-01T1{}:00:00Z\nACGT\n+\nIIII\n".format(i, i)
|
|
15
|
+
for i in range(3)
|
|
16
|
+
)
|
|
17
|
+
(folder / "run.fastq").write_text(reads)
|
|
18
|
+
return folder
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_main_success(fastq_folder, tmp_path):
|
|
22
|
+
out_dir = tmp_path / "out"
|
|
23
|
+
rc = main(["-f", str(fastq_folder), "-o", str(out_dir), "-i", "1h", "-p", "cli"])
|
|
24
|
+
assert rc == 0
|
|
25
|
+
outputs = sorted(out_dir.glob("cli_0-*.fastq.gz"))
|
|
26
|
+
assert len(outputs) == 3
|
|
27
|
+
with gzip.open(outputs[-1], "rt") as handle:
|
|
28
|
+
assert sum(1 for line in handle if line.startswith("@read")) >= 1
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_main_no_fastq(tmp_path):
|
|
32
|
+
empty = tmp_path / "empty"
|
|
33
|
+
empty.mkdir()
|
|
34
|
+
rc = main(["-f", str(empty), "-o", str(tmp_path / "out"), "-i", "1h"])
|
|
35
|
+
assert rc == 1
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_main_bad_interval(fastq_folder, tmp_path):
|
|
39
|
+
rc = main(["-f", str(fastq_folder), "-o", str(tmp_path / "out"), "-i", "1x"])
|
|
40
|
+
assert rc == 1
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def test_main_no_timestamps(tmp_path):
|
|
44
|
+
folder = tmp_path / "fastq"
|
|
45
|
+
folder.mkdir()
|
|
46
|
+
(folder / "run.fastq").write_text("@read1 length=4\nACGT\n+\nIIII\n")
|
|
47
|
+
rc = main(["-f", str(folder), "-o", str(tmp_path / "out"), "-i", "1h"])
|
|
48
|
+
assert rc == 1
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def test_version(capsys):
|
|
52
|
+
with pytest.raises(SystemExit) as exc:
|
|
53
|
+
main(["--version"])
|
|
54
|
+
assert exc.value.code == 0
|
|
55
|
+
assert __version__ in capsys.readouterr().out
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from datetime import datetime, timezone
|
|
2
|
+
|
|
3
|
+
from nanotimesort.timestamps import extract_start_time, parse_timestamp
|
|
4
|
+
|
|
5
|
+
GUPPY_HEADER = (
|
|
6
|
+
b"@c041234f-1234-4a81-9a5a-1234567890ab runid=abc123 read=42 ch=133 "
|
|
7
|
+
b"start_time=2019-07-16T19:51:22Z flow_cell_id=FAK12345"
|
|
8
|
+
)
|
|
9
|
+
MINKNOW_HEADER = (
|
|
10
|
+
b"@bd8655fb-383c-45cc-bff3-eb1dc86533e0 runid=abc parent_read_id=bd8655fb "
|
|
11
|
+
b"start_time=2025-01-13T10:45:28.681306+00:00 protocol_group_id=test"
|
|
12
|
+
)
|
|
13
|
+
DORADO_HEADER = (
|
|
14
|
+
b"@0000813e-1111-4c35-8f57-222233334444 qs:f:21.5 du:f:12.44 ns:i:62205 "
|
|
15
|
+
b"ts:i:10 mx:i:3 ch:i:942 st:Z:2023-09-01T11:13:45.731+00:00 rn:i:9566 "
|
|
16
|
+
b"fn:Z:PAO12345_pass_0.pod5 sm:f:421.3 sd:f:93.0 sv:Z:quantile dx:i:0 "
|
|
17
|
+
b"RG:Z:abc_dna_r10.4.1_e8.2_400bps_hac@v4.2.0"
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_guppy_header():
|
|
22
|
+
dt = extract_start_time(GUPPY_HEADER)
|
|
23
|
+
assert dt == datetime(2019, 7, 16, 19, 51, 22, tzinfo=timezone.utc)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def test_minknow_header_with_microseconds():
|
|
27
|
+
dt = extract_start_time(MINKNOW_HEADER)
|
|
28
|
+
assert dt == datetime(2025, 1, 13, 10, 45, 28, 681306, tzinfo=timezone.utc)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_dorado_header():
|
|
32
|
+
dt = extract_start_time(DORADO_HEADER)
|
|
33
|
+
assert dt == datetime(2023, 9, 1, 11, 13, 45, 731000, tzinfo=timezone.utc)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_header_without_time_returns_none():
|
|
37
|
+
assert extract_start_time(b"@read1 length=100") is None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_naive_timestamp_assumed_utc():
|
|
41
|
+
dt = parse_timestamp("2023-09-01T11:13:45")
|
|
42
|
+
assert dt.tzinfo is not None
|
|
43
|
+
assert dt.utcoffset().total_seconds() == 0
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_z_suffix_with_milliseconds():
|
|
47
|
+
dt = parse_timestamp("2023-09-01T11:13:45.7Z")
|
|
48
|
+
assert dt == datetime(2023, 9, 1, 11, 13, 45, 700000, tzinfo=timezone.utc)
|