nanotimesort 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2019 Marc-Olivier Duceppe
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,94 @@
1
+ Metadata-Version: 2.4
2
+ Name: nanotimesort
3
+ Version: 1.0.0
4
+ Summary: Bin Oxford Nanopore reads by cumulative sequencing time intervals (Guppy and Dorado compatible)
5
+ Author-email: Marc-Olivier Duceppe <duceppemo@gmail.com>
6
+ License: MIT License
7
+
8
+ Copyright (c) 2019 Marc-Olivier Duceppe
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Project-URL: Homepage, https://github.com/duceppemo/nanoTimeSort
29
+ Project-URL: Issues, https://github.com/duceppemo/nanoTimeSort/issues
30
+ Project-URL: Wiki, https://github.com/duceppemo/nanoTimeSort/wiki
31
+ Keywords: nanopore,fastq,bioinformatics,minion,dorado,guppy,binning
32
+ Classifier: Development Status :: 5 - Production/Stable
33
+ Classifier: Environment :: Console
34
+ Classifier: Intended Audience :: Science/Research
35
+ Classifier: License :: OSI Approved :: MIT License
36
+ Classifier: Operating System :: OS Independent
37
+ Classifier: Programming Language :: Python :: 3
38
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
39
+ Requires-Python: >=3.8
40
+ Description-Content-Type: text/markdown
41
+ License-File: LICENSE
42
+ Provides-Extra: dev
43
+ Requires-Dist: pytest; extra == "dev"
44
+ Requires-Dist: pytest-cov; extra == "dev"
45
+ Requires-Dist: ruff; extra == "dev"
46
+ Dynamic: license-file
47
+
48
+ <p align="center">
49
+ <img src="docs/images/logo.svg" alt="nanoTimeSort" width="520">
50
+ </p>
51
+
52
+ <p align="center">
53
+ <a href="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
54
+ <a href="https://codecov.io/gh/duceppemo/nanoTimeSort"><img src="https://codecov.io/gh/duceppemo/nanoTimeSort/branch/master/graph/badge.svg" alt="codecov"></a>
55
+ <a href="LICENSE"><img src="https://img.shields.io/badge/License-MIT-yellow.svg" alt="License: MIT"></a>
56
+ <a href="https://www.python.org/downloads/"><img src="https://img.shields.io/badge/python-3.8%2B-blue.svg" alt="Python 3.8+"></a>
57
+ <a href="pyproject.toml"><img src="https://img.shields.io/badge/dependencies-none-brightgreen.svg" alt="No dependencies"></a>
58
+ </p>
59
+
60
+ Bin Oxford Nanopore reads by **cumulative sequencing time intervals**, using the read start time
61
+ that MinKNOW, Guppy and Dorado embed in every FASTQ header. Useful to answer *"how long did I
62
+ actually need to sequence?"* — for time-to-detection studies, assembly saturation curves, or
63
+ benchmarking real-time pipelines.
64
+
65
+ ```text
66
+ fastq_pass/ ──► sample_0-1h_125437reads_1103093674bp.fastq.gz
67
+ sample_0-2h_225891reads_2087456221bp.fastq.gz (bins are cumulative)
68
+ sample_0-3h_301255reads_2812345678bp.fastq.gz
69
+ ```
70
+
71
+ Compatible with **Guppy/MinKNOW** (`start_time=`) and **Dorado** (`st:Z:`) headers. Fast
72
+ (each read compressed once, ~100× faster than v0.1), low-memory (streaming), parallel, and
73
+ pure standard library.
74
+
75
+ ## Install
76
+
77
+ ```bash
78
+ pip install git+https://github.com/duceppemo/nanoTimeSort.git
79
+ ```
80
+
81
+ Requires Python ≥ 3.8. No other dependencies.
82
+
83
+ ## Usage
84
+
85
+ ```bash
86
+ nanotimesort -f /path/to/fastq_pass/ -o /path/to/output/ -i 1h -p my_sample
87
+ ```
88
+
89
+ 📖 **Full documentation** — CLI reference, tutorial with example data, header compatibility,
90
+ design notes and FAQ — is in the [**wiki**](https://github.com/duceppemo/nanoTimeSort/wiki).
91
+
92
+ ## License
93
+
94
+ [MIT](LICENSE) © Marc-Olivier Duceppe
@@ -0,0 +1,47 @@
1
+ <p align="center">
2
+ <img src="docs/images/logo.svg" alt="nanoTimeSort" width="520">
3
+ </p>
4
+
5
+ <p align="center">
6
+ <a href="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
7
+ <a href="https://codecov.io/gh/duceppemo/nanoTimeSort"><img src="https://codecov.io/gh/duceppemo/nanoTimeSort/branch/master/graph/badge.svg" alt="codecov"></a>
8
+ <a href="LICENSE"><img src="https://img.shields.io/badge/License-MIT-yellow.svg" alt="License: MIT"></a>
9
+ <a href="https://www.python.org/downloads/"><img src="https://img.shields.io/badge/python-3.8%2B-blue.svg" alt="Python 3.8+"></a>
10
+ <a href="pyproject.toml"><img src="https://img.shields.io/badge/dependencies-none-brightgreen.svg" alt="No dependencies"></a>
11
+ </p>
12
+
13
+ Bin Oxford Nanopore reads by **cumulative sequencing time intervals**, using the read start time
14
+ that MinKNOW, Guppy and Dorado embed in every FASTQ header. Useful to answer *"how long did I
15
+ actually need to sequence?"* — for time-to-detection studies, assembly saturation curves, or
16
+ benchmarking real-time pipelines.
17
+
18
+ ```text
19
+ fastq_pass/ ──► sample_0-1h_125437reads_1103093674bp.fastq.gz
20
+ sample_0-2h_225891reads_2087456221bp.fastq.gz (bins are cumulative)
21
+ sample_0-3h_301255reads_2812345678bp.fastq.gz
22
+ ```
23
+
24
+ Compatible with **Guppy/MinKNOW** (`start_time=`) and **Dorado** (`st:Z:`) headers. Fast
25
+ (each read compressed once, ~100× faster than v0.1), low-memory (streaming), parallel, and
26
+ pure standard library.
27
+
28
+ ## Install
29
+
30
+ ```bash
31
+ pip install git+https://github.com/duceppemo/nanoTimeSort.git
32
+ ```
33
+
34
+ Requires Python ≥ 3.8. No other dependencies.
35
+
36
+ ## Usage
37
+
38
+ ```bash
39
+ nanotimesort -f /path/to/fastq_pass/ -o /path/to/output/ -i 1h -p my_sample
40
+ ```
41
+
42
+ 📖 **Full documentation** — CLI reference, tutorial with example data, header compatibility,
43
+ design notes and FAQ — is in the [**wiki**](https://github.com/duceppemo/nanoTimeSort/wiki).
44
+
45
+ ## License
46
+
47
+ [MIT](LICENSE) © Marc-Olivier Duceppe
@@ -0,0 +1,4 @@
1
+ """nanoTimeSort: bin Oxford Nanopore reads by cumulative sequencing time intervals."""
2
+
3
+ __version__ = "1.0.0"
4
+ __author__ = "duceppemo"
@@ -0,0 +1,315 @@
1
+ """Core binning engine.
2
+
3
+ The pipeline runs in three stages, none of which holds reads in memory:
4
+
5
+ 1. **Scan** (parallel): stream every FASTQ once to find the earliest and
6
+ latest read start time and the total read count.
7
+ 2. **Chunk** (parallel): stream every FASTQ again; each read is gzip-
8
+ compressed exactly once, into the chunk file of the single interval it
9
+ belongs to.
10
+ 3. **Assemble** (sequential): cumulative output *k* is built by raw byte
11
+ concatenation of output *k-1* and the chunks of interval *k*. Gzip members
12
+ concatenate into a valid gzip stream, so no data is ever recompressed.
13
+
14
+ Compared to the original implementation (which recompressed every read into
15
+ every cumulative bin at gzip level 9), this is typically one to two orders of
16
+ magnitude faster.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import gzip
22
+ import os
23
+ import shutil
24
+ import sys
25
+ import tempfile
26
+ from concurrent.futures import ProcessPoolExecutor
27
+ from dataclasses import dataclass
28
+ from datetime import datetime
29
+ from math import floor
30
+ from time import time
31
+ from typing import IO, Dict, Iterator, List, Optional, Tuple
32
+
33
+ from .timestamps import extract_start_time
34
+
35
+ UNIT_SECONDS = {"h": 3600.0, "m": 60.0, "s": 1.0}
36
+ UNIT_NAMES = {"h": "hour", "m": "minute", "s": "second"}
37
+
38
+ _COPY_BUFFER = 4 * 1024 * 1024
39
+
40
+
41
+ @dataclass
42
+ class ScanResult:
43
+ """Summary of one scan pass over a FASTQ file."""
44
+
45
+ reads: int = 0
46
+ missing: int = 0 # reads without a recognizable start time
47
+ t_min: Optional[datetime] = None
48
+ t_max: Optional[datetime] = None
49
+
50
+
51
+ def open_fastq(path: str) -> IO[bytes]:
52
+ """Open a plain or gzipped FASTQ for binary reading."""
53
+ if path.endswith(".gz"):
54
+ return gzip.open(path, "rb")
55
+ return open(path, "rb", buffering=1024 * 1024)
56
+
57
+
58
+ def fastq_records(handle: IO[bytes]) -> Iterator[Tuple[bytes, bytes, bytes]]:
59
+ """Yield (header, sequence, quality) byte lines, newline-stripped."""
60
+ while True:
61
+ header = handle.readline()
62
+ if not header:
63
+ return
64
+ seq = handle.readline()
65
+ handle.readline() # '+' separator line
66
+ qual = handle.readline()
67
+ if not qual:
68
+ return # truncated final record: skip it
69
+ yield header.rstrip(), seq.rstrip(), qual.rstrip()
70
+
71
+
72
+ def find_fastq_files(input_path: str) -> List[str]:
73
+ """Collect FASTQ files from a folder (recursively) or a single file path."""
74
+ extensions = (".fastq", ".fastq.gz", ".fq", ".fq.gz")
75
+ if os.path.isfile(input_path):
76
+ return [input_path] if input_path.endswith(extensions) else []
77
+ fastq_list = []
78
+ for root, _dirs, filenames in os.walk(input_path):
79
+ for filename in filenames:
80
+ if filename.endswith(extensions):
81
+ fastq_list.append(os.path.join(root, filename))
82
+ return sorted(fastq_list)
83
+
84
+
85
+ def scan_file(path: str) -> ScanResult:
86
+ """Pass 1: stream one file and record read count and time extremes."""
87
+ result = ScanResult()
88
+ if os.path.getsize(path) == 0:
89
+ return result
90
+ with open_fastq(path) as handle:
91
+ for header, _seq, _qual in fastq_records(handle):
92
+ start_time = extract_start_time(header)
93
+ if start_time is None:
94
+ result.missing += 1
95
+ continue
96
+ result.reads += 1
97
+ if result.t_min is None or start_time < result.t_min:
98
+ result.t_min = start_time
99
+ if result.t_max is None or start_time > result.t_max:
100
+ result.t_max = start_time
101
+ return result
102
+
103
+
104
+ def chunk_file(
105
+ path: str,
106
+ file_index: int,
107
+ t_min: datetime,
108
+ bin_seconds: float,
109
+ num_bins: int,
110
+ chunk_dir: str,
111
+ compresslevel: int,
112
+ ) -> Tuple[List[int], List[int]]:
113
+ """Pass 2: write each read of one file into its interval chunk.
114
+
115
+ Chunk files are opened lazily, so intervals with no reads in this file
116
+ cost nothing. Returns per-interval read and base-pair counts.
117
+ """
118
+ reads_per_bin = [0] * num_bins
119
+ bp_per_bin = [0] * num_bins
120
+ handles: Dict[int, IO[bytes]] = {}
121
+
122
+ if os.path.getsize(path) == 0:
123
+ return reads_per_bin, bp_per_bin
124
+
125
+ try:
126
+ with open_fastq(path) as handle:
127
+ for header, seq, qual in fastq_records(handle):
128
+ start_time = extract_start_time(header)
129
+ if start_time is None:
130
+ continue
131
+ elapsed = (start_time - t_min).total_seconds()
132
+ bin_index = min(floor(elapsed / bin_seconds), num_bins - 1)
133
+ out = handles.get(bin_index)
134
+ if out is None:
135
+ chunk_path = os.path.join(
136
+ chunk_dir, "chunk_b{:06d}_f{:06d}.fastq.gz".format(bin_index, file_index)
137
+ )
138
+ out = gzip.open(chunk_path, "wb", compresslevel=compresslevel)
139
+ handles[bin_index] = out
140
+ out.write(header + b"\n" + seq + b"\n+\n" + qual + b"\n")
141
+ reads_per_bin[bin_index] += 1
142
+ bp_per_bin[bin_index] += len(seq)
143
+ finally:
144
+ for out in handles.values():
145
+ out.close()
146
+
147
+ return reads_per_bin, bp_per_bin
148
+
149
+
150
+ def format_number(value: float) -> str:
151
+ """Render 2.0 as '2' and 0.5 as '0.5' for use in file names."""
152
+ return str(int(value)) if float(value).is_integer() else str(value)
153
+
154
+
155
+ class NanoTimeSort:
156
+ """Bin Nanopore reads into cumulative sequencing-time interval files."""
157
+
158
+ def __init__(
159
+ self,
160
+ input_path: str,
161
+ output_folder: str,
162
+ interval: str,
163
+ prefix: str = "interval",
164
+ threads: int = 1,
165
+ compresslevel: int = 4,
166
+ ):
167
+ self.input_path = input_path
168
+ self.output_folder = output_folder
169
+ self.prefix = prefix
170
+ self.threads = max(1, threads)
171
+ self.compresslevel = compresslevel
172
+ self.bin_size, self.units = self._parse_interval(interval)
173
+ self.bin_seconds = self.bin_size * UNIT_SECONDS[self.units]
174
+
175
+ @staticmethod
176
+ def _parse_interval(interval: str) -> Tuple[float, str]:
177
+ units = interval[-1].lower()
178
+ if units not in UNIT_SECONDS:
179
+ raise ValueError(
180
+ "Invalid interval unit '{}'. Use one of: {}".format(
181
+ units, ", ".join(sorted(UNIT_SECONDS))
182
+ )
183
+ )
184
+ try:
185
+ bin_size = float(interval[:-1])
186
+ except ValueError:
187
+ raise ValueError("Invalid interval value: '{}'".format(interval)) from None
188
+ if bin_size <= 0:
189
+ raise ValueError("Interval must be greater than zero.")
190
+ return bin_size, units
191
+
192
+ def run(self) -> List[str]:
193
+ """Execute the full pipeline. Returns the list of output file paths."""
194
+ overall_start = time()
195
+ os.makedirs(self.output_folder, exist_ok=True)
196
+
197
+ fastq_list = find_fastq_files(self.input_path)
198
+ if not fastq_list:
199
+ raise FileNotFoundError("No FASTQ files found in {}".format(self.input_path))
200
+ workers = min(self.threads, len(fastq_list))
201
+
202
+ # Pass 1: find the time range of the run.
203
+ print("Scanning {} FASTQ file(s)...".format(len(fastq_list)), end="", flush=True)
204
+ start = time()
205
+ scans = self._map(scan_file, fastq_list, workers)
206
+ total_reads = sum(s.reads for s in scans)
207
+ total_missing = sum(s.missing for s in scans)
208
+ t_mins = [s.t_min for s in scans if s.t_min is not None]
209
+ t_maxs = [s.t_max for s in scans if s.t_max is not None]
210
+ print(" {} reads in {}".format(total_reads, elapsed_time(time() - start)))
211
+ if total_missing:
212
+ print(
213
+ "Warning: {} read(s) without a 'start_time=' or 'st:Z:' header field "
214
+ "were skipped.".format(total_missing),
215
+ file=sys.stderr,
216
+ )
217
+ if not t_mins:
218
+ raise ValueError(
219
+ "No read start times found. Headers must contain a Guppy/MinKNOW "
220
+ "'start_time=' field or a Dorado 'st:Z:' tag."
221
+ )
222
+
223
+ t_min, t_max = min(t_mins), max(t_maxs)
224
+ run_seconds = (t_max - t_min).total_seconds()
225
+ num_bins = int(run_seconds // self.bin_seconds) + 1
226
+ print(
227
+ "Run spans {} -> {} intervals of {}{}".format(
228
+ elapsed_time(run_seconds), num_bins, format_number(self.bin_size), self.units
229
+ )
230
+ )
231
+
232
+ # Pass 2: compress each read once into its interval chunk.
233
+ print("Binning reads...", end="", flush=True)
234
+ start = time()
235
+ with tempfile.TemporaryDirectory(prefix="nanotimesort_", dir=self.output_folder) as chunk_dir:
236
+ jobs = [
237
+ (path, i, t_min, self.bin_seconds, num_bins, chunk_dir, self.compresslevel)
238
+ for i, path in enumerate(fastq_list)
239
+ ]
240
+ results = self._starmap(chunk_file, jobs, workers)
241
+ reads_per_bin = [0] * num_bins
242
+ bp_per_bin = [0] * num_bins
243
+ for file_reads, file_bp in results:
244
+ for b in range(num_bins):
245
+ reads_per_bin[b] += file_reads[b]
246
+ bp_per_bin[b] += file_bp[b]
247
+ print(" done in {}".format(elapsed_time(time() - start)))
248
+
249
+ # Pass 3: assemble cumulative outputs by gzip member concatenation.
250
+ print("Writing cumulative interval files...", end="", flush=True)
251
+ start = time()
252
+ outputs = self._assemble(chunk_dir, num_bins, reads_per_bin, bp_per_bin)
253
+ print(" done in {}".format(elapsed_time(time() - start)))
254
+
255
+ print("Total run time: {}".format(elapsed_time(time() - overall_start)))
256
+ return outputs
257
+
258
+ def _assemble(
259
+ self,
260
+ chunk_dir: str,
261
+ num_bins: int,
262
+ reads_per_bin: List[int],
263
+ bp_per_bin: List[int],
264
+ ) -> List[str]:
265
+ chunk_names = sorted(os.listdir(chunk_dir))
266
+ outputs: List[str] = []
267
+ previous_path: Optional[str] = None
268
+ cumulative_reads = 0
269
+ cumulative_bp = 0
270
+
271
+ for b in range(num_bins):
272
+ cumulative_reads += reads_per_bin[b]
273
+ cumulative_bp += bp_per_bin[b]
274
+ label = format_number((b + 1) * self.bin_size)
275
+ out_name = "{}_0-{}{}_{}reads_{}bp.fastq.gz".format(
276
+ self.prefix, label, self.units, cumulative_reads, cumulative_bp
277
+ )
278
+ out_path = os.path.join(self.output_folder, out_name)
279
+ prefix = "chunk_b{:06d}_".format(b)
280
+ with open(out_path, "wb") as out:
281
+ if previous_path is not None:
282
+ with open(previous_path, "rb") as prev:
283
+ shutil.copyfileobj(prev, out, _COPY_BUFFER)
284
+ for name in chunk_names:
285
+ if name.startswith(prefix):
286
+ with open(os.path.join(chunk_dir, name), "rb") as chunk:
287
+ shutil.copyfileobj(chunk, out, _COPY_BUFFER)
288
+ outputs.append(out_path)
289
+ previous_path = out_path
290
+
291
+ return outputs
292
+
293
+ def _map(self, func, items, workers):
294
+ if workers <= 1 or len(items) <= 1:
295
+ return [func(item) for item in items]
296
+ with ProcessPoolExecutor(max_workers=workers) as pool:
297
+ return list(pool.map(func, items))
298
+
299
+ def _starmap(self, func, jobs, workers):
300
+ if workers <= 1 or len(jobs) <= 1:
301
+ return [func(*job) for job in jobs]
302
+ with ProcessPoolExecutor(max_workers=workers) as pool:
303
+ futures = [pool.submit(func, *job) for job in jobs]
304
+ return [f.result() for f in futures]
305
+
306
+
307
+ def elapsed_time(seconds: float) -> str:
308
+ """Format a duration in seconds as a compact '1d2h3m4s' style string."""
309
+ if seconds < 1:
310
+ return "{:.2f}s".format(seconds)
311
+ minutes, secs = divmod(int(round(seconds)), 60)
312
+ hours, minutes = divmod(minutes, 60)
313
+ days, hours = divmod(hours, 24)
314
+ parts = [("d", days), ("h", hours), ("m", minutes), ("s", secs)]
315
+ return "".join("{}{}".format(value, name) for name, value in parts if value) or "0s"
@@ -0,0 +1,78 @@
1
+ """Command-line interface for nanoTimeSort."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+ from argparse import ArgumentParser, RawTextHelpFormatter
7
+ from multiprocessing import cpu_count
8
+
9
+ from . import __version__
10
+ from .binner import NanoTimeSort
11
+
12
+
13
+ def build_parser() -> ArgumentParser:
14
+ cpu = cpu_count()
15
+ parser = ArgumentParser(
16
+ prog="nanotimesort",
17
+ description="Bin Oxford Nanopore reads by cumulative sequencing time intervals.\n"
18
+ "Supports Guppy/MinKNOW ('start_time=') and Dorado ('st:Z:') FASTQ headers.",
19
+ formatter_class=RawTextHelpFormatter,
20
+ )
21
+ parser.add_argument(
22
+ "-f", "--fastq", metavar="/basecalled/folder/", required=True,
23
+ help="Input folder (searched recursively) or single FASTQ file.\n"
24
+ "Accepts .fastq, .fq, .fastq.gz and .fq.gz.",
25
+ )
26
+ parser.add_argument(
27
+ "-o", "--output", metavar="/output/folder/", required=True,
28
+ help="Output folder. Created if it does not exist.",
29
+ )
30
+ parser.add_argument(
31
+ "-i", "--interval", metavar="1h", required=True,
32
+ help="Time interval for the bins, e.g. '1h', '30m' or '90s'.\n"
33
+ "Bins are cumulative: with '-i 1h', the second file also\n"
34
+ "contains the reads of the first hour.",
35
+ )
36
+ parser.add_argument(
37
+ "-p", "--prefix", metavar="my_sample", default="interval",
38
+ help="Output file prefix. Files are named like\n"
39
+ "'my_sample_0-1h_123reads_456789bp.fastq.gz'.\n"
40
+ "Default: interval",
41
+ )
42
+ parser.add_argument(
43
+ "-t", "--threads", metavar=str(cpu), type=int, default=cpu,
44
+ help="Number of FASTQ files to process in parallel.\n"
45
+ "Default: {}".format(cpu),
46
+ )
47
+ parser.add_argument(
48
+ "-c", "--compression-level", metavar="4", type=int, default=4,
49
+ choices=range(1, 10),
50
+ help="Gzip compression level for output files (1=fastest, 9=smallest).\n"
51
+ "Default: 4",
52
+ )
53
+ parser.add_argument(
54
+ "-v", "--version", action="version", version="nanoTimeSort v{}".format(__version__),
55
+ )
56
+ return parser
57
+
58
+
59
+ def main(argv=None) -> int:
60
+ args = build_parser().parse_args(argv)
61
+ try:
62
+ binner = NanoTimeSort(
63
+ input_path=args.fastq,
64
+ output_folder=args.output,
65
+ interval=args.interval,
66
+ prefix=args.prefix,
67
+ threads=args.threads,
68
+ compresslevel=args.compression_level,
69
+ )
70
+ binner.run()
71
+ except (ValueError, FileNotFoundError) as err:
72
+ print("Error: {}".format(err), file=sys.stderr)
73
+ return 1
74
+ return 0
75
+
76
+
77
+ if __name__ == "__main__":
78
+ sys.exit(main())
@@ -0,0 +1,69 @@
1
+ """Extraction and parsing of read start times from Nanopore FASTQ headers.
2
+
3
+ Two header dialects are supported:
4
+
5
+ * Guppy / MinKNOW (key=value fields)::
6
+
7
+ @<read_id> runid=... read=... ch=... start_time=2019-07-16T19:51:22Z
8
+ @<read_id> ... start_time=2025-01-13T10:45:28.681306+00:00 ...
9
+
10
+ * Dorado (SAM-style tags)::
11
+
12
+ @<read_id> qs:f:21.3 du:f:12.44 ch:i:942 st:Z:2023-09-01T11:13:45.731+00:00 ...
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import re
18
+ from datetime import datetime, timezone
19
+ from typing import Optional, Union
20
+
21
+ # Guppy/MinKNOW style field.
22
+ _GUPPY_PREFIX = b"start_time="
23
+ # Dorado SAM-tag style field (st:Z:<ISO 8601>).
24
+ _DORADO_PREFIX = b"st:Z:"
25
+
26
+ _FRACTION_RE = re.compile(r"\.(\d+)")
27
+
28
+
29
+ def extract_start_time(header: bytes) -> Optional[datetime]:
30
+ """Return the read start time from a FASTQ header line, or None if absent.
31
+
32
+ :param header: raw FASTQ header line (bytes, with or without trailing newline)
33
+ :return: timezone-aware datetime (UTC assumed when the timestamp is naive)
34
+ """
35
+ for item in header.split():
36
+ if item.startswith(_GUPPY_PREFIX):
37
+ return parse_timestamp(item[len(_GUPPY_PREFIX):])
38
+ if item.startswith(_DORADO_PREFIX):
39
+ return parse_timestamp(item[len(_DORADO_PREFIX):])
40
+ return None
41
+
42
+
43
+ def parse_timestamp(raw: Union[bytes, str]) -> datetime:
44
+ """Parse an ISO 8601 / RFC 3339 timestamp into a timezone-aware datetime.
45
+
46
+ Handles 'Z' suffixes and unusual fractional-second precision on Python
47
+ versions where datetime.fromisoformat() is strict (< 3.11).
48
+ """
49
+ if isinstance(raw, bytes):
50
+ raw = raw.decode("ascii")
51
+ try:
52
+ dt = datetime.fromisoformat(raw)
53
+ except ValueError:
54
+ dt = datetime.fromisoformat(_normalize(raw))
55
+ if dt.tzinfo is None:
56
+ dt = dt.replace(tzinfo=timezone.utc)
57
+ return dt
58
+
59
+
60
+ def _normalize(s: str) -> str:
61
+ """Rewrite a timestamp so strict fromisoformat() implementations accept it."""
62
+ if s.endswith(("Z", "z")):
63
+ s = s[:-1] + "+00:00"
64
+
65
+ def _pad(match: "re.Match[str]") -> str:
66
+ digits = match.group(1)[:6]
67
+ return "." + digits.ljust(6, "0")
68
+
69
+ return _FRACTION_RE.sub(_pad, s, count=1)
@@ -0,0 +1,94 @@
1
+ Metadata-Version: 2.4
2
+ Name: nanotimesort
3
+ Version: 1.0.0
4
+ Summary: Bin Oxford Nanopore reads by cumulative sequencing time intervals (Guppy and Dorado compatible)
5
+ Author-email: Marc-Olivier Duceppe <duceppemo@gmail.com>
6
+ License: MIT License
7
+
8
+ Copyright (c) 2019 Marc-Olivier Duceppe
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Project-URL: Homepage, https://github.com/duceppemo/nanoTimeSort
29
+ Project-URL: Issues, https://github.com/duceppemo/nanoTimeSort/issues
30
+ Project-URL: Wiki, https://github.com/duceppemo/nanoTimeSort/wiki
31
+ Keywords: nanopore,fastq,bioinformatics,minion,dorado,guppy,binning
32
+ Classifier: Development Status :: 5 - Production/Stable
33
+ Classifier: Environment :: Console
34
+ Classifier: Intended Audience :: Science/Research
35
+ Classifier: License :: OSI Approved :: MIT License
36
+ Classifier: Operating System :: OS Independent
37
+ Classifier: Programming Language :: Python :: 3
38
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
39
+ Requires-Python: >=3.8
40
+ Description-Content-Type: text/markdown
41
+ License-File: LICENSE
42
+ Provides-Extra: dev
43
+ Requires-Dist: pytest; extra == "dev"
44
+ Requires-Dist: pytest-cov; extra == "dev"
45
+ Requires-Dist: ruff; extra == "dev"
46
+ Dynamic: license-file
47
+
48
+ <p align="center">
49
+ <img src="docs/images/logo.svg" alt="nanoTimeSort" width="520">
50
+ </p>
51
+
52
+ <p align="center">
53
+ <a href="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/nanoTimeSort/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
54
+ <a href="https://codecov.io/gh/duceppemo/nanoTimeSort"><img src="https://codecov.io/gh/duceppemo/nanoTimeSort/branch/master/graph/badge.svg" alt="codecov"></a>
55
+ <a href="LICENSE"><img src="https://img.shields.io/badge/License-MIT-yellow.svg" alt="License: MIT"></a>
56
+ <a href="https://www.python.org/downloads/"><img src="https://img.shields.io/badge/python-3.8%2B-blue.svg" alt="Python 3.8+"></a>
57
+ <a href="pyproject.toml"><img src="https://img.shields.io/badge/dependencies-none-brightgreen.svg" alt="No dependencies"></a>
58
+ </p>
59
+
60
+ Bin Oxford Nanopore reads by **cumulative sequencing time intervals**, using the read start time
61
+ that MinKNOW, Guppy and Dorado embed in every FASTQ header. Useful to answer *"how long did I
62
+ actually need to sequence?"* — for time-to-detection studies, assembly saturation curves, or
63
+ benchmarking real-time pipelines.
64
+
65
+ ```text
66
+ fastq_pass/ ──► sample_0-1h_125437reads_1103093674bp.fastq.gz
67
+ sample_0-2h_225891reads_2087456221bp.fastq.gz (bins are cumulative)
68
+ sample_0-3h_301255reads_2812345678bp.fastq.gz
69
+ ```
70
+
71
+ Compatible with **Guppy/MinKNOW** (`start_time=`) and **Dorado** (`st:Z:`) headers. Fast
72
+ (each read compressed once, ~100× faster than v0.1), low-memory (streaming), parallel, and
73
+ pure standard library.
74
+
75
+ ## Install
76
+
77
+ ```bash
78
+ pip install git+https://github.com/duceppemo/nanoTimeSort.git
79
+ ```
80
+
81
+ Requires Python ≥ 3.8. No other dependencies.
82
+
83
+ ## Usage
84
+
85
+ ```bash
86
+ nanotimesort -f /path/to/fastq_pass/ -o /path/to/output/ -i 1h -p my_sample
87
+ ```
88
+
89
+ 📖 **Full documentation** — CLI reference, tutorial with example data, header compatibility,
90
+ design notes and FAQ — is in the [**wiki**](https://github.com/duceppemo/nanoTimeSort/wiki).
91
+
92
+ ## License
93
+
94
+ [MIT](LICENSE) © Marc-Olivier Duceppe
@@ -0,0 +1,16 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ nanotimesort/__init__.py
5
+ nanotimesort/binner.py
6
+ nanotimesort/cli.py
7
+ nanotimesort/timestamps.py
8
+ nanotimesort.egg-info/PKG-INFO
9
+ nanotimesort.egg-info/SOURCES.txt
10
+ nanotimesort.egg-info/dependency_links.txt
11
+ nanotimesort.egg-info/entry_points.txt
12
+ nanotimesort.egg-info/requires.txt
13
+ nanotimesort.egg-info/top_level.txt
14
+ tests/test_binner.py
15
+ tests/test_cli.py
16
+ tests/test_timestamps.py
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ nanotimesort = nanotimesort.cli:main
@@ -0,0 +1,5 @@
1
+
2
+ [dev]
3
+ pytest
4
+ pytest-cov
5
+ ruff
@@ -0,0 +1 @@
1
+ nanotimesort
@@ -0,0 +1,47 @@
1
+ [build-system]
2
+ requires = ["setuptools>=64"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "nanotimesort"
7
+ version = "1.0.0"
8
+ description = "Bin Oxford Nanopore reads by cumulative sequencing time intervals (Guppy and Dorado compatible)"
9
+ readme = "README.md"
10
+ license = { file = "LICENSE" }
11
+ authors = [{ name = "Marc-Olivier Duceppe", email = "duceppemo@gmail.com" }]
12
+ requires-python = ">=3.8"
13
+ keywords = ["nanopore", "fastq", "bioinformatics", "minion", "dorado", "guppy", "binning"]
14
+ classifiers = [
15
+ "Development Status :: 5 - Production/Stable",
16
+ "Environment :: Console",
17
+ "Intended Audience :: Science/Research",
18
+ "License :: OSI Approved :: MIT License",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
22
+ ]
23
+ dependencies = []
24
+
25
+ [project.optional-dependencies]
26
+ dev = ["pytest", "pytest-cov", "ruff"]
27
+
28
+ [project.urls]
29
+ Homepage = "https://github.com/duceppemo/nanoTimeSort"
30
+ Issues = "https://github.com/duceppemo/nanoTimeSort/issues"
31
+ Wiki = "https://github.com/duceppemo/nanoTimeSort/wiki"
32
+
33
+ [project.scripts]
34
+ nanotimesort = "nanotimesort.cli:main"
35
+
36
+ [tool.setuptools]
37
+ packages = ["nanotimesort"]
38
+
39
+ [tool.ruff]
40
+ line-length = 110
41
+ target-version = "py38"
42
+
43
+ [tool.ruff.lint]
44
+ select = ["E4", "E7", "E9", "F", "B"]
45
+
46
+ [tool.pytest.ini_options]
47
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,125 @@
1
+ import gzip
2
+ import os
3
+
4
+ import pytest
5
+
6
+ from nanotimesort.binner import NanoTimeSort, find_fastq_files
7
+
8
+ BASE = "2024-05-01T{:02d}:{:02d}:00Z"
9
+
10
+
11
+ def make_read(read_id, minutes, style="guppy", seq="ACGT" * 10):
12
+ """Build one FASTQ record with a start time `minutes` after 10:00 UTC."""
13
+ hour, minute = divmod(600 + minutes, 60)
14
+ stamp = BASE.format(hour, minute)
15
+ if style == "guppy":
16
+ header = "@{} runid=abc ch=1 start_time={}".format(read_id, stamp)
17
+ else: # dorado
18
+ stamp = stamp.replace("Z", ".000+00:00")
19
+ header = "@{} qs:f:20.0 ch:i:1 st:Z:{}".format(read_id, stamp)
20
+ return "{}\n{}\n+\n{}\n".format(header, seq, "I" * len(seq))
21
+
22
+
23
+ @pytest.fixture
24
+ def run_folder(tmp_path):
25
+ """Two FASTQ files (one gzipped, one plain) spanning ~2.5 hours."""
26
+ fastq_dir = tmp_path / "fastq"
27
+ fastq_dir.mkdir()
28
+ # File 1 (gzipped, Guppy headers): reads at 0, 30 and 70 minutes.
29
+ content1 = (
30
+ make_read("read1", 0)
31
+ + make_read("read2", 30)
32
+ + make_read("read3", 70)
33
+ )
34
+ with gzip.open(fastq_dir / "part1.fastq.gz", "wt") as handle:
35
+ handle.write(content1)
36
+ # File 2 (plain, Dorado headers): reads at 90 and 150 minutes.
37
+ content2 = make_read("read4", 90, style="dorado") + make_read("read5", 150, style="dorado")
38
+ (fastq_dir / "part2.fastq").write_text(content2)
39
+ return fastq_dir
40
+
41
+
42
+ def read_ids(path):
43
+ with gzip.open(path, "rt") as handle:
44
+ return [line.split()[0][1:] for i, line in enumerate(handle) if i % 4 == 0]
45
+
46
+
47
+ def test_find_fastq_files(run_folder):
48
+ files = find_fastq_files(str(run_folder))
49
+ assert [os.path.basename(f) for f in files] == ["part1.fastq.gz", "part2.fastq"]
50
+
51
+
52
+ def test_cumulative_binning(run_folder, tmp_path):
53
+ out_dir = tmp_path / "out"
54
+ binner = NanoTimeSort(
55
+ input_path=str(run_folder),
56
+ output_folder=str(out_dir),
57
+ interval="1h",
58
+ prefix="test",
59
+ threads=1,
60
+ )
61
+ outputs = binner.run()
62
+
63
+ # Run spans 150 minutes -> three cumulative 1 h bins.
64
+ names = [os.path.basename(p) for p in outputs]
65
+ bp = 40 # each read is 40 bp
66
+ assert names == [
67
+ "test_0-1h_2reads_{}bp.fastq.gz".format(2 * bp),
68
+ "test_0-2h_4reads_{}bp.fastq.gz".format(4 * bp),
69
+ "test_0-3h_5reads_{}bp.fastq.gz".format(5 * bp),
70
+ ]
71
+
72
+ # Bin contents are cumulative and mix Guppy + Dorado reads.
73
+ assert sorted(read_ids(outputs[0])) == ["read1", "read2"]
74
+ assert sorted(read_ids(outputs[1])) == ["read1", "read2", "read3", "read4"]
75
+ assert sorted(read_ids(outputs[2])) == ["read1", "read2", "read3", "read4", "read5"]
76
+
77
+
78
+ def test_parallel_matches_serial(run_folder, tmp_path):
79
+ results = {}
80
+ for threads in (1, 2):
81
+ out_dir = tmp_path / "out_t{}".format(threads)
82
+ binner = NanoTimeSort(
83
+ input_path=str(run_folder),
84
+ output_folder=str(out_dir),
85
+ interval="30m",
86
+ prefix="par",
87
+ threads=threads,
88
+ )
89
+ outputs = binner.run()
90
+ results[threads] = {
91
+ os.path.basename(p): sorted(read_ids(p)) for p in outputs
92
+ }
93
+ assert results[1] == results[2]
94
+
95
+
96
+ def test_single_file_input(run_folder, tmp_path):
97
+ out_dir = tmp_path / "out_single"
98
+ binner = NanoTimeSort(
99
+ input_path=str(run_folder / "part2.fastq"),
100
+ output_folder=str(out_dir),
101
+ interval="2h",
102
+ prefix="single",
103
+ threads=1,
104
+ )
105
+ outputs = binner.run()
106
+ assert len(outputs) == 1
107
+ assert sorted(read_ids(outputs[0])) == ["read4", "read5"]
108
+
109
+
110
+ def test_invalid_interval():
111
+ with pytest.raises(ValueError):
112
+ NanoTimeSort("in", "out", interval="1x")
113
+ with pytest.raises(ValueError):
114
+ NanoTimeSort("in", "out", interval="-5m")
115
+
116
+
117
+ def test_reads_without_timestamp_are_skipped(tmp_path, capsys):
118
+ fastq_dir = tmp_path / "fastq"
119
+ fastq_dir.mkdir()
120
+ content = make_read("good1", 0) + "@orphan length=4\nACGT\n+\nIIII\n" + make_read("good2", 10)
121
+ (fastq_dir / "mixed.fastq").write_text(content)
122
+ out_dir = tmp_path / "out"
123
+ binner = NanoTimeSort(str(fastq_dir), str(out_dir), interval="1h", threads=1)
124
+ outputs = binner.run()
125
+ assert sorted(read_ids(outputs[-1])) == ["good1", "good2"]
@@ -0,0 +1,55 @@
1
+ import gzip
2
+
3
+ import pytest
4
+
5
+ from nanotimesort import __version__
6
+ from nanotimesort.cli import main
7
+
8
+
9
+ @pytest.fixture
10
+ def fastq_folder(tmp_path):
11
+ folder = tmp_path / "fastq"
12
+ folder.mkdir()
13
+ reads = "".join(
14
+ "@read{} runid=abc start_time=2024-05-01T1{}:00:00Z\nACGT\n+\nIIII\n".format(i, i)
15
+ for i in range(3)
16
+ )
17
+ (folder / "run.fastq").write_text(reads)
18
+ return folder
19
+
20
+
21
+ def test_main_success(fastq_folder, tmp_path):
22
+ out_dir = tmp_path / "out"
23
+ rc = main(["-f", str(fastq_folder), "-o", str(out_dir), "-i", "1h", "-p", "cli"])
24
+ assert rc == 0
25
+ outputs = sorted(out_dir.glob("cli_0-*.fastq.gz"))
26
+ assert len(outputs) == 3
27
+ with gzip.open(outputs[-1], "rt") as handle:
28
+ assert sum(1 for line in handle if line.startswith("@read")) >= 1
29
+
30
+
31
+ def test_main_no_fastq(tmp_path):
32
+ empty = tmp_path / "empty"
33
+ empty.mkdir()
34
+ rc = main(["-f", str(empty), "-o", str(tmp_path / "out"), "-i", "1h"])
35
+ assert rc == 1
36
+
37
+
38
+ def test_main_bad_interval(fastq_folder, tmp_path):
39
+ rc = main(["-f", str(fastq_folder), "-o", str(tmp_path / "out"), "-i", "1x"])
40
+ assert rc == 1
41
+
42
+
43
+ def test_main_no_timestamps(tmp_path):
44
+ folder = tmp_path / "fastq"
45
+ folder.mkdir()
46
+ (folder / "run.fastq").write_text("@read1 length=4\nACGT\n+\nIIII\n")
47
+ rc = main(["-f", str(folder), "-o", str(tmp_path / "out"), "-i", "1h"])
48
+ assert rc == 1
49
+
50
+
51
+ def test_version(capsys):
52
+ with pytest.raises(SystemExit) as exc:
53
+ main(["--version"])
54
+ assert exc.value.code == 0
55
+ assert __version__ in capsys.readouterr().out
@@ -0,0 +1,48 @@
1
+ from datetime import datetime, timezone
2
+
3
+ from nanotimesort.timestamps import extract_start_time, parse_timestamp
4
+
5
+ GUPPY_HEADER = (
6
+ b"@c041234f-1234-4a81-9a5a-1234567890ab runid=abc123 read=42 ch=133 "
7
+ b"start_time=2019-07-16T19:51:22Z flow_cell_id=FAK12345"
8
+ )
9
+ MINKNOW_HEADER = (
10
+ b"@bd8655fb-383c-45cc-bff3-eb1dc86533e0 runid=abc parent_read_id=bd8655fb "
11
+ b"start_time=2025-01-13T10:45:28.681306+00:00 protocol_group_id=test"
12
+ )
13
+ DORADO_HEADER = (
14
+ b"@0000813e-1111-4c35-8f57-222233334444 qs:f:21.5 du:f:12.44 ns:i:62205 "
15
+ b"ts:i:10 mx:i:3 ch:i:942 st:Z:2023-09-01T11:13:45.731+00:00 rn:i:9566 "
16
+ b"fn:Z:PAO12345_pass_0.pod5 sm:f:421.3 sd:f:93.0 sv:Z:quantile dx:i:0 "
17
+ b"RG:Z:abc_dna_r10.4.1_e8.2_400bps_hac@v4.2.0"
18
+ )
19
+
20
+
21
+ def test_guppy_header():
22
+ dt = extract_start_time(GUPPY_HEADER)
23
+ assert dt == datetime(2019, 7, 16, 19, 51, 22, tzinfo=timezone.utc)
24
+
25
+
26
+ def test_minknow_header_with_microseconds():
27
+ dt = extract_start_time(MINKNOW_HEADER)
28
+ assert dt == datetime(2025, 1, 13, 10, 45, 28, 681306, tzinfo=timezone.utc)
29
+
30
+
31
+ def test_dorado_header():
32
+ dt = extract_start_time(DORADO_HEADER)
33
+ assert dt == datetime(2023, 9, 1, 11, 13, 45, 731000, tzinfo=timezone.utc)
34
+
35
+
36
+ def test_header_without_time_returns_none():
37
+ assert extract_start_time(b"@read1 length=100") is None
38
+
39
+
40
+ def test_naive_timestamp_assumed_utc():
41
+ dt = parse_timestamp("2023-09-01T11:13:45")
42
+ assert dt.tzinfo is not None
43
+ assert dt.utcoffset().total_seconds() == 0
44
+
45
+
46
+ def test_z_suffix_with_milliseconds():
47
+ dt = parse_timestamp("2023-09-01T11:13:45.7Z")
48
+ assert dt == datetime(2023, 9, 1, 11, 13, 45, 700000, tzinfo=timezone.utc)