jamofetch 3.7.8__tar.gz → 3.7.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jamofetch-3.7.9/CHANGELOG.md +53 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/PKG-INFO +1 -1
- {jamofetch-3.7.8 → jamofetch-3.7.9}/pyproject.toml +1 -1
- {jamofetch-3.7.8 → jamofetch-3.7.9}/src/jamofetch/jamofetch.py +174 -45
- jamofetch-3.7.9/tests/test_jamofetch.py +184 -0
- jamofetch-3.7.8/CHANGELOG.md +0 -7
- jamofetch-3.7.8/tests/test_jamofetch.py +0 -47
- {jamofetch-3.7.8 → jamofetch-3.7.9}/.gitignore +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/.gitlab-ci.yml +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/.readthedocs.yml +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/CONDUCT.md +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/CONTRIBUTING.md +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/README.md +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/Makefile +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/changelog.md +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/conduct.md +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/conf.py +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/contributing.md +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/demo_script.py +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/index.md +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/make.bat +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/docs/requirements.txt +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/prepare-gitlab-publish.sh +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/push-version-tag.sh +0 -0
- {jamofetch-3.7.8 → jamofetch-3.7.9}/src/jamofetch/__init__.py +0 -0
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
<!--next-version-placeholder-->
|
|
4
|
+
|
|
5
|
+
## v3.7.9 (17/09/2026)
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- **A library sequenced more than once no longer resolves to an arbitrary file.**
|
|
10
|
+
`_find_fastq_path()` sorted the link directory by `ST_CTIME` and kept the last
|
|
11
|
+
match. Every link comes from one `jamo link` call within the same second and
|
|
12
|
+
`ST_CTIME` is whole seconds, so the candidates all tie and Python's stable sort
|
|
13
|
+
returns whichever `os.listdir()` listed last - directory hash order on Linux,
|
|
14
|
+
so not alphabetical, not newest, and not necessarily the same on two hosts.
|
|
15
|
+
It now returns the single match, or raises and names every candidate.
|
|
16
|
+
|
|
17
|
+
Observed on library LBBCZRJ: four matching records across three pools - cell
|
|
18
|
+
37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608 reads,
|
|
19
|
+
registered twice under different `AUTO-` folders with identical md5s) and cell
|
|
20
|
+
37964 (run 3230, 948,348 reads). Any of the three could be analysed, silently,
|
|
21
|
+
with a 2x spread in read count.
|
|
22
|
+
|
|
23
|
+
Links pointing at the same target are not ambiguous and are still accepted.
|
|
24
|
+
|
|
25
|
+
### Added
|
|
26
|
+
|
|
27
|
+
- `get_cmd()` and `JamoFetcher.fetch_lib_seq()` take `smrt_cell_id` and
|
|
28
|
+
`physical_run_id`, which narrow the query to one sequencing of the library.
|
|
29
|
+
`smrt_cell_id` is the one to prefer: it identifies a library's membership in a
|
|
30
|
+
single pool, and equals the PacBio pipeline service's `libraries[].id`. For
|
|
31
|
+
LBBCZRJ it reduces four matches to one (or, for cell 37892, to two records
|
|
32
|
+
that are byte-identical and share a `file_name`).
|
|
33
|
+
- CLI: `--smrt-cell-id` (single `--library` only) and `--physical-run-id`
|
|
34
|
+
(applies to every `--library`).
|
|
35
|
+
|
|
36
|
+
### Changed
|
|
37
|
+
|
|
38
|
+
- The CLI no longer swallows per-library fetch errors. `_fetch_seq()` caught
|
|
39
|
+
every exception with a bare `pass`, which hid the new ambiguity error whose
|
|
40
|
+
whole purpose is to be seen.
|
|
41
|
+
- `JamoFetcher.fetch_lib_seq()` logs every candidate link when more than one
|
|
42
|
+
matches, before selection.
|
|
43
|
+
|
|
44
|
+
### Compatibility
|
|
45
|
+
|
|
46
|
+
- Calls that pass no id produce the byte-identical command 3.7.8 produced;
|
|
47
|
+
asserted in `tests/test_jamofetch.py`. Callers that relied on a library with
|
|
48
|
+
several matches silently resolving to one of them now get a `RuntimeError`
|
|
49
|
+
instead - that is the point of the release.
|
|
50
|
+
|
|
51
|
+
## v0.1.0 (07/07/2023)
|
|
52
|
+
|
|
53
|
+
- First release of `jamofetch`!
|
|
@@ -9,7 +9,6 @@ import subprocess
|
|
|
9
9
|
import sys
|
|
10
10
|
import time
|
|
11
11
|
from argparse import ArgumentParser
|
|
12
|
-
from stat import ST_CTIME
|
|
13
12
|
|
|
14
13
|
# Base commands; get_cmd() appends the query.
|
|
15
14
|
#
|
|
@@ -22,7 +21,7 @@ TWO_HOURS = 7200 # seconds
|
|
|
22
21
|
ONE_MINUTE = 60 # seconds
|
|
23
22
|
|
|
24
23
|
|
|
25
|
-
def get_cmd(clean_lib_name) -> str:
|
|
24
|
+
def get_cmd(clean_lib_name, smrt_cell_id=None, physical_run_id=None) -> str:
|
|
26
25
|
"""
|
|
27
26
|
Build the site-appropriate `jamo link` command for a library's SDM fastq.
|
|
28
27
|
|
|
@@ -34,22 +33,83 @@ def get_cmd(clean_lib_name) -> str:
|
|
|
34
33
|
|
|
35
34
|
Key order matters: jamo names each symlink '<first string-valued query key's
|
|
36
35
|
value>.<file_name>', so metadata.library_name must stay first or
|
|
37
|
-
_find_fastq_path() will not find the link.
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
36
|
+
_find_fastq_path() will not find the link. The disambiguators below are
|
|
37
|
+
integers, so they cannot take that position, but they are appended last
|
|
38
|
+
anyway.
|
|
39
|
+
|
|
40
|
+
A library sequenced on more than one run matches one fastq per run, and
|
|
41
|
+
without a disambiguator NOTHING here chooses between them - the caller ends
|
|
42
|
+
up with several symlinks and _find_fastq_path() refuses to guess. Pass
|
|
43
|
+
smrt_cell_id (preferred) or physical_run_id to narrow the query to the
|
|
44
|
+
sequencing this analysis is actually about.
|
|
45
|
+
|
|
46
|
+
Observed case, library LBBCZRJ: four matching records across three pools -
|
|
47
|
+
cell 37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608
|
|
48
|
+
reads, registered twice under different AUTO- folders with identical md5s)
|
|
49
|
+
and cell 37964 (run 3230, 948,348 reads). Adding the cell id returns exactly
|
|
50
|
+
one record in the first and third cases, and in the second the two records
|
|
51
|
+
are byte-identical and share a file_name, so they collapse to one symlink.
|
|
52
|
+
|
|
53
|
+
smrt_cell_id is the finer key and is the one to prefer: it identifies a
|
|
54
|
+
library's membership in one specific pool. It is JAMO's
|
|
55
|
+
metadata.sdm_smrt_cell_id, and equals the PacBio pipeline service's
|
|
56
|
+
libraries[].id for the same library.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
clean_lib_name: the library name, e.g. 'LBBCZRJ'.
|
|
60
|
+
smrt_cell_id: optional metadata.sdm_smrt_cell_id to narrow to.
|
|
61
|
+
physical_run_id: optional metadata.pacbio_physical_run_id to narrow to.
|
|
62
|
+
Coarser than smrt_cell_id; a run holds many cells.
|
|
63
|
+
|
|
64
|
+
Returns:
|
|
65
|
+
The full shell command, with the JSON query single-quoted.
|
|
41
66
|
"""
|
|
42
67
|
# Validated because the name is interpolated into a shell command.
|
|
43
68
|
lib_name = _clean_library_name(clean_lib_name)
|
|
44
|
-
query =
|
|
69
|
+
query = {
|
|
45
70
|
"metadata.library_name": lib_name,
|
|
46
71
|
"metadata.fastq_type": "sdm_normal",
|
|
47
72
|
"group": "sdm",
|
|
48
|
-
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
cell = _check_id(smrt_cell_id, 'smrt_cell_id')
|
|
76
|
+
if cell is not None:
|
|
77
|
+
query["metadata.sdm_smrt_cell_id"] = cell
|
|
78
|
+
|
|
79
|
+
run = _check_id(physical_run_id, 'physical_run_id')
|
|
80
|
+
if run is not None:
|
|
81
|
+
query["metadata.pacbio_physical_run_id"] = run
|
|
82
|
+
|
|
83
|
+
query = json.dumps(query, separators=(',', ':'))
|
|
49
84
|
base = JAMO_CMD_DORI if os.getenv('SLURM_PARTITION') == 'dori' else JAMO_CMD_NERSC
|
|
50
85
|
return f"{base} custom '{query}'"
|
|
51
86
|
|
|
52
87
|
|
|
88
|
+
def _check_id(value, name):
|
|
89
|
+
"""
|
|
90
|
+
Normalise an optional integer query value, or raise.
|
|
91
|
+
|
|
92
|
+
Accepts an int or a string of digits (ids arrive from JSON APIs and from
|
|
93
|
+
argparse as either) and returns an int, or None when nothing was given.
|
|
94
|
+
Anything else raises: these values are interpolated into a shell command, so
|
|
95
|
+
a stray string must never reach it.
|
|
96
|
+
"""
|
|
97
|
+
if value is None:
|
|
98
|
+
return None
|
|
99
|
+
if isinstance(value, bool):
|
|
100
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
101
|
+
if isinstance(value, str):
|
|
102
|
+
stripped = value.strip()
|
|
103
|
+
if not stripped.isdigit():
|
|
104
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
105
|
+
value = int(stripped)
|
|
106
|
+
if not isinstance(value, int):
|
|
107
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
108
|
+
if value <= 0:
|
|
109
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
110
|
+
return value
|
|
111
|
+
|
|
112
|
+
|
|
53
113
|
def _clean_library_name(library_name):
|
|
54
114
|
clean_name = f"{library_name}".strip()
|
|
55
115
|
if not re.match(r'^[A-Z]+$', clean_name):
|
|
@@ -101,42 +161,68 @@ def _wait_for_seq(seq_path, wait_interval_secs, wait_max_secs):
|
|
|
101
161
|
raise RuntimeError(f"sequence for link {seq_path} not provisioned within max wait seconds: {wait_max}")
|
|
102
162
|
|
|
103
163
|
|
|
104
|
-
def
|
|
164
|
+
def _matching_links(lib_name, seq_dir):
|
|
105
165
|
"""
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
166
|
+
Return every symlink in seq_dir that jamo created for this library, as a
|
|
167
|
+
list of (link_path, target) sorted by link name.
|
|
168
|
+
|
|
169
|
+
jamo names each link '<library>.<file_name>', so the prefix is what
|
|
170
|
+
identifies ownership. Targets are read with os.readlink rather than
|
|
171
|
+
os.path.realpath: a link whose target has not been copied to this host yet
|
|
172
|
+
is still a real candidate, and realpath would obscure that.
|
|
109
173
|
"""
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
logging.debug(f"file paths: {file_paths}")
|
|
115
|
-
# get file stats
|
|
116
|
-
path_stats = [(path, os.lstat(path)) for path in file_paths]
|
|
117
|
-
path_stats.sort(key=lambda x: x[1][ST_CTIME])
|
|
118
|
-
for path_stat in path_stats:
|
|
119
|
-
logging.debug(f"{path_stat[0]}:")
|
|
120
|
-
for stat in sorted(filter(lambda a: a.startswith('st_'), dir(path_stat[1]))):
|
|
121
|
-
logging.debug(f" {stat}: {getattr(path_stat[1], stat)}")
|
|
122
|
-
|
|
123
|
-
matched_link = None
|
|
124
|
-
for path_stat in path_stats:
|
|
125
|
-
filepath = path_stat[0]
|
|
126
|
-
filetime = path_stat[1][ST_CTIME]
|
|
127
|
-
filename = os.path.basename(filepath)
|
|
128
|
-
if not os.path.islink(filepath):
|
|
174
|
+
links = []
|
|
175
|
+
for file_name in sorted(os.listdir(os.path.realpath(seq_dir))):
|
|
176
|
+
file_path = os.path.join(seq_dir, file_name)
|
|
177
|
+
if not os.path.islink(file_path):
|
|
129
178
|
continue
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
logging.debug(f"link match: {filename}")
|
|
179
|
+
if file_name.startswith(f"{lib_name}.") and file_name.endswith('.fastq.gz'):
|
|
180
|
+
links.append((file_path, os.readlink(file_path)))
|
|
181
|
+
return links
|
|
134
182
|
|
|
135
|
-
if matched_link is not None:
|
|
136
|
-
logging.debug(f"returning library seq file: {matched_link}")
|
|
137
|
-
return matched_link
|
|
138
183
|
|
|
139
|
-
|
|
184
|
+
def _find_fastq_path(lib_name, seq_dir):
|
|
185
|
+
"""
|
|
186
|
+
Return the one sequence symlink jamo created for this library.
|
|
187
|
+
|
|
188
|
+
This used to sort the directory by ctime and keep the last match, which was
|
|
189
|
+
a coin flip whenever a library matched more than one fastq: every link comes
|
|
190
|
+
from a single `jamo link` call within the same second, ST_CTIME is whole
|
|
191
|
+
seconds, so all the candidates tie and Python's stable sort just hands back
|
|
192
|
+
whichever os.listdir() happened to return last. That is directory hash order
|
|
193
|
+
on Linux - not alphabetical, not newest, and not the same on two hosts. A
|
|
194
|
+
library sequenced three times could therefore be analysed against any of the
|
|
195
|
+
three, silently, with a 2x spread in read count and nothing in the log to
|
|
196
|
+
say which.
|
|
197
|
+
|
|
198
|
+
So it no longer guesses. One candidate is returned; several distinct targets
|
|
199
|
+
raise, naming every one of them. Narrow the query instead - pass
|
|
200
|
+
smrt_cell_id or physical_run_id to get_cmd()/fetch_lib_seq() - or clear out
|
|
201
|
+
stale links from an earlier fetch.
|
|
202
|
+
|
|
203
|
+
Several links pointing at the SAME target are not ambiguous and are accepted:
|
|
204
|
+
SDM registers the same file under more than one path (observed for LBBCZRJ,
|
|
205
|
+
two AUTO- folders, identical md5).
|
|
206
|
+
"""
|
|
207
|
+
links = _matching_links(lib_name, seq_dir)
|
|
208
|
+
|
|
209
|
+
if not links:
|
|
210
|
+
raise RuntimeError(f"failed to find sequence file for library {lib_name}")
|
|
211
|
+
|
|
212
|
+
distinct_targets = {target for _, target in links}
|
|
213
|
+
if len(distinct_targets) > 1:
|
|
214
|
+
listing = "\n".join(
|
|
215
|
+
f" {os.path.basename(link)} -> {target}" for link, target in links
|
|
216
|
+
)
|
|
217
|
+
raise RuntimeError(
|
|
218
|
+
f"library {lib_name} matches {len(distinct_targets)} different "
|
|
219
|
+
f"sequence files in {seq_dir}; refusing to pick one:\n{listing}\n"
|
|
220
|
+
" Narrow the query with smrt_cell_id (preferred) or "
|
|
221
|
+
"physical_run_id, or remove links left over from an earlier fetch."
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
logging.debug(f"returning library seq file: {links[0][0]}")
|
|
225
|
+
return links[0][0]
|
|
140
226
|
|
|
141
227
|
|
|
142
228
|
class LibSeq:
|
|
@@ -190,33 +276,58 @@ class JamoFetcher():
|
|
|
190
276
|
self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
|
|
191
277
|
self._link_dir = os.path.realpath(link_dir)
|
|
192
278
|
|
|
193
|
-
def fetch_lib_seq(self, lib_name, out_file=sys.stderr
|
|
279
|
+
def fetch_lib_seq(self, lib_name, out_file=sys.stderr, smrt_cell_id=None,
|
|
280
|
+
physical_run_id=None) -> JamoLibSeq:
|
|
194
281
|
"""
|
|
195
282
|
Execute JAMO command to link sequence for library
|
|
196
283
|
and return a LibSeq object containing the path to the sequence.
|
|
284
|
+
|
|
285
|
+
smrt_cell_id (preferred) or physical_run_id narrow the query to one
|
|
286
|
+
sequencing of this library. Without one, a library sequenced more than
|
|
287
|
+
once links several files and _find_fastq_path() raises rather than pick
|
|
288
|
+
between them - see get_cmd().
|
|
289
|
+
|
|
290
|
+
They are per-call rather than per-fetcher because one JamoFetcher is
|
|
291
|
+
commonly shared across a pool's libraries, and each library has its own
|
|
292
|
+
cell id.
|
|
197
293
|
"""
|
|
198
294
|
pathlib.Path(self._link_dir).mkdir(parents=True, exist_ok=True, mode=0o775)
|
|
199
295
|
os.chdir(self._link_dir)
|
|
200
296
|
clean_lib_name = _clean_library_name(lib_name)
|
|
201
297
|
|
|
202
|
-
cmd = get_cmd(clean_lib_name
|
|
298
|
+
cmd = get_cmd(clean_lib_name, smrt_cell_id=smrt_cell_id,
|
|
299
|
+
physical_run_id=physical_run_id)
|
|
203
300
|
print(f"\n{cmd}", file=out_file)
|
|
204
301
|
output = subprocess.check_output(cmd, shell=True)
|
|
205
302
|
for line in output.splitlines():
|
|
206
303
|
print(line.decode("utf-8"), file=out_file)
|
|
304
|
+
|
|
305
|
+
# Report what was linked before choosing, so an ambiguous fetch is
|
|
306
|
+
# visible in the log even though the next line raises on it.
|
|
307
|
+
links = _matching_links(clean_lib_name, self._link_dir)
|
|
308
|
+
if len(links) > 1:
|
|
309
|
+
print(f"{clean_lib_name}: {len(links)} candidate links in "
|
|
310
|
+
f"{self._link_dir}", file=out_file)
|
|
311
|
+
for link, target in links:
|
|
312
|
+
print(f" {os.path.basename(link)} -> {target}", file=out_file)
|
|
313
|
+
|
|
207
314
|
seq_path = _find_fastq_path(clean_lib_name, self._link_dir)
|
|
208
315
|
print(f"{clean_lib_name} {seq_path}", file=out_file)
|
|
209
316
|
return JamoLibSeq(clean_lib_name, seq_path, self._wait_interval_secs, self._wait_max_secs)
|
|
210
317
|
|
|
211
318
|
|
|
212
|
-
def _fetch_seq(fetcher: JamoFetcher, libs):
|
|
319
|
+
def _fetch_seq(fetcher: JamoFetcher, libs, smrt_cell_id=None, physical_run_id=None):
|
|
213
320
|
lib_seq_dict = {}
|
|
214
321
|
for lib in libs:
|
|
215
322
|
try:
|
|
216
|
-
lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(
|
|
323
|
+
lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(
|
|
324
|
+
lib, out_file=sys.stderr, smrt_cell_id=smrt_cell_id,
|
|
325
|
+
physical_run_id=physical_run_id)
|
|
217
326
|
lib_seq_dict[lib_seq.get_lib_name()] = lib_seq
|
|
218
327
|
except Exception as e:
|
|
219
|
-
pass
|
|
328
|
+
# Was a bare `pass`, which hid every failure - including the
|
|
329
|
+
# ambiguous-match error, whose whole purpose is to be seen.
|
|
330
|
+
print(f"{lib}: {e}", file=sys.stderr)
|
|
220
331
|
return lib_seq_dict
|
|
221
332
|
|
|
222
333
|
|
|
@@ -229,12 +340,22 @@ def main(args):
|
|
|
229
340
|
wait_interval_secs = _check_int_not_negative(args.interval, "interval")
|
|
230
341
|
wait_max_secs = _check_int_not_negative(args.max, "max", allow_minus_1=True)
|
|
231
342
|
|
|
343
|
+
# A cell id identifies one library's membership in one pool, so it cannot be
|
|
344
|
+
# shared across several -l libraries. A run id can: a run holds many cells.
|
|
345
|
+
if args.smrt_cell_id is not None and len(args.library) > 1:
|
|
346
|
+
raise ValueError(
|
|
347
|
+
"--smrt-cell-id applies to a single library; it was given with "
|
|
348
|
+
f"{len(args.library)} -l options. Use --physical-run-id, or fetch "
|
|
349
|
+
"one library at a time.")
|
|
350
|
+
|
|
232
351
|
link_dir = os.path.realpath(args.directory if args.directory else '.')
|
|
233
352
|
fetcher: JamoFetcher = JamoFetcher(link_dir=link_dir, wait_interval_secs=wait_interval_secs,
|
|
234
353
|
wait_max_secs=wait_max_secs)
|
|
235
354
|
|
|
236
355
|
print("fetching sequence:")
|
|
237
|
-
lib_seq_dict = _fetch_seq(fetcher, args.library
|
|
356
|
+
lib_seq_dict = _fetch_seq(fetcher, args.library,
|
|
357
|
+
smrt_cell_id=args.smrt_cell_id,
|
|
358
|
+
physical_run_id=args.physical_run_id)
|
|
238
359
|
|
|
239
360
|
if not lib_seq_dict:
|
|
240
361
|
return # no sequence was fetched
|
|
@@ -273,6 +394,14 @@ def cli():
|
|
|
273
394
|
parser.add_argument('-d', '--directory', required=False, default='.',
|
|
274
395
|
help="directory where to link sequence, defaults to current directory. " +
|
|
275
396
|
"Directory will be created if it doesn't exit.")
|
|
397
|
+
parser.add_argument('--smrt-cell-id', required=False, type=int, default=None,
|
|
398
|
+
help="metadata.sdm_smrt_cell_id to fetch, for a library sequenced " +
|
|
399
|
+
"more than once. Identifies one library in one pool, so it " +
|
|
400
|
+
"applies to a single --library.")
|
|
401
|
+
parser.add_argument('--physical-run-id', required=False, type=int, default=None,
|
|
402
|
+
help="metadata.pacbio_physical_run_id to fetch, for a library " +
|
|
403
|
+
"sequenced more than once. Coarser than --smrt-cell-id; " +
|
|
404
|
+
"applies to every --library given.")
|
|
276
405
|
parser.add_argument('-i', '--interval', required=False, type=int, default=10,
|
|
277
406
|
help="wait interval in seconds to check if sequence has been fetched, " +
|
|
278
407
|
"ignored if wait flag not set")
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
import shlex
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from jamofetch import jamofetch
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _query(cmd):
|
|
11
|
+
# The query is the single-quoted final argument; shlex undoes the shell quoting.
|
|
12
|
+
return json.loads(shlex.split(cmd)[-1])
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_nersc_cmd(monkeypatch):
|
|
16
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
17
|
+
cmd = jamofetch.get_cmd('LBBDHFZ')
|
|
18
|
+
assert cmd.startswith('module load jamo; jamo link custom ')
|
|
19
|
+
assert _query(cmd) == {
|
|
20
|
+
"metadata.library_name": "LBBDHFZ",
|
|
21
|
+
"metadata.fastq_type": "sdm_normal",
|
|
22
|
+
"group": "sdm",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def test_dori_cmd_uses_exec(monkeypatch):
|
|
27
|
+
monkeypatch.setenv('SLURM_PARTITION', 'dori')
|
|
28
|
+
cmd = jamofetch.get_cmd('LBBDHFZ')
|
|
29
|
+
assert cmd.startswith('apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom ')
|
|
30
|
+
assert _query(cmd)["metadata.library_name"] == "LBBDHFZ"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_library_name_is_first_query_key(monkeypatch):
|
|
34
|
+
# jamo names the symlink after the first string-valued key; _find_fastq_path
|
|
35
|
+
# relies on that being the library name.
|
|
36
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
37
|
+
assert next(iter(_query(jamofetch.get_cmd('LBBDHFZ')))) == "metadata.library_name"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_query_does_not_filter_on_user(monkeypatch):
|
|
41
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
42
|
+
assert "user" not in _query(jamofetch.get_cmd('LBBDHFZ'))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@pytest.mark.parametrize('bad', ["LBB'DHFZ", 'lbbdhfz', 'LBB DHFZ', 'LBB;rm'])
|
|
46
|
+
def test_invalid_library_name_rejected(bad):
|
|
47
|
+
with pytest.raises(ValueError):
|
|
48
|
+
jamofetch.get_cmd(bad)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# --- disambiguating a library sequenced more than once -----------------------
|
|
52
|
+
#
|
|
53
|
+
# Real case this exists for, library LBBCZRJ: four records matching the plain
|
|
54
|
+
# query, across three pools -- cell 37397 (run 3207), cell 37892 (run 3232,
|
|
55
|
+
# registered twice with identical md5s) and cell 37964 (run 3230). Read counts
|
|
56
|
+
# 1,296,169 / 1,965,608 / 948,348. Before 3.7.9 the one analysed was whichever
|
|
57
|
+
# os.listdir() returned last.
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def test_smrt_cell_id_narrows_query(monkeypatch):
|
|
61
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
62
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
|
|
63
|
+
assert query == {
|
|
64
|
+
"metadata.library_name": "LBBCZRJ",
|
|
65
|
+
"metadata.fastq_type": "sdm_normal",
|
|
66
|
+
"group": "sdm",
|
|
67
|
+
"metadata.sdm_smrt_cell_id": 37892,
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_physical_run_id_narrows_query(monkeypatch):
|
|
72
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
73
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', physical_run_id=3232))
|
|
74
|
+
assert query["metadata.pacbio_physical_run_id"] == 3232
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def test_both_ids_may_be_given(monkeypatch):
|
|
78
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
79
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892, physical_run_id=3232))
|
|
80
|
+
assert query["metadata.sdm_smrt_cell_id"] == 37892
|
|
81
|
+
assert query["metadata.pacbio_physical_run_id"] == 3232
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_library_name_still_first_key_with_ids(monkeypatch):
|
|
85
|
+
# jamo names the symlink after the first string-valued key. The ids are
|
|
86
|
+
# ints so they cannot take that slot, but the ordering is load-bearing
|
|
87
|
+
# enough to assert directly.
|
|
88
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
89
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
|
|
90
|
+
assert next(iter(query)) == "metadata.library_name"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_no_ids_produces_the_3_7_8_command(monkeypatch):
|
|
94
|
+
# Backward compatibility: callers that pass no id must get the byte-identical
|
|
95
|
+
# command 3.7.8 produced. synbioqc-pbj pins these strings in its own tests.
|
|
96
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
97
|
+
assert jamofetch.get_cmd('LBBDHFZ') == (
|
|
98
|
+
"module load jamo; jamo link custom "
|
|
99
|
+
'\'{"metadata.library_name":"LBBDHFZ","metadata.fastq_type":"sdm_normal","group":"sdm"}\''
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
@pytest.mark.parametrize('bad', ['37892; rm -rf /', 'abc', '', -1, 0, 3.5, True, [37892]])
|
|
104
|
+
def test_invalid_ids_rejected(bad):
|
|
105
|
+
with pytest.raises(ValueError):
|
|
106
|
+
jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=bad)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def test_numeric_string_id_accepted(monkeypatch):
|
|
110
|
+
# Ids arrive from JSON APIs and argparse as either int or str.
|
|
111
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
112
|
+
assert _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id='37892'))[
|
|
113
|
+
"metadata.sdm_smrt_cell_id"] == 37892
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# --- _find_fastq_path no longer guesses --------------------------------------
|
|
117
|
+
|
|
118
|
+
LBBCZRJ_TARGETS = {
|
|
119
|
+
"pbio-3207.37397.bc2080_OA--bc2080_OA.hifi_reads.bc2080_OA.ccs.fastq.gz":
|
|
120
|
+
"/global/dna/dm_archive/sdm/pacbio/00/32/07",
|
|
121
|
+
"pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz":
|
|
122
|
+
"/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306",
|
|
123
|
+
"pbr6186.3230.37667.37964.bc2080_OA--bc2080_OA.pacbio.processed_well.37667.37964.fastq.gz":
|
|
124
|
+
"/global/dna/dm_archive/sdm/analyses-242/AUTO-2421406",
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _link(seq_dir, lib, file_name, folder):
|
|
129
|
+
link = os.path.join(seq_dir, f"{lib}.{file_name}")
|
|
130
|
+
os.symlink(os.path.join(folder, file_name), link)
|
|
131
|
+
return link
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def test_single_match_is_returned(tmp_path):
|
|
135
|
+
name, folder = next(iter(LBBCZRJ_TARGETS.items()))
|
|
136
|
+
_link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
137
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def test_ambiguous_match_raises_and_names_every_candidate(tmp_path):
|
|
141
|
+
for name, folder in LBBCZRJ_TARGETS.items():
|
|
142
|
+
_link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
143
|
+
|
|
144
|
+
with pytest.raises(RuntimeError) as excinfo:
|
|
145
|
+
jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
|
|
146
|
+
|
|
147
|
+
message = str(excinfo.value)
|
|
148
|
+
assert "3 different" in message
|
|
149
|
+
for name, folder in LBBCZRJ_TARGETS.items():
|
|
150
|
+
assert os.path.join(folder, name) in message
|
|
151
|
+
assert "smrt_cell_id" in message
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def test_duplicate_registrations_of_one_file_are_not_ambiguous(tmp_path):
|
|
155
|
+
# SDM registered the same file under two AUTO- folders with identical md5s.
|
|
156
|
+
# Same file_name means one link name, so only one link can exist -- but if
|
|
157
|
+
# two links ever point at the same target, that is not a real choice.
|
|
158
|
+
name = "pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz"
|
|
159
|
+
target = "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306/" + name
|
|
160
|
+
os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.{name}"))
|
|
161
|
+
os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.copy.{name}"))
|
|
162
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) is not None
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_other_libraries_links_are_ignored(tmp_path):
|
|
166
|
+
name, folder = next(iter(LBBCZRJ_TARGETS.items()))
|
|
167
|
+
_link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
168
|
+
for other_name, other_folder in list(LBBCZRJ_TARGETS.items())[1:]:
|
|
169
|
+
_link(str(tmp_path), 'LBBDHFZ', other_name, other_folder)
|
|
170
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def test_no_match_raises(tmp_path):
|
|
174
|
+
with pytest.raises(RuntimeError, match="failed to find sequence file"):
|
|
175
|
+
jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def test_broken_link_is_still_a_candidate(tmp_path):
|
|
179
|
+
# On dori the target is copied into a local cache and the link is broken
|
|
180
|
+
# until that lands; it is still the file this library resolved to.
|
|
181
|
+
name, folder = next(iter(LBBCZRJ_TARGETS.items()))
|
|
182
|
+
link = _link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
183
|
+
assert not os.path.exists(link)
|
|
184
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) == link
|
jamofetch-3.7.8/CHANGELOG.md
DELETED
|
@@ -1,47 +0,0 @@
|
|
|
1
|
-
import json
|
|
2
|
-
import shlex
|
|
3
|
-
|
|
4
|
-
import pytest
|
|
5
|
-
|
|
6
|
-
from jamofetch import jamofetch
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
def _query(cmd):
|
|
10
|
-
# The query is the single-quoted final argument; shlex undoes the shell quoting.
|
|
11
|
-
return json.loads(shlex.split(cmd)[-1])
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
def test_nersc_cmd(monkeypatch):
|
|
15
|
-
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
16
|
-
cmd = jamofetch.get_cmd('LBBDHFZ')
|
|
17
|
-
assert cmd.startswith('module load jamo; jamo link custom ')
|
|
18
|
-
assert _query(cmd) == {
|
|
19
|
-
"metadata.library_name": "LBBDHFZ",
|
|
20
|
-
"metadata.fastq_type": "sdm_normal",
|
|
21
|
-
"group": "sdm",
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
def test_dori_cmd_uses_exec(monkeypatch):
|
|
26
|
-
monkeypatch.setenv('SLURM_PARTITION', 'dori')
|
|
27
|
-
cmd = jamofetch.get_cmd('LBBDHFZ')
|
|
28
|
-
assert cmd.startswith('apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom ')
|
|
29
|
-
assert _query(cmd)["metadata.library_name"] == "LBBDHFZ"
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
def test_library_name_is_first_query_key(monkeypatch):
|
|
33
|
-
# jamo names the symlink after the first string-valued key; _find_fastq_path
|
|
34
|
-
# relies on that being the library name.
|
|
35
|
-
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
36
|
-
assert next(iter(_query(jamofetch.get_cmd('LBBDHFZ')))) == "metadata.library_name"
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
def test_query_does_not_filter_on_user(monkeypatch):
|
|
40
|
-
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
41
|
-
assert "user" not in _query(jamofetch.get_cmd('LBBDHFZ'))
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
@pytest.mark.parametrize('bad', ["LBB'DHFZ", 'lbbdhfz', 'LBB DHFZ', 'LBB;rm'])
|
|
45
|
-
def test_invalid_library_name_rejected(bad):
|
|
46
|
-
with pytest.raises(ValueError):
|
|
47
|
-
jamofetch.get_cmd(bad)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|