jamofetch 3.7.8__tar.gz → 3.7.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,53 @@
1
+ # Changelog
2
+
3
+ <!--next-version-placeholder-->
4
+
5
+ ## v3.7.9 (17/09/2026)
6
+
7
+ ### Fixed
8
+
9
+ - **A library sequenced more than once no longer resolves to an arbitrary file.**
10
+ `_find_fastq_path()` sorted the link directory by `ST_CTIME` and kept the last
11
+ match. Every link comes from one `jamo link` call within the same second and
12
+ `ST_CTIME` is whole seconds, so the candidates all tie and Python's stable sort
13
+ returns whichever `os.listdir()` listed last - directory hash order on Linux,
14
+ so not alphabetical, not newest, and not necessarily the same on two hosts.
15
+ It now returns the single match, or raises and names every candidate.
16
+
17
+ Observed on library LBBCZRJ: four matching records across three pools - cell
18
+ 37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608 reads,
19
+ registered twice under different `AUTO-` folders with identical md5s) and cell
20
+ 37964 (run 3230, 948,348 reads). Any of the three could be analysed, silently,
21
+ with a 2x spread in read count.
22
+
23
+ Links pointing at the same target are not ambiguous and are still accepted.
24
+
25
+ ### Added
26
+
27
+ - `get_cmd()` and `JamoFetcher.fetch_lib_seq()` take `smrt_cell_id` and
28
+ `physical_run_id`, which narrow the query to one sequencing of the library.
29
+ `smrt_cell_id` is the one to prefer: it identifies a library's membership in a
30
+ single pool, and equals the PacBio pipeline service's `libraries[].id`. For
31
+ LBBCZRJ it reduces four matches to one (or, for cell 37892, to two records
32
+ that are byte-identical and share a `file_name`).
33
+ - CLI: `--smrt-cell-id` (single `--library` only) and `--physical-run-id`
34
+ (applies to every `--library`).
35
+
36
+ ### Changed
37
+
38
+ - The CLI no longer swallows per-library fetch errors. `_fetch_seq()` caught
39
+ every exception with a bare `pass`, which hid the new ambiguity error whose
40
+ whole purpose is to be seen.
41
+ - `JamoFetcher.fetch_lib_seq()` logs every candidate link when more than one
42
+ matches, before selection.
43
+
44
+ ### Compatibility
45
+
46
+ - Calls that pass no id produce the byte-identical command 3.7.8 produced;
47
+ asserted in `tests/test_jamofetch.py`. Callers that relied on a library with
48
+ several matches silently resolving to one of them now get a `RuntimeError`
49
+ instead - that is the point of the release.
50
+
51
+ ## v0.1.0 (07/07/2023)
52
+
53
+ - First release of `jamofetch`!
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: jamofetch
3
- Version: 3.7.8
3
+ Version: 3.7.9
4
4
  Summary: A thin wrapper to retrieve sequence from JAMO at NERSC and on Dori.
5
5
  Author: Duncan Scott
6
6
  Requires-Python: >=3.9
@@ -4,7 +4,7 @@ build-backend = "flit_core.buildapi"
4
4
 
5
5
  [project]
6
6
  name = "jamofetch"
7
- version = "3.7.8"
7
+ version = "3.7.9"
8
8
  description = "A thin wrapper to retrieve sequence from JAMO at NERSC and on Dori."
9
9
  authors = [{ name = "Duncan Scott" }]
10
10
  readme = "README.md"
@@ -9,7 +9,6 @@ import subprocess
9
9
  import sys
10
10
  import time
11
11
  from argparse import ArgumentParser
12
- from stat import ST_CTIME
13
12
 
14
13
  # Base commands; get_cmd() appends the query.
15
14
  #
@@ -22,7 +21,7 @@ TWO_HOURS = 7200 # seconds
22
21
  ONE_MINUTE = 60 # seconds
23
22
 
24
23
 
25
- def get_cmd(clean_lib_name) -> str:
24
+ def get_cmd(clean_lib_name, smrt_cell_id=None, physical_run_id=None) -> str:
26
25
  """
27
26
  Build the site-appropriate `jamo link` command for a library's SDM fastq.
28
27
 
@@ -34,22 +33,83 @@ def get_cmd(clean_lib_name) -> str:
34
33
 
35
34
  Key order matters: jamo names each symlink '<first string-valued query key's
36
35
  value>.<file_name>', so metadata.library_name must stay first or
37
- _find_fastq_path() will not find the link.
38
-
39
- A library sequenced on more than one run matches one fastq per run; nothing
40
- here chooses between them.
36
+ _find_fastq_path() will not find the link. The disambiguators below are
37
+ integers, so they cannot take that position, but they are appended last
38
+ anyway.
39
+
40
+ A library sequenced on more than one run matches one fastq per run, and
41
+ without a disambiguator NOTHING here chooses between them - the caller ends
42
+ up with several symlinks and _find_fastq_path() refuses to guess. Pass
43
+ smrt_cell_id (preferred) or physical_run_id to narrow the query to the
44
+ sequencing this analysis is actually about.
45
+
46
+ Observed case, library LBBCZRJ: four matching records across three pools -
47
+ cell 37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608
48
+ reads, registered twice under different AUTO- folders with identical md5s)
49
+ and cell 37964 (run 3230, 948,348 reads). Adding the cell id returns exactly
50
+ one record in the first and third cases, and in the second the two records
51
+ are byte-identical and share a file_name, so they collapse to one symlink.
52
+
53
+ smrt_cell_id is the finer key and is the one to prefer: it identifies a
54
+ library's membership in one specific pool. It is JAMO's
55
+ metadata.sdm_smrt_cell_id, and equals the PacBio pipeline service's
56
+ libraries[].id for the same library.
57
+
58
+ Args:
59
+ clean_lib_name: the library name, e.g. 'LBBCZRJ'.
60
+ smrt_cell_id: optional metadata.sdm_smrt_cell_id to narrow to.
61
+ physical_run_id: optional metadata.pacbio_physical_run_id to narrow to.
62
+ Coarser than smrt_cell_id; a run holds many cells.
63
+
64
+ Returns:
65
+ The full shell command, with the JSON query single-quoted.
41
66
  """
42
67
  # Validated because the name is interpolated into a shell command.
43
68
  lib_name = _clean_library_name(clean_lib_name)
44
- query = json.dumps({
69
+ query = {
45
70
  "metadata.library_name": lib_name,
46
71
  "metadata.fastq_type": "sdm_normal",
47
72
  "group": "sdm",
48
- }, separators=(',', ':'))
73
+ }
74
+
75
+ cell = _check_id(smrt_cell_id, 'smrt_cell_id')
76
+ if cell is not None:
77
+ query["metadata.sdm_smrt_cell_id"] = cell
78
+
79
+ run = _check_id(physical_run_id, 'physical_run_id')
80
+ if run is not None:
81
+ query["metadata.pacbio_physical_run_id"] = run
82
+
83
+ query = json.dumps(query, separators=(',', ':'))
49
84
  base = JAMO_CMD_DORI if os.getenv('SLURM_PARTITION') == 'dori' else JAMO_CMD_NERSC
50
85
  return f"{base} custom '{query}'"
51
86
 
52
87
 
88
+ def _check_id(value, name):
89
+ """
90
+ Normalise an optional integer query value, or raise.
91
+
92
+ Accepts an int or a string of digits (ids arrive from JSON APIs and from
93
+ argparse as either) and returns an int, or None when nothing was given.
94
+ Anything else raises: these values are interpolated into a shell command, so
95
+ a stray string must never reach it.
96
+ """
97
+ if value is None:
98
+ return None
99
+ if isinstance(value, bool):
100
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
101
+ if isinstance(value, str):
102
+ stripped = value.strip()
103
+ if not stripped.isdigit():
104
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
105
+ value = int(stripped)
106
+ if not isinstance(value, int):
107
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
108
+ if value <= 0:
109
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
110
+ return value
111
+
112
+
53
113
  def _clean_library_name(library_name):
54
114
  clean_name = f"{library_name}".strip()
55
115
  if not re.match(r'^[A-Z]+$', clean_name):
@@ -101,42 +161,68 @@ def _wait_for_seq(seq_path, wait_interval_secs, wait_max_secs):
101
161
  raise RuntimeError(f"sequence for link {seq_path} not provisioned within max wait seconds: {wait_max}")
102
162
 
103
163
 
104
- def _find_fastq_path(lib_name, seq_dir):
164
+ def _matching_links(lib_name, seq_dir):
105
165
  """
106
- Find symlink created for library in sequence directory
107
- by matching pattern similar to:
108
- HHCHO.52485.1.359430.GTGCTTA-GTAAGCA.fastq.gz
166
+ Return every symlink in seq_dir that jamo created for this library, as a
167
+ list of (link_path, target) sorted by link name.
168
+
169
+ jamo names each link '<library>.<file_name>', so the prefix is what
170
+ identifies ownership. Targets are read with os.readlink rather than
171
+ os.path.realpath: a link whose target has not been copied to this host yet
172
+ is still a real candidate, and realpath would obscure that.
109
173
  """
110
- # sort contents in directory by modification time (we want the last modified)
111
- # adapted from:
112
- # https://www.tutorialspoint.com/How-do-you-get-a-directory-listing-sorted-by-creation-date-in-Python
113
- file_paths = [os.path.join(seq_dir, file_name) for file_name in os.listdir(os.path.realpath(seq_dir))]
114
- logging.debug(f"file paths: {file_paths}")
115
- # get file stats
116
- path_stats = [(path, os.lstat(path)) for path in file_paths]
117
- path_stats.sort(key=lambda x: x[1][ST_CTIME])
118
- for path_stat in path_stats:
119
- logging.debug(f"{path_stat[0]}:")
120
- for stat in sorted(filter(lambda a: a.startswith('st_'), dir(path_stat[1]))):
121
- logging.debug(f" {stat}: {getattr(path_stat[1], stat)}")
122
-
123
- matched_link = None
124
- for path_stat in path_stats:
125
- filepath = path_stat[0]
126
- filetime = path_stat[1][ST_CTIME]
127
- filename = os.path.basename(filepath)
128
- if not os.path.islink(filepath):
174
+ links = []
175
+ for file_name in sorted(os.listdir(os.path.realpath(seq_dir))):
176
+ file_path = os.path.join(seq_dir, file_name)
177
+ if not os.path.islink(file_path):
129
178
  continue
130
- logging.debug(f"link candidate: {filename}; creation time: {filetime}")
131
- if str(filename).startswith(f"{lib_name}.") and str(filename).endswith('.fastq.gz'):
132
- matched_link = filepath
133
- logging.debug(f"link match: {filename}")
179
+ if file_name.startswith(f"{lib_name}.") and file_name.endswith('.fastq.gz'):
180
+ links.append((file_path, os.readlink(file_path)))
181
+ return links
134
182
 
135
- if matched_link is not None:
136
- logging.debug(f"returning library seq file: {matched_link}")
137
- return matched_link
138
183
 
139
- raise RuntimeError(f"failed to find sequence file for library {lib_name}")
184
+ def _find_fastq_path(lib_name, seq_dir):
185
+ """
186
+ Return the one sequence symlink jamo created for this library.
187
+
188
+ This used to sort the directory by ctime and keep the last match, which was
189
+ a coin flip whenever a library matched more than one fastq: every link comes
190
+ from a single `jamo link` call within the same second, ST_CTIME is whole
191
+ seconds, so all the candidates tie and Python's stable sort just hands back
192
+ whichever os.listdir() happened to return last. That is directory hash order
193
+ on Linux - not alphabetical, not newest, and not the same on two hosts. A
194
+ library sequenced three times could therefore be analysed against any of the
195
+ three, silently, with a 2x spread in read count and nothing in the log to
196
+ say which.
197
+
198
+ So it no longer guesses. One candidate is returned; several distinct targets
199
+ raise, naming every one of them. Narrow the query instead - pass
200
+ smrt_cell_id or physical_run_id to get_cmd()/fetch_lib_seq() - or clear out
201
+ stale links from an earlier fetch.
202
+
203
+ Several links pointing at the SAME target are not ambiguous and are accepted:
204
+ SDM registers the same file under more than one path (observed for LBBCZRJ,
205
+ two AUTO- folders, identical md5).
206
+ """
207
+ links = _matching_links(lib_name, seq_dir)
208
+
209
+ if not links:
210
+ raise RuntimeError(f"failed to find sequence file for library {lib_name}")
211
+
212
+ distinct_targets = {target for _, target in links}
213
+ if len(distinct_targets) > 1:
214
+ listing = "\n".join(
215
+ f" {os.path.basename(link)} -> {target}" for link, target in links
216
+ )
217
+ raise RuntimeError(
218
+ f"library {lib_name} matches {len(distinct_targets)} different "
219
+ f"sequence files in {seq_dir}; refusing to pick one:\n{listing}\n"
220
+ " Narrow the query with smrt_cell_id (preferred) or "
221
+ "physical_run_id, or remove links left over from an earlier fetch."
222
+ )
223
+
224
+ logging.debug(f"returning library seq file: {links[0][0]}")
225
+ return links[0][0]
140
226
 
141
227
 
142
228
  class LibSeq:
@@ -190,33 +276,58 @@ class JamoFetcher():
190
276
  self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
191
277
  self._link_dir = os.path.realpath(link_dir)
192
278
 
193
- def fetch_lib_seq(self, lib_name, out_file=sys.stderr) -> JamoLibSeq:
279
+ def fetch_lib_seq(self, lib_name, out_file=sys.stderr, smrt_cell_id=None,
280
+ physical_run_id=None) -> JamoLibSeq:
194
281
  """
195
282
  Execute JAMO command to link sequence for library
196
283
  and return a LibSeq object containing the path to the sequence.
284
+
285
+ smrt_cell_id (preferred) or physical_run_id narrow the query to one
286
+ sequencing of this library. Without one, a library sequenced more than
287
+ once links several files and _find_fastq_path() raises rather than pick
288
+ between them - see get_cmd().
289
+
290
+ They are per-call rather than per-fetcher because one JamoFetcher is
291
+ commonly shared across a pool's libraries, and each library has its own
292
+ cell id.
197
293
  """
198
294
  pathlib.Path(self._link_dir).mkdir(parents=True, exist_ok=True, mode=0o775)
199
295
  os.chdir(self._link_dir)
200
296
  clean_lib_name = _clean_library_name(lib_name)
201
297
 
202
- cmd = get_cmd(clean_lib_name)
298
+ cmd = get_cmd(clean_lib_name, smrt_cell_id=smrt_cell_id,
299
+ physical_run_id=physical_run_id)
203
300
  print(f"\n{cmd}", file=out_file)
204
301
  output = subprocess.check_output(cmd, shell=True)
205
302
  for line in output.splitlines():
206
303
  print(line.decode("utf-8"), file=out_file)
304
+
305
+ # Report what was linked before choosing, so an ambiguous fetch is
306
+ # visible in the log even though the next line raises on it.
307
+ links = _matching_links(clean_lib_name, self._link_dir)
308
+ if len(links) > 1:
309
+ print(f"{clean_lib_name}: {len(links)} candidate links in "
310
+ f"{self._link_dir}", file=out_file)
311
+ for link, target in links:
312
+ print(f" {os.path.basename(link)} -> {target}", file=out_file)
313
+
207
314
  seq_path = _find_fastq_path(clean_lib_name, self._link_dir)
208
315
  print(f"{clean_lib_name} {seq_path}", file=out_file)
209
316
  return JamoLibSeq(clean_lib_name, seq_path, self._wait_interval_secs, self._wait_max_secs)
210
317
 
211
318
 
212
- def _fetch_seq(fetcher: JamoFetcher, libs):
319
+ def _fetch_seq(fetcher: JamoFetcher, libs, smrt_cell_id=None, physical_run_id=None):
213
320
  lib_seq_dict = {}
214
321
  for lib in libs:
215
322
  try:
216
- lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(lib, out_file=sys.stderr)
323
+ lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(
324
+ lib, out_file=sys.stderr, smrt_cell_id=smrt_cell_id,
325
+ physical_run_id=physical_run_id)
217
326
  lib_seq_dict[lib_seq.get_lib_name()] = lib_seq
218
327
  except Exception as e:
219
- pass
328
+ # Was a bare `pass`, which hid every failure - including the
329
+ # ambiguous-match error, whose whole purpose is to be seen.
330
+ print(f"{lib}: {e}", file=sys.stderr)
220
331
  return lib_seq_dict
221
332
 
222
333
 
@@ -229,12 +340,22 @@ def main(args):
229
340
  wait_interval_secs = _check_int_not_negative(args.interval, "interval")
230
341
  wait_max_secs = _check_int_not_negative(args.max, "max", allow_minus_1=True)
231
342
 
343
+ # A cell id identifies one library's membership in one pool, so it cannot be
344
+ # shared across several -l libraries. A run id can: a run holds many cells.
345
+ if args.smrt_cell_id is not None and len(args.library) > 1:
346
+ raise ValueError(
347
+ "--smrt-cell-id applies to a single library; it was given with "
348
+ f"{len(args.library)} -l options. Use --physical-run-id, or fetch "
349
+ "one library at a time.")
350
+
232
351
  link_dir = os.path.realpath(args.directory if args.directory else '.')
233
352
  fetcher: JamoFetcher = JamoFetcher(link_dir=link_dir, wait_interval_secs=wait_interval_secs,
234
353
  wait_max_secs=wait_max_secs)
235
354
 
236
355
  print("fetching sequence:")
237
- lib_seq_dict = _fetch_seq(fetcher, args.library)
356
+ lib_seq_dict = _fetch_seq(fetcher, args.library,
357
+ smrt_cell_id=args.smrt_cell_id,
358
+ physical_run_id=args.physical_run_id)
238
359
 
239
360
  if not lib_seq_dict:
240
361
  return # no sequence was fetched
@@ -273,6 +394,14 @@ def cli():
273
394
  parser.add_argument('-d', '--directory', required=False, default='.',
274
395
  help="directory where to link sequence, defaults to current directory. " +
275
396
  "Directory will be created if it doesn't exit.")
397
+ parser.add_argument('--smrt-cell-id', required=False, type=int, default=None,
398
+ help="metadata.sdm_smrt_cell_id to fetch, for a library sequenced " +
399
+ "more than once. Identifies one library in one pool, so it " +
400
+ "applies to a single --library.")
401
+ parser.add_argument('--physical-run-id', required=False, type=int, default=None,
402
+ help="metadata.pacbio_physical_run_id to fetch, for a library " +
403
+ "sequenced more than once. Coarser than --smrt-cell-id; " +
404
+ "applies to every --library given.")
276
405
  parser.add_argument('-i', '--interval', required=False, type=int, default=10,
277
406
  help="wait interval in seconds to check if sequence has been fetched, " +
278
407
  "ignored if wait flag not set")
@@ -0,0 +1,184 @@
1
+ import json
2
+ import os
3
+ import shlex
4
+
5
+ import pytest
6
+
7
+ from jamofetch import jamofetch
8
+
9
+
10
+ def _query(cmd):
11
+ # The query is the single-quoted final argument; shlex undoes the shell quoting.
12
+ return json.loads(shlex.split(cmd)[-1])
13
+
14
+
15
+ def test_nersc_cmd(monkeypatch):
16
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
17
+ cmd = jamofetch.get_cmd('LBBDHFZ')
18
+ assert cmd.startswith('module load jamo; jamo link custom ')
19
+ assert _query(cmd) == {
20
+ "metadata.library_name": "LBBDHFZ",
21
+ "metadata.fastq_type": "sdm_normal",
22
+ "group": "sdm",
23
+ }
24
+
25
+
26
+ def test_dori_cmd_uses_exec(monkeypatch):
27
+ monkeypatch.setenv('SLURM_PARTITION', 'dori')
28
+ cmd = jamofetch.get_cmd('LBBDHFZ')
29
+ assert cmd.startswith('apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom ')
30
+ assert _query(cmd)["metadata.library_name"] == "LBBDHFZ"
31
+
32
+
33
+ def test_library_name_is_first_query_key(monkeypatch):
34
+ # jamo names the symlink after the first string-valued key; _find_fastq_path
35
+ # relies on that being the library name.
36
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
37
+ assert next(iter(_query(jamofetch.get_cmd('LBBDHFZ')))) == "metadata.library_name"
38
+
39
+
40
+ def test_query_does_not_filter_on_user(monkeypatch):
41
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
42
+ assert "user" not in _query(jamofetch.get_cmd('LBBDHFZ'))
43
+
44
+
45
+ @pytest.mark.parametrize('bad', ["LBB'DHFZ", 'lbbdhfz', 'LBB DHFZ', 'LBB;rm'])
46
+ def test_invalid_library_name_rejected(bad):
47
+ with pytest.raises(ValueError):
48
+ jamofetch.get_cmd(bad)
49
+
50
+
51
+ # --- disambiguating a library sequenced more than once -----------------------
52
+ #
53
+ # Real case this exists for, library LBBCZRJ: four records matching the plain
54
+ # query, across three pools -- cell 37397 (run 3207), cell 37892 (run 3232,
55
+ # registered twice with identical md5s) and cell 37964 (run 3230). Read counts
56
+ # 1,296,169 / 1,965,608 / 948,348. Before 3.7.9 the one analysed was whichever
57
+ # os.listdir() returned last.
58
+
59
+
60
+ def test_smrt_cell_id_narrows_query(monkeypatch):
61
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
62
+ query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
63
+ assert query == {
64
+ "metadata.library_name": "LBBCZRJ",
65
+ "metadata.fastq_type": "sdm_normal",
66
+ "group": "sdm",
67
+ "metadata.sdm_smrt_cell_id": 37892,
68
+ }
69
+
70
+
71
+ def test_physical_run_id_narrows_query(monkeypatch):
72
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
73
+ query = _query(jamofetch.get_cmd('LBBCZRJ', physical_run_id=3232))
74
+ assert query["metadata.pacbio_physical_run_id"] == 3232
75
+
76
+
77
+ def test_both_ids_may_be_given(monkeypatch):
78
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
79
+ query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892, physical_run_id=3232))
80
+ assert query["metadata.sdm_smrt_cell_id"] == 37892
81
+ assert query["metadata.pacbio_physical_run_id"] == 3232
82
+
83
+
84
+ def test_library_name_still_first_key_with_ids(monkeypatch):
85
+ # jamo names the symlink after the first string-valued key. The ids are
86
+ # ints so they cannot take that slot, but the ordering is load-bearing
87
+ # enough to assert directly.
88
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
89
+ query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
90
+ assert next(iter(query)) == "metadata.library_name"
91
+
92
+
93
+ def test_no_ids_produces_the_3_7_8_command(monkeypatch):
94
+ # Backward compatibility: callers that pass no id must get the byte-identical
95
+ # command 3.7.8 produced. synbioqc-pbj pins these strings in its own tests.
96
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
97
+ assert jamofetch.get_cmd('LBBDHFZ') == (
98
+ "module load jamo; jamo link custom "
99
+ '\'{"metadata.library_name":"LBBDHFZ","metadata.fastq_type":"sdm_normal","group":"sdm"}\''
100
+ )
101
+
102
+
103
+ @pytest.mark.parametrize('bad', ['37892; rm -rf /', 'abc', '', -1, 0, 3.5, True, [37892]])
104
+ def test_invalid_ids_rejected(bad):
105
+ with pytest.raises(ValueError):
106
+ jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=bad)
107
+
108
+
109
+ def test_numeric_string_id_accepted(monkeypatch):
110
+ # Ids arrive from JSON APIs and argparse as either int or str.
111
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
112
+ assert _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id='37892'))[
113
+ "metadata.sdm_smrt_cell_id"] == 37892
114
+
115
+
116
+ # --- _find_fastq_path no longer guesses --------------------------------------
117
+
118
+ LBBCZRJ_TARGETS = {
119
+ "pbio-3207.37397.bc2080_OA--bc2080_OA.hifi_reads.bc2080_OA.ccs.fastq.gz":
120
+ "/global/dna/dm_archive/sdm/pacbio/00/32/07",
121
+ "pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz":
122
+ "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306",
123
+ "pbr6186.3230.37667.37964.bc2080_OA--bc2080_OA.pacbio.processed_well.37667.37964.fastq.gz":
124
+ "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421406",
125
+ }
126
+
127
+
128
+ def _link(seq_dir, lib, file_name, folder):
129
+ link = os.path.join(seq_dir, f"{lib}.{file_name}")
130
+ os.symlink(os.path.join(folder, file_name), link)
131
+ return link
132
+
133
+
134
+ def test_single_match_is_returned(tmp_path):
135
+ name, folder = next(iter(LBBCZRJ_TARGETS.items()))
136
+ _link(str(tmp_path), 'LBBCZRJ', name, folder)
137
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
138
+
139
+
140
+ def test_ambiguous_match_raises_and_names_every_candidate(tmp_path):
141
+ for name, folder in LBBCZRJ_TARGETS.items():
142
+ _link(str(tmp_path), 'LBBCZRJ', name, folder)
143
+
144
+ with pytest.raises(RuntimeError) as excinfo:
145
+ jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
146
+
147
+ message = str(excinfo.value)
148
+ assert "3 different" in message
149
+ for name, folder in LBBCZRJ_TARGETS.items():
150
+ assert os.path.join(folder, name) in message
151
+ assert "smrt_cell_id" in message
152
+
153
+
154
+ def test_duplicate_registrations_of_one_file_are_not_ambiguous(tmp_path):
155
+ # SDM registered the same file under two AUTO- folders with identical md5s.
156
+ # Same file_name means one link name, so only one link can exist -- but if
157
+ # two links ever point at the same target, that is not a real choice.
158
+ name = "pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz"
159
+ target = "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306/" + name
160
+ os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.{name}"))
161
+ os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.copy.{name}"))
162
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) is not None
163
+
164
+
165
+ def test_other_libraries_links_are_ignored(tmp_path):
166
+ name, folder = next(iter(LBBCZRJ_TARGETS.items()))
167
+ _link(str(tmp_path), 'LBBCZRJ', name, folder)
168
+ for other_name, other_folder in list(LBBCZRJ_TARGETS.items())[1:]:
169
+ _link(str(tmp_path), 'LBBDHFZ', other_name, other_folder)
170
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
171
+
172
+
173
+ def test_no_match_raises(tmp_path):
174
+ with pytest.raises(RuntimeError, match="failed to find sequence file"):
175
+ jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
176
+
177
+
178
+ def test_broken_link_is_still_a_candidate(tmp_path):
179
+ # On dori the target is copied into a local cache and the link is broken
180
+ # until that lands; it is still the file this library resolved to.
181
+ name, folder = next(iter(LBBCZRJ_TARGETS.items()))
182
+ link = _link(str(tmp_path), 'LBBCZRJ', name, folder)
183
+ assert not os.path.exists(link)
184
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) == link
@@ -1,7 +0,0 @@
1
- # Changelog
2
-
3
- <!--next-version-placeholder-->
4
-
5
- ## v0.1.0 (07/07/2023)
6
-
7
- - First release of `jamofetch`!
@@ -1,47 +0,0 @@
1
- import json
2
- import shlex
3
-
4
- import pytest
5
-
6
- from jamofetch import jamofetch
7
-
8
-
9
- def _query(cmd):
10
- # The query is the single-quoted final argument; shlex undoes the shell quoting.
11
- return json.loads(shlex.split(cmd)[-1])
12
-
13
-
14
- def test_nersc_cmd(monkeypatch):
15
- monkeypatch.delenv('SLURM_PARTITION', raising=False)
16
- cmd = jamofetch.get_cmd('LBBDHFZ')
17
- assert cmd.startswith('module load jamo; jamo link custom ')
18
- assert _query(cmd) == {
19
- "metadata.library_name": "LBBDHFZ",
20
- "metadata.fastq_type": "sdm_normal",
21
- "group": "sdm",
22
- }
23
-
24
-
25
- def test_dori_cmd_uses_exec(monkeypatch):
26
- monkeypatch.setenv('SLURM_PARTITION', 'dori')
27
- cmd = jamofetch.get_cmd('LBBDHFZ')
28
- assert cmd.startswith('apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom ')
29
- assert _query(cmd)["metadata.library_name"] == "LBBDHFZ"
30
-
31
-
32
- def test_library_name_is_first_query_key(monkeypatch):
33
- # jamo names the symlink after the first string-valued key; _find_fastq_path
34
- # relies on that being the library name.
35
- monkeypatch.delenv('SLURM_PARTITION', raising=False)
36
- assert next(iter(_query(jamofetch.get_cmd('LBBDHFZ')))) == "metadata.library_name"
37
-
38
-
39
- def test_query_does_not_filter_on_user(monkeypatch):
40
- monkeypatch.delenv('SLURM_PARTITION', raising=False)
41
- assert "user" not in _query(jamofetch.get_cmd('LBBDHFZ'))
42
-
43
-
44
- @pytest.mark.parametrize('bad', ["LBB'DHFZ", 'lbbdhfz', 'LBB DHFZ', 'LBB;rm'])
45
- def test_invalid_library_name_rejected(bad):
46
- with pytest.raises(ValueError):
47
- jamofetch.get_cmd(bad)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes