variantgrid-api 1.6.0__tar.gz → 1.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {variantgrid_api-1.6.0/src/variantgrid_api.egg-info → variantgrid_api-1.7.0}/PKG-INFO +1 -1
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/pyproject.toml +1 -1
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api/api_client.py +17 -3
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api/data_models.py +23 -2
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0/src/variantgrid_api.egg-info}/PKG-INFO +1 -1
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_api_client.py +59 -12
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/LICENSE +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/README.md +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/setup.cfg +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api/cli.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api/mock_variantgrid_api.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api.egg-info/SOURCES.txt +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api.egg-info/dependency_links.txt +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api.egg-info/entry_points.txt +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api.egg-info/requires.txt +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api.egg-info/top_level.txt +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_api_client_annotation.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_api_client_bulk.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_api_client_capabilities.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_api_client_patients.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_api_client_validation.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_cli.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_data_models.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_mock_variantgrid_api.py +0 -0
- {variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/tests/test_sequencer_model_from_name.py +0 -0
|
@@ -243,8 +243,20 @@ class VariantGridAPI:
|
|
|
243
243
|
for sf in sequencing_files:
|
|
244
244
|
# The server requires both paths - catch it here, naming the record, rather than a 400 for the batch
|
|
245
245
|
self._validate_string(f"SequencingFile '{sf.sample_name}' bam_file.path", sf.bam_file and sf.bam_file.path)
|
|
246
|
-
|
|
246
|
+
vcf_files = sf.get_vcf_files()
|
|
247
|
+
self._validate_list(f"SequencingFile '{sf.sample_name}' vcf_files", vcf_files)
|
|
248
|
+
for i, vcf_file in enumerate(vcf_files):
|
|
249
|
+
self._validate_string(f"SequencingFile '{sf.sample_name}' vcf_files[{i}].path",
|
|
250
|
+
vcf_file and vcf_file.path)
|
|
251
|
+
# The server keeps one VCF per BAM and caller - a repeated caller would silently replace a path
|
|
252
|
+
callers = [f"{vc.name} {vc.version}" for vcf_file in vcf_files
|
|
253
|
+
if vcf_file and (vc := vcf_file.variant_caller)]
|
|
254
|
+
if repeated := {c for c in callers if callers.count(c) > 1}:
|
|
255
|
+
raise ValueError(f"SequencingFile '{sf.sample_name}' has more than one VCF from variant caller(s) "
|
|
256
|
+
f"{', '.join(sorted(repeated))} - each VCF off a BAM needs its own caller")
|
|
247
257
|
data = sf.to_dict()
|
|
258
|
+
data.pop("vcf_file", None)
|
|
259
|
+
data.pop("vcf_files", None)
|
|
248
260
|
# put into hierarchial JSON DRF expects
|
|
249
261
|
fastq_r1 = data.pop("fastq_r1", None)
|
|
250
262
|
fastq_r2 = data.pop("fastq_r2", None)
|
|
@@ -255,8 +267,10 @@ class VariantGridAPI:
|
|
|
255
267
|
data["unaligned_reads"] = unaligned_reads
|
|
256
268
|
elif fastq_r2:
|
|
257
269
|
raise ValueError(f"SequencingFile '{sf.sample_name}' has fastq_r2 without fastq_r1")
|
|
258
|
-
# No FastQs (BAM-first run) - server resolves the sample from sample_name
|
|
259
|
-
|
|
270
|
+
# No FastQs (BAM-first run) - server resolves the sample from sample_name.
|
|
271
|
+
# The server takes one VCF per record, so each is a record sharing the BAM and FastQs
|
|
272
|
+
for vcf_file in vcf_files:
|
|
273
|
+
records.append({**data, "vcf_file": vcf_file.to_dict() if vcf_file else None})
|
|
260
274
|
|
|
261
275
|
json_data = {
|
|
262
276
|
"sample_sheet": sample_sheet_lookup.to_dict(),
|
|
@@ -196,12 +196,33 @@ class VCFFile(SingleSampleVCF):
|
|
|
196
196
|
@dataclass_json
|
|
197
197
|
@dataclass
|
|
198
198
|
class SequencingFile:
|
|
199
|
-
""" FastQs are optional - BAM-first runs (sequencer emits BAM, or FastQs not kept) send just BAM + VCF
|
|
199
|
+
""" FastQs are optional - BAM-first runs (sequencer emits BAM, or FastQs not kept) send just BAM + VCF
|
|
200
|
+
|
|
201
|
+
vcf_files: the VCFs called off this BAM, one per variant caller, eg DRAGEN TSO 500's small variant VCF and
|
|
202
|
+
its gene-level CNV VCF. The server keeps one VCF per BAM and caller, so a second with the same caller
|
|
203
|
+
would replace the first - create_sequencing_data() raises instead.
|
|
204
|
+
|
|
205
|
+
vcf_file is deprecated - use vcf_files. It still works (set it and it is sent, read it back as before),
|
|
206
|
+
and get_vcf_files() gives both """
|
|
200
207
|
sample_name: str
|
|
201
208
|
bam_file: BamFile
|
|
202
|
-
vcf_file: SingleSampleVCF
|
|
209
|
+
vcf_file: Optional[SingleSampleVCF] = field(default=None, metadata=config(exclude=lambda x: x is None))
|
|
203
210
|
fastq_r1: Optional[str] = field(default=None, metadata=config(exclude=lambda x: x is None))
|
|
204
211
|
fastq_r2: Optional[str] = field(default=None, metadata=config(exclude=lambda x: x is None))
|
|
212
|
+
vcf_files: Optional[List[SingleSampleVCF]] = field(default=None, metadata=config(exclude=lambda x: x is None))
|
|
213
|
+
|
|
214
|
+
def __post_init__(self):
|
|
215
|
+
if self.vcf_file is not None:
|
|
216
|
+
warnings.warn(
|
|
217
|
+
"SequencingFile.vcf_file is deprecated; use vcf_files instead.",
|
|
218
|
+
DeprecationWarning,
|
|
219
|
+
stacklevel=3,
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
def get_vcf_files(self) -> List[SingleSampleVCF]:
|
|
223
|
+
""" vcf_file (deprecated) then vcf_files """
|
|
224
|
+
vcf_files = [self.vcf_file] if self.vcf_file is not None else []
|
|
225
|
+
return vcf_files + list(self.vcf_files or [])
|
|
205
226
|
|
|
206
227
|
|
|
207
228
|
@dataclass_json
|
|
@@ -7,7 +7,7 @@ import pytest
|
|
|
7
7
|
import responses
|
|
8
8
|
|
|
9
9
|
from variantgrid_api.api_client import VariantGridAPI, DateTimeEncoder, EmptyInputPolicy
|
|
10
|
-
from variantgrid_api.data_models import BamFile, SingleSampleVCF
|
|
10
|
+
from variantgrid_api.data_models import BamFile, SequencingFile, SingleSampleVCF, VariantCaller
|
|
11
11
|
|
|
12
12
|
|
|
13
13
|
def _last_json():
|
|
@@ -67,19 +67,21 @@ def test_create_sequencing_data_fastq_r2_without_r1_raises(api, vg_objects):
|
|
|
67
67
|
with pytest.raises(ValueError):
|
|
68
68
|
api.create_sequencing_data(vg_objects["sample_sheet_lookup"], [sf])
|
|
69
69
|
|
|
70
|
-
@pytest.mark.parametrize("
|
|
71
|
-
("
|
|
72
|
-
("
|
|
73
|
-
("
|
|
74
|
-
("
|
|
75
|
-
("
|
|
70
|
+
@pytest.mark.parametrize("changes, name", [
|
|
71
|
+
({"vcf_files": None}, "vcf_files"),
|
|
72
|
+
({"vcf_files": []}, "vcf_files"),
|
|
73
|
+
({"vcf_files": [None]}, r"vcf_files\[0\].path"),
|
|
74
|
+
({"vcf_files": [SingleSampleVCF(path=None)]}, r"vcf_files\[0\].path"),
|
|
75
|
+
({"vcf_files": [SingleSampleVCF(path="")]}, r"vcf_files\[0\].path"),
|
|
76
|
+
({"bam_file": None}, "bam_file.path"),
|
|
77
|
+
({"bam_file": BamFile(path=None)}, "bam_file.path"),
|
|
76
78
|
])
|
|
77
79
|
@responses.activate
|
|
78
|
-
def test_create_sequencing_data_missing_path_names_record(api, vg_objects,
|
|
80
|
+
def test_create_sequencing_data_missing_path_names_record(api, vg_objects, changes, name):
|
|
79
81
|
""" SACGF/variantgrid_api#23 - caught before sending, naming the record, rather than a 400 for the batch """
|
|
80
82
|
sequencing_files = list(vg_objects["sequencing_files"])
|
|
81
|
-
sequencing_files[1] = dataclasses.replace(sequencing_files[1], **
|
|
82
|
-
with pytest.raises(ValueError, match=f"SequencingFile 'fake_sample_2' {
|
|
83
|
+
sequencing_files[1] = dataclasses.replace(sequencing_files[1], **changes)
|
|
84
|
+
with pytest.raises(ValueError, match=f"SequencingFile 'fake_sample_2' {name}"):
|
|
83
85
|
api.create_sequencing_data(vg_objects["sample_sheet_lookup"], sequencing_files)
|
|
84
86
|
assert len(responses.calls) == 0
|
|
85
87
|
|
|
@@ -89,10 +91,55 @@ def test_create_sequencing_data_missing_vcf_path_warns_and_posts(server, api_tok
|
|
|
89
91
|
url = f"{server}/seqauto/api/v1/sequencing_files/bulk_create"
|
|
90
92
|
responses.add(responses.POST, url, json={"created": 2}, status=200)
|
|
91
93
|
sequencing_files = list(vg_objects["sequencing_files"])
|
|
92
|
-
sequencing_files[0] = dataclasses.replace(sequencing_files[0],
|
|
94
|
+
sequencing_files[0] = dataclasses.replace(sequencing_files[0], vcf_files=[SingleSampleVCF(path=None)])
|
|
93
95
|
api.create_sequencing_data(vg_objects["sample_sheet_lookup"], sequencing_files)
|
|
94
96
|
assert len(responses.calls) == 1
|
|
95
|
-
assert any("SequencingFile 'fake_sample_1'
|
|
97
|
+
assert any("SequencingFile 'fake_sample_1' vcf_files[0].path" in r.message for r in caplog.records)
|
|
98
|
+
|
|
99
|
+
@responses.activate
|
|
100
|
+
def test_create_sequencing_data_vcf_files_share_the_bam(api, server, vg_objects):
|
|
101
|
+
""" eg DRAGEN TSO 500's CNV VCF beside its small variant VCF - one record per VCF, same BAM and FastQs """
|
|
102
|
+
url = f"{server}/seqauto/api/v1/sequencing_files/bulk_create"
|
|
103
|
+
responses.add(responses.POST, url, json={"created": 3}, status=200)
|
|
104
|
+
cnv_vcf = SingleSampleVCF(path="/data/fake_sample_1.cnv.vcf", variant_caller=VariantCaller(name="cnv", version="1"))
|
|
105
|
+
sequencing_files = list(vg_objects["sequencing_files"])
|
|
106
|
+
sf = sequencing_files[0]
|
|
107
|
+
sequencing_files[0] = dataclasses.replace(sf, vcf_files=sf.vcf_files + [cnv_vcf])
|
|
108
|
+
api.create_sequencing_data(vg_objects["sample_sheet_lookup"], sequencing_files)
|
|
109
|
+
|
|
110
|
+
records = _last_json()["records"]
|
|
111
|
+
assert [r["sample_name"] for r in records] == ["fake_sample_1", "fake_sample_1", "fake_sample_2"]
|
|
112
|
+
first, second = records[0], records[1]
|
|
113
|
+
assert "vcf_files" not in first
|
|
114
|
+
assert first["vcf_file"]["path"] == sf.vcf_files[0].path
|
|
115
|
+
assert second["vcf_file"]["path"] == "/data/fake_sample_1.cnv.vcf"
|
|
116
|
+
assert second["bam_file"] == first["bam_file"]
|
|
117
|
+
assert second["unaligned_reads"] == first["unaligned_reads"]
|
|
118
|
+
|
|
119
|
+
def test_create_sequencing_data_vcf_files_same_caller_raises(api, vg_objects):
|
|
120
|
+
""" The server keeps one VCF per BAM and caller, so the second would silently replace the first's path """
|
|
121
|
+
sf = vg_objects["sequencing_files"][0]
|
|
122
|
+
same_caller = SingleSampleVCF(path="/data/other.vcf", variant_caller=sf.vcf_files[0].variant_caller)
|
|
123
|
+
sf = dataclasses.replace(sf, vcf_files=sf.vcf_files + [same_caller])
|
|
124
|
+
with pytest.raises(ValueError, match="fake_sample_1.*more than one VCF"):
|
|
125
|
+
api.create_sequencing_data(vg_objects["sample_sheet_lookup"], [sf])
|
|
126
|
+
|
|
127
|
+
@responses.activate
|
|
128
|
+
def test_create_sequencing_data_deprecated_vcf_file_still_sent(api, server, vg_objects):
|
|
129
|
+
""" vcf_file is deprecated for vcf_files, but a client still using it sends the same records as before """
|
|
130
|
+
url = f"{server}/seqauto/api/v1/sequencing_files/bulk_create"
|
|
131
|
+
responses.add(responses.POST, url, json={"created": 2}, status=200)
|
|
132
|
+
old_style = []
|
|
133
|
+
for sf in vg_objects["sequencing_files"]:
|
|
134
|
+
with pytest.warns(DeprecationWarning, match="vcf_file is deprecated"):
|
|
135
|
+
old_style.append(SequencingFile(sample_name=sf.sample_name, bam_file=sf.bam_file,
|
|
136
|
+
vcf_file=sf.vcf_files[0], fastq_r1=sf.fastq_r1, fastq_r2=sf.fastq_r2))
|
|
137
|
+
api.create_sequencing_data(vg_objects["sample_sheet_lookup"], old_style)
|
|
138
|
+
old_records = _last_json()["records"]
|
|
139
|
+
|
|
140
|
+
api.create_sequencing_data(vg_objects["sample_sheet_lookup"], vg_objects["sequencing_files"])
|
|
141
|
+
assert old_records == _last_json()["records"]
|
|
142
|
+
assert old_style[0].vcf_file.path == old_style[0].get_vcf_files()[0].path
|
|
96
143
|
|
|
97
144
|
def assert_post(api_call, url):
|
|
98
145
|
responses.add(responses.POST, url, json={"ok": True}, status=200)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{variantgrid_api-1.6.0 → variantgrid_api-1.7.0}/src/variantgrid_api.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|