variantgrid-api 1.7.0__tar.gz → 1.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {variantgrid_api-1.7.0/src/variantgrid_api.egg-info → variantgrid_api-1.9.0}/PKG-INFO +59 -26
  2. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/README.md +57 -2
  3. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/pyproject.toml +4 -4
  4. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api/api_client.py +102 -47
  5. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api/cli.py +68 -8
  6. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api/data_models.py +98 -53
  7. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api/mock_variantgrid_api.py +9 -14
  8. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0/src/variantgrid_api.egg-info}/PKG-INFO +59 -26
  9. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api.egg-info/SOURCES.txt +1 -0
  10. variantgrid_api-1.9.0/tests/test_alignment_files.py +204 -0
  11. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_api_client.py +29 -23
  12. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_api_client_bulk.py +6 -6
  13. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_api_client_capabilities.py +5 -5
  14. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_api_client_patients.py +1 -45
  15. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_cli.py +53 -0
  16. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_mock_variantgrid_api.py +27 -17
  17. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/LICENSE +0 -0
  18. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/setup.cfg +0 -0
  19. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api.egg-info/dependency_links.txt +0 -0
  20. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api.egg-info/entry_points.txt +0 -0
  21. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api.egg-info/requires.txt +0 -0
  22. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/src/variantgrid_api.egg-info/top_level.txt +0 -0
  23. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_api_client_annotation.py +0 -0
  24. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_api_client_validation.py +0 -0
  25. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_data_models.py +0 -0
  26. {variantgrid_api-1.7.0 → variantgrid_api-1.9.0}/tests/test_sequencer_model_from_name.py +0 -0
@@ -1,34 +1,12 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: variantgrid_api
3
- Version: 1.7.0
3
+ Version: 1.9.0
4
4
  Summary: A Python API client for VariantGrid
5
5
  Author-email: Dave Lawrence <davmlaw@gmail.com>
6
- License: MIT License
7
-
8
- Copyright (c) 2024
9
-
10
- Permission is hereby granted, free of charge, to any person obtaining a copy
11
- of this software and associated documentation files (the "Software"), to deal
12
- in the Software without restriction, including without limitation the rights
13
- to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
- copies of the Software, and to permit persons to whom the Software is
15
- furnished to do so, subject to the following conditions:
16
-
17
- The above copyright notice and this permission notice shall be included in all
18
- copies or substantial portions of the Software.
19
-
20
- THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
- IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
- FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
- AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
- LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
- OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
- SOFTWARE.
27
-
6
+ License-Expression: MIT
28
7
  Project-URL: Homepage, https://github.com/SACGF/variantgrid_api
29
8
  Classifier: Development Status :: 3 - Alpha
30
9
  Classifier: Intended Audience :: Developers
31
- Classifier: License :: OSI Approved :: MIT License
32
10
  Classifier: Programming Language :: Python :: 3
33
11
  Requires-Python: >=3.8
34
12
  Description-Content-Type: text/markdown
@@ -79,10 +57,15 @@ $ vg_api annotate_vcf input.vcf.gz -o results/
79
57
  Uploaded input.vcf.gz (id=13256).
80
58
  Annotating input.vcf.gz - run the same command again later to download.
81
59
 
60
+ $ vg_api annotate_vcf input.vcf.gz -o results/ # still annotating
61
+ input.vcf.gz: not ready yet - pipeline Success, import 100.0%, 2 annotation runs remaining. Run the same command again later.
62
+
82
63
  $ vg_api annotate_vcf input.vcf.gz -o results/ # once it's done
83
64
  Annotated vcf written to results/input.vcf_annotated_v254_GRCh38.vcf.gz
84
65
  ```
85
66
 
67
+ Add `--verbose` to log each request, response and the full upload status to stderr.
68
+
86
69
  From Python it's `upload_file()` / `poll_upload_status()` / `download_annotated()`, or the blocking
87
70
  `annotate_vcf()` one-liner. See
88
71
  **[Annotate a VCF](https://github.com/SACGF/variantgrid_api/wiki/Annotate-a-VCF)** on the wiki for batches, the
@@ -90,8 +73,8 @@ submit-now/download-later pattern, and all the options.
90
73
 
91
74
  ## Patients, specimens and extractions
92
75
 
93
- VariantGrid can record which patient, specimen and extraction a lab's sequencing came from, and specimen-level
94
- measures such as TMB, MSI and GIS. This needs a server at or after SACGF/variantgrid#1716.
76
+ VariantGrid can record which patient, specimen and extraction a lab's sequencing came from. This needs a server
77
+ at or after SACGF/variantgrid#1716.
95
78
 
96
79
  ```
97
80
  from variantgrid_api.data_models import Patient, Specimen, Extraction, ExternalReference, NucleicAcid
@@ -107,6 +90,25 @@ api.upload_file("sample.vcf.gz", path=None,
107
90
  A bare string names a record by its local reference. Use `ExternalReference(code=..., external_type=...)` to
108
91
  name it by a LIMS identifier instead. See `examples/example_tso500.py` for a full run.
109
92
 
93
+ ## BAMs and CRAMs
94
+
95
+ A sequencing sample can have several alignment files, eg a BAM and its recalibrated BAM, or a BAM plus a CRAM.
96
+ List them in `SequencingFile.alignment_files` (and `QC.alignment_files`), the file the variant caller ran on first:
97
+
98
+ ```
99
+ from variantgrid_api.data_models import AlignmentFile, SequencingFile, SingleSampleVCF
100
+
101
+ SequencingFile(sample_name="sample_1",
102
+ alignment_files=[AlignmentFile(path="/data/sample_1.bam", aligner=aligner),
103
+ AlignmentFile(path="/data/sample_1.cram", aligner=aligner)],
104
+ vcf_files=[SingleSampleVCF(path="/data/sample_1.vcf.gz", variant_caller=variant_caller)])
105
+ ```
106
+
107
+ `AlignmentFile.file_type` (`AlignmentFileType.BAM` / `CRAM`) is optional and inferred from the path. A server
108
+ without the `alignment_files` feature gets the old wire format instead: one sequencing file record per alignment
109
+ file as `bam_file`, and a QC's first alignment file as its `bam_file`. A CRAM there also needs the
110
+ `cram_alignment_files` feature. `BamFile` and the `bam_file` fields still work but are deprecated.
111
+
110
112
  ## Talking to more than one VariantGrid version
111
113
 
112
114
  Servers of different ages accept different calls. The client asks the server which features it has
@@ -135,6 +137,37 @@ Where the fallback isn't simply "do nothing", branch on `api.supports("feature")
135
137
  `api.accepts_upload("file_type")`. `api.capabilities.version` is useful for logging which server you reached.
136
138
  For tests, `MockVariantGridAPI(capabilities=ServerCapabilities.LEGACY)` behaves like an old server.
137
139
 
140
+ ## DRAGEN TSO 500 run-level files
141
+
142
+ A TSO 500 run's `CombinedVariantOutput.tsv` (the pair's TMB / MSI / GIS) and `MetricsOutput.tsv` (library QC)
143
+ are uploaded with the run's name as `sequencing_run` metadata, the only metadata they take:
144
+
145
+ ```
146
+ from variantgrid_api.data_models import UploadFileType
147
+
148
+ api.upload_file("ExampleSample_CombinedVariantOutput.tsv", path=None,
149
+ metadata={"sequencing_run": sequencing_run.name},
150
+ file_type=UploadFileType.DRAGEN_TSO500_COMBINED_VARIANT_OUTPUT)
151
+ api.upload_file("MetricsOutput.tsv", path=None, metadata={"sequencing_run": sequencing_run.name},
152
+ file_type=UploadFileType.DRAGEN_TSO500_METRICS_OUTPUT)
153
+ ```
154
+
155
+ - **MetricsOutput** requires `sequencing_run` - the file never names its run.
156
+ - **CombinedVariantOutput** names its run 'NA', so the server takes the run from `sequencing_run`. Without it,
157
+ the server uses the registered run whose current sample sheet names the pair's sample IDs, and if no registered
158
+ run names the pair the import fails. Register the run and sample sheet first, and send `sequencing_run` anyway.
159
+
160
+ The server reads TMB / MSI / GIS from the CombinedVariantOutput (SACGF/variantgrid#1904), which replaced
161
+ specimen measures. See `examples/example_tso500.py` for a full run.
162
+
163
+ ## Permission to write sequencing data
164
+
165
+ Posts to the server's `seqauto/api/` endpoints - the sequencing `create_*` calls from `create_experiment` through
166
+ `create_sequencing_data` and the QC calls, and `link_sequencing_sample_extraction` - need a superuser's API token,
167
+ or one belonging to a member of the server's SeqAuto write group (`seqauto_api_write` unless the server's
168
+ `SEQAUTO_API_WRITE_GROUP` setting says otherwise). Anyone else gets a 403 (`requests.HTTPError`). Reads, such as
169
+ `sequencing_run_has_vcf()`, need only a valid token.
170
+
138
171
  ## Testing
139
172
 
140
173
  ```
@@ -36,10 +36,15 @@ $ vg_api annotate_vcf input.vcf.gz -o results/
36
36
  Uploaded input.vcf.gz (id=13256).
37
37
  Annotating input.vcf.gz - run the same command again later to download.
38
38
 
39
+ $ vg_api annotate_vcf input.vcf.gz -o results/ # still annotating
40
+ input.vcf.gz: not ready yet - pipeline Success, import 100.0%, 2 annotation runs remaining. Run the same command again later.
41
+
39
42
  $ vg_api annotate_vcf input.vcf.gz -o results/ # once it's done
40
43
  Annotated vcf written to results/input.vcf_annotated_v254_GRCh38.vcf.gz
41
44
  ```
42
45
 
46
+ Add `--verbose` to log each request, response and the full upload status to stderr.
47
+
43
48
  From Python it's `upload_file()` / `poll_upload_status()` / `download_annotated()`, or the blocking
44
49
  `annotate_vcf()` one-liner. See
45
50
  **[Annotate a VCF](https://github.com/SACGF/variantgrid_api/wiki/Annotate-a-VCF)** on the wiki for batches, the
@@ -47,8 +52,8 @@ submit-now/download-later pattern, and all the options.
47
52
 
48
53
  ## Patients, specimens and extractions
49
54
 
50
- VariantGrid can record which patient, specimen and extraction a lab's sequencing came from, and specimen-level
51
- measures such as TMB, MSI and GIS. This needs a server at or after SACGF/variantgrid#1716.
55
+ VariantGrid can record which patient, specimen and extraction a lab's sequencing came from. This needs a server
56
+ at or after SACGF/variantgrid#1716.
52
57
 
53
58
  ```
54
59
  from variantgrid_api.data_models import Patient, Specimen, Extraction, ExternalReference, NucleicAcid
@@ -64,6 +69,25 @@ api.upload_file("sample.vcf.gz", path=None,
64
69
  A bare string names a record by its local reference. Use `ExternalReference(code=..., external_type=...)` to
65
70
  name it by a LIMS identifier instead. See `examples/example_tso500.py` for a full run.
66
71
 
72
+ ## BAMs and CRAMs
73
+
74
+ A sequencing sample can have several alignment files, eg a BAM and its recalibrated BAM, or a BAM plus a CRAM.
75
+ List them in `SequencingFile.alignment_files` (and `QC.alignment_files`), the file the variant caller ran on first:
76
+
77
+ ```
78
+ from variantgrid_api.data_models import AlignmentFile, SequencingFile, SingleSampleVCF
79
+
80
+ SequencingFile(sample_name="sample_1",
81
+ alignment_files=[AlignmentFile(path="/data/sample_1.bam", aligner=aligner),
82
+ AlignmentFile(path="/data/sample_1.cram", aligner=aligner)],
83
+ vcf_files=[SingleSampleVCF(path="/data/sample_1.vcf.gz", variant_caller=variant_caller)])
84
+ ```
85
+
86
+ `AlignmentFile.file_type` (`AlignmentFileType.BAM` / `CRAM`) is optional and inferred from the path. A server
87
+ without the `alignment_files` feature gets the old wire format instead: one sequencing file record per alignment
88
+ file as `bam_file`, and a QC's first alignment file as its `bam_file`. A CRAM there also needs the
89
+ `cram_alignment_files` feature. `BamFile` and the `bam_file` fields still work but are deprecated.
90
+
67
91
  ## Talking to more than one VariantGrid version
68
92
 
69
93
  Servers of different ages accept different calls. The client asks the server which features it has
@@ -92,6 +116,37 @@ Where the fallback isn't simply "do nothing", branch on `api.supports("feature")
92
116
  `api.accepts_upload("file_type")`. `api.capabilities.version` is useful for logging which server you reached.
93
117
  For tests, `MockVariantGridAPI(capabilities=ServerCapabilities.LEGACY)` behaves like an old server.
94
118
 
119
+ ## DRAGEN TSO 500 run-level files
120
+
121
+ A TSO 500 run's `CombinedVariantOutput.tsv` (the pair's TMB / MSI / GIS) and `MetricsOutput.tsv` (library QC)
122
+ are uploaded with the run's name as `sequencing_run` metadata, the only metadata they take:
123
+
124
+ ```
125
+ from variantgrid_api.data_models import UploadFileType
126
+
127
+ api.upload_file("ExampleSample_CombinedVariantOutput.tsv", path=None,
128
+ metadata={"sequencing_run": sequencing_run.name},
129
+ file_type=UploadFileType.DRAGEN_TSO500_COMBINED_VARIANT_OUTPUT)
130
+ api.upload_file("MetricsOutput.tsv", path=None, metadata={"sequencing_run": sequencing_run.name},
131
+ file_type=UploadFileType.DRAGEN_TSO500_METRICS_OUTPUT)
132
+ ```
133
+
134
+ - **MetricsOutput** requires `sequencing_run` - the file never names its run.
135
+ - **CombinedVariantOutput** names its run 'NA', so the server takes the run from `sequencing_run`. Without it,
136
+ the server uses the registered run whose current sample sheet names the pair's sample IDs, and if no registered
137
+ run names the pair the import fails. Register the run and sample sheet first, and send `sequencing_run` anyway.
138
+
139
+ The server reads TMB / MSI / GIS from the CombinedVariantOutput (SACGF/variantgrid#1904), which replaced
140
+ specimen measures. See `examples/example_tso500.py` for a full run.
141
+
142
+ ## Permission to write sequencing data
143
+
144
+ Posts to the server's `seqauto/api/` endpoints - the sequencing `create_*` calls from `create_experiment` through
145
+ `create_sequencing_data` and the QC calls, and `link_sequencing_sample_extraction` - need a superuser's API token,
146
+ or one belonging to a member of the server's SeqAuto write group (`seqauto_api_write` unless the server's
147
+ `SEQAUTO_API_WRITE_GROUP` setting says otherwise). Anyone else gets a 403 (`requests.HTTPError`). Reads, such as
148
+ `sequencing_run_has_vcf()`, need only a valid token.
149
+
95
150
  ## Testing
96
151
 
97
152
  ```
@@ -1,21 +1,21 @@
1
1
  [build-system]
2
- requires = ["setuptools", "wheel"]
2
+ requires = ["setuptools>=77.0.0", "wheel"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "variantgrid_api"
7
- version = "1.7.0"
7
+ version = "1.9.0"
8
8
  description = "A Python API client for VariantGrid"
9
9
  authors = [
10
10
  { name = "Dave Lawrence", email = "davmlaw@gmail.com" }
11
11
  ]
12
12
  readme = "README.md"
13
- license = {file = "LICENSE"}
13
+ license = "MIT"
14
+ license-files = ["LICENSE"]
14
15
  requires-python = ">=3.8"
15
16
  classifiers = [
16
17
  "Development Status :: 3 - Alpha",
17
18
  "Intended Audience :: Developers",
18
- "License :: OSI Approved :: MIT License",
19
19
  "Programming Language :: Python :: 3",
20
20
  ]
21
21
  dependencies = [
@@ -13,7 +13,7 @@ import requests
13
13
 
14
14
  from variantgrid_api.data_models import EnrichmentKit, SequencingRun, SampleSheet, JointCalledVCF, \
15
15
  SampleSheetLookup, SequencingFile, QCGeneList, QCExecStats, QCGeneCoverage, SequencerModel, Sequencer, \
16
- SequencingSampleLookup, Patient, Specimen, Extraction, SpecimenMeasure, ExternalReference, ReferenceLike, \
16
+ SequencingSampleLookup, Patient, Specimen, Extraction, ExternalReference, ReferenceLike, \
17
17
  reference_json, ServerCapabilities, ServerFeature, UploadFileType
18
18
 
19
19
 
@@ -237,26 +237,24 @@ class VariantGridAPI:
237
237
  return self.create_joint_called_vcf(sample_sheet_combined_vcf_file)
238
238
 
239
239
  def create_sequencing_data(self, sample_sheet_lookup: SampleSheetLookup, sequencing_files: List[SequencingFile]):
240
+ """ One record per VCF, each carrying the sample's alignment files and FastQs.
241
+
242
+ A SequencingFile using alignment_files is sent as 'alignment_files' when the server supports
243
+ ServerFeature.ALIGNMENT_FILES. Otherwise each alignment file goes in its own record as 'bam_file' (what
244
+ older servers take) - a CRAM there also needs ServerFeature.CRAM_ALIGNMENT_FILES, or it is handled by
245
+ unsupported_feature_policy (SKIP leaves the CRAM out). A SequencingFile using only the deprecated
246
+ bam_file is sent exactly as before, without asking the server for its capabilities """
240
247
  self._validate_object("sample_sheet_lookup", sample_sheet_lookup)
241
248
  self._validate_list("sequencing_files", sequencing_files)
249
+ # Check every record before building any, as building one can ask the server for its capabilities
250
+ for sf in sequencing_files:
251
+ self._validate_sequencing_file(sf)
252
+
242
253
  records = []
243
254
  for sf in sequencing_files:
244
- # The server requires both paths - catch it here, naming the record, rather than a 400 for the batch
245
- self._validate_string(f"SequencingFile '{sf.sample_name}' bam_file.path", sf.bam_file and sf.bam_file.path)
246
- vcf_files = sf.get_vcf_files()
247
- self._validate_list(f"SequencingFile '{sf.sample_name}' vcf_files", vcf_files)
248
- for i, vcf_file in enumerate(vcf_files):
249
- self._validate_string(f"SequencingFile '{sf.sample_name}' vcf_files[{i}].path",
250
- vcf_file and vcf_file.path)
251
- # The server keeps one VCF per BAM and caller - a repeated caller would silently replace a path
252
- callers = [f"{vc.name} {vc.version}" for vcf_file in vcf_files
253
- if vcf_file and (vc := vcf_file.variant_caller)]
254
- if repeated := {c for c in callers if callers.count(c) > 1}:
255
- raise ValueError(f"SequencingFile '{sf.sample_name}' has more than one VCF from variant caller(s) "
256
- f"{', '.join(sorted(repeated))} - each VCF off a BAM needs its own caller")
257
255
  data = sf.to_dict()
258
- data.pop("vcf_file", None)
259
- data.pop("vcf_files", None)
256
+ for key in ("bam_file", "vcf_file", "vcf_files", "alignment_files"):
257
+ data.pop(key, None)
260
258
  # put into hierarchial JSON DRF expects
261
259
  fastq_r1 = data.pop("fastq_r1", None)
262
260
  fastq_r2 = data.pop("fastq_r2", None)
@@ -265,12 +263,12 @@ class VariantGridAPI:
265
263
  if fastq_r2:
266
264
  unaligned_reads["fastq_r2"] = {"path": fastq_r2}
267
265
  data["unaligned_reads"] = unaligned_reads
268
- elif fastq_r2:
269
- raise ValueError(f"SequencingFile '{sf.sample_name}' has fastq_r2 without fastq_r1")
270
266
  # No FastQs (BAM-first run) - server resolves the sample from sample_name.
271
- # The server takes one VCF per record, so each is a record sharing the BAM and FastQs
272
- for vcf_file in vcf_files:
273
- records.append({**data, "vcf_file": vcf_file.to_dict() if vcf_file else None})
267
+ # The server takes one VCF per record, so each is a record sharing the alignment files and FastQs
268
+ for alignment_data in self._sequencing_file_alignment_data(sf):
269
+ for vcf_file in sf.get_vcf_files():
270
+ records.append({"sample_name": sf.sample_name, **alignment_data, **data,
271
+ "vcf_file": vcf_file.to_dict() if vcf_file else None})
274
272
 
275
273
  json_data = {
276
274
  "sample_sheet": sample_sheet_lookup.to_dict(),
@@ -279,9 +277,83 @@ class VariantGridAPI:
279
277
  return self._post("seqauto/api/v1/sequencing_files/bulk_create",
280
278
  json_data)
281
279
 
280
+ def _validate_sequencing_file(self, sf: SequencingFile):
281
+ # The server requires both paths - catch it here, naming the record, rather than a 400 for the batch
282
+ if sf.alignment_files is None:
283
+ hint = "" if sf.bam_file else " (or set alignment_files)"
284
+ self._validate_string(f"SequencingFile '{sf.sample_name}' bam_file.path{hint}",
285
+ sf.bam_file and sf.bam_file.path)
286
+ else:
287
+ alignment_files = sf.get_alignment_files()
288
+ self._validate_list(f"SequencingFile '{sf.sample_name}' alignment_files", alignment_files)
289
+ for i, alignment_file in enumerate(alignment_files):
290
+ self._validate_string(f"SequencingFile '{sf.sample_name}' alignment_files[{i}].path",
291
+ alignment_file and alignment_file.path)
292
+ vcf_files = sf.get_vcf_files()
293
+ self._validate_list(f"SequencingFile '{sf.sample_name}' vcf_files", vcf_files)
294
+ for i, vcf_file in enumerate(vcf_files):
295
+ self._validate_string(f"SequencingFile '{sf.sample_name}' vcf_files[{i}].path",
296
+ vcf_file and vcf_file.path)
297
+ # The server keeps one VCF per BAM and caller - a repeated caller would silently replace a path
298
+ callers = [f"{vc.name} {vc.version}" for vcf_file in vcf_files
299
+ if vcf_file and (vc := vcf_file.variant_caller)]
300
+ if repeated := {c for c in callers if callers.count(c) > 1}:
301
+ raise ValueError(f"SequencingFile '{sf.sample_name}' has more than one VCF from variant caller(s) "
302
+ f"{', '.join(sorted(repeated))} - each VCF off a BAM needs its own caller")
303
+ if sf.fastq_r2 and not sf.fastq_r1:
304
+ raise ValueError(f"SequencingFile '{sf.sample_name}' has fastq_r2 without fastq_r1")
305
+
306
+ def _sequencing_file_alignment_data(self, sf: SequencingFile) -> List[dict]:
307
+ """ The alignment file part of sf's records - one dict per record """
308
+ if sf.alignment_files is None:
309
+ # Only the deprecated bam_file - sent as before
310
+ return [{"bam_file": sf.bam_file.to_dict() if sf.bam_file else None}]
311
+ alignment_files = sf.get_alignment_files()
312
+ if self.supports(ServerFeature.ALIGNMENT_FILES):
313
+ return [{"alignment_files": [af.to_dict() if af else None for af in alignment_files]}]
314
+
315
+ # Older server - one record per alignment file, sent as bam_file
316
+ alignment_data = []
317
+ for af in alignment_files:
318
+ if af is None:
319
+ alignment_data.append({"bam_file": None}) # Already reported by _validate_sequencing_file
320
+ elif af.is_cram() and not self.supports(ServerFeature.CRAM_ALIGNMENT_FILES):
321
+ self._unsupported(f"SequencingFile '{sf.sample_name}' CRAM '{af.path}' "
322
+ f"(server doesn't support feature '{ServerFeature.CRAM_ALIGNMENT_FILES}')")
323
+ else:
324
+ alignment_data.append({"bam_file": self._legacy_bam_file_json(af.to_dict())})
325
+ return alignment_data
326
+
327
+ @staticmethod
328
+ def _legacy_bam_file_json(alignment_file_data: dict) -> dict:
329
+ """ An alignment file as an older server's 'bam_file' - it has no file_type """
330
+ return {k: v for k, v in alignment_file_data.items() if k != "file_type"}
331
+
332
+ def _qc_record_json(self, qc_record) -> dict:
333
+ """ qc_record.to_dict(), with its QC's alignment files as 'alignment_files' (bam_file first) to a server
334
+ with ServerFeature.ALIGNMENT_FILES. An older server takes one 'bam_file' - it finds the QC by
335
+ sequencing sample, VCF path and that path, so the first alignment file (the one the VCF was called
336
+ from) is sent. A QC using only the deprecated bam_file is sent as before, with no probe """
337
+ json_data = qc_record.to_dict()
338
+ qc = qc_record.qc
339
+ if qc is None or qc.alignment_files is None:
340
+ return json_data
341
+ qc_data = json_data["qc"]
342
+ qc_data.pop("bam_file", None)
343
+ qc_data.pop("alignment_files", None)
344
+ alignment_files = qc.get_alignment_files()
345
+ if self.supports(ServerFeature.ALIGNMENT_FILES):
346
+ qc_data["alignment_files"] = [af.to_dict() if af else None for af in alignment_files]
347
+ else:
348
+ first = alignment_files[0] if alignment_files else None
349
+ json_data["qc"] = {"sequencing_sample": qc_data.pop("sequencing_sample"),
350
+ "bam_file": self._legacy_bam_file_json(first.to_dict()) if first else None,
351
+ **qc_data}
352
+ return json_data
353
+
282
354
  def create_qc_gene_list(self, qc_gene_list: QCGeneList):
283
355
  self._validate_object("qc_gene_list", qc_gene_list)
284
- json_data = qc_gene_list.to_dict()
356
+ json_data = self._qc_record_json(qc_gene_list)
285
357
  return self._post("seqauto/api/v1/qc_gene_list/",
286
358
  json_data)
287
359
 
@@ -290,7 +362,7 @@ class VariantGridAPI:
290
362
  self._validate_list("qc_gene_lists", qc_gene_lists)
291
363
  json_data = {
292
364
  "records": [
293
- qcgl.to_dict() for qcgl in qc_gene_lists
365
+ self._qc_record_json(qcgl) for qcgl in qc_gene_lists
294
366
  ]
295
367
  }
296
368
  return self._post("seqauto/api/v1/qc_gene_list/bulk_create",
@@ -298,7 +370,7 @@ class VariantGridAPI:
298
370
 
299
371
  def create_qc_exec_stats(self, qc_exec_stats: QCExecStats):
300
372
  self._validate_object("qc_exec_stats", qc_exec_stats)
301
- json_data = qc_exec_stats.to_dict()
373
+ json_data = self._qc_record_json(qc_exec_stats)
302
374
  return self._post("seqauto/api/v1/qc_exec_summary/",
303
375
  json_data)
304
376
 
@@ -306,7 +378,7 @@ class VariantGridAPI:
306
378
  self._validate_list("qc_exec_stats", qc_exec_stats)
307
379
  json_data = {
308
380
  "records": [
309
- qces.to_dict() for qces in qc_exec_stats
381
+ self._qc_record_json(qces) for qces in qc_exec_stats
310
382
  ]
311
383
  }
312
384
  return self._post("seqauto/api/v1/qc_exec_summary/bulk_create",
@@ -316,7 +388,7 @@ class VariantGridAPI:
316
388
  self._validate_list("qc_gene_coverage_list", qc_gene_coverage_list)
317
389
  json_data = {
318
390
  "records": [
319
- qcgc.to_dict() for qcgc in qc_gene_coverage_list
391
+ self._qc_record_json(qcgc) for qcgc in qc_gene_coverage_list
320
392
  ]
321
393
  }
322
394
  return self._post("seqauto/api/v1/qc_gene_coverage/bulk_create",
@@ -346,27 +418,6 @@ class VariantGridAPI:
346
418
  self._validate_object("extraction", extraction)
347
419
  return self._post("patients/api/v1/extraction/", extraction.to_dict())
348
420
 
349
- def create_specimen_measure(self, specimen_reference: ReferenceLike, measure: SpecimenMeasure):
350
- """ An unknown specimen is a 400. Replaces any existing measure of the same type for the specimen """
351
- if not self._require(ServerFeature.SPECIMEN_MEASURES):
352
- return None
353
- self._validate_reference("specimen_reference", specimen_reference)
354
- self._validate_object("measure", measure)
355
- json_data = {"specimen": reference_json(specimen_reference), **measure.to_dict()}
356
- return self._post("patients/api/v1/specimen_measure/", json_data)
357
-
358
- def create_specimen_measures(self, specimen_reference: ReferenceLike, measures: List[SpecimenMeasure]):
359
- """ A run's measures (TMB, MSI, GIS etc) against one specimen in one call """
360
- if not self._require(ServerFeature.SPECIMEN_MEASURES):
361
- return None
362
- self._validate_reference("specimen_reference", specimen_reference)
363
- self._validate_list("measures", measures)
364
- json_data = {
365
- "specimen": reference_json(specimen_reference),
366
- "measures": [measure.to_dict() for measure in measures],
367
- }
368
- return self._post("patients/api/v1/specimen_measure/bulk_create", json_data)
369
-
370
421
  def link_sequencing_sample_extraction(self, sequencing_sample_lookup: SequencingSampleLookup,
371
422
  extraction_reference: ReferenceLike) -> Optional[dict]:
372
423
  """ Name the extraction a sequencing sample was made from. One call per sequencing sample is
@@ -407,6 +458,10 @@ class VariantGridAPI:
407
458
  'sample_extractions' - {vcf_sample_name: reference} for a multi-sample VCF
408
459
  Send 'extraction' or 'sample_extractions', not both. An unknown key is a 400. An extraction
409
460
  the server doesn't have yet is not - it attaches once the extraction is created.
461
+ A run-level DRAGEN TSO 500 file takes only 'sequencing_run' (the run's name): required for a
462
+ MetricsOutput.tsv, and for a CombinedVariantOutput.tsv the run it belongs to - otherwise the
463
+ server uses the registered run whose current sample sheet names the pair's sample IDs, and the
464
+ import fails if there's none (SACGF/variantgrid#1904).
410
465
  Needs the server feature 'upload_metadata'. Without it, SKIP uploads the file without the
411
466
  metadata (as older clients did) rather than not at all, and ERROR raises
412
467
 
@@ -7,6 +7,7 @@ state (upload ids, pending.json, ...) is kept, so you can poll from any machine
7
7
  """
8
8
  import argparse
9
9
  import hashlib
10
+ import json
10
11
  import logging
11
12
  import os
12
13
  import sys
@@ -31,6 +32,22 @@ def _sha256(path):
31
32
  return h.hexdigest()
32
33
 
33
34
 
35
+ def _enable_verbose_logging():
36
+ """Send the vg_api loggers (the CLI's and the client's requests/responses) to stderr."""
37
+ vg_api_logger = logging.getLogger("vg_api")
38
+ vg_api_logger.setLevel(logging.DEBUG)
39
+ # main() can run more than once in a process (tests) - reuse our handler, pointed at the current stderr
40
+ for handler in vg_api_logger.handlers:
41
+ if getattr(handler, "_vg_api_verbose", False):
42
+ handler.setStream(sys.stderr)
43
+ return
44
+ handler = logging.StreamHandler(sys.stderr)
45
+ handler.setLevel(logging.DEBUG)
46
+ handler.setFormatter(logging.Formatter("%(asctime)s %(levelname)s %(message)s"))
47
+ handler._vg_api_verbose = True
48
+ vg_api_logger.addHandler(handler)
49
+
50
+
34
51
  def _build_api(args):
35
52
  token = args.token or os.environ.get("VARIANTGRID_API_TOKEN")
36
53
  if not token:
@@ -38,7 +55,42 @@ def _build_api(args):
38
55
  server = args.server or os.environ.get("VARIANTGRID_API_SERVER") or DEFAULT_SERVER
39
56
  # Own logger so we can silence the expected 404 when a file hasn't been uploaded yet.
40
57
  logger = logging.getLogger("vg_api.cli")
41
- return VariantGridAPI(server=server, api_token=token, logger=logger), logger
58
+ verbose = args.verbose
59
+ if verbose:
60
+ _enable_verbose_logging()
61
+ api = VariantGridAPI(server=server, api_token=token, logger=logger,
62
+ log_request=verbose, log_response=verbose)
63
+ return api, logger
64
+
65
+
66
+ def _log_status(args, logger, status):
67
+ if args.verbose:
68
+ logger.debug("Upload status: %s", json.dumps(status, indent=2, sort_keys=True))
69
+
70
+
71
+ def _pending_reason(status):
72
+ """Which of the gates before download hasn't passed yet, from the upload_status dict, eg
73
+ 'pipeline Success, import 100%, 2 annotation runs remaining'."""
74
+ parts = []
75
+ if pipeline_status := status.get("pipeline_status"):
76
+ parts.append(f"pipeline {pipeline_status}")
77
+ import_status = status.get("import_status")
78
+ progress = status.get("progress_percent")
79
+ if import_status or progress is not None:
80
+ import_part = "import"
81
+ if import_status:
82
+ import_part += f" {import_status}"
83
+ if progress is not None:
84
+ import_part += f" {progress}%"
85
+ parts.append(import_part)
86
+ remaining = status.get("remaining_annotation_runs")
87
+ if remaining:
88
+ parts.append(f"{remaining} annotation run{'s' if remaining != 1 else ''} remaining")
89
+ elif status.get("annotation_complete") is False:
90
+ parts.append("annotation not complete")
91
+ if status.get("downloads_available") is False:
92
+ parts.append("downloads not available")
93
+ return ", ".join(parts)
42
94
 
43
95
 
44
96
  def _is_not_found(exc):
@@ -46,11 +98,13 @@ def _is_not_found(exc):
46
98
  return resp is not None and resp.status_code == 404
47
99
 
48
100
 
49
- def _probe_status(api, logger, sha256):
101
+ def _probe_status(api, logger, sha256, verbose=False):
50
102
  """Poll status by content hash. Returns the status dict, or None if the server has never
51
- seen this file (404) - a 404 here just means "not uploaded yet", so we silence its logging."""
103
+ seen this file (404) - a 404 here just means "not uploaded yet", so we silence its logging
104
+ unless verbose."""
52
105
  prev_level = logger.level
53
- logger.setLevel(logging.CRITICAL)
106
+ if not verbose:
107
+ logger.setLevel(logging.CRITICAL)
54
108
  try:
55
109
  return api.poll_upload_status(sha256=sha256)
56
110
  except requests.HTTPError as e:
@@ -76,7 +130,7 @@ def annotate_vcf_cmd(args):
76
130
  name = os.path.basename(args.vcf)
77
131
  sha256 = _sha256(args.vcf)
78
132
 
79
- status = _probe_status(api, logger, sha256)
133
+ status = _probe_status(api, logger, sha256, verbose=args.verbose)
80
134
  if status is None:
81
135
  # We've never uploaded this file - do it now.
82
136
  up = api.upload_file(args.vcf, path=None)
@@ -94,14 +148,16 @@ def annotate_vcf_cmd(args):
94
148
  print(f"Annotating {name} - run the same command again later to download.")
95
149
  return EXIT_PENDING
96
150
  if err := status.get("error"):
151
+ _log_status(args, logger, status)
97
152
  print(f"{name}: annotation error - {err}", file=sys.stderr)
98
153
  return EXIT_ERROR
99
154
  if status.get("annotation_complete"):
100
155
  return _download(api, args, sha256)
101
156
 
102
- progress = status.get("progress_percent")
103
- suffix = f" (progress {progress}%)" if progress is not None else ""
104
- print(f"{name}: not ready yet{suffix} - run the same command again later.")
157
+ _log_status(args, logger, status)
158
+ reason = _pending_reason(status)
159
+ reason = f" - {reason}" if reason else ""
160
+ print(f"{name}: not ready yet{reason}. Run the same command again later.")
105
161
  return EXIT_PENDING
106
162
 
107
163
 
@@ -111,6 +167,8 @@ def build_parser():
111
167
  help="VariantGrid server URL (default: $VARIANTGRID_API_SERVER or "
112
168
  f"{DEFAULT_SERVER})")
113
169
  common.add_argument("--token", help="API token (default: $VARIANTGRID_API_TOKEN)")
170
+ common.add_argument("-v", "--verbose", action="store_true",
171
+ help="Log requests, responses and the full upload status to stderr")
114
172
 
115
173
  parser = argparse.ArgumentParser(prog="vg_api", description="VariantGrid API command line tool")
116
174
  sub = parser.add_subparsers(dest="command", required=True)
@@ -142,6 +200,8 @@ def main(argv=None):
142
200
  try:
143
201
  return args.func(args)
144
202
  except AnnotationError as e:
203
+ if e.status is not None:
204
+ _log_status(args, logging.getLogger("vg_api.cli"), e.status)
145
205
  print(f"Annotation error: {e}", file=sys.stderr)
146
206
  return EXIT_ERROR
147
207
  except requests.HTTPError as e: