variantgrid-api 1.3.1__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {variantgrid_api-1.3.1/src/variantgrid_api.egg-info → variantgrid_api-1.4.0}/PKG-INFO +16 -14
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/README.md +17 -15
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/pyproject.toml +1 -1
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api/cli.py +42 -26
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api/data_models.py +12 -8
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0/src/variantgrid_api.egg-info}/PKG-INFO +16 -14
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_api_client.py +20 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_cli.py +29 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/LICENSE +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/setup.cfg +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api/api_client.py +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api/mock_variantgrid_api.py +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api.egg-info/SOURCES.txt +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api.egg-info/dependency_links.txt +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api.egg-info/entry_points.txt +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api.egg-info/requires.txt +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api.egg-info/top_level.txt +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_api_client_annotation.py +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_api_client_bulk.py +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_api_client_validation.py +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_data_models.py +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_mock_variantgrid_api.py +0 -0
- {variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/tests/test_sequencer_model_from_name.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: variantgrid_api
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: A Python API client for VariantGrid
|
|
5
5
|
Author-email: Dave Lawrence <davmlaw@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -68,23 +68,25 @@ result = api.create_enrichment_kit(enrichment_kit)
|
|
|
68
68
|
|
|
69
69
|
## Annotate a VCF and download it back
|
|
70
70
|
|
|
71
|
-
Upload a VCF,
|
|
72
|
-
|
|
71
|
+
Upload a VCF, have VariantGrid import + annotate any novel variants, then download the cohort-level annotated
|
|
72
|
+
export (all samples, single-sample VCFs included). Annotation can take a while, so the quickest way in is the
|
|
73
|
+
`vg_api` command line tool: the first call uploads, and running the same command again downloads the result
|
|
74
|
+
once it's ready.
|
|
73
75
|
|
|
74
|
-
```
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
76
|
+
```console
|
|
77
|
+
$ export VARIANTGRID_API_TOKEN=YOUR_API_TOKEN
|
|
78
|
+
$ vg_api annotate_vcf input.vcf.gz -o results/
|
|
79
|
+
Uploaded input.vcf.gz (id=13256).
|
|
80
|
+
Annotating input.vcf.gz - run the same command again later to download.
|
|
78
81
|
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
print(f"Annotated VCF written to {path}")
|
|
82
|
+
$ vg_api annotate_vcf input.vcf.gz -o results/ # once it's done
|
|
83
|
+
Annotated vcf written to results/input.vcf_annotated_v254_GRCh38.vcf.gz
|
|
82
84
|
```
|
|
83
85
|
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
86
|
+
From Python it's `upload_file()` / `poll_upload_status()` / `download_annotated()`, or the blocking
|
|
87
|
+
`annotate_vcf()` one-liner. See
|
|
88
|
+
**[Annotate a VCF](https://github.com/SACGF/variantgrid_api/wiki/Annotate-a-VCF)** on the wiki for batches, the
|
|
89
|
+
submit-now/download-later pattern, and all the options.
|
|
88
90
|
|
|
89
91
|
## Testing
|
|
90
92
|
|
|
@@ -25,23 +25,25 @@ result = api.create_enrichment_kit(enrichment_kit)
|
|
|
25
25
|
|
|
26
26
|
## Annotate a VCF and download it back
|
|
27
27
|
|
|
28
|
-
Upload a VCF,
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
28
|
+
Upload a VCF, have VariantGrid import + annotate any novel variants, then download the cohort-level annotated
|
|
29
|
+
export (all samples, single-sample VCFs included). Annotation can take a while, so the quickest way in is the
|
|
30
|
+
`vg_api` command line tool: the first call uploads, and running the same command again downloads the result
|
|
31
|
+
once it's ready.
|
|
32
|
+
|
|
33
|
+
```console
|
|
34
|
+
$ export VARIANTGRID_API_TOKEN=YOUR_API_TOKEN
|
|
35
|
+
$ vg_api annotate_vcf input.vcf.gz -o results/
|
|
36
|
+
Uploaded input.vcf.gz (id=13256).
|
|
37
|
+
Annotating input.vcf.gz - run the same command again later to download.
|
|
38
|
+
|
|
39
|
+
$ vg_api annotate_vcf input.vcf.gz -o results/ # once it's done
|
|
40
|
+
Annotated vcf written to results/input.vcf_annotated_v254_GRCh38.vcf.gz
|
|
39
41
|
```
|
|
40
42
|
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
43
|
+
From Python it's `upload_file()` / `poll_upload_status()` / `download_annotated()`, or the blocking
|
|
44
|
+
`annotate_vcf()` one-liner. See
|
|
45
|
+
**[Annotate a VCF](https://github.com/SACGF/variantgrid_api/wiki/Annotate-a-VCF)** on the wiki for batches, the
|
|
46
|
+
submit-now/download-later pattern, and all the options.
|
|
45
47
|
|
|
46
48
|
## Testing
|
|
47
49
|
|
|
@@ -46,6 +46,27 @@ def _is_not_found(exc):
|
|
|
46
46
|
return resp is not None and resp.status_code == 404
|
|
47
47
|
|
|
48
48
|
|
|
49
|
+
def _probe_status(api, logger, sha256):
|
|
50
|
+
"""Poll status by content hash. Returns the status dict, or None if the server has never
|
|
51
|
+
seen this file (404) - a 404 here just means "not uploaded yet", so we silence its logging."""
|
|
52
|
+
prev_level = logger.level
|
|
53
|
+
logger.setLevel(logging.CRITICAL)
|
|
54
|
+
try:
|
|
55
|
+
return api.poll_upload_status(sha256=sha256)
|
|
56
|
+
except requests.HTTPError as e:
|
|
57
|
+
if _is_not_found(e):
|
|
58
|
+
return None
|
|
59
|
+
raise
|
|
60
|
+
finally:
|
|
61
|
+
logger.setLevel(prev_level)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _download(api, args, sha256):
|
|
65
|
+
path = api.download_annotated(sha256=sha256, export_type=args.export_type, dest_path=args.dest)
|
|
66
|
+
print(f"Annotated {args.export_type} written to {path}")
|
|
67
|
+
return EXIT_OK
|
|
68
|
+
|
|
69
|
+
|
|
49
70
|
def annotate_vcf_cmd(args):
|
|
50
71
|
if not os.path.isfile(args.vcf):
|
|
51
72
|
print(f"No such file: {args.vcf}", file=sys.stderr)
|
|
@@ -53,38 +74,30 @@ def annotate_vcf_cmd(args):
|
|
|
53
74
|
|
|
54
75
|
api, logger = _build_api(args)
|
|
55
76
|
name = os.path.basename(args.vcf)
|
|
56
|
-
|
|
57
|
-
if args.wait:
|
|
58
|
-
# Blocking one-shot: upload, wait (can take hours), download.
|
|
59
|
-
path = api.annotate_vcf(args.vcf, export_type=args.export_type, dest_path=args.dest,
|
|
60
|
-
poll_interval=args.poll_interval)
|
|
61
|
-
print(f"Annotated {args.export_type} written to {path}")
|
|
62
|
-
return EXIT_OK
|
|
63
|
-
|
|
64
77
|
sha256 = _sha256(args.vcf)
|
|
65
78
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
except requests.HTTPError as e:
|
|
72
|
-
if _is_not_found(e):
|
|
73
|
-
up = api.upload_file(args.vcf, path=None)
|
|
74
|
-
print(f"Uploaded {name} (id={up['uploaded_file_id']}). "
|
|
75
|
-
f"Annotating - run the same command again later to download.")
|
|
76
|
-
return EXIT_PENDING
|
|
77
|
-
raise
|
|
78
|
-
finally:
|
|
79
|
-
logger.setLevel(prev_level)
|
|
79
|
+
status = _probe_status(api, logger, sha256)
|
|
80
|
+
if status is None:
|
|
81
|
+
# We've never uploaded this file - do it now.
|
|
82
|
+
up = api.upload_file(args.vcf, path=None)
|
|
83
|
+
print(f"Uploaded {name} (id={up['uploaded_file_id']}).")
|
|
80
84
|
|
|
85
|
+
if args.wait:
|
|
86
|
+
# Poll the server ourselves until annotation finishes (can take hours), then download.
|
|
87
|
+
if status is None or not status.get("annotation_complete"):
|
|
88
|
+
print("Waiting for annotation to finish - this can take a while...")
|
|
89
|
+
api.wait_for_annotation(sha256=sha256, timeout=args.timeout, poll_interval=args.poll_interval)
|
|
90
|
+
return _download(api, args, sha256)
|
|
91
|
+
|
|
92
|
+
# Single-shot: report where it's at, download only if it's ready.
|
|
93
|
+
if status is None:
|
|
94
|
+
print(f"Annotating {name} - run the same command again later to download.")
|
|
95
|
+
return EXIT_PENDING
|
|
81
96
|
if err := status.get("error"):
|
|
82
97
|
print(f"{name}: annotation error - {err}", file=sys.stderr)
|
|
83
98
|
return EXIT_ERROR
|
|
84
99
|
if status.get("annotation_complete"):
|
|
85
|
-
|
|
86
|
-
print(f"Annotated {args.export_type} written to {path}")
|
|
87
|
-
return EXIT_OK
|
|
100
|
+
return _download(api, args, sha256)
|
|
88
101
|
|
|
89
102
|
progress = status.get("progress_percent")
|
|
90
103
|
suffix = f" (progress {progress}%)" if progress is not None else ""
|
|
@@ -113,9 +126,12 @@ def build_parser():
|
|
|
113
126
|
p.add_argument("-o", "--dest", default=".",
|
|
114
127
|
help="Destination directory or file for the download (default: current directory)")
|
|
115
128
|
p.add_argument("--wait", action="store_true",
|
|
116
|
-
help="
|
|
129
|
+
help="Poll until annotation finishes and download it, instead of returning immediately "
|
|
130
|
+
"(can take hours)")
|
|
117
131
|
p.add_argument("--poll-interval", type=float, default=10,
|
|
118
132
|
help="Seconds between status polls when using --wait (default: 10)")
|
|
133
|
+
p.add_argument("--timeout", type=float, default=86400,
|
|
134
|
+
help="Give up after this many seconds when using --wait (default: 86400 = 24h)")
|
|
119
135
|
p.set_defaults(func=annotate_vcf_cmd)
|
|
120
136
|
return parser
|
|
121
137
|
|
|
@@ -132,6 +132,14 @@ class VariantCaller:
|
|
|
132
132
|
run_params: Optional[str] = None
|
|
133
133
|
|
|
134
134
|
|
|
135
|
+
@dataclass_json
|
|
136
|
+
@dataclass
|
|
137
|
+
class SequencingSampleLookup:
|
|
138
|
+
""" Only used as arguments to find existing sequencing sample on server - not enough details to create one """
|
|
139
|
+
sample_sheet_lookup: SampleSheetLookup = field(metadata=config(field_name="sample_sheet"))
|
|
140
|
+
sample_name: str
|
|
141
|
+
|
|
142
|
+
|
|
135
143
|
@dataclass_json
|
|
136
144
|
@dataclass
|
|
137
145
|
class JointCalledVCF:
|
|
@@ -139,6 +147,10 @@ class JointCalledVCF:
|
|
|
139
147
|
path: str
|
|
140
148
|
sample_sheet_lookup: SampleSheetLookup = field(metadata=config(field_name="sample_sheet"))
|
|
141
149
|
variant_caller: VariantCaller
|
|
150
|
+
# Set for joint calls that draw samples from more than one sequencing run, eg a family trio.
|
|
151
|
+
# sample_sheet_lookup stays the owning run - the one the path sits under.
|
|
152
|
+
sequencing_samples: Optional[List[SequencingSampleLookup]] = \
|
|
153
|
+
field(default=None, metadata=config(exclude=lambda x: x is None))
|
|
142
154
|
|
|
143
155
|
|
|
144
156
|
@dataclass_json
|
|
@@ -190,14 +202,6 @@ class SequencingFile:
|
|
|
190
202
|
vcf_file: SingleSampleVCF
|
|
191
203
|
|
|
192
204
|
|
|
193
|
-
@dataclass_json
|
|
194
|
-
@dataclass
|
|
195
|
-
class SequencingSampleLookup:
|
|
196
|
-
""" Only used as arguments to find existing sequencing sample on server - not enough details to create one """
|
|
197
|
-
sample_sheet_lookup: SampleSheetLookup = field(metadata=config(field_name="sample_sheet"))
|
|
198
|
-
sample_name: str
|
|
199
|
-
|
|
200
|
-
|
|
201
205
|
@dataclass_json
|
|
202
206
|
@dataclass
|
|
203
207
|
class QC:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: variantgrid_api
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: A Python API client for VariantGrid
|
|
5
5
|
Author-email: Dave Lawrence <davmlaw@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -68,23 +68,25 @@ result = api.create_enrichment_kit(enrichment_kit)
|
|
|
68
68
|
|
|
69
69
|
## Annotate a VCF and download it back
|
|
70
70
|
|
|
71
|
-
Upload a VCF,
|
|
72
|
-
|
|
71
|
+
Upload a VCF, have VariantGrid import + annotate any novel variants, then download the cohort-level annotated
|
|
72
|
+
export (all samples, single-sample VCFs included). Annotation can take a while, so the quickest way in is the
|
|
73
|
+
`vg_api` command line tool: the first call uploads, and running the same command again downloads the result
|
|
74
|
+
once it's ready.
|
|
73
75
|
|
|
74
|
-
```
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
76
|
+
```console
|
|
77
|
+
$ export VARIANTGRID_API_TOKEN=YOUR_API_TOKEN
|
|
78
|
+
$ vg_api annotate_vcf input.vcf.gz -o results/
|
|
79
|
+
Uploaded input.vcf.gz (id=13256).
|
|
80
|
+
Annotating input.vcf.gz - run the same command again later to download.
|
|
78
81
|
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
print(f"Annotated VCF written to {path}")
|
|
82
|
+
$ vg_api annotate_vcf input.vcf.gz -o results/ # once it's done
|
|
83
|
+
Annotated vcf written to results/input.vcf_annotated_v254_GRCh38.vcf.gz
|
|
82
84
|
```
|
|
83
85
|
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
86
|
+
From Python it's `upload_file()` / `poll_upload_status()` / `download_annotated()`, or the blocking
|
|
87
|
+
`annotate_vcf()` one-liner. See
|
|
88
|
+
**[Annotate a VCF](https://github.com/SACGF/variantgrid_api/wiki/Annotate-a-VCF)** on the wiki for batches, the
|
|
89
|
+
submit-now/download-later pattern, and all the options.
|
|
88
90
|
|
|
89
91
|
## Testing
|
|
90
92
|
|
|
@@ -97,6 +97,26 @@ def test_create_joint_called_vcf_posts_json(api, server, vg_objects):
|
|
|
97
97
|
url
|
|
98
98
|
)
|
|
99
99
|
assert body["path"].endswith(".vcf.gz")
|
|
100
|
+
# A single-run joint call leaves the key off entirely, so the server keeps its existing behaviour
|
|
101
|
+
assert "sequencing_samples" not in body
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@responses.activate
|
|
105
|
+
def test_create_joint_called_vcf_posts_cross_run_members(api, server, vg_objects):
|
|
106
|
+
url = f"{server}/seqauto/api/v1/joint_called_vcf/"
|
|
107
|
+
body = assert_post(
|
|
108
|
+
lambda: api.create_joint_called_vcf(
|
|
109
|
+
vg_objects["cross_run_joint_called_vcf"]
|
|
110
|
+
),
|
|
111
|
+
url
|
|
112
|
+
)
|
|
113
|
+
members = body["sequencing_samples"]
|
|
114
|
+
assert [m["sample_name"] for m in members] == ["fake_sample_1", "fake_sample_mum", "fake_sample_dad"]
|
|
115
|
+
# Members carry their own sample sheet, so two of them point at the run that sequenced the parents
|
|
116
|
+
runs = {m["sample_sheet"]["sequencing_run"] for m in members}
|
|
117
|
+
assert len(runs) == 2
|
|
118
|
+
# The owning sheet is still the run the path sits under
|
|
119
|
+
assert body["sample_sheet"]["sequencing_run"] == vg_objects["SEQUENCING_RUN_NAME"]
|
|
100
120
|
|
|
101
121
|
|
|
102
122
|
@responses.activate
|
|
@@ -87,6 +87,35 @@ def test_csv_export_type(vcf, tmp_path):
|
|
|
87
87
|
assert download_call.request.url.endswith("/csv")
|
|
88
88
|
|
|
89
89
|
|
|
90
|
+
@responses.activate
|
|
91
|
+
def test_wait_uploads_polls_then_downloads(vcf, tmp_path):
|
|
92
|
+
"""--wait on a new file: upload, poll until complete, then download - no external loop."""
|
|
93
|
+
responses.add(responses.GET, STATUS_RE, json={"detail": "not found"}, status=404) # probe
|
|
94
|
+
responses.add(responses.POST, UPLOAD_URL, json={"uploaded_file_id": 7}, status=200)
|
|
95
|
+
responses.add(responses.GET, STATUS_RE, json={"annotation_complete": False, "error": None}, status=200)
|
|
96
|
+
responses.add(responses.GET, STATUS_RE, json={"annotation_complete": True, "error": None}, status=200)
|
|
97
|
+
responses.add(responses.GET, DOWNLOAD_RE, body=b"data", status=200,
|
|
98
|
+
headers={"Content-Disposition": 'attachment; filename="out.vcf.gz"'})
|
|
99
|
+
|
|
100
|
+
rc = cli.main(_argv(vcf, "--wait", "-o", str(tmp_path), "--poll-interval", "0"))
|
|
101
|
+
|
|
102
|
+
assert rc == cli.EXIT_OK
|
|
103
|
+
assert (tmp_path / "out.vcf.gz").read_bytes() == b"data"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@responses.activate
|
|
107
|
+
def test_wait_downloads_without_reupload_when_already_known(vcf, tmp_path):
|
|
108
|
+
"""--wait on an already-uploaded, already-complete file must not re-upload."""
|
|
109
|
+
responses.add(responses.GET, STATUS_RE, json={"annotation_complete": True, "error": None}, status=200)
|
|
110
|
+
responses.add(responses.GET, DOWNLOAD_RE, body=b"data", status=200,
|
|
111
|
+
headers={"Content-Disposition": 'attachment; filename="out.vcf.gz"'})
|
|
112
|
+
|
|
113
|
+
rc = cli.main(_argv(vcf, "--wait", "-o", str(tmp_path), "--poll-interval", "0"))
|
|
114
|
+
|
|
115
|
+
assert rc == cli.EXIT_OK
|
|
116
|
+
assert not any(c.request.method == "POST" for c in responses.calls)
|
|
117
|
+
|
|
118
|
+
|
|
90
119
|
def test_missing_file_is_error(tmp_path):
|
|
91
120
|
rc = cli.main(_argv(str(tmp_path / "nope.vcf")))
|
|
92
121
|
assert rc == cli.EXIT_ERROR
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{variantgrid_api-1.3.1 → variantgrid_api-1.4.0}/src/variantgrid_api.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|