fhr 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fhr-0.1.1/LICENSE +73 -0
- fhr-0.1.1/PKG-INFO +255 -0
- fhr-0.1.1/README.md +234 -0
- fhr-0.1.1/fhr.py +452 -0
- fhr-0.1.1/pyproject.toml +40 -0
fhr-0.1.1/LICENSE
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
CONTENTS
|
|
2
|
+
|
|
3
|
+
Public Domain Notice
|
|
4
|
+
Exceptions (for bundled 3rd-party code)
|
|
5
|
+
Copyright F.A.Q.
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
==============================================================
|
|
9
|
+
PUBLIC DOMAIN NOTICE
|
|
10
|
+
United States Department of Agriculture
|
|
11
|
+
Agricultural Research Service
|
|
12
|
+
|
|
13
|
+
With the exception of certain third-party files summarized below, this
|
|
14
|
+
software is a "United States Government Work" under the terms of the
|
|
15
|
+
United States Copyright Act. It was written as part of the authors'
|
|
16
|
+
official duties as United States Government employees and thus cannot
|
|
17
|
+
be copyrighted. This software is freely available to the public for
|
|
18
|
+
use. The United States Department of Agriculture, Agricultural
|
|
19
|
+
Research Service (USDA - ARS) and the U.S. Government have not placed
|
|
20
|
+
any restriction on its use or reproduction.
|
|
21
|
+
|
|
22
|
+
Although all reasonable efforts have been taken to ensure the accuracy
|
|
23
|
+
and reliability of the software and data, the USDA ARS and the U.S.
|
|
24
|
+
Government do not and cannot warrant the performance or results tha may
|
|
25
|
+
be obtained by using this software or data. The USDA ARS and the U.S.
|
|
26
|
+
Government disclaim all warranties, express or implied, including
|
|
27
|
+
warranties of performance, merchantability or fitness for any particular
|
|
28
|
+
purpose.
|
|
29
|
+
|
|
30
|
+
Please cite the authors in any work or product based on this material.
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
==============================================================
|
|
34
|
+
EXCEPTIONS (in all cases excluding USDA-ARS-written makefiles):
|
|
35
|
+
|
|
36
|
+
Location:
|
|
37
|
+
Author:
|
|
38
|
+
License:
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
==============================================================
|
|
42
|
+
Copyright F.A.Q.
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
--------------------------------------------------------------
|
|
46
|
+
Q. Our product makes use of the USDA - ARS source code, and we made changes
|
|
47
|
+
and additions to that version of the USDA - ARS code to better fit it to
|
|
48
|
+
our needs. Can we copyright the code, and how?
|
|
49
|
+
|
|
50
|
+
A. You can copyright only the *changes* or the *additions* you made to the
|
|
51
|
+
NCBI source code. You should identify unambiguously those sections of
|
|
52
|
+
the code that were modified, e.g. by commenting any changes you made
|
|
53
|
+
in the code you distribute. Therefore, your license has to make clear
|
|
54
|
+
to users that your product is a combination of code that is public domain
|
|
55
|
+
within the U.S. (but may be subject to copyright by the U.S. in foreign
|
|
56
|
+
countries) and code that has been created or modified by you.
|
|
57
|
+
|
|
58
|
+
--------------------------------------------------------------
|
|
59
|
+
Q. Can we (re)license all or part of the USDA - ARS source code?
|
|
60
|
+
|
|
61
|
+
A. No, you cannot license or relicense the source code written by USDA - ARS
|
|
62
|
+
since you cannot claim any copyright in the software that was developed
|
|
63
|
+
at USDA - ARS as a 'government work' and consequently is in the public
|
|
64
|
+
domain within the U.S.
|
|
65
|
+
|
|
66
|
+
--------------------------------------------------------------
|
|
67
|
+
Q. What if these copyright guidelines are not clear enough or are not
|
|
68
|
+
applicable to my particular case?
|
|
69
|
+
|
|
70
|
+
A. Contact us. Send your questions to 'answers@usda.gov'.
|
|
71
|
+
--------------------------------------------------------------
|
|
72
|
+
|
|
73
|
+
This file was modified from the NCBI Boilerplate LICENSE file
|
fhr-0.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: fhr
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: This tool is used to validate and convert between different FHR header serializations
|
|
5
|
+
License: USDA-ARS
|
|
6
|
+
Author: David Molik
|
|
7
|
+
Author-email: david.molik@usda.gov
|
|
8
|
+
Requires-Python: >=3.9,<4.0
|
|
9
|
+
Classifier: License :: Other/Proprietary License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Requires-Dist: argparse (>=1.4.0,<2.0.0)
|
|
16
|
+
Requires-Dist: jsonschema (>=4.21.1,<5.0.0)
|
|
17
|
+
Requires-Dist: microdata (>=0.8.0,<0.9.0)
|
|
18
|
+
Requires-Dist: pyyaml (>=6.0.1,<7.0.0)
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
# FHR-File-Converter
|
|
22
|
+
[](https://doi.org/10.5281/zenodo.6762547)
|
|
23
|
+
|
|
24
|
+
This is the fhr file converter, it can convert fhr inbetween json, fasta, microdata, and fasta header. If you would like a detailed specification of fhr, see [FHR-Specification](https://github.com/FAIR-bioHeaders/FHR-Specification)
|
|
25
|
+
|
|
26
|
+
## Installation
|
|
27
|
+
|
|
28
|
+
You can install the FHR file converter and its dependencies using Poetry:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
poetry install
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Usage
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
### Commnand line Usage
|
|
38
|
+
|
|
39
|
+
Using FHR on the command line:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
fhr-convert <input>.<yaml|json|fasta|html> <output>.<yaml|json|fasta|html>
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Detailed Usage:
|
|
46
|
+
|
|
47
|
+
```
|
|
48
|
+
usage: fhr-convert [-h] [--version] <file> <file>
|
|
49
|
+
|
|
50
|
+
Convert from one FHR supported file type to another
|
|
51
|
+
|
|
52
|
+
positional arguments:
|
|
53
|
+
<file> input followed by output
|
|
54
|
+
|
|
55
|
+
optional arguments:
|
|
56
|
+
-h, --help show this help message and exit
|
|
57
|
+
--version show program's version number and exit
|
|
58
|
+
|
|
59
|
+
positional <file> input and output files
|
|
60
|
+
input files can be one of:
|
|
61
|
+
<input>.yml
|
|
62
|
+
<input>.fasta - fasta contining a fhr header
|
|
63
|
+
<input>.html - html containing microdata
|
|
64
|
+
|
|
65
|
+
output files can be one of:
|
|
66
|
+
<output>.yml
|
|
67
|
+
<output>.fasta - fasta output type will be made as a fasta header without sequences
|
|
68
|
+
<output>.html - microdata output type will be made into generic html output
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Validating an FHR file on command line
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
fhr-validate <input>.<yaml|json|fasta|html>
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Detailed Usage:
|
|
79
|
+
|
|
80
|
+
```
|
|
81
|
+
usage: fhr-validate [-h] [--version] <file>
|
|
82
|
+
|
|
83
|
+
Validate a fhr containing file
|
|
84
|
+
|
|
85
|
+
positional <file> input and output files
|
|
86
|
+
input files can be one of:
|
|
87
|
+
<input>.yml
|
|
88
|
+
<input>.fasta - fasta contining a fhr header
|
|
89
|
+
<input>.html - html containing microdata
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
As such validating a yaml file named "important\_genome.fhr.yml" would be:
|
|
94
|
+
|
|
95
|
+
`fhr-validate important_genome.fhr.yml`
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
## Using FHR in Python
|
|
100
|
+
|
|
101
|
+
To use FHR libabry in Python
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
>>> from fhr import fhr
|
|
105
|
+
>>> file = open("example.yaml")
|
|
106
|
+
>>> data = fhr()
|
|
107
|
+
>>> data.input_yaml(file.read())
|
|
108
|
+
>>> data.output_fasta()
|
|
109
|
+
";~schema: https://raw.githubusercontent.com/FFRGS/FFRGS-Specification/main/fhr.json\n;~schemaVersion: 1\n;~genome: Bombas huntii\n;~version: 0.0.1\n;~author:;~ name:Adam Wright\n;~ url:https://wormbase.org/resource/person/WBPerson30813\n;~assembler:;~ name:David Molik\n;~ url:https:/david.molik.co/person\n;~place:;~ name:PBARC\n;~ url:https://www.ars.usda.gov/pacific-west-area/hilo-hi/daniel-k-inouye-us-pacific-basin-agricultural-research-center/\n;~taxa: Bombas huntii\n;~assemblySoftware: HiFiASM\n;~physicalSample: Located in Freezer 33, Drawer 137\n;~dateCreated: 2022-03-21\n;~instrument: ['Sequel IIe', 'Nanopore']\n;~scholarlyArticle: https://doi.org/10.1371/journal.pntd.0008755\n;~documentation: Built assembly from... \n;~identifier: ['gkx10242566416842']\n;~relatedLink: ['https/david.molik.co/genome']\n;~funding: some\n;~reuseConditions: public domain\n"
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Checksums
|
|
113
|
+
|
|
114
|
+
The FHR stores checksums, allowing the FASTA header of the reference genome to contain the checksum for the FASTA file without the header.
|
|
115
|
+
|
|
116
|
+
To utilize the checksum, strip the FASTA header:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
cat example.fasta | grep -E -v '^;~\s?checksum' > example.check.fasta
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
To strip the checksum:
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
cat example.fasta | grep -E ';~\s?checksum' | sed 's/^;~checksum://g' | sed '/\'//g'
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## Docker Support
|
|
130
|
+
|
|
131
|
+
You can also run the FHR file converter in a Docker container. To build the Docker image:
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
docker build -t fhr-file-converter .
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
And then run the Docker container:
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
docker run -it --rm fhr-file-converter
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
## Running Code Quality Checks
|
|
145
|
+
|
|
146
|
+
Ensuring code quality is crucial for maintaining a healthy and sustainable codebase. The following tools help enforce coding standards and best practices:
|
|
147
|
+
|
|
148
|
+
### isort
|
|
149
|
+
|
|
150
|
+
`isort` is a tool that sorts Python imports alphabetically within each section and separated by a blank line. It ensures consistent import styles across your project.
|
|
151
|
+
|
|
152
|
+
To run isort, use the following command:
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
poetry run isort .
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
### ruff
|
|
159
|
+
ruff is a lightweight linter for Python that aims to detect common programming errors, stylistic issues, and code smells. It provides quick feedback on potential issues in your code.
|
|
160
|
+
|
|
161
|
+
To run ruff, use the following command:
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
poetry run ruff .
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
### black
|
|
168
|
+
|
|
169
|
+
`black` is an uncompromising Python code formatter. It reformats entire files in place to ensure a consistent and readable code style. It's opinionated and strives for the smallest diffs possible.
|
|
170
|
+
|
|
171
|
+
To run black, use the following command:
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
poetry run black .
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
Running these code quality checks regularly helps maintain a clean and consistent codebase, making it easier to collaborate with others and ensuring code readability and maintainability. These checks are required to pass in order to pull changes into the main branch.
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
### pytest
|
|
181
|
+
|
|
182
|
+
Make sure you install depedencies first and then run the tests with poetry
|
|
183
|
+
```bash
|
|
184
|
+
poetry run install
|
|
185
|
+
poetry run pytest
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
## Citing FHR
|
|
189
|
+
Information on Citations of FHR
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
### Citing the Validation Tool
|
|
193
|
+
cite the validation tool when directly interacting with the tool or library
|
|
194
|
+
The APA citation for the [FHR validation/converter software](https://github.com/FAIR-bioHeaders/FHR-File-Converter) is:
|
|
195
|
+
|
|
196
|
+
```
|
|
197
|
+
Molik, D., & Wright, A. FHR File Converster [Computer software]. https://github.com/FAIR-bioHeaders/FHR-File-Converter
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
Or in bibtex:
|
|
201
|
+
```bibtex
|
|
202
|
+
% Citation For FHR Validation/Converter Software
|
|
203
|
+
@software{FHR_File_Converter,
|
|
204
|
+
author = {Molik, David and Wright, Adam},
|
|
205
|
+
year = {2023},
|
|
206
|
+
license = {PDDL-1.0},
|
|
207
|
+
title = {{FHR File Converster}},
|
|
208
|
+
url = {https://github.com/FAIR-bioHeaders/FHR-File-Converter},
|
|
209
|
+
doi = {10.5281/zenodo.6762547}
|
|
210
|
+
}
|
|
211
|
+
```
|
|
212
|
+
### Citing the Specification
|
|
213
|
+
cite the specification when directly interacting with the specification (pull requests, comments on schema)
|
|
214
|
+
The APA citation for the [FHR specification](https://github.com/FAIR-bioHeaders/FHR-Specification) is:
|
|
215
|
+
|
|
216
|
+
```
|
|
217
|
+
Molik, D., & Wright, A. FHR Specification [Data set]. https://github.com/FAIR-bioHeaders/FHR-Specification
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
Or in bibtex:
|
|
221
|
+
```bibtex
|
|
222
|
+
% Citation For FHR Specification
|
|
223
|
+
@misc{FHR_Specification,
|
|
224
|
+
author = {Molik, David and Wright, Adam},
|
|
225
|
+
year = {2023},
|
|
226
|
+
title = {{FHR Specification}},
|
|
227
|
+
url = {https://github.com/FAIR-bioHeaders/FHR-Specification},
|
|
228
|
+
doi = {10.5281/zenodo.6762549}
|
|
229
|
+
}
|
|
230
|
+
```
|
|
231
|
+
### Citing the Preprint
|
|
232
|
+
**(best option)** cite the preprint talking about the effort, or want a broad citation of FHR
|
|
233
|
+
The APA citation for the [FHR preprint](https://www.biorxiv.org/content/10.1101/2023.11.29.569306v1) is:
|
|
234
|
+
|
|
235
|
+
```
|
|
236
|
+
Wright, A., Wilkinson, M. D., Mungall, C., Cain, S., Richards, S., Sternberg, P., ... & Molik, D. C. (2023). Data Resources and Analyses Fair Header Reference genome: A Trustworthy standard. bioRxiv, 2023-11.
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
Or in bibtex:
|
|
240
|
+
```bibtex
|
|
241
|
+
% Citation For FHR Pre-print
|
|
242
|
+
@article {Wright2023,
|
|
243
|
+
author = {Adam Wright and Mark D Wilkinson and Chris Mungall and Scott Cain and Stephen Richards and Paul Sternberg and Ellen Provin and Jonathan L Jacobs and Scott Geib and Daniela Raciti and Karen Yook and Lincoln Stein and David C Molik},
|
|
244
|
+
title = {DATA RESOURCES AND ANALYSES FAIR Header Reference genome: A TRUSTworthy standard},
|
|
245
|
+
elocation-id = {2023.11.29.569306},
|
|
246
|
+
year = {2023},
|
|
247
|
+
doi = {10.1101/2023.11.29.569306},
|
|
248
|
+
publisher = {Cold Spring Harbor Laboratory},
|
|
249
|
+
abstract = {The lack of interoperable data standards among reference genome data-sharing platforms inhibits cross-platform analysis while increasing the risk of data provenance loss. Here, we describe the FAIR-bioHeaders Reference genome (FHR), a metadata standard guided by the principles of Findability, Accessibility, Interoperability, and Reuse (FAIR) in addition to the principles of Transparency, Responsibility, User focus, Sustainability, and Technology (TRUST). The objective of FHR is to provide an extensive set of data serialisation methods and minimum data field requirements while still maintaining extensibility, flexibility, and expressivity in an increasingly decentralised genomic data ecosystem. The effort needed to implement FHR is low; FHR{\textquoteright}s design philosophy ensures easy implementation while retaining the benefits gained from recording both machine and human-readable provenance.Competing Interest StatementThe authors have declared no competing interest.},
|
|
250
|
+
URL = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306},
|
|
251
|
+
eprint = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306.full.pdf},
|
|
252
|
+
journal = {bioRxiv}
|
|
253
|
+
}
|
|
254
|
+
```
|
|
255
|
+
|
fhr-0.1.1/README.md
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
# FHR-File-Converter
|
|
2
|
+
[](https://doi.org/10.5281/zenodo.6762547)
|
|
3
|
+
|
|
4
|
+
This is the fhr file converter, it can convert fhr inbetween json, fasta, microdata, and fasta header. If you would like a detailed specification of fhr, see [FHR-Specification](https://github.com/FAIR-bioHeaders/FHR-Specification)
|
|
5
|
+
|
|
6
|
+
## Installation
|
|
7
|
+
|
|
8
|
+
You can install the FHR file converter and its dependencies using Poetry:
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
poetry install
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
## Usage
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
### Commnand line Usage
|
|
18
|
+
|
|
19
|
+
Using FHR on the command line:
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
fhr-convert <input>.<yaml|json|fasta|html> <output>.<yaml|json|fasta|html>
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Detailed Usage:
|
|
26
|
+
|
|
27
|
+
```
|
|
28
|
+
usage: fhr-convert [-h] [--version] <file> <file>
|
|
29
|
+
|
|
30
|
+
Convert from one FHR supported file type to another
|
|
31
|
+
|
|
32
|
+
positional arguments:
|
|
33
|
+
<file> input followed by output
|
|
34
|
+
|
|
35
|
+
optional arguments:
|
|
36
|
+
-h, --help show this help message and exit
|
|
37
|
+
--version show program's version number and exit
|
|
38
|
+
|
|
39
|
+
positional <file> input and output files
|
|
40
|
+
input files can be one of:
|
|
41
|
+
<input>.yml
|
|
42
|
+
<input>.fasta - fasta contining a fhr header
|
|
43
|
+
<input>.html - html containing microdata
|
|
44
|
+
|
|
45
|
+
output files can be one of:
|
|
46
|
+
<output>.yml
|
|
47
|
+
<output>.fasta - fasta output type will be made as a fasta header without sequences
|
|
48
|
+
<output>.html - microdata output type will be made into generic html output
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Validating an FHR file on command line
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
fhr-validate <input>.<yaml|json|fasta|html>
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Detailed Usage:
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
usage: fhr-validate [-h] [--version] <file>
|
|
62
|
+
|
|
63
|
+
Validate a fhr containing file
|
|
64
|
+
|
|
65
|
+
positional <file> input and output files
|
|
66
|
+
input files can be one of:
|
|
67
|
+
<input>.yml
|
|
68
|
+
<input>.fasta - fasta contining a fhr header
|
|
69
|
+
<input>.html - html containing microdata
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
As such validating a yaml file named "important\_genome.fhr.yml" would be:
|
|
74
|
+
|
|
75
|
+
`fhr-validate important_genome.fhr.yml`
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
## Using FHR in Python
|
|
80
|
+
|
|
81
|
+
To use FHR libabry in Python
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
>>> from fhr import fhr
|
|
85
|
+
>>> file = open("example.yaml")
|
|
86
|
+
>>> data = fhr()
|
|
87
|
+
>>> data.input_yaml(file.read())
|
|
88
|
+
>>> data.output_fasta()
|
|
89
|
+
";~schema: https://raw.githubusercontent.com/FFRGS/FFRGS-Specification/main/fhr.json\n;~schemaVersion: 1\n;~genome: Bombas huntii\n;~version: 0.0.1\n;~author:;~ name:Adam Wright\n;~ url:https://wormbase.org/resource/person/WBPerson30813\n;~assembler:;~ name:David Molik\n;~ url:https:/david.molik.co/person\n;~place:;~ name:PBARC\n;~ url:https://www.ars.usda.gov/pacific-west-area/hilo-hi/daniel-k-inouye-us-pacific-basin-agricultural-research-center/\n;~taxa: Bombas huntii\n;~assemblySoftware: HiFiASM\n;~physicalSample: Located in Freezer 33, Drawer 137\n;~dateCreated: 2022-03-21\n;~instrument: ['Sequel IIe', 'Nanopore']\n;~scholarlyArticle: https://doi.org/10.1371/journal.pntd.0008755\n;~documentation: Built assembly from... \n;~identifier: ['gkx10242566416842']\n;~relatedLink: ['https/david.molik.co/genome']\n;~funding: some\n;~reuseConditions: public domain\n"
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Checksums
|
|
93
|
+
|
|
94
|
+
The FHR stores checksums, allowing the FASTA header of the reference genome to contain the checksum for the FASTA file without the header.
|
|
95
|
+
|
|
96
|
+
To utilize the checksum, strip the FASTA header:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
cat example.fasta | grep -E -v '^;~\s?checksum' > example.check.fasta
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
To strip the checksum:
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
cat example.fasta | grep -E ';~\s?checksum' | sed 's/^;~checksum://g' | sed '/\'//g'
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Docker Support
|
|
110
|
+
|
|
111
|
+
You can also run the FHR file converter in a Docker container. To build the Docker image:
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
docker build -t fhr-file-converter .
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
And then run the Docker container:
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
docker run -it --rm fhr-file-converter
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
## Running Code Quality Checks
|
|
125
|
+
|
|
126
|
+
Ensuring code quality is crucial for maintaining a healthy and sustainable codebase. The following tools help enforce coding standards and best practices:
|
|
127
|
+
|
|
128
|
+
### isort
|
|
129
|
+
|
|
130
|
+
`isort` is a tool that sorts Python imports alphabetically within each section and separated by a blank line. It ensures consistent import styles across your project.
|
|
131
|
+
|
|
132
|
+
To run isort, use the following command:
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
poetry run isort .
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
### ruff
|
|
139
|
+
ruff is a lightweight linter for Python that aims to detect common programming errors, stylistic issues, and code smells. It provides quick feedback on potential issues in your code.
|
|
140
|
+
|
|
141
|
+
To run ruff, use the following command:
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
poetry run ruff .
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
### black
|
|
148
|
+
|
|
149
|
+
`black` is an uncompromising Python code formatter. It reformats entire files in place to ensure a consistent and readable code style. It's opinionated and strives for the smallest diffs possible.
|
|
150
|
+
|
|
151
|
+
To run black, use the following command:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
poetry run black .
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
Running these code quality checks regularly helps maintain a clean and consistent codebase, making it easier to collaborate with others and ensuring code readability and maintainability. These checks are required to pass in order to pull changes into the main branch.
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
### pytest
|
|
161
|
+
|
|
162
|
+
Make sure you install depedencies first and then run the tests with poetry
|
|
163
|
+
```bash
|
|
164
|
+
poetry run install
|
|
165
|
+
poetry run pytest
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
## Citing FHR
|
|
169
|
+
Information on Citations of FHR
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
### Citing the Validation Tool
|
|
173
|
+
cite the validation tool when directly interacting with the tool or library
|
|
174
|
+
The APA citation for the [FHR validation/converter software](https://github.com/FAIR-bioHeaders/FHR-File-Converter) is:
|
|
175
|
+
|
|
176
|
+
```
|
|
177
|
+
Molik, D., & Wright, A. FHR File Converster [Computer software]. https://github.com/FAIR-bioHeaders/FHR-File-Converter
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
Or in bibtex:
|
|
181
|
+
```bibtex
|
|
182
|
+
% Citation For FHR Validation/Converter Software
|
|
183
|
+
@software{FHR_File_Converter,
|
|
184
|
+
author = {Molik, David and Wright, Adam},
|
|
185
|
+
year = {2023},
|
|
186
|
+
license = {PDDL-1.0},
|
|
187
|
+
title = {{FHR File Converster}},
|
|
188
|
+
url = {https://github.com/FAIR-bioHeaders/FHR-File-Converter},
|
|
189
|
+
doi = {10.5281/zenodo.6762547}
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
### Citing the Specification
|
|
193
|
+
cite the specification when directly interacting with the specification (pull requests, comments on schema)
|
|
194
|
+
The APA citation for the [FHR specification](https://github.com/FAIR-bioHeaders/FHR-Specification) is:
|
|
195
|
+
|
|
196
|
+
```
|
|
197
|
+
Molik, D., & Wright, A. FHR Specification [Data set]. https://github.com/FAIR-bioHeaders/FHR-Specification
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
Or in bibtex:
|
|
201
|
+
```bibtex
|
|
202
|
+
% Citation For FHR Specification
|
|
203
|
+
@misc{FHR_Specification,
|
|
204
|
+
author = {Molik, David and Wright, Adam},
|
|
205
|
+
year = {2023},
|
|
206
|
+
title = {{FHR Specification}},
|
|
207
|
+
url = {https://github.com/FAIR-bioHeaders/FHR-Specification},
|
|
208
|
+
doi = {10.5281/zenodo.6762549}
|
|
209
|
+
}
|
|
210
|
+
```
|
|
211
|
+
### Citing the Preprint
|
|
212
|
+
**(best option)** cite the preprint talking about the effort, or want a broad citation of FHR
|
|
213
|
+
The APA citation for the [FHR preprint](https://www.biorxiv.org/content/10.1101/2023.11.29.569306v1) is:
|
|
214
|
+
|
|
215
|
+
```
|
|
216
|
+
Wright, A., Wilkinson, M. D., Mungall, C., Cain, S., Richards, S., Sternberg, P., ... & Molik, D. C. (2023). Data Resources and Analyses Fair Header Reference genome: A Trustworthy standard. bioRxiv, 2023-11.
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
Or in bibtex:
|
|
220
|
+
```bibtex
|
|
221
|
+
% Citation For FHR Pre-print
|
|
222
|
+
@article {Wright2023,
|
|
223
|
+
author = {Adam Wright and Mark D Wilkinson and Chris Mungall and Scott Cain and Stephen Richards and Paul Sternberg and Ellen Provin and Jonathan L Jacobs and Scott Geib and Daniela Raciti and Karen Yook and Lincoln Stein and David C Molik},
|
|
224
|
+
title = {DATA RESOURCES AND ANALYSES FAIR Header Reference genome: A TRUSTworthy standard},
|
|
225
|
+
elocation-id = {2023.11.29.569306},
|
|
226
|
+
year = {2023},
|
|
227
|
+
doi = {10.1101/2023.11.29.569306},
|
|
228
|
+
publisher = {Cold Spring Harbor Laboratory},
|
|
229
|
+
abstract = {The lack of interoperable data standards among reference genome data-sharing platforms inhibits cross-platform analysis while increasing the risk of data provenance loss. Here, we describe the FAIR-bioHeaders Reference genome (FHR), a metadata standard guided by the principles of Findability, Accessibility, Interoperability, and Reuse (FAIR) in addition to the principles of Transparency, Responsibility, User focus, Sustainability, and Technology (TRUST). The objective of FHR is to provide an extensive set of data serialisation methods and minimum data field requirements while still maintaining extensibility, flexibility, and expressivity in an increasingly decentralised genomic data ecosystem. The effort needed to implement FHR is low; FHR{\textquoteright}s design philosophy ensures easy implementation while retaining the benefits gained from recording both machine and human-readable provenance.Competing Interest StatementThe authors have declared no competing interest.},
|
|
230
|
+
URL = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306},
|
|
231
|
+
eprint = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306.full.pdf},
|
|
232
|
+
journal = {bioRxiv}
|
|
233
|
+
}
|
|
234
|
+
```
|
fhr-0.1.1/fhr.py
ADDED
|
@@ -0,0 +1,452 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import re
|
|
3
|
+
from typing import Dict, List, Union
|
|
4
|
+
|
|
5
|
+
import microdata # type: ignore
|
|
6
|
+
import yaml
|
|
7
|
+
from jsonschema import validate
|
|
8
|
+
|
|
9
|
+
with open("fhr_schema.json", "r") as f:
|
|
10
|
+
schema = json.load(f)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class fhr:
|
|
14
|
+
|
|
15
|
+
def __init__(
|
|
16
|
+
self,
|
|
17
|
+
schema: str = "",
|
|
18
|
+
version: str = "",
|
|
19
|
+
schemaVersion: int = 0,
|
|
20
|
+
genome: str = "",
|
|
21
|
+
assemblySoftware: str = "",
|
|
22
|
+
voucherSpecimen: str = "",
|
|
23
|
+
dateCreated: str = "",
|
|
24
|
+
scholarlyArticle: str = "",
|
|
25
|
+
documentation: str = "",
|
|
26
|
+
reuseConditions: str = "",
|
|
27
|
+
vitalStats: Dict[str, Union[int, str]] = {},
|
|
28
|
+
masking: str = "",
|
|
29
|
+
checksum: str = "",
|
|
30
|
+
) -> None:
|
|
31
|
+
self.schema: str = schema
|
|
32
|
+
self.version: str = version
|
|
33
|
+
self.schemaVersion: int = schemaVersion
|
|
34
|
+
self.genome: str = genome
|
|
35
|
+
self.genomeSynonym: List[str] = []
|
|
36
|
+
self.metadataAuthor: List[Dict[str, str]] = []
|
|
37
|
+
self.assemblyAuthor: List[Dict[str, str]] = []
|
|
38
|
+
self.accessionID: Dict[str, str] = {}
|
|
39
|
+
self.taxon: Dict[str, str] = {}
|
|
40
|
+
self.assemblySoftware: str = assemblySoftware
|
|
41
|
+
self.voucherSpecimen: str = voucherSpecimen
|
|
42
|
+
self.dateCreated: str = dateCreated
|
|
43
|
+
self.instrument: List[str] = []
|
|
44
|
+
self.scholarlyArticle: str = scholarlyArticle
|
|
45
|
+
self.documentation: str = documentation
|
|
46
|
+
self.identifier: List[str] = []
|
|
47
|
+
self.relatedLink: List[str] = []
|
|
48
|
+
self.funding: List[str] = []
|
|
49
|
+
self.masking: str = masking
|
|
50
|
+
self.vitalStats: Dict[str, Union[int, str]] = vitalStats
|
|
51
|
+
self.reuseConditions: str = reuseConditions
|
|
52
|
+
self.checksum: str = checksum
|
|
53
|
+
|
|
54
|
+
def input_yaml(self, stream: str):
|
|
55
|
+
data = yaml.safe_load(stream)
|
|
56
|
+
self.schema = data["schema"]
|
|
57
|
+
self.version = data["version"]
|
|
58
|
+
self.schemaVersion = data["schemaVersion"]
|
|
59
|
+
self.genome = data["genome"]
|
|
60
|
+
self.taxon["name"] = data["taxon"]["name"]
|
|
61
|
+
self.taxon["uri"] = data["taxon"]["uri"]
|
|
62
|
+
self.metadataAuthor = data["metadataAuthor"]
|
|
63
|
+
self.assemblyAuthor = data["assemblyAuthor"]
|
|
64
|
+
self.dateCreated = data["dateCreated"]
|
|
65
|
+
self.masking = data["masking"]
|
|
66
|
+
self.checksum = data["checksum"]
|
|
67
|
+
|
|
68
|
+
try:
|
|
69
|
+
self.genomeSynonym = data["genomeSynonym"]
|
|
70
|
+
except KeyError:
|
|
71
|
+
self.genomeSynonym = None
|
|
72
|
+
|
|
73
|
+
try:
|
|
74
|
+
self.accessionID["url"] = data["accessionID"]["url"]
|
|
75
|
+
except (KeyError, TypeError):
|
|
76
|
+
self.accessionID["url"] = None
|
|
77
|
+
|
|
78
|
+
try:
|
|
79
|
+
self.taxon["name"] = data["taxon"]["name"]
|
|
80
|
+
self.taxon["uri"] = data["taxon"]["uri"]
|
|
81
|
+
except KeyError:
|
|
82
|
+
self.taxon["name"] = None
|
|
83
|
+
self.taxon["uri"] = None
|
|
84
|
+
|
|
85
|
+
try:
|
|
86
|
+
self.assemblySoftware = data["assemblySoftware"]
|
|
87
|
+
except KeyError:
|
|
88
|
+
self.assemblySoftware = None
|
|
89
|
+
|
|
90
|
+
try:
|
|
91
|
+
self.voucherSpecimen = data["voucherSpecimen"]
|
|
92
|
+
except KeyError:
|
|
93
|
+
self.voucherSpecimen = None
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
self.instrument = data["instrument"]
|
|
97
|
+
except KeyError:
|
|
98
|
+
self.instrument = None
|
|
99
|
+
|
|
100
|
+
try:
|
|
101
|
+
self.scholarlyArticle = data["scholarlyArticle"]
|
|
102
|
+
except KeyError:
|
|
103
|
+
self.scholarlyArticle = None
|
|
104
|
+
|
|
105
|
+
try:
|
|
106
|
+
self.documentation = data["documentation"]
|
|
107
|
+
except KeyError:
|
|
108
|
+
self.documentation = None
|
|
109
|
+
|
|
110
|
+
try:
|
|
111
|
+
self.identifier = data["identifier"]
|
|
112
|
+
except KeyError:
|
|
113
|
+
self.identifier = None
|
|
114
|
+
|
|
115
|
+
try:
|
|
116
|
+
self.relatedLink = data["relatedLink"]
|
|
117
|
+
except KeyError:
|
|
118
|
+
self.relatedLink = None
|
|
119
|
+
|
|
120
|
+
try:
|
|
121
|
+
self.funding = data["funding"]
|
|
122
|
+
except KeyError:
|
|
123
|
+
self.funding = None
|
|
124
|
+
|
|
125
|
+
try:
|
|
126
|
+
self.vitalStats["N50"] = data["vitalStats"]["N50"]
|
|
127
|
+
self.vitalStats["L50"] = data["vitalStats"]["L50"]
|
|
128
|
+
self.vitalStats["L90"] = data["vitalStats"]["L90"]
|
|
129
|
+
self.vitalStats["totalBasePairs"] = data["vitalStats"]["totalBasePairs"]
|
|
130
|
+
self.vitalStats["numberContigs"] = data["vitalStats"]["numberContigs"]
|
|
131
|
+
self.vitalStats["numberScaffolds"] = data["vitalStats"]["numberScaffolds"]
|
|
132
|
+
self.vitalStats["readTechnology"] = data["vitalStats"]["readTechnology"]
|
|
133
|
+
except (KeyError, TypeError):
|
|
134
|
+
self.vitalStats = {
|
|
135
|
+
"N50": None,
|
|
136
|
+
"L50": None,
|
|
137
|
+
"L90": None,
|
|
138
|
+
"totalBasePairs": None,
|
|
139
|
+
"numberContigs": None,
|
|
140
|
+
"numberScaffolds": None,
|
|
141
|
+
"readTechnology": None,
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
try:
|
|
145
|
+
self.reuseConditions = data["reuseConditions"]
|
|
146
|
+
except KeyError:
|
|
147
|
+
self.reuseConditions = None
|
|
148
|
+
|
|
149
|
+
def output_yaml(self) -> str:
|
|
150
|
+
return yaml.dump(self.__dict__)
|
|
151
|
+
|
|
152
|
+
def input_fasta(self, stream: str) -> None:
|
|
153
|
+
formulated = "\n".join(
|
|
154
|
+
re.sub(";~", "", line)
|
|
155
|
+
for line in stream.splitlines()
|
|
156
|
+
if line.startswith(";~")
|
|
157
|
+
)
|
|
158
|
+
self.input_yaml(formulated)
|
|
159
|
+
|
|
160
|
+
def output_fasta(self) -> str:
|
|
161
|
+
array = ";~- "
|
|
162
|
+
name = "\n;~- name:"
|
|
163
|
+
uri = "\n;~ uri:"
|
|
164
|
+
end_span = ""
|
|
165
|
+
|
|
166
|
+
data = (
|
|
167
|
+
f";~schema: {self.schema}\n"
|
|
168
|
+
f";~schemaVersion: {self.schemaVersion}\n"
|
|
169
|
+
f";~genome: {self.genome}\n"
|
|
170
|
+
f";~genomeSynonym:\n"
|
|
171
|
+
f"{array + array.join(x + end_span for x in self.genomeSynonym)}"
|
|
172
|
+
f";~version: {self.version}\n"
|
|
173
|
+
f";~metadataAuthor:"
|
|
174
|
+
f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.metadataAuthor)}'
|
|
175
|
+
f"\n;~assemblyAuthor:"
|
|
176
|
+
f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.assemblyAuthor)}'
|
|
177
|
+
f";~accessionID:\n"
|
|
178
|
+
f';~ name:{self.accessionID["name"]}\n'
|
|
179
|
+
f';~ url:{self.accessionID["url"]}\n'
|
|
180
|
+
f";~taxon:\n"
|
|
181
|
+
f';~ name:{self.taxon["name"]}\n'
|
|
182
|
+
f';~ uri:{self.taxon["uri"]}\n'
|
|
183
|
+
f";~assemblySoftware: {self.assemblySoftware}\n"
|
|
184
|
+
f";~voucherSpecimen: {self.voucherSpecimen}\n"
|
|
185
|
+
f";~dateCreated: {self.dateCreated}\n"
|
|
186
|
+
f";~instrument:\n"
|
|
187
|
+
f"{array + array.join(x + end_span for x in self.instrument)}"
|
|
188
|
+
f";~scholarlyArticle: {self.scholarlyArticle}\n"
|
|
189
|
+
f";~documentation: {self.documentation}\n"
|
|
190
|
+
f";~identifier:\n"
|
|
191
|
+
f"{array + array.join(x + end_span for x in self.identifier)}"
|
|
192
|
+
f";~relatedLink:\n"
|
|
193
|
+
f"{array + array.join(x + end_span for x in self.relatedLink)}"
|
|
194
|
+
f";~funding:\n"
|
|
195
|
+
f"{array + array.join(x + end_span for x in self.funding)}"
|
|
196
|
+
f";~masking {self.masking}\n"
|
|
197
|
+
f";~vitalStats:\n"
|
|
198
|
+
f';~-N50: {self.vitalStats["N50"]}\n'
|
|
199
|
+
f';~-L50: {self.vitalStats["L50"]}\n'
|
|
200
|
+
f';~-L90: {self.vitalStats["L90"]}\n'
|
|
201
|
+
f';~-totalBasePairs: {self.vitalStats["totalBasePairs"]}\n'
|
|
202
|
+
f';~-numberContigs: {self.vitalStats["numberContigs"]}\n'
|
|
203
|
+
f';~-numberScaffolds: {self.vitalStats["numberScaffolds"]}\n'
|
|
204
|
+
f';~-readTechnology: {self.vitalStats["readTechnology"]}\n'
|
|
205
|
+
f";~reuseConditions: {self.reuseConditions}\n"
|
|
206
|
+
f";~checksum: {self.checksum}\n"
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
return data
|
|
210
|
+
|
|
211
|
+
def input_microdata(self, stream: str) -> None:
|
|
212
|
+
data = microdata.get_items(stream)
|
|
213
|
+
data = data[0]
|
|
214
|
+
self.schema = data.schema
|
|
215
|
+
self.version = data.version
|
|
216
|
+
self.schemaVersion = data.schemaVersion
|
|
217
|
+
self.genome = data.genome
|
|
218
|
+
self.genomeSynonym = data.get_all("genomeSynonym")
|
|
219
|
+
self.metadataAuthor = data.get_all("metadataAuthor")
|
|
220
|
+
self.assemblyAuthor = data.get_all("assemblyAuthor")
|
|
221
|
+
self.accessionID = data.get_all("accessionID")
|
|
222
|
+
self.taxon = data.get_all("taxon")
|
|
223
|
+
self.assemblySoftware = data.assemblySoftware
|
|
224
|
+
self.voucherSpecimen = data.voucherSpecimen
|
|
225
|
+
self.dateCreated = data.dateCreated
|
|
226
|
+
self.instrument = data.get_all("instrument")
|
|
227
|
+
self.scholarlyArticle = data.scholarlyArticle
|
|
228
|
+
self.documentation = data.documentation
|
|
229
|
+
self.identifier = data.get_all("identifier")
|
|
230
|
+
self.relatedLink = data.get_all("relatedLink")
|
|
231
|
+
self.funding = data.get_all("funding")
|
|
232
|
+
self.masking = data.masking
|
|
233
|
+
self.vitalStats = data.get_all("vitalStats")
|
|
234
|
+
self.reuseConditions = data.reuseConditions
|
|
235
|
+
self.checksum = data.checksum
|
|
236
|
+
|
|
237
|
+
def output_microdata(self) -> str:
|
|
238
|
+
instrument = '<span itemprop="instrument">'
|
|
239
|
+
identifier = '<span itemprop="identifier">'
|
|
240
|
+
relatedLink = '<span itemprop="relatedLink">'
|
|
241
|
+
funding = '<span itemprop="funding">'
|
|
242
|
+
metadataAuthor = '<span itemprop="metadataAuthor">'
|
|
243
|
+
assemblyAuthor = '<span itemprop="assemblyAuthor">'
|
|
244
|
+
genomeSynonym = '<span itemprop="genomeSynonym">'
|
|
245
|
+
name = '<span itemprop="name">'
|
|
246
|
+
uri = '<span itemprop="uri">'
|
|
247
|
+
end_span = "</span>"
|
|
248
|
+
|
|
249
|
+
data = (
|
|
250
|
+
f'<div itemscope itemtype="https://raw.githubusercontent.com/FAIR-bioHeaders/FHR-Specification/main/fhr.json" version="{self.schemaVersion}">'
|
|
251
|
+
f'<span itemprop="schema">{self.schema}</span>'
|
|
252
|
+
f'<span itemprop="schemaVersion">{self.schemaVersion}</span>'
|
|
253
|
+
f'<span itemprop="version">{self.version}</span>'
|
|
254
|
+
f'<span itemprop="genome">{self.genome}</span>'
|
|
255
|
+
f"{genomeSynonym + genomeSynonym.join(x + end_span for x in self.genomeSynonym)}"
|
|
256
|
+
f'{metadataAuthor + metadataAuthor.join(name + x["name"] + end_span + uri + x["uri"] + end_span for x in self.metadataAuthor)}'
|
|
257
|
+
f"</span>"
|
|
258
|
+
f'{assemblyAuthor + assemblyAuthor.join(name + x["name"] + end_span + uri + x["uri"] + end_span for x in self.assemblyAuthor)}'
|
|
259
|
+
f"</span>"
|
|
260
|
+
f'<span itemprop="accessionID">'
|
|
261
|
+
f' <span itemprop="name">{self.accessionID["name"]}</span>'
|
|
262
|
+
f' <span itemprop="url">{self.accessionID["url"]}"</span>'
|
|
263
|
+
f"</span>"
|
|
264
|
+
f'<span itemprop="taxon">'
|
|
265
|
+
f' <span itemprop="name">{self.taxon["name"]}</span>'
|
|
266
|
+
f' <span itemprop="uri">{self.taxon["uri"]}</span>'
|
|
267
|
+
f"</span>"
|
|
268
|
+
f'<span itemprop="assemblySoftware">{self.assemblySoftware}</span>'
|
|
269
|
+
f'<span itemprop="voucherSpecimen">{self.voucherSpecimen}</span>'
|
|
270
|
+
f'<span itemprop="dateCreated">{self.dateCreated}</span>'
|
|
271
|
+
f"{instrument + instrument.join(x + end_span for x in self.instrument)}"
|
|
272
|
+
f'<span itemprop="scholarlyArticle">{self.scholarlyArticle}</span>'
|
|
273
|
+
f'<span itemprop="documentation">{self.documentation}</span>'
|
|
274
|
+
f"{identifier + identifier.join(x + end_span for x in self.identifier)}"
|
|
275
|
+
f"{relatedLink + relatedLink.join(x + end_span for x in self.relatedLink)}"
|
|
276
|
+
f"{funding + funding.join(x + end_span for x in self.funding)}"
|
|
277
|
+
f'<span itemprop="masking">{self.masking}</span>'
|
|
278
|
+
f'<span itemprop="vitalStats">'
|
|
279
|
+
f' <span itemprop="N50">{self.vitalStats["N50"]}</span>'
|
|
280
|
+
f' <span itemprop="L50">{self.vitalStats["L50"]}</span>'
|
|
281
|
+
f' <span itemprop="L90">{self.vitalStats["L90"]}</span>'
|
|
282
|
+
f' <span itemprop="totalBasePairs">{self.vitalStats["totalBasePairs"]}</span>'
|
|
283
|
+
f' <span itemprop="numberContigs">{self.vitalStats["numberContigs"]}</span>'
|
|
284
|
+
f' <span itemprop="numberScaffolds">{self.vitalStats["numberScaffolds"]}</span>'
|
|
285
|
+
f' <span itemprop="readTechnology">{self.vitalStats["readTechnology"]}</span>'
|
|
286
|
+
f"</span>"
|
|
287
|
+
f'<span itemprop="reuseConditions">{self.reuseConditions}</span>'
|
|
288
|
+
f'<span itemprop="checksum">{self.checksum}</span>'
|
|
289
|
+
f"</div>"
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
return data
|
|
293
|
+
|
|
294
|
+
def input_json(self, stream: str) -> None:
|
|
295
|
+
data = json.loads(stream)
|
|
296
|
+
self.schema = data["schema"]
|
|
297
|
+
self.version = data["version"]
|
|
298
|
+
self.schemaVersion = data["schemaVersion"]
|
|
299
|
+
self.genome = data["genome"]
|
|
300
|
+
self.taxon["name"] = data["taxon"]["name"]
|
|
301
|
+
self.taxon["uri"] = data["taxon"]["uri"]
|
|
302
|
+
self.metadataAuthor = data["metadataAuthor"]
|
|
303
|
+
self.assemblyAuthor = data["assemblyAuthor"]
|
|
304
|
+
self.dateCreated = data["dateCreated"]
|
|
305
|
+
self.masking = data["masking"]
|
|
306
|
+
self.checksum = data["checksum"]
|
|
307
|
+
try:
|
|
308
|
+
self.genomeSynonym = data["genomeSynonym"]
|
|
309
|
+
except KeyError:
|
|
310
|
+
self.genomeSynonym = None
|
|
311
|
+
|
|
312
|
+
try:
|
|
313
|
+
self.accessionID["url"] = data["accessionID"]["url"]
|
|
314
|
+
except (KeyError, TypeError):
|
|
315
|
+
self.accessionID["url"] = None
|
|
316
|
+
|
|
317
|
+
try:
|
|
318
|
+
self.taxon["name"] = data["taxon"]["name"]
|
|
319
|
+
self.taxon["uri"] = data["taxon"]["uri"]
|
|
320
|
+
except KeyError:
|
|
321
|
+
self.taxon["name"] = None
|
|
322
|
+
self.taxon["uri"] = None
|
|
323
|
+
|
|
324
|
+
try:
|
|
325
|
+
self.assemblySoftware = data["assemblySoftware"]
|
|
326
|
+
except KeyError:
|
|
327
|
+
self.assemblySoftware = None
|
|
328
|
+
|
|
329
|
+
try:
|
|
330
|
+
self.voucherSpecimen = data["voucherSpecimen"]
|
|
331
|
+
except KeyError:
|
|
332
|
+
self.voucherSpecimen = None
|
|
333
|
+
|
|
334
|
+
try:
|
|
335
|
+
self.instrument = data["instrument"]
|
|
336
|
+
except KeyError:
|
|
337
|
+
self.instrument = None
|
|
338
|
+
|
|
339
|
+
try:
|
|
340
|
+
self.scholarlyArticle = data["scholarlyArticle"]
|
|
341
|
+
except KeyError:
|
|
342
|
+
self.scholarlyArticle = None
|
|
343
|
+
|
|
344
|
+
try:
|
|
345
|
+
self.documentation = data["documentation"]
|
|
346
|
+
except KeyError:
|
|
347
|
+
self.documentation = None
|
|
348
|
+
|
|
349
|
+
try:
|
|
350
|
+
self.identifier = data["identifier"]
|
|
351
|
+
except KeyError:
|
|
352
|
+
self.identifier = None
|
|
353
|
+
|
|
354
|
+
try:
|
|
355
|
+
self.relatedLink = data["relatedLink"]
|
|
356
|
+
except KeyError:
|
|
357
|
+
self.relatedLink = None
|
|
358
|
+
|
|
359
|
+
try:
|
|
360
|
+
self.funding = data["funding"]
|
|
361
|
+
except KeyError:
|
|
362
|
+
self.funding = None
|
|
363
|
+
|
|
364
|
+
try:
|
|
365
|
+
self.vitalStats["N50"] = data["vitalStats"]["N50"]
|
|
366
|
+
self.vitalStats["L50"] = data["vitalStats"]["L50"]
|
|
367
|
+
self.vitalStats["L90"] = data["vitalStats"]["L90"]
|
|
368
|
+
self.vitalStats["totalBasePairs"] = data["vitalStats"]["totalBasePairs"]
|
|
369
|
+
self.vitalStats["numberContigs"] = data["vitalStats"]["numberContigs"]
|
|
370
|
+
self.vitalStats["numberScaffolds"] = data["vitalStats"]["numberScaffolds"]
|
|
371
|
+
self.vitalStats["readTechnology"] = data["vitalStats"]["readTechnology"]
|
|
372
|
+
except (KeyError, TypeError):
|
|
373
|
+
self.vitalStats = {
|
|
374
|
+
"N50": None,
|
|
375
|
+
"L50": None,
|
|
376
|
+
"L90": None,
|
|
377
|
+
"totalBasePairs": None,
|
|
378
|
+
"numberContigs": None,
|
|
379
|
+
"numberScaffolds": None,
|
|
380
|
+
"readTechnology": None,
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
try:
|
|
384
|
+
self.reuseConditions = data["reuseConditions"]
|
|
385
|
+
except KeyError:
|
|
386
|
+
self.reuseConditions = None
|
|
387
|
+
|
|
388
|
+
def output_json(self) -> str:
|
|
389
|
+
return json.dumps(self.__dict__)
|
|
390
|
+
|
|
391
|
+
def input_gfa(self, stream: str) -> None:
|
|
392
|
+
formulated = "\n".join(
|
|
393
|
+
re.sub("#~", "", line)
|
|
394
|
+
for line in stream.splitlines()
|
|
395
|
+
if line.startswith("#~")
|
|
396
|
+
)
|
|
397
|
+
self.input_yaml(formulated)
|
|
398
|
+
|
|
399
|
+
def output_gfa(self) -> str:
|
|
400
|
+
array = "#~- "
|
|
401
|
+
name = "\n;~- name:"
|
|
402
|
+
uri = "\n;~ uri:"
|
|
403
|
+
end_span = ""
|
|
404
|
+
|
|
405
|
+
data = (
|
|
406
|
+
f"#~schema: {self.schema}\n"
|
|
407
|
+
f"#~schemaVersion: {self.schemaVersion}\n"
|
|
408
|
+
f"#~genome: {self.genome}\n"
|
|
409
|
+
f"#~genomeSynonym:\n"
|
|
410
|
+
f"{array + array.join(x + end_span for x in self.genomeSynonym)}"
|
|
411
|
+
f"#~version: {self.version}\n"
|
|
412
|
+
f"#~metadataAuthor:"
|
|
413
|
+
f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.metadataAuthor)}'
|
|
414
|
+
f"\n;~assemblyAuthor:"
|
|
415
|
+
f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.assemblyAuthor)}'
|
|
416
|
+
f"#~accessionID:\n"
|
|
417
|
+
f'#~ name:{self.accessionID["name"]}\n'
|
|
418
|
+
f'#~ url:{self.accessionID["url"]}\n'
|
|
419
|
+
f"#~taxon:\n"
|
|
420
|
+
f'#~ name:{self.taxon["name"]}\n'
|
|
421
|
+
f'#~ uri:{self.taxon["uri"]}\n'
|
|
422
|
+
f"#~assemblySoftware: {self.assemblySoftware}\n"
|
|
423
|
+
f"#~voucherSpecimen: {self.voucherSpecimen}\n"
|
|
424
|
+
f"#~dateCreated: {self.dateCreated}\n"
|
|
425
|
+
f"#~instrument:\n"
|
|
426
|
+
f"{array + array.join(x + end_span for x in self.instrument)}"
|
|
427
|
+
f"#~scholarlyArticle: {self.scholarlyArticle}\n"
|
|
428
|
+
f"#~documentation: {self.documentation}\n"
|
|
429
|
+
f"#~identifier:\n"
|
|
430
|
+
f"{array + array.join(x + end_span for x in self.identifier)}"
|
|
431
|
+
f"#~relatedLink:\n"
|
|
432
|
+
f"{array + array.join(x + end_span for x in self.relatedLink)}"
|
|
433
|
+
f"#~funding:\n"
|
|
434
|
+
f"{array + array.join(x + end_span for x in self.funding)}"
|
|
435
|
+
f"#~masking {self.masking}\n"
|
|
436
|
+
f"#~vitalStats:\n"
|
|
437
|
+
f'#~-N50: {self.vitalStats["N50"]}\n'
|
|
438
|
+
f'#~-L50: {self.vitalStats["L50"]}\n'
|
|
439
|
+
f'#~-L90: {self.vitalStats["L90"]}\n'
|
|
440
|
+
f'#~-totalBasePairs: {self.vitalStats["totalBasePairs"]}\n'
|
|
441
|
+
f'#~-numberContigs: {self.vitalStats["numberContigs"]}\n'
|
|
442
|
+
f'#~-numberScaffolds: {self.vitalStats["numberScaffolds"]}\n'
|
|
443
|
+
f'#~-readTechnology: {self.vitalStats["readTechnology"]}\n'
|
|
444
|
+
f"#~reuseConditions: {self.reuseConditions}\n"
|
|
445
|
+
f"#~checksum: {self.checksum}\n"
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
return data
|
|
449
|
+
|
|
450
|
+
def fhr_validate(self) -> None:
|
|
451
|
+
fhr_instance = json.dumps(self.__dict__)
|
|
452
|
+
validate(instance=fhr_instance, schema=schema)
|
fhr-0.1.1/pyproject.toml
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[tool.flake8]
|
|
2
|
+
max-line-length = 120
|
|
3
|
+
ignore = "E203, E266, E501, W503"
|
|
4
|
+
|
|
5
|
+
[tool.poetry]
|
|
6
|
+
name = "fhr"
|
|
7
|
+
version = "0.1.1"
|
|
8
|
+
description = "This tool is used to validate and convert between different FHR header serializations"
|
|
9
|
+
authors = ["David Molik <david.molik@usda.gov>","Adam Wright <adam.wright@oicr.on.ca>"]
|
|
10
|
+
license = "USDA-ARS"
|
|
11
|
+
readme = "README.md"
|
|
12
|
+
|
|
13
|
+
[tool.poetry.dependencies]
|
|
14
|
+
python = "^3.9"
|
|
15
|
+
argparse = "^1.4.0"
|
|
16
|
+
microdata = "^0.8.0"
|
|
17
|
+
jsonschema = "^4.21.1"
|
|
18
|
+
pyyaml = "^6.0.1"
|
|
19
|
+
|
|
20
|
+
[tool.poetry.dev-dependencies]
|
|
21
|
+
mypy = "^1.8.0"
|
|
22
|
+
types-pyyaml = "^6.0.12.12"
|
|
23
|
+
types-jsonschema = "^4.21.0.20240118"
|
|
24
|
+
isort = "^5.13.2"
|
|
25
|
+
black = "^24.1.1"
|
|
26
|
+
ruff = "^0.2.1"
|
|
27
|
+
|
|
28
|
+
[tool.poetry.scripts]
|
|
29
|
+
fhr-convert = "fhr_convert:main"
|
|
30
|
+
fhr-validate = "fhr_validate:main"
|
|
31
|
+
fhr-fasta-strip = "fasta.fhr_fasta_strip:main"
|
|
32
|
+
fhr-fasta-combine = "fasta.fhr_fasta_combine:main"
|
|
33
|
+
fhr-fasta-validate = "fasta.fhr_fasta_validate:main"
|
|
34
|
+
fhr-gfa-strip = "gfa.fhr_gfa_strip:main"
|
|
35
|
+
fhr-gfa-combine = "gfa.fhr_gfa_combine:main"
|
|
36
|
+
fhr-gfa-validate = "gfa.fhr_gfa_validate:main"
|
|
37
|
+
|
|
38
|
+
[tool.poetry.group.dev.dependencies]
|
|
39
|
+
pytest = "^8.0.0"
|
|
40
|
+
|