fhr 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
fhr-0.1.1/LICENSE ADDED
@@ -0,0 +1,73 @@
1
+ CONTENTS
2
+
3
+ Public Domain Notice
4
+ Exceptions (for bundled 3rd-party code)
5
+ Copyright F.A.Q.
6
+
7
+
8
+ ==============================================================
9
+ PUBLIC DOMAIN NOTICE
10
+ United States Department of Agriculture
11
+ Agricultural Research Service
12
+
13
+ With the exception of certain third-party files summarized below, this
14
+ software is a "United States Government Work" under the terms of the
15
+ United States Copyright Act. It was written as part of the authors'
16
+ official duties as United States Government employees and thus cannot
17
+ be copyrighted. This software is freely available to the public for
18
+ use. The United States Department of Agriculture, Agricultural
19
+ Research Service (USDA - ARS) and the U.S. Government have not placed
20
+ any restriction on its use or reproduction.
21
+
22
+ Although all reasonable efforts have been taken to ensure the accuracy
23
+ and reliability of the software and data, the USDA ARS and the U.S.
24
+ Government do not and cannot warrant the performance or results tha may
25
+ be obtained by using this software or data. The USDA ARS and the U.S.
26
+ Government disclaim all warranties, express or implied, including
27
+ warranties of performance, merchantability or fitness for any particular
28
+ purpose.
29
+
30
+ Please cite the authors in any work or product based on this material.
31
+
32
+
33
+ ==============================================================
34
+ EXCEPTIONS (in all cases excluding USDA-ARS-written makefiles):
35
+
36
+ Location:
37
+ Author:
38
+ License:
39
+
40
+
41
+ ==============================================================
42
+ Copyright F.A.Q.
43
+
44
+
45
+ --------------------------------------------------------------
46
+ Q. Our product makes use of the USDA - ARS source code, and we made changes
47
+ and additions to that version of the USDA - ARS code to better fit it to
48
+ our needs. Can we copyright the code, and how?
49
+
50
+ A. You can copyright only the *changes* or the *additions* you made to the
51
+ NCBI source code. You should identify unambiguously those sections of
52
+ the code that were modified, e.g. by commenting any changes you made
53
+ in the code you distribute. Therefore, your license has to make clear
54
+ to users that your product is a combination of code that is public domain
55
+ within the U.S. (but may be subject to copyright by the U.S. in foreign
56
+ countries) and code that has been created or modified by you.
57
+
58
+ --------------------------------------------------------------
59
+ Q. Can we (re)license all or part of the USDA - ARS source code?
60
+
61
+ A. No, you cannot license or relicense the source code written by USDA - ARS
62
+ since you cannot claim any copyright in the software that was developed
63
+ at USDA - ARS as a 'government work' and consequently is in the public
64
+ domain within the U.S.
65
+
66
+ --------------------------------------------------------------
67
+ Q. What if these copyright guidelines are not clear enough or are not
68
+ applicable to my particular case?
69
+
70
+ A. Contact us. Send your questions to 'answers@usda.gov'.
71
+ --------------------------------------------------------------
72
+
73
+ This file was modified from the NCBI Boilerplate LICENSE file
fhr-0.1.1/PKG-INFO ADDED
@@ -0,0 +1,255 @@
1
+ Metadata-Version: 2.1
2
+ Name: fhr
3
+ Version: 0.1.1
4
+ Summary: This tool is used to validate and convert between different FHR header serializations
5
+ License: USDA-ARS
6
+ Author: David Molik
7
+ Author-email: david.molik@usda.gov
8
+ Requires-Python: >=3.9,<4.0
9
+ Classifier: License :: Other/Proprietary License
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.9
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Requires-Dist: argparse (>=1.4.0,<2.0.0)
16
+ Requires-Dist: jsonschema (>=4.21.1,<5.0.0)
17
+ Requires-Dist: microdata (>=0.8.0,<0.9.0)
18
+ Requires-Dist: pyyaml (>=6.0.1,<7.0.0)
19
+ Description-Content-Type: text/markdown
20
+
21
+ # FHR-File-Converter
22
+ [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.6762547.svg)](https://doi.org/10.5281/zenodo.6762547)
23
+
24
+ This is the fhr file converter, it can convert fhr inbetween json, fasta, microdata, and fasta header. If you would like a detailed specification of fhr, see [FHR-Specification](https://github.com/FAIR-bioHeaders/FHR-Specification)
25
+
26
+ ## Installation
27
+
28
+ You can install the FHR file converter and its dependencies using Poetry:
29
+
30
+ ```bash
31
+ poetry install
32
+ ```
33
+
34
+ ## Usage
35
+
36
+
37
+ ### Commnand line Usage
38
+
39
+ Using FHR on the command line:
40
+
41
+ ```bash
42
+ fhr-convert <input>.<yaml|json|fasta|html> <output>.<yaml|json|fasta|html>
43
+ ```
44
+
45
+ Detailed Usage:
46
+
47
+ ```
48
+ usage: fhr-convert [-h] [--version] <file> <file>
49
+
50
+ Convert from one FHR supported file type to another
51
+
52
+ positional arguments:
53
+ <file> input followed by output
54
+
55
+ optional arguments:
56
+ -h, --help show this help message and exit
57
+ --version show program's version number and exit
58
+
59
+ positional <file> input and output files
60
+ input files can be one of:
61
+ <input>.yml
62
+ <input>.fasta - fasta contining a fhr header
63
+ <input>.html - html containing microdata
64
+
65
+ output files can be one of:
66
+ <output>.yml
67
+ <output>.fasta - fasta output type will be made as a fasta header without sequences
68
+ <output>.html - microdata output type will be made into generic html output
69
+ ```
70
+
71
+ ## Validating an FHR file on command line
72
+
73
+
74
+ ```bash
75
+ fhr-validate <input>.<yaml|json|fasta|html>
76
+ ```
77
+
78
+ Detailed Usage:
79
+
80
+ ```
81
+ usage: fhr-validate [-h] [--version] <file>
82
+
83
+ Validate a fhr containing file
84
+
85
+ positional <file> input and output files
86
+ input files can be one of:
87
+ <input>.yml
88
+ <input>.fasta - fasta contining a fhr header
89
+ <input>.html - html containing microdata
90
+ ```
91
+
92
+
93
+ As such validating a yaml file named "important\_genome.fhr.yml" would be:
94
+
95
+ `fhr-validate important_genome.fhr.yml`
96
+
97
+
98
+
99
+ ## Using FHR in Python
100
+
101
+ To use FHR libabry in Python
102
+
103
+ ```python
104
+ >>> from fhr import fhr
105
+ >>> file = open("example.yaml")
106
+ >>> data = fhr()
107
+ >>> data.input_yaml(file.read())
108
+ >>> data.output_fasta()
109
+ ";~schema: https://raw.githubusercontent.com/FFRGS/FFRGS-Specification/main/fhr.json\n;~schemaVersion: 1\n;~genome: Bombas huntii\n;~version: 0.0.1\n;~author:;~ name:Adam Wright\n;~ url:https://wormbase.org/resource/person/WBPerson30813\n;~assembler:;~ name:David Molik\n;~ url:https:/david.molik.co/person\n;~place:;~ name:PBARC\n;~ url:https://www.ars.usda.gov/pacific-west-area/hilo-hi/daniel-k-inouye-us-pacific-basin-agricultural-research-center/\n;~taxa: Bombas huntii\n;~assemblySoftware: HiFiASM\n;~physicalSample: Located in Freezer 33, Drawer 137\n;~dateCreated: 2022-03-21\n;~instrument: ['Sequel IIe', 'Nanopore']\n;~scholarlyArticle: https://doi.org/10.1371/journal.pntd.0008755\n;~documentation: Built assembly from... \n;~identifier: ['gkx10242566416842']\n;~relatedLink: ['https/david.molik.co/genome']\n;~funding: some\n;~reuseConditions: public domain\n"
110
+ ```
111
+
112
+ ## Checksums
113
+
114
+ The FHR stores checksums, allowing the FASTA header of the reference genome to contain the checksum for the FASTA file without the header.
115
+
116
+ To utilize the checksum, strip the FASTA header:
117
+
118
+ ```bash
119
+ cat example.fasta | grep -E -v '^;~\s?checksum' > example.check.fasta
120
+ ```
121
+
122
+ To strip the checksum:
123
+
124
+
125
+ ```bash
126
+ cat example.fasta | grep -E ';~\s?checksum' | sed 's/^;~checksum://g' | sed '/\'//g'
127
+ ```
128
+
129
+ ## Docker Support
130
+
131
+ You can also run the FHR file converter in a Docker container. To build the Docker image:
132
+
133
+ ```bash
134
+ docker build -t fhr-file-converter .
135
+ ```
136
+
137
+ And then run the Docker container:
138
+
139
+ ```bash
140
+ docker run -it --rm fhr-file-converter
141
+ ```
142
+
143
+
144
+ ## Running Code Quality Checks
145
+
146
+ Ensuring code quality is crucial for maintaining a healthy and sustainable codebase. The following tools help enforce coding standards and best practices:
147
+
148
+ ### isort
149
+
150
+ `isort` is a tool that sorts Python imports alphabetically within each section and separated by a blank line. It ensures consistent import styles across your project.
151
+
152
+ To run isort, use the following command:
153
+
154
+ ```bash
155
+ poetry run isort .
156
+ ```
157
+
158
+ ### ruff
159
+ ruff is a lightweight linter for Python that aims to detect common programming errors, stylistic issues, and code smells. It provides quick feedback on potential issues in your code.
160
+
161
+ To run ruff, use the following command:
162
+
163
+ ```bash
164
+ poetry run ruff .
165
+ ```
166
+
167
+ ### black
168
+
169
+ `black` is an uncompromising Python code formatter. It reformats entire files in place to ensure a consistent and readable code style. It's opinionated and strives for the smallest diffs possible.
170
+
171
+ To run black, use the following command:
172
+
173
+ ```bash
174
+ poetry run black .
175
+ ```
176
+
177
+ Running these code quality checks regularly helps maintain a clean and consistent codebase, making it easier to collaborate with others and ensuring code readability and maintainability. These checks are required to pass in order to pull changes into the main branch.
178
+
179
+
180
+ ### pytest
181
+
182
+ Make sure you install depedencies first and then run the tests with poetry
183
+ ```bash
184
+ poetry run install
185
+ poetry run pytest
186
+ ```
187
+
188
+ ## Citing FHR
189
+ Information on Citations of FHR
190
+
191
+
192
+ ### Citing the Validation Tool
193
+ cite the validation tool when directly interacting with the tool or library
194
+ The APA citation for the [FHR validation/converter software](https://github.com/FAIR-bioHeaders/FHR-File-Converter) is:
195
+
196
+ ```
197
+ Molik, D., & Wright, A. FHR File Converster [Computer software]. https://github.com/FAIR-bioHeaders/FHR-File-Converter
198
+ ```
199
+
200
+ Or in bibtex:
201
+ ```bibtex
202
+ % Citation For FHR Validation/Converter Software
203
+ @software{FHR_File_Converter,
204
+ author = {Molik, David and Wright, Adam},
205
+ year = {2023},
206
+ license = {PDDL-1.0},
207
+ title = {{FHR File Converster}},
208
+ url = {https://github.com/FAIR-bioHeaders/FHR-File-Converter},
209
+ doi = {10.5281/zenodo.6762547}
210
+ }
211
+ ```
212
+ ### Citing the Specification
213
+ cite the specification when directly interacting with the specification (pull requests, comments on schema)
214
+ The APA citation for the [FHR specification](https://github.com/FAIR-bioHeaders/FHR-Specification) is:
215
+
216
+ ```
217
+ Molik, D., & Wright, A. FHR Specification [Data set]. https://github.com/FAIR-bioHeaders/FHR-Specification
218
+ ```
219
+
220
+ Or in bibtex:
221
+ ```bibtex
222
+ % Citation For FHR Specification
223
+ @misc{FHR_Specification,
224
+ author = {Molik, David and Wright, Adam},
225
+ year = {2023},
226
+ title = {{FHR Specification}},
227
+ url = {https://github.com/FAIR-bioHeaders/FHR-Specification},
228
+ doi = {10.5281/zenodo.6762549}
229
+ }
230
+ ```
231
+ ### Citing the Preprint
232
+ **(best option)** cite the preprint talking about the effort, or want a broad citation of FHR
233
+ The APA citation for the [FHR preprint](https://www.biorxiv.org/content/10.1101/2023.11.29.569306v1) is:
234
+
235
+ ```
236
+ Wright, A., Wilkinson, M. D., Mungall, C., Cain, S., Richards, S., Sternberg, P., ... & Molik, D. C. (2023). Data Resources and Analyses Fair Header Reference genome: A Trustworthy standard. bioRxiv, 2023-11.
237
+ ```
238
+
239
+ Or in bibtex:
240
+ ```bibtex
241
+ % Citation For FHR Pre-print
242
+ @article {Wright2023,
243
+ author = {Adam Wright and Mark D Wilkinson and Chris Mungall and Scott Cain and Stephen Richards and Paul Sternberg and Ellen Provin and Jonathan L Jacobs and Scott Geib and Daniela Raciti and Karen Yook and Lincoln Stein and David C Molik},
244
+ title = {DATA RESOURCES AND ANALYSES FAIR Header Reference genome: A TRUSTworthy standard},
245
+ elocation-id = {2023.11.29.569306},
246
+ year = {2023},
247
+ doi = {10.1101/2023.11.29.569306},
248
+ publisher = {Cold Spring Harbor Laboratory},
249
+ abstract = {The lack of interoperable data standards among reference genome data-sharing platforms inhibits cross-platform analysis while increasing the risk of data provenance loss. Here, we describe the FAIR-bioHeaders Reference genome (FHR), a metadata standard guided by the principles of Findability, Accessibility, Interoperability, and Reuse (FAIR) in addition to the principles of Transparency, Responsibility, User focus, Sustainability, and Technology (TRUST). The objective of FHR is to provide an extensive set of data serialisation methods and minimum data field requirements while still maintaining extensibility, flexibility, and expressivity in an increasingly decentralised genomic data ecosystem. The effort needed to implement FHR is low; FHR{\textquoteright}s design philosophy ensures easy implementation while retaining the benefits gained from recording both machine and human-readable provenance.Competing Interest StatementThe authors have declared no competing interest.},
250
+ URL = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306},
251
+ eprint = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306.full.pdf},
252
+ journal = {bioRxiv}
253
+ }
254
+ ```
255
+
fhr-0.1.1/README.md ADDED
@@ -0,0 +1,234 @@
1
+ # FHR-File-Converter
2
+ [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.6762547.svg)](https://doi.org/10.5281/zenodo.6762547)
3
+
4
+ This is the fhr file converter, it can convert fhr inbetween json, fasta, microdata, and fasta header. If you would like a detailed specification of fhr, see [FHR-Specification](https://github.com/FAIR-bioHeaders/FHR-Specification)
5
+
6
+ ## Installation
7
+
8
+ You can install the FHR file converter and its dependencies using Poetry:
9
+
10
+ ```bash
11
+ poetry install
12
+ ```
13
+
14
+ ## Usage
15
+
16
+
17
+ ### Commnand line Usage
18
+
19
+ Using FHR on the command line:
20
+
21
+ ```bash
22
+ fhr-convert <input>.<yaml|json|fasta|html> <output>.<yaml|json|fasta|html>
23
+ ```
24
+
25
+ Detailed Usage:
26
+
27
+ ```
28
+ usage: fhr-convert [-h] [--version] <file> <file>
29
+
30
+ Convert from one FHR supported file type to another
31
+
32
+ positional arguments:
33
+ <file> input followed by output
34
+
35
+ optional arguments:
36
+ -h, --help show this help message and exit
37
+ --version show program's version number and exit
38
+
39
+ positional <file> input and output files
40
+ input files can be one of:
41
+ <input>.yml
42
+ <input>.fasta - fasta contining a fhr header
43
+ <input>.html - html containing microdata
44
+
45
+ output files can be one of:
46
+ <output>.yml
47
+ <output>.fasta - fasta output type will be made as a fasta header without sequences
48
+ <output>.html - microdata output type will be made into generic html output
49
+ ```
50
+
51
+ ## Validating an FHR file on command line
52
+
53
+
54
+ ```bash
55
+ fhr-validate <input>.<yaml|json|fasta|html>
56
+ ```
57
+
58
+ Detailed Usage:
59
+
60
+ ```
61
+ usage: fhr-validate [-h] [--version] <file>
62
+
63
+ Validate a fhr containing file
64
+
65
+ positional <file> input and output files
66
+ input files can be one of:
67
+ <input>.yml
68
+ <input>.fasta - fasta contining a fhr header
69
+ <input>.html - html containing microdata
70
+ ```
71
+
72
+
73
+ As such validating a yaml file named "important\_genome.fhr.yml" would be:
74
+
75
+ `fhr-validate important_genome.fhr.yml`
76
+
77
+
78
+
79
+ ## Using FHR in Python
80
+
81
+ To use FHR libabry in Python
82
+
83
+ ```python
84
+ >>> from fhr import fhr
85
+ >>> file = open("example.yaml")
86
+ >>> data = fhr()
87
+ >>> data.input_yaml(file.read())
88
+ >>> data.output_fasta()
89
+ ";~schema: https://raw.githubusercontent.com/FFRGS/FFRGS-Specification/main/fhr.json\n;~schemaVersion: 1\n;~genome: Bombas huntii\n;~version: 0.0.1\n;~author:;~ name:Adam Wright\n;~ url:https://wormbase.org/resource/person/WBPerson30813\n;~assembler:;~ name:David Molik\n;~ url:https:/david.molik.co/person\n;~place:;~ name:PBARC\n;~ url:https://www.ars.usda.gov/pacific-west-area/hilo-hi/daniel-k-inouye-us-pacific-basin-agricultural-research-center/\n;~taxa: Bombas huntii\n;~assemblySoftware: HiFiASM\n;~physicalSample: Located in Freezer 33, Drawer 137\n;~dateCreated: 2022-03-21\n;~instrument: ['Sequel IIe', 'Nanopore']\n;~scholarlyArticle: https://doi.org/10.1371/journal.pntd.0008755\n;~documentation: Built assembly from... \n;~identifier: ['gkx10242566416842']\n;~relatedLink: ['https/david.molik.co/genome']\n;~funding: some\n;~reuseConditions: public domain\n"
90
+ ```
91
+
92
+ ## Checksums
93
+
94
+ The FHR stores checksums, allowing the FASTA header of the reference genome to contain the checksum for the FASTA file without the header.
95
+
96
+ To utilize the checksum, strip the FASTA header:
97
+
98
+ ```bash
99
+ cat example.fasta | grep -E -v '^;~\s?checksum' > example.check.fasta
100
+ ```
101
+
102
+ To strip the checksum:
103
+
104
+
105
+ ```bash
106
+ cat example.fasta | grep -E ';~\s?checksum' | sed 's/^;~checksum://g' | sed '/\'//g'
107
+ ```
108
+
109
+ ## Docker Support
110
+
111
+ You can also run the FHR file converter in a Docker container. To build the Docker image:
112
+
113
+ ```bash
114
+ docker build -t fhr-file-converter .
115
+ ```
116
+
117
+ And then run the Docker container:
118
+
119
+ ```bash
120
+ docker run -it --rm fhr-file-converter
121
+ ```
122
+
123
+
124
+ ## Running Code Quality Checks
125
+
126
+ Ensuring code quality is crucial for maintaining a healthy and sustainable codebase. The following tools help enforce coding standards and best practices:
127
+
128
+ ### isort
129
+
130
+ `isort` is a tool that sorts Python imports alphabetically within each section and separated by a blank line. It ensures consistent import styles across your project.
131
+
132
+ To run isort, use the following command:
133
+
134
+ ```bash
135
+ poetry run isort .
136
+ ```
137
+
138
+ ### ruff
139
+ ruff is a lightweight linter for Python that aims to detect common programming errors, stylistic issues, and code smells. It provides quick feedback on potential issues in your code.
140
+
141
+ To run ruff, use the following command:
142
+
143
+ ```bash
144
+ poetry run ruff .
145
+ ```
146
+
147
+ ### black
148
+
149
+ `black` is an uncompromising Python code formatter. It reformats entire files in place to ensure a consistent and readable code style. It's opinionated and strives for the smallest diffs possible.
150
+
151
+ To run black, use the following command:
152
+
153
+ ```bash
154
+ poetry run black .
155
+ ```
156
+
157
+ Running these code quality checks regularly helps maintain a clean and consistent codebase, making it easier to collaborate with others and ensuring code readability and maintainability. These checks are required to pass in order to pull changes into the main branch.
158
+
159
+
160
+ ### pytest
161
+
162
+ Make sure you install depedencies first and then run the tests with poetry
163
+ ```bash
164
+ poetry run install
165
+ poetry run pytest
166
+ ```
167
+
168
+ ## Citing FHR
169
+ Information on Citations of FHR
170
+
171
+
172
+ ### Citing the Validation Tool
173
+ cite the validation tool when directly interacting with the tool or library
174
+ The APA citation for the [FHR validation/converter software](https://github.com/FAIR-bioHeaders/FHR-File-Converter) is:
175
+
176
+ ```
177
+ Molik, D., & Wright, A. FHR File Converster [Computer software]. https://github.com/FAIR-bioHeaders/FHR-File-Converter
178
+ ```
179
+
180
+ Or in bibtex:
181
+ ```bibtex
182
+ % Citation For FHR Validation/Converter Software
183
+ @software{FHR_File_Converter,
184
+ author = {Molik, David and Wright, Adam},
185
+ year = {2023},
186
+ license = {PDDL-1.0},
187
+ title = {{FHR File Converster}},
188
+ url = {https://github.com/FAIR-bioHeaders/FHR-File-Converter},
189
+ doi = {10.5281/zenodo.6762547}
190
+ }
191
+ ```
192
+ ### Citing the Specification
193
+ cite the specification when directly interacting with the specification (pull requests, comments on schema)
194
+ The APA citation for the [FHR specification](https://github.com/FAIR-bioHeaders/FHR-Specification) is:
195
+
196
+ ```
197
+ Molik, D., & Wright, A. FHR Specification [Data set]. https://github.com/FAIR-bioHeaders/FHR-Specification
198
+ ```
199
+
200
+ Or in bibtex:
201
+ ```bibtex
202
+ % Citation For FHR Specification
203
+ @misc{FHR_Specification,
204
+ author = {Molik, David and Wright, Adam},
205
+ year = {2023},
206
+ title = {{FHR Specification}},
207
+ url = {https://github.com/FAIR-bioHeaders/FHR-Specification},
208
+ doi = {10.5281/zenodo.6762549}
209
+ }
210
+ ```
211
+ ### Citing the Preprint
212
+ **(best option)** cite the preprint talking about the effort, or want a broad citation of FHR
213
+ The APA citation for the [FHR preprint](https://www.biorxiv.org/content/10.1101/2023.11.29.569306v1) is:
214
+
215
+ ```
216
+ Wright, A., Wilkinson, M. D., Mungall, C., Cain, S., Richards, S., Sternberg, P., ... & Molik, D. C. (2023). Data Resources and Analyses Fair Header Reference genome: A Trustworthy standard. bioRxiv, 2023-11.
217
+ ```
218
+
219
+ Or in bibtex:
220
+ ```bibtex
221
+ % Citation For FHR Pre-print
222
+ @article {Wright2023,
223
+ author = {Adam Wright and Mark D Wilkinson and Chris Mungall and Scott Cain and Stephen Richards and Paul Sternberg and Ellen Provin and Jonathan L Jacobs and Scott Geib and Daniela Raciti and Karen Yook and Lincoln Stein and David C Molik},
224
+ title = {DATA RESOURCES AND ANALYSES FAIR Header Reference genome: A TRUSTworthy standard},
225
+ elocation-id = {2023.11.29.569306},
226
+ year = {2023},
227
+ doi = {10.1101/2023.11.29.569306},
228
+ publisher = {Cold Spring Harbor Laboratory},
229
+ abstract = {The lack of interoperable data standards among reference genome data-sharing platforms inhibits cross-platform analysis while increasing the risk of data provenance loss. Here, we describe the FAIR-bioHeaders Reference genome (FHR), a metadata standard guided by the principles of Findability, Accessibility, Interoperability, and Reuse (FAIR) in addition to the principles of Transparency, Responsibility, User focus, Sustainability, and Technology (TRUST). The objective of FHR is to provide an extensive set of data serialisation methods and minimum data field requirements while still maintaining extensibility, flexibility, and expressivity in an increasingly decentralised genomic data ecosystem. The effort needed to implement FHR is low; FHR{\textquoteright}s design philosophy ensures easy implementation while retaining the benefits gained from recording both machine and human-readable provenance.Competing Interest StatementThe authors have declared no competing interest.},
230
+ URL = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306},
231
+ eprint = {https://www.biorxiv.org/content/early/2023/12/01/2023.11.29.569306.full.pdf},
232
+ journal = {bioRxiv}
233
+ }
234
+ ```
fhr-0.1.1/fhr.py ADDED
@@ -0,0 +1,452 @@
1
+ import json
2
+ import re
3
+ from typing import Dict, List, Union
4
+
5
+ import microdata # type: ignore
6
+ import yaml
7
+ from jsonschema import validate
8
+
9
+ with open("fhr_schema.json", "r") as f:
10
+ schema = json.load(f)
11
+
12
+
13
+ class fhr:
14
+
15
+ def __init__(
16
+ self,
17
+ schema: str = "",
18
+ version: str = "",
19
+ schemaVersion: int = 0,
20
+ genome: str = "",
21
+ assemblySoftware: str = "",
22
+ voucherSpecimen: str = "",
23
+ dateCreated: str = "",
24
+ scholarlyArticle: str = "",
25
+ documentation: str = "",
26
+ reuseConditions: str = "",
27
+ vitalStats: Dict[str, Union[int, str]] = {},
28
+ masking: str = "",
29
+ checksum: str = "",
30
+ ) -> None:
31
+ self.schema: str = schema
32
+ self.version: str = version
33
+ self.schemaVersion: int = schemaVersion
34
+ self.genome: str = genome
35
+ self.genomeSynonym: List[str] = []
36
+ self.metadataAuthor: List[Dict[str, str]] = []
37
+ self.assemblyAuthor: List[Dict[str, str]] = []
38
+ self.accessionID: Dict[str, str] = {}
39
+ self.taxon: Dict[str, str] = {}
40
+ self.assemblySoftware: str = assemblySoftware
41
+ self.voucherSpecimen: str = voucherSpecimen
42
+ self.dateCreated: str = dateCreated
43
+ self.instrument: List[str] = []
44
+ self.scholarlyArticle: str = scholarlyArticle
45
+ self.documentation: str = documentation
46
+ self.identifier: List[str] = []
47
+ self.relatedLink: List[str] = []
48
+ self.funding: List[str] = []
49
+ self.masking: str = masking
50
+ self.vitalStats: Dict[str, Union[int, str]] = vitalStats
51
+ self.reuseConditions: str = reuseConditions
52
+ self.checksum: str = checksum
53
+
54
+ def input_yaml(self, stream: str):
55
+ data = yaml.safe_load(stream)
56
+ self.schema = data["schema"]
57
+ self.version = data["version"]
58
+ self.schemaVersion = data["schemaVersion"]
59
+ self.genome = data["genome"]
60
+ self.taxon["name"] = data["taxon"]["name"]
61
+ self.taxon["uri"] = data["taxon"]["uri"]
62
+ self.metadataAuthor = data["metadataAuthor"]
63
+ self.assemblyAuthor = data["assemblyAuthor"]
64
+ self.dateCreated = data["dateCreated"]
65
+ self.masking = data["masking"]
66
+ self.checksum = data["checksum"]
67
+
68
+ try:
69
+ self.genomeSynonym = data["genomeSynonym"]
70
+ except KeyError:
71
+ self.genomeSynonym = None
72
+
73
+ try:
74
+ self.accessionID["url"] = data["accessionID"]["url"]
75
+ except (KeyError, TypeError):
76
+ self.accessionID["url"] = None
77
+
78
+ try:
79
+ self.taxon["name"] = data["taxon"]["name"]
80
+ self.taxon["uri"] = data["taxon"]["uri"]
81
+ except KeyError:
82
+ self.taxon["name"] = None
83
+ self.taxon["uri"] = None
84
+
85
+ try:
86
+ self.assemblySoftware = data["assemblySoftware"]
87
+ except KeyError:
88
+ self.assemblySoftware = None
89
+
90
+ try:
91
+ self.voucherSpecimen = data["voucherSpecimen"]
92
+ except KeyError:
93
+ self.voucherSpecimen = None
94
+
95
+ try:
96
+ self.instrument = data["instrument"]
97
+ except KeyError:
98
+ self.instrument = None
99
+
100
+ try:
101
+ self.scholarlyArticle = data["scholarlyArticle"]
102
+ except KeyError:
103
+ self.scholarlyArticle = None
104
+
105
+ try:
106
+ self.documentation = data["documentation"]
107
+ except KeyError:
108
+ self.documentation = None
109
+
110
+ try:
111
+ self.identifier = data["identifier"]
112
+ except KeyError:
113
+ self.identifier = None
114
+
115
+ try:
116
+ self.relatedLink = data["relatedLink"]
117
+ except KeyError:
118
+ self.relatedLink = None
119
+
120
+ try:
121
+ self.funding = data["funding"]
122
+ except KeyError:
123
+ self.funding = None
124
+
125
+ try:
126
+ self.vitalStats["N50"] = data["vitalStats"]["N50"]
127
+ self.vitalStats["L50"] = data["vitalStats"]["L50"]
128
+ self.vitalStats["L90"] = data["vitalStats"]["L90"]
129
+ self.vitalStats["totalBasePairs"] = data["vitalStats"]["totalBasePairs"]
130
+ self.vitalStats["numberContigs"] = data["vitalStats"]["numberContigs"]
131
+ self.vitalStats["numberScaffolds"] = data["vitalStats"]["numberScaffolds"]
132
+ self.vitalStats["readTechnology"] = data["vitalStats"]["readTechnology"]
133
+ except (KeyError, TypeError):
134
+ self.vitalStats = {
135
+ "N50": None,
136
+ "L50": None,
137
+ "L90": None,
138
+ "totalBasePairs": None,
139
+ "numberContigs": None,
140
+ "numberScaffolds": None,
141
+ "readTechnology": None,
142
+ }
143
+
144
+ try:
145
+ self.reuseConditions = data["reuseConditions"]
146
+ except KeyError:
147
+ self.reuseConditions = None
148
+
149
+ def output_yaml(self) -> str:
150
+ return yaml.dump(self.__dict__)
151
+
152
+ def input_fasta(self, stream: str) -> None:
153
+ formulated = "\n".join(
154
+ re.sub(";~", "", line)
155
+ for line in stream.splitlines()
156
+ if line.startswith(";~")
157
+ )
158
+ self.input_yaml(formulated)
159
+
160
+ def output_fasta(self) -> str:
161
+ array = ";~- "
162
+ name = "\n;~- name:"
163
+ uri = "\n;~ uri:"
164
+ end_span = ""
165
+
166
+ data = (
167
+ f";~schema: {self.schema}\n"
168
+ f";~schemaVersion: {self.schemaVersion}\n"
169
+ f";~genome: {self.genome}\n"
170
+ f";~genomeSynonym:\n"
171
+ f"{array + array.join(x + end_span for x in self.genomeSynonym)}"
172
+ f";~version: {self.version}\n"
173
+ f";~metadataAuthor:"
174
+ f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.metadataAuthor)}'
175
+ f"\n;~assemblyAuthor:"
176
+ f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.assemblyAuthor)}'
177
+ f";~accessionID:\n"
178
+ f';~ name:{self.accessionID["name"]}\n'
179
+ f';~ url:{self.accessionID["url"]}\n'
180
+ f";~taxon:\n"
181
+ f';~ name:{self.taxon["name"]}\n'
182
+ f';~ uri:{self.taxon["uri"]}\n'
183
+ f";~assemblySoftware: {self.assemblySoftware}\n"
184
+ f";~voucherSpecimen: {self.voucherSpecimen}\n"
185
+ f";~dateCreated: {self.dateCreated}\n"
186
+ f";~instrument:\n"
187
+ f"{array + array.join(x + end_span for x in self.instrument)}"
188
+ f";~scholarlyArticle: {self.scholarlyArticle}\n"
189
+ f";~documentation: {self.documentation}\n"
190
+ f";~identifier:\n"
191
+ f"{array + array.join(x + end_span for x in self.identifier)}"
192
+ f";~relatedLink:\n"
193
+ f"{array + array.join(x + end_span for x in self.relatedLink)}"
194
+ f";~funding:\n"
195
+ f"{array + array.join(x + end_span for x in self.funding)}"
196
+ f";~masking {self.masking}\n"
197
+ f";~vitalStats:\n"
198
+ f';~-N50: {self.vitalStats["N50"]}\n'
199
+ f';~-L50: {self.vitalStats["L50"]}\n'
200
+ f';~-L90: {self.vitalStats["L90"]}\n'
201
+ f';~-totalBasePairs: {self.vitalStats["totalBasePairs"]}\n'
202
+ f';~-numberContigs: {self.vitalStats["numberContigs"]}\n'
203
+ f';~-numberScaffolds: {self.vitalStats["numberScaffolds"]}\n'
204
+ f';~-readTechnology: {self.vitalStats["readTechnology"]}\n'
205
+ f";~reuseConditions: {self.reuseConditions}\n"
206
+ f";~checksum: {self.checksum}\n"
207
+ )
208
+
209
+ return data
210
+
211
+ def input_microdata(self, stream: str) -> None:
212
+ data = microdata.get_items(stream)
213
+ data = data[0]
214
+ self.schema = data.schema
215
+ self.version = data.version
216
+ self.schemaVersion = data.schemaVersion
217
+ self.genome = data.genome
218
+ self.genomeSynonym = data.get_all("genomeSynonym")
219
+ self.metadataAuthor = data.get_all("metadataAuthor")
220
+ self.assemblyAuthor = data.get_all("assemblyAuthor")
221
+ self.accessionID = data.get_all("accessionID")
222
+ self.taxon = data.get_all("taxon")
223
+ self.assemblySoftware = data.assemblySoftware
224
+ self.voucherSpecimen = data.voucherSpecimen
225
+ self.dateCreated = data.dateCreated
226
+ self.instrument = data.get_all("instrument")
227
+ self.scholarlyArticle = data.scholarlyArticle
228
+ self.documentation = data.documentation
229
+ self.identifier = data.get_all("identifier")
230
+ self.relatedLink = data.get_all("relatedLink")
231
+ self.funding = data.get_all("funding")
232
+ self.masking = data.masking
233
+ self.vitalStats = data.get_all("vitalStats")
234
+ self.reuseConditions = data.reuseConditions
235
+ self.checksum = data.checksum
236
+
237
+ def output_microdata(self) -> str:
238
+ instrument = '<span itemprop="instrument">'
239
+ identifier = '<span itemprop="identifier">'
240
+ relatedLink = '<span itemprop="relatedLink">'
241
+ funding = '<span itemprop="funding">'
242
+ metadataAuthor = '<span itemprop="metadataAuthor">'
243
+ assemblyAuthor = '<span itemprop="assemblyAuthor">'
244
+ genomeSynonym = '<span itemprop="genomeSynonym">'
245
+ name = '<span itemprop="name">'
246
+ uri = '<span itemprop="uri">'
247
+ end_span = "</span>"
248
+
249
+ data = (
250
+ f'<div itemscope itemtype="https://raw.githubusercontent.com/FAIR-bioHeaders/FHR-Specification/main/fhr.json" version="{self.schemaVersion}">'
251
+ f'<span itemprop="schema">{self.schema}</span>'
252
+ f'<span itemprop="schemaVersion">{self.schemaVersion}</span>'
253
+ f'<span itemprop="version">{self.version}</span>'
254
+ f'<span itemprop="genome">{self.genome}</span>'
255
+ f"{genomeSynonym + genomeSynonym.join(x + end_span for x in self.genomeSynonym)}"
256
+ f'{metadataAuthor + metadataAuthor.join(name + x["name"] + end_span + uri + x["uri"] + end_span for x in self.metadataAuthor)}'
257
+ f"</span>"
258
+ f'{assemblyAuthor + assemblyAuthor.join(name + x["name"] + end_span + uri + x["uri"] + end_span for x in self.assemblyAuthor)}'
259
+ f"</span>"
260
+ f'<span itemprop="accessionID">'
261
+ f' <span itemprop="name">{self.accessionID["name"]}</span>'
262
+ f' <span itemprop="url">{self.accessionID["url"]}"</span>'
263
+ f"</span>"
264
+ f'<span itemprop="taxon">'
265
+ f' <span itemprop="name">{self.taxon["name"]}</span>'
266
+ f' <span itemprop="uri">{self.taxon["uri"]}</span>'
267
+ f"</span>"
268
+ f'<span itemprop="assemblySoftware">{self.assemblySoftware}</span>'
269
+ f'<span itemprop="voucherSpecimen">{self.voucherSpecimen}</span>'
270
+ f'<span itemprop="dateCreated">{self.dateCreated}</span>'
271
+ f"{instrument + instrument.join(x + end_span for x in self.instrument)}"
272
+ f'<span itemprop="scholarlyArticle">{self.scholarlyArticle}</span>'
273
+ f'<span itemprop="documentation">{self.documentation}</span>'
274
+ f"{identifier + identifier.join(x + end_span for x in self.identifier)}"
275
+ f"{relatedLink + relatedLink.join(x + end_span for x in self.relatedLink)}"
276
+ f"{funding + funding.join(x + end_span for x in self.funding)}"
277
+ f'<span itemprop="masking">{self.masking}</span>'
278
+ f'<span itemprop="vitalStats">'
279
+ f' <span itemprop="N50">{self.vitalStats["N50"]}</span>'
280
+ f' <span itemprop="L50">{self.vitalStats["L50"]}</span>'
281
+ f' <span itemprop="L90">{self.vitalStats["L90"]}</span>'
282
+ f' <span itemprop="totalBasePairs">{self.vitalStats["totalBasePairs"]}</span>'
283
+ f' <span itemprop="numberContigs">{self.vitalStats["numberContigs"]}</span>'
284
+ f' <span itemprop="numberScaffolds">{self.vitalStats["numberScaffolds"]}</span>'
285
+ f' <span itemprop="readTechnology">{self.vitalStats["readTechnology"]}</span>'
286
+ f"</span>"
287
+ f'<span itemprop="reuseConditions">{self.reuseConditions}</span>'
288
+ f'<span itemprop="checksum">{self.checksum}</span>'
289
+ f"</div>"
290
+ )
291
+
292
+ return data
293
+
294
+ def input_json(self, stream: str) -> None:
295
+ data = json.loads(stream)
296
+ self.schema = data["schema"]
297
+ self.version = data["version"]
298
+ self.schemaVersion = data["schemaVersion"]
299
+ self.genome = data["genome"]
300
+ self.taxon["name"] = data["taxon"]["name"]
301
+ self.taxon["uri"] = data["taxon"]["uri"]
302
+ self.metadataAuthor = data["metadataAuthor"]
303
+ self.assemblyAuthor = data["assemblyAuthor"]
304
+ self.dateCreated = data["dateCreated"]
305
+ self.masking = data["masking"]
306
+ self.checksum = data["checksum"]
307
+ try:
308
+ self.genomeSynonym = data["genomeSynonym"]
309
+ except KeyError:
310
+ self.genomeSynonym = None
311
+
312
+ try:
313
+ self.accessionID["url"] = data["accessionID"]["url"]
314
+ except (KeyError, TypeError):
315
+ self.accessionID["url"] = None
316
+
317
+ try:
318
+ self.taxon["name"] = data["taxon"]["name"]
319
+ self.taxon["uri"] = data["taxon"]["uri"]
320
+ except KeyError:
321
+ self.taxon["name"] = None
322
+ self.taxon["uri"] = None
323
+
324
+ try:
325
+ self.assemblySoftware = data["assemblySoftware"]
326
+ except KeyError:
327
+ self.assemblySoftware = None
328
+
329
+ try:
330
+ self.voucherSpecimen = data["voucherSpecimen"]
331
+ except KeyError:
332
+ self.voucherSpecimen = None
333
+
334
+ try:
335
+ self.instrument = data["instrument"]
336
+ except KeyError:
337
+ self.instrument = None
338
+
339
+ try:
340
+ self.scholarlyArticle = data["scholarlyArticle"]
341
+ except KeyError:
342
+ self.scholarlyArticle = None
343
+
344
+ try:
345
+ self.documentation = data["documentation"]
346
+ except KeyError:
347
+ self.documentation = None
348
+
349
+ try:
350
+ self.identifier = data["identifier"]
351
+ except KeyError:
352
+ self.identifier = None
353
+
354
+ try:
355
+ self.relatedLink = data["relatedLink"]
356
+ except KeyError:
357
+ self.relatedLink = None
358
+
359
+ try:
360
+ self.funding = data["funding"]
361
+ except KeyError:
362
+ self.funding = None
363
+
364
+ try:
365
+ self.vitalStats["N50"] = data["vitalStats"]["N50"]
366
+ self.vitalStats["L50"] = data["vitalStats"]["L50"]
367
+ self.vitalStats["L90"] = data["vitalStats"]["L90"]
368
+ self.vitalStats["totalBasePairs"] = data["vitalStats"]["totalBasePairs"]
369
+ self.vitalStats["numberContigs"] = data["vitalStats"]["numberContigs"]
370
+ self.vitalStats["numberScaffolds"] = data["vitalStats"]["numberScaffolds"]
371
+ self.vitalStats["readTechnology"] = data["vitalStats"]["readTechnology"]
372
+ except (KeyError, TypeError):
373
+ self.vitalStats = {
374
+ "N50": None,
375
+ "L50": None,
376
+ "L90": None,
377
+ "totalBasePairs": None,
378
+ "numberContigs": None,
379
+ "numberScaffolds": None,
380
+ "readTechnology": None,
381
+ }
382
+
383
+ try:
384
+ self.reuseConditions = data["reuseConditions"]
385
+ except KeyError:
386
+ self.reuseConditions = None
387
+
388
+ def output_json(self) -> str:
389
+ return json.dumps(self.__dict__)
390
+
391
+ def input_gfa(self, stream: str) -> None:
392
+ formulated = "\n".join(
393
+ re.sub("#~", "", line)
394
+ for line in stream.splitlines()
395
+ if line.startswith("#~")
396
+ )
397
+ self.input_yaml(formulated)
398
+
399
+ def output_gfa(self) -> str:
400
+ array = "#~- "
401
+ name = "\n;~- name:"
402
+ uri = "\n;~ uri:"
403
+ end_span = ""
404
+
405
+ data = (
406
+ f"#~schema: {self.schema}\n"
407
+ f"#~schemaVersion: {self.schemaVersion}\n"
408
+ f"#~genome: {self.genome}\n"
409
+ f"#~genomeSynonym:\n"
410
+ f"{array + array.join(x + end_span for x in self.genomeSynonym)}"
411
+ f"#~version: {self.version}\n"
412
+ f"#~metadataAuthor:"
413
+ f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.metadataAuthor)}'
414
+ f"\n;~assemblyAuthor:"
415
+ f'{name + name.join(name + x["name"] + uri + x["uri"] for x in self.assemblyAuthor)}'
416
+ f"#~accessionID:\n"
417
+ f'#~ name:{self.accessionID["name"]}\n'
418
+ f'#~ url:{self.accessionID["url"]}\n'
419
+ f"#~taxon:\n"
420
+ f'#~ name:{self.taxon["name"]}\n'
421
+ f'#~ uri:{self.taxon["uri"]}\n'
422
+ f"#~assemblySoftware: {self.assemblySoftware}\n"
423
+ f"#~voucherSpecimen: {self.voucherSpecimen}\n"
424
+ f"#~dateCreated: {self.dateCreated}\n"
425
+ f"#~instrument:\n"
426
+ f"{array + array.join(x + end_span for x in self.instrument)}"
427
+ f"#~scholarlyArticle: {self.scholarlyArticle}\n"
428
+ f"#~documentation: {self.documentation}\n"
429
+ f"#~identifier:\n"
430
+ f"{array + array.join(x + end_span for x in self.identifier)}"
431
+ f"#~relatedLink:\n"
432
+ f"{array + array.join(x + end_span for x in self.relatedLink)}"
433
+ f"#~funding:\n"
434
+ f"{array + array.join(x + end_span for x in self.funding)}"
435
+ f"#~masking {self.masking}\n"
436
+ f"#~vitalStats:\n"
437
+ f'#~-N50: {self.vitalStats["N50"]}\n'
438
+ f'#~-L50: {self.vitalStats["L50"]}\n'
439
+ f'#~-L90: {self.vitalStats["L90"]}\n'
440
+ f'#~-totalBasePairs: {self.vitalStats["totalBasePairs"]}\n'
441
+ f'#~-numberContigs: {self.vitalStats["numberContigs"]}\n'
442
+ f'#~-numberScaffolds: {self.vitalStats["numberScaffolds"]}\n'
443
+ f'#~-readTechnology: {self.vitalStats["readTechnology"]}\n'
444
+ f"#~reuseConditions: {self.reuseConditions}\n"
445
+ f"#~checksum: {self.checksum}\n"
446
+ )
447
+
448
+ return data
449
+
450
+ def fhr_validate(self) -> None:
451
+ fhr_instance = json.dumps(self.__dict__)
452
+ validate(instance=fhr_instance, schema=schema)
@@ -0,0 +1,40 @@
1
+ [tool.flake8]
2
+ max-line-length = 120
3
+ ignore = "E203, E266, E501, W503"
4
+
5
+ [tool.poetry]
6
+ name = "fhr"
7
+ version = "0.1.1"
8
+ description = "This tool is used to validate and convert between different FHR header serializations"
9
+ authors = ["David Molik <david.molik@usda.gov>","Adam Wright <adam.wright@oicr.on.ca>"]
10
+ license = "USDA-ARS"
11
+ readme = "README.md"
12
+
13
+ [tool.poetry.dependencies]
14
+ python = "^3.9"
15
+ argparse = "^1.4.0"
16
+ microdata = "^0.8.0"
17
+ jsonschema = "^4.21.1"
18
+ pyyaml = "^6.0.1"
19
+
20
+ [tool.poetry.dev-dependencies]
21
+ mypy = "^1.8.0"
22
+ types-pyyaml = "^6.0.12.12"
23
+ types-jsonschema = "^4.21.0.20240118"
24
+ isort = "^5.13.2"
25
+ black = "^24.1.1"
26
+ ruff = "^0.2.1"
27
+
28
+ [tool.poetry.scripts]
29
+ fhr-convert = "fhr_convert:main"
30
+ fhr-validate = "fhr_validate:main"
31
+ fhr-fasta-strip = "fasta.fhr_fasta_strip:main"
32
+ fhr-fasta-combine = "fasta.fhr_fasta_combine:main"
33
+ fhr-fasta-validate = "fasta.fhr_fasta_validate:main"
34
+ fhr-gfa-strip = "gfa.fhr_gfa_strip:main"
35
+ fhr-gfa-combine = "gfa.fhr_gfa_combine:main"
36
+ fhr-gfa-validate = "gfa.fhr_gfa_validate:main"
37
+
38
+ [tool.poetry.group.dev.dependencies]
39
+ pytest = "^8.0.0"
40
+