niagads-metadata-validator 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- niagads_metadata_validator-0.2.0/PKG-INFO +231 -0
- niagads_metadata_validator-0.2.0/README.md +207 -0
- niagads_metadata_validator-0.2.0/niagads/arg_parser/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/arg_parser/core.py +41 -0
- niagads_metadata_validator-0.2.0/niagads/csv_parser/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/csv_parser/core.py +123 -0
- niagads_metadata_validator-0.2.0/niagads/csv_validator/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/csv_validator/core.py +218 -0
- niagads_metadata_validator-0.2.0/niagads/dict_utils/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/dict_utils/core.py +169 -0
- niagads_metadata_validator-0.2.0/niagads/enums/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/enums/core.py +36 -0
- niagads_metadata_validator-0.2.0/niagads/excel_parser/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/excel_parser/core.py +212 -0
- niagads_metadata_validator-0.2.0/niagads/exceptions/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/exceptions/core.py +73 -0
- niagads_metadata_validator-0.2.0/niagads/json_validator/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/json_validator/core.py +192 -0
- niagads_metadata_validator-0.2.0/niagads/json_validator/format_checkers.py +75 -0
- niagads_metadata_validator-0.2.0/niagads/list_utils/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/list_utils/core.py +192 -0
- niagads_metadata_validator-0.2.0/niagads/logging_utils/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/logging_utils/core.py +68 -0
- niagads_metadata_validator-0.2.0/niagads/metadata_validator/README.md +194 -0
- niagads_metadata_validator-0.2.0/niagads/metadata_validator/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/metadata_validator/core.py +134 -0
- niagads_metadata_validator-0.2.0/niagads/metadata_validator_tool/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/metadata_validator_tool/core.py +280 -0
- niagads_metadata_validator-0.2.0/niagads/pd_dataframe/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/pd_dataframe/core.py +46 -0
- niagads_metadata_validator-0.2.0/niagads/string_utils/__init__.py +4 -0
- niagads_metadata_validator-0.2.0/niagads/string_utils/core.py +437 -0
- niagads_metadata_validator-0.2.0/niagads/string_utils/regular_expressions.py +13 -0
- niagads_metadata_validator-0.2.0/niagads/sys_utils/__init__.py +3 -0
- niagads_metadata_validator-0.2.0/niagads/sys_utils/core.py +349 -0
- niagads_metadata_validator-0.2.0/pyproject.toml +39 -0
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: niagads-metadata-validator
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: JSON Schema based validation of dataset metadata developed to support submissions to the NIAGADS Data Sharing Service
|
|
5
|
+
License: GNU GPLv3
|
|
6
|
+
Author: fossilfriend
|
|
7
|
+
Author-email: egreenfest@gmail.com
|
|
8
|
+
Requires-Python: >=3.11,<4.0
|
|
9
|
+
Classifier: License :: Other/Proprietary License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Requires-Dist: jsonschema (>=4.23.0,<5.0.0)
|
|
15
|
+
Requires-Dist: openpyxl (>=3.1.5,<4.0.0)
|
|
16
|
+
Requires-Dist: pandas (>=2.2.3,<3.0.0)
|
|
17
|
+
Requires-Dist: python-dateutil (>=2.9.0.post0,<3.0.0)
|
|
18
|
+
Requires-Dist: strenum (>=0.4.15,<0.5.0)
|
|
19
|
+
Project-URL: Bug Reports, https://github.com/NIAGADS/niagads-pylib/issues
|
|
20
|
+
Project-URL: Homepage, https://github.com/NIAGADS/niagads-pylib
|
|
21
|
+
Project-URL: Source, https://github.com/NIAGADS/niagads-pylib
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
<!-- markdownlint-disable -->
|
|
25
|
+
|
|
26
|
+
<a href="https://github.com/NIAGADS/niagads-pylib/blob/main/bases/niagads/metadata_validator_tool/core.py#L0"><img align="right" style="float:right;" src="https://img.shields.io/badge/-source-cccccc?style=flat-square"></a>
|
|
27
|
+
|
|
28
|
+
# NIAGADS JSON Schema based metadata validation tools
|
|
29
|
+
|
|
30
|
+
This tool allows the user to perform [JSON Schema](https://json-schema.org/)-based validation of a sample or file manifest metadata file arranged in tabular format (with a header row that has field names matching the validation schema).
|
|
31
|
+
|
|
32
|
+
The tool works for delimited text files (.tab, .csv., .txt) as well as excel (.xls, .xlsx) files.
|
|
33
|
+
|
|
34
|
+
This tool can be run as a script or can also be imported as a module. When run as a script, results are piped to STDOUT unless the `--log` option is specified.
|
|
35
|
+
|
|
36
|
+
## Requirements
|
|
37
|
+
|
|
38
|
+
* Python: >3.12,<4.0
|
|
39
|
+
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
## Usage
|
|
43
|
+
|
|
44
|
+
### command-line
|
|
45
|
+
|
|
46
|
+
Run with the `--help` option to get full USAGE information
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
validate-metadata --help
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
### module
|
|
53
|
+
|
|
54
|
+
Import package into your python script.
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
import niagads.metadata_validator_tool.core as mv_tool
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Use [`mv_tool.initialize_validator`](#function-initialize_validator) to initialize and retrieve a validator object for further manipulation. Use [`mv_tool.run`](#function-run) to initialize and run a validation with default configuration. See [validator documentation](https://github.com/NIAGADS/niagads-pylib/blob/main/components/niagads/metadata_validator/README.md) for more information about validator properties and member functions.
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
Example code, with schema and metadata files are availble in the code repostory: [examples/niagads-metadata-validator](https://github.com/NIAGADS/niagads-pylib/blob/e58808f2ef2b412e68ef66ff214683783d2f7576/projects/examples/niagads-metadata-validator/example.ipynb).
|
|
65
|
+
|
|
66
|
+
---
|
|
67
|
+
|
|
68
|
+
## API Reference
|
|
69
|
+
|
|
70
|
+
---
|
|
71
|
+
|
|
72
|
+
## <kbd>function</kbd> `get_templated_schema_file`
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
get_templated_schema_file(dir: str, template: str) → str
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Verify that templated schema file `{schemaDir}/{vType}.json` exists.
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
**Args:**
|
|
83
|
+
|
|
84
|
+
- <b>`path`</b> (str): path to directory containing schema file
|
|
85
|
+
- <b>`template`</b> (str): template name
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
**Raises:**
|
|
90
|
+
|
|
91
|
+
- <b>`FileExistsError`</b>: if the schema file does not exist
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
**Returns:**
|
|
96
|
+
|
|
97
|
+
- <b>`str`</b>: schema file name
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
---
|
|
101
|
+
|
|
102
|
+
## <kbd>function</kbd> `get_templated_metadata_file`
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
get_templated_metadata_file(
|
|
106
|
+
prefix: str,
|
|
107
|
+
template: str,
|
|
108
|
+
extensions: List[str] = ['xlsx', 'xls', 'txt', 'csv', 'tab']
|
|
109
|
+
) → str
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Find metadata file based on templated name `{prefix}{validator_type}.{ext}`.
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
**Args:**
|
|
117
|
+
|
|
118
|
+
- <b>`path`</b> (str): file path; may include prefix/file pattern to match (e.g. /files/study1/experiment1-)
|
|
119
|
+
- <b>`template`</b> (str): template name
|
|
120
|
+
- <b>`extensions`</b> (List[str], optional): allowable file extensions. Defaults to ["xlsx", "xls", "txt", "csv", "tab"].
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
**Raises:**
|
|
125
|
+
|
|
126
|
+
- <b>`FileNotFoundError`</b>: if metadata file does not exist
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
**Returns:**
|
|
131
|
+
|
|
132
|
+
- <b>`str`</b>: metadata file name
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
---
|
|
136
|
+
|
|
137
|
+
## <kbd>function</kbd> `initialize_validator`
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
initialize_validator(
|
|
141
|
+
file: str,
|
|
142
|
+
schema: str,
|
|
143
|
+
metadataType: MetadataValidatorType,
|
|
144
|
+
idField: str = None
|
|
145
|
+
) → Union[BiosourcePropertiesValidator, FileManifestValidator]
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
Initialize and return a metadata validator.
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
**Args:**
|
|
153
|
+
|
|
154
|
+
- <b>`file`</b> (str): metadata file name
|
|
155
|
+
- <b>`schema`</b> (str): JSONschema file name
|
|
156
|
+
- <b>`metadataType`</b> (MetadataValidatorType): type of metadata to be validated
|
|
157
|
+
- <b>`idField`</b> (str, optional): biosource id field in the metadata file; required for `BIOSOURCE_PROPERTIES` validation. Defaults to None.
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
**Raises:**
|
|
162
|
+
|
|
163
|
+
- <b>`RuntimeError`</b>: if `metadataType == 'BIOSOURCE_PROPERTIES'` and no `idField` was provided
|
|
164
|
+
- <b>`ValueError`</b>: if invalid `metadataType` is specified
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
**Returns:**
|
|
169
|
+
|
|
170
|
+
- <b>`Union[BiosourcePropertiesValidator, FileManifestValidator]`</b>: the validator object
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
---
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
## <kbd>function</kbd> `run`
|
|
177
|
+
|
|
178
|
+
```python
|
|
179
|
+
run(
|
|
180
|
+
file: str,
|
|
181
|
+
schema: str,
|
|
182
|
+
metadataType: str,
|
|
183
|
+
idField: str = None,
|
|
184
|
+
failOnError: bool = False
|
|
185
|
+
)
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Run validation.
|
|
189
|
+
|
|
190
|
+
Validator initialization fully encapsulated. Returns validation result.
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
**Args:**
|
|
195
|
+
|
|
196
|
+
- <b>`file`</b> (str): metadata file name
|
|
197
|
+
- <b>`schema`</b> (str): JSONschema file name
|
|
198
|
+
- <b>`metadataType`</b> (MetadataValidatorType): type of metadata to be validated
|
|
199
|
+
- <b>`idField`</b> (str, optional): biosource id field in the metadata file; required for `BIOSOURCE_PROPERTIES` valdiatoin. Defaults to None.
|
|
200
|
+
- <b>`failOnError`</b> (bool, optional): raise an exception on validation error if true, otherwise returns list of validation errors. Defaults to False.
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
**Returns:**
|
|
205
|
+
|
|
206
|
+
- <b>`list`</b>: list of validation errors
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
---
|
|
210
|
+
|
|
211
|
+
## <kbd>class</kbd> `MetadataValidatorType`
|
|
212
|
+
Enum defining types of supported tabular metadata files.
|
|
213
|
+
|
|
214
|
+
```python
|
|
215
|
+
BIOSOURCE_PROPERTIES = '''biosource properties file;
|
|
216
|
+
a file that maps a sample or participant to descriptive properties
|
|
217
|
+
(e.g., phenotype or material) or a ISA-TAB-like sample file'''
|
|
218
|
+
|
|
219
|
+
FILE_MANIFEST = "file manifest or a sample-data-relationship (SDRF) file"
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
---
|
|
229
|
+
|
|
230
|
+
_This file was automatically generated via [lazydocs](https://github.com/ml-tooling/lazydocs)._
|
|
231
|
+
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
<!-- markdownlint-disable -->
|
|
2
|
+
|
|
3
|
+
<a href="https://github.com/NIAGADS/niagads-pylib/blob/main/bases/niagads/metadata_validator_tool/core.py#L0"><img align="right" style="float:right;" src="https://img.shields.io/badge/-source-cccccc?style=flat-square"></a>
|
|
4
|
+
|
|
5
|
+
# NIAGADS JSON Schema based metadata validation tools
|
|
6
|
+
|
|
7
|
+
This tool allows the user to perform [JSON Schema](https://json-schema.org/)-based validation of a sample or file manifest metadata file arranged in tabular format (with a header row that has field names matching the validation schema).
|
|
8
|
+
|
|
9
|
+
The tool works for delimited text files (.tab, .csv., .txt) as well as excel (.xls, .xlsx) files.
|
|
10
|
+
|
|
11
|
+
This tool can be run as a script or can also be imported as a module. When run as a script, results are piped to STDOUT unless the `--log` option is specified.
|
|
12
|
+
|
|
13
|
+
## Requirements
|
|
14
|
+
|
|
15
|
+
* Python: >3.12,<4.0
|
|
16
|
+
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
## Usage
|
|
20
|
+
|
|
21
|
+
### command-line
|
|
22
|
+
|
|
23
|
+
Run with the `--help` option to get full USAGE information
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
validate-metadata --help
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
### module
|
|
30
|
+
|
|
31
|
+
Import package into your python script.
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
import niagads.metadata_validator_tool.core as mv_tool
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Use [`mv_tool.initialize_validator`](#function-initialize_validator) to initialize and retrieve a validator object for further manipulation. Use [`mv_tool.run`](#function-run) to initialize and run a validation with default configuration. See [validator documentation](https://github.com/NIAGADS/niagads-pylib/blob/main/components/niagads/metadata_validator/README.md) for more information about validator properties and member functions.
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
Example code, with schema and metadata files are availble in the code repostory: [examples/niagads-metadata-validator](https://github.com/NIAGADS/niagads-pylib/blob/e58808f2ef2b412e68ef66ff214683783d2f7576/projects/examples/niagads-metadata-validator/example.ipynb).
|
|
42
|
+
|
|
43
|
+
---
|
|
44
|
+
|
|
45
|
+
## API Reference
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## <kbd>function</kbd> `get_templated_schema_file`
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
get_templated_schema_file(dir: str, template: str) → str
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Verify that templated schema file `{schemaDir}/{vType}.json` exists.
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
**Args:**
|
|
60
|
+
|
|
61
|
+
- <b>`path`</b> (str): path to directory containing schema file
|
|
62
|
+
- <b>`template`</b> (str): template name
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
**Raises:**
|
|
67
|
+
|
|
68
|
+
- <b>`FileExistsError`</b>: if the schema file does not exist
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
**Returns:**
|
|
73
|
+
|
|
74
|
+
- <b>`str`</b>: schema file name
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## <kbd>function</kbd> `get_templated_metadata_file`
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
get_templated_metadata_file(
|
|
83
|
+
prefix: str,
|
|
84
|
+
template: str,
|
|
85
|
+
extensions: List[str] = ['xlsx', 'xls', 'txt', 'csv', 'tab']
|
|
86
|
+
) → str
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Find metadata file based on templated name `{prefix}{validator_type}.{ext}`.
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
**Args:**
|
|
94
|
+
|
|
95
|
+
- <b>`path`</b> (str): file path; may include prefix/file pattern to match (e.g. /files/study1/experiment1-)
|
|
96
|
+
- <b>`template`</b> (str): template name
|
|
97
|
+
- <b>`extensions`</b> (List[str], optional): allowable file extensions. Defaults to ["xlsx", "xls", "txt", "csv", "tab"].
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
**Raises:**
|
|
102
|
+
|
|
103
|
+
- <b>`FileNotFoundError`</b>: if metadata file does not exist
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
**Returns:**
|
|
108
|
+
|
|
109
|
+
- <b>`str`</b>: metadata file name
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
---
|
|
113
|
+
|
|
114
|
+
## <kbd>function</kbd> `initialize_validator`
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
initialize_validator(
|
|
118
|
+
file: str,
|
|
119
|
+
schema: str,
|
|
120
|
+
metadataType: MetadataValidatorType,
|
|
121
|
+
idField: str = None
|
|
122
|
+
) → Union[BiosourcePropertiesValidator, FileManifestValidator]
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Initialize and return a metadata validator.
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
**Args:**
|
|
130
|
+
|
|
131
|
+
- <b>`file`</b> (str): metadata file name
|
|
132
|
+
- <b>`schema`</b> (str): JSONschema file name
|
|
133
|
+
- <b>`metadataType`</b> (MetadataValidatorType): type of metadata to be validated
|
|
134
|
+
- <b>`idField`</b> (str, optional): biosource id field in the metadata file; required for `BIOSOURCE_PROPERTIES` validation. Defaults to None.
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
**Raises:**
|
|
139
|
+
|
|
140
|
+
- <b>`RuntimeError`</b>: if `metadataType == 'BIOSOURCE_PROPERTIES'` and no `idField` was provided
|
|
141
|
+
- <b>`ValueError`</b>: if invalid `metadataType` is specified
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
**Returns:**
|
|
146
|
+
|
|
147
|
+
- <b>`Union[BiosourcePropertiesValidator, FileManifestValidator]`</b>: the validator object
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
---
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
## <kbd>function</kbd> `run`
|
|
154
|
+
|
|
155
|
+
```python
|
|
156
|
+
run(
|
|
157
|
+
file: str,
|
|
158
|
+
schema: str,
|
|
159
|
+
metadataType: str,
|
|
160
|
+
idField: str = None,
|
|
161
|
+
failOnError: bool = False
|
|
162
|
+
)
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
Run validation.
|
|
166
|
+
|
|
167
|
+
Validator initialization fully encapsulated. Returns validation result.
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
**Args:**
|
|
172
|
+
|
|
173
|
+
- <b>`file`</b> (str): metadata file name
|
|
174
|
+
- <b>`schema`</b> (str): JSONschema file name
|
|
175
|
+
- <b>`metadataType`</b> (MetadataValidatorType): type of metadata to be validated
|
|
176
|
+
- <b>`idField`</b> (str, optional): biosource id field in the metadata file; required for `BIOSOURCE_PROPERTIES` valdiatoin. Defaults to None.
|
|
177
|
+
- <b>`failOnError`</b> (bool, optional): raise an exception on validation error if true, otherwise returns list of validation errors. Defaults to False.
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
**Returns:**
|
|
182
|
+
|
|
183
|
+
- <b>`list`</b>: list of validation errors
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
---
|
|
187
|
+
|
|
188
|
+
## <kbd>class</kbd> `MetadataValidatorType`
|
|
189
|
+
Enum defining types of supported tabular metadata files.
|
|
190
|
+
|
|
191
|
+
```python
|
|
192
|
+
BIOSOURCE_PROPERTIES = '''biosource properties file;
|
|
193
|
+
a file that maps a sample or participant to descriptive properties
|
|
194
|
+
(e.g., phenotype or material) or a ISA-TAB-like sample file'''
|
|
195
|
+
|
|
196
|
+
FILE_MANIFEST = "file manifest or a sample-data-relationship (SDRF) file"
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
---
|
|
206
|
+
|
|
207
|
+
_This file was automatically generated via [lazydocs](https://github.com/ml-tooling/lazydocs)._
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""helpers for argparse args, including custom actions"""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from argparse import ArgumentTypeError
|
|
5
|
+
|
|
6
|
+
from niagads.enums.core import CaseInsensitiveEnum
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def json_type(value: str) -> dict:
|
|
10
|
+
"""
|
|
11
|
+
convert a JSON string argument value to an object
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
value (str): JSON string
|
|
15
|
+
|
|
16
|
+
Raises:
|
|
17
|
+
argparse.ArgumentTypeError
|
|
18
|
+
|
|
19
|
+
Returns:
|
|
20
|
+
dict: decoded JSON
|
|
21
|
+
"""
|
|
22
|
+
try:
|
|
23
|
+
return json.decodes(value)
|
|
24
|
+
except:
|
|
25
|
+
raise ArgumentTypeError("Invalid JSON: " + value)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def case_insensitive_enum_type(enumType: CaseInsensitiveEnum):
|
|
29
|
+
"""check that the string belongs to the `enumType`"""
|
|
30
|
+
|
|
31
|
+
def type_func(value):
|
|
32
|
+
try:
|
|
33
|
+
matchedEnum: CaseInsensitiveEnum = enumType(value)
|
|
34
|
+
return matchedEnum.value
|
|
35
|
+
|
|
36
|
+
except:
|
|
37
|
+
raise ArgumentTypeError(
|
|
38
|
+
f"invalid choice: '{value} (choose from [{', '.join(enumType.list())}]"
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
return type_func
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import json
|
|
3
|
+
|
|
4
|
+
from csv import Sniffer, Dialect
|
|
5
|
+
from pandas import read_csv, DataFrame
|
|
6
|
+
|
|
7
|
+
from niagads.dict_utils.core import convert_str2numeric_values
|
|
8
|
+
from niagads.pd_dataframe.core import strip
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class CSVFileParser:
|
|
12
|
+
"""
|
|
13
|
+
parser for CSV files; mainly to add the following functionality:
|
|
14
|
+
|
|
15
|
+
* infer delimiter
|
|
16
|
+
* to_json (leveraging pandas)
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
def __init__(self, file: str, sep: str = None, debug: bool = False):
|
|
20
|
+
"""
|
|
21
|
+
init new CSVParser
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
file (str): file name (full path)
|
|
25
|
+
sep (str, optional): delimiter; if None will attempt to infer. Defaults to None.
|
|
26
|
+
debug (bool, optional): enable debug mode. Defaults to False.
|
|
27
|
+
"""
|
|
28
|
+
self._debug = debug
|
|
29
|
+
self.logger = logging.getLogger(__name__)
|
|
30
|
+
self.__file = file
|
|
31
|
+
self.__sep = sep
|
|
32
|
+
self.__na = None # missing value string representation
|
|
33
|
+
self.__strip = False # flag for trimming leading & trailing whitespace
|
|
34
|
+
|
|
35
|
+
def na(self, value: str):
|
|
36
|
+
"""
|
|
37
|
+
fill NA's with specified value when using pandas conversions
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
value (str): value to fill (e.g., 'NULL', 'NA', '.')
|
|
41
|
+
"""
|
|
42
|
+
self.__na = value
|
|
43
|
+
|
|
44
|
+
def strip(self, strip=True):
|
|
45
|
+
"""
|
|
46
|
+
flag indicating whether to iterate over all fields and
|
|
47
|
+
trim leading and trailing spaces when converting to JSON or CSV
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
strip (bool, optional): trim leading and trailing spaces from all fields. Defaults to True.
|
|
51
|
+
"""
|
|
52
|
+
self.__strip = strip
|
|
53
|
+
|
|
54
|
+
def to_json(self, transpose=False, returnStr=False, **kwargs):
|
|
55
|
+
"""
|
|
56
|
+
converts the CSV file to JSON
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
transpose (bool, optional): transpose the worksheet?
|
|
60
|
+
returnStr (bool, optional): return jsonStr instead of object
|
|
61
|
+
**kwargs (optional): arguments to pass to `pandas` `read_csv` see
|
|
62
|
+
(see https://pandas.pydata.org/docs/reference/api/pandas.read_excel.html))
|
|
63
|
+
|
|
64
|
+
Returns:
|
|
65
|
+
if `returnStr` returns JSON string instead of object
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
# orient='records' returns indexes; e.g. [index: {row data}] so need to extract the values
|
|
69
|
+
jsonStr = self.to_pandas_df(transpose, **kwargs).to_json(orient="records")
|
|
70
|
+
|
|
71
|
+
# convert strings to numeric so can do typing validation
|
|
72
|
+
jsonObj = json.loads(jsonStr)
|
|
73
|
+
if isinstance(jsonObj, list):
|
|
74
|
+
jsonObj = [convert_str2numeric_values(r) for r in json.loads(jsonStr)]
|
|
75
|
+
else:
|
|
76
|
+
jsonObj = convert_str2numeric_values(jsonObj)
|
|
77
|
+
|
|
78
|
+
return json.dumps(jsonObj) if returnStr else json.loads(jsonStr)
|
|
79
|
+
|
|
80
|
+
def __trim(self, df: DataFrame):
|
|
81
|
+
"""
|
|
82
|
+
trims trailing spaces if set in options
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
df (DataFrame): pandas data frame
|
|
86
|
+
"""
|
|
87
|
+
return strip(df) if self.__strip else df
|
|
88
|
+
|
|
89
|
+
def sniff(self):
|
|
90
|
+
"""
|
|
91
|
+
'sniff' out / infer the delimitier
|
|
92
|
+
"""
|
|
93
|
+
if self.__sep is not None:
|
|
94
|
+
return self.__sep
|
|
95
|
+
else:
|
|
96
|
+
with open(self.__file, "r") as fh:
|
|
97
|
+
dialect: Dialect = Sniffer().sniff(fh.read(1024))
|
|
98
|
+
fh.seek(0)
|
|
99
|
+
return dialect.delimiter
|
|
100
|
+
|
|
101
|
+
def to_pandas_df(self, transpose=False, **kwargs) -> DataFrame:
|
|
102
|
+
"""
|
|
103
|
+
_summary_
|
|
104
|
+
|
|
105
|
+
Args:
|
|
106
|
+
transpose (str): transpose the worksheet
|
|
107
|
+
**kwargs: must match expected args for pandas.read_excel
|
|
108
|
+
(see https://pandas.pydata.org/docs/reference/api/pandas.read_excel.html)
|
|
109
|
+
|
|
110
|
+
Returns:
|
|
111
|
+
DataFrame: CSV data in data frame format
|
|
112
|
+
"""
|
|
113
|
+
if kwargs is None:
|
|
114
|
+
kwargs = {}
|
|
115
|
+
|
|
116
|
+
if "delimiter" not in kwargs:
|
|
117
|
+
kwargs["delimiter"] = self.sniff() if self.__sep is None else self.__sep
|
|
118
|
+
|
|
119
|
+
# raise error if False
|
|
120
|
+
df: DataFrame = read_csv(self.__file, **kwargs)
|
|
121
|
+
if self.__na is not None:
|
|
122
|
+
df.fillna(self.__na)
|
|
123
|
+
return self.__trim(df.T) if transpose else self.__trim(df)
|