biodata-models 0.0.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- biodata_models/__init__.py +3 -0
- biodata_models/_generators/__init__.py +1 -0
- biodata_models/_generators/dev_utils.py +40 -0
- biodata_models/_generators/generator.py +109 -0
- biodata_models/_generators/models/atlas.csv +3 -0
- biodata_models/_generators/models/brain_atlas.csv +841 -0
- biodata_models/_generators/models/celegans_developmental_stage.csv +775 -0
- biodata_models/_generators/models/drosophila_developmental_stage.csv +211 -0
- biodata_models/_generators/models/harp_types.csv +46 -0
- biodata_models/_generators/models/human_developmental_stage.csv +240 -0
- biodata_models/_generators/models/modalities.csv +22 -0
- biodata_models/_generators/models/mouse_anatomy.csv +8037 -0
- biodata_models/_generators/models/mouse_developmental_stage.csv +135 -0
- biodata_models/_generators/models/organizations.csv +128 -0
- biodata_models/_generators/models/process_names.csv +48 -0
- biodata_models/_generators/models/protocols.csv +68 -0
- biodata_models/_generators/models/registries.csv +15 -0
- biodata_models/_generators/models/slap2_acquisition_type.csv +3 -0
- biodata_models/_generators/models/species.csv +18 -0
- biodata_models/_generators/models/specimen_procedure_types.csv +18 -0
- biodata_models/_generators/models/stimulus_modality.csv +9 -0
- biodata_models/_generators/templates/atlas.txt +11 -0
- biodata_models/_generators/templates/brain_atlas.txt +52 -0
- biodata_models/_generators/templates/celegans_developmental_stage.txt +124 -0
- biodata_models/_generators/templates/drosophila_developmental_stage.txt +124 -0
- biodata_models/_generators/templates/harp_types.txt +29 -0
- biodata_models/_generators/templates/human_developmental_stage.txt +124 -0
- biodata_models/_generators/templates/modalities.txt +37 -0
- biodata_models/_generators/templates/mouse_anatomy.txt +185 -0
- biodata_models/_generators/templates/mouse_developmental_stage.txt +124 -0
- biodata_models/_generators/templates/organizations.txt +52 -0
- biodata_models/_generators/templates/process_names.txt +11 -0
- biodata_models/_generators/templates/protocols.txt +52 -0
- biodata_models/_generators/templates/registries.txt +11 -0
- biodata_models/_generators/templates/slap2_acquisition_type.txt +11 -0
- biodata_models/_generators/templates/species.txt +81 -0
- biodata_models/_generators/templates/specimen_procedure_types.txt +11 -0
- biodata_models/_generators/templates/stimulus_modality.txt +11 -0
- biodata_models/_generators/update_harp_types.py +8 -0
- biodata_models/atlas.py +10 -0
- biodata_models/brain_atlas.py +5085 -0
- biodata_models/celegans_developmental_stage.py +902 -0
- biodata_models/coordinates.py +70 -0
- biodata_models/data_name_patterns.py +116 -0
- biodata_models/devices.py +181 -0
- biodata_models/drosophila_developmental_stage.py +332 -0
- biodata_models/gene.py +54 -0
- biodata_models/harp_types.py +432 -0
- biodata_models/human_developmental_stage.py +361 -0
- biodata_models/licenses.py +10 -0
- biodata_models/modalities.py +231 -0
- biodata_models/mouse_anatomy.py +8796 -0
- biodata_models/mouse_developmental_stage.py +256 -0
- biodata_models/organizations.py +1448 -0
- biodata_models/pid_names.py +25 -0
- biodata_models/process_names.py +55 -0
- biodata_models/protocols.py +904 -0
- biodata_models/reagent.py +25 -0
- biodata_models/registries.py +22 -0
- biodata_models/slap2_acquisition_type.py +10 -0
- biodata_models/species.py +260 -0
- biodata_models/specimen_procedure_types.py +25 -0
- biodata_models/stimulus_modality.py +16 -0
- biodata_models/system_architecture.py +53 -0
- biodata_models/units.py +173 -0
- biodata_models-0.0.4.dist-info/METADATA +80 -0
- biodata_models-0.0.4.dist-info/RECORD +70 -0
- biodata_models-0.0.4.dist-info/WHEEL +5 -0
- biodata_models-0.0.4.dist-info/licenses/LICENSE +21 -0
- biodata_models-0.0.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Generators"""
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Dev utilities for constructing models from CSV files"""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
import yaml
|
|
5
|
+
import requests
|
|
6
|
+
import pandas as pd
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def to_class_name_underscored(name: str) -> str:
|
|
11
|
+
"""Convert a name to a class name by capitalizing and removing non-alphanumeric characters.
|
|
12
|
+
|
|
13
|
+
Always prefixes the string with an underscore."""
|
|
14
|
+
name = str(name)
|
|
15
|
+
return "_" + re.sub(r"\W+", "_", name.title()).replace(" ", "")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def to_class_name(name: str) -> str:
|
|
19
|
+
"""Convert a name to a valid class name by capitalizing and removing non-alphanumeric characters.
|
|
20
|
+
|
|
21
|
+
Replace any non alphanumeric characters at the beginning of the string with a single _."""
|
|
22
|
+
name = str(name)
|
|
23
|
+
return re.sub(r"\W|^(?=\d)", "_", name.title()).replace(" ", "")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def update_harp_types(
|
|
27
|
+
url: str = "https://raw.githubusercontent.com/harp-tech/whoami/refs/heads/main/whoami.yml",
|
|
28
|
+
):
|
|
29
|
+
"""Pull the latest harp types from the whoami.yml file and save them to a CSV file."""
|
|
30
|
+
response = requests.get(url, allow_redirects=True, timeout=5)
|
|
31
|
+
content = response.content.decode("utf-8")
|
|
32
|
+
content = yaml.safe_load(content)
|
|
33
|
+
|
|
34
|
+
devices = content["devices"]
|
|
35
|
+
data = [{"name": device["name"], "whoami": str(whoami)} for whoami, device in devices.items()]
|
|
36
|
+
|
|
37
|
+
df = pd.DataFrame(data)
|
|
38
|
+
|
|
39
|
+
current_dir = Path(__file__).parent.resolve()
|
|
40
|
+
df.to_csv(current_dir / "models/harp_types.csv", index=False)
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Code generator for data schema models."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
from jinja2 import Environment
|
|
5
|
+
import pandas as pd
|
|
6
|
+
from biodata_models._generators.dev_utils import to_class_name, to_class_name_underscored
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
import subprocess
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
SKIP_SORT = ["mouse_anatomy"]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def check_black_version():
|
|
15
|
+
"""Check that the version of the black package is >= 25.0.0"""
|
|
16
|
+
import black
|
|
17
|
+
from packaging import version
|
|
18
|
+
|
|
19
|
+
if version.parse(black.__version__) < version.parse("25.0.0"):
|
|
20
|
+
raise AssertionError("Please upgrade the black package to version 25.0.0 or later.")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def load_data(data_type: str, root_path: str) -> pd.DataFrame:
|
|
24
|
+
"""Load the data for the given data type"""
|
|
25
|
+
|
|
26
|
+
ROOT_DIR = Path(root_path)
|
|
27
|
+
data_file = ROOT_DIR / "_generators" / "models" / f"{data_type}.csv"
|
|
28
|
+
data = pd.read_csv(data_file)
|
|
29
|
+
|
|
30
|
+
# If there's a name field, sort A->Z
|
|
31
|
+
if "name" in data.columns and data_type not in SKIP_SORT:
|
|
32
|
+
data = data.sort_values("name")
|
|
33
|
+
|
|
34
|
+
return data
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def regex_search(value, pattern):
|
|
38
|
+
"""Perform regex search on a value and return matched groups."""
|
|
39
|
+
import re
|
|
40
|
+
|
|
41
|
+
match = re.search(pattern, value)
|
|
42
|
+
if match:
|
|
43
|
+
return match.groups()
|
|
44
|
+
return []
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def generate_code(data_type: str, root_path: str, isort: bool = True, black: bool = True):
|
|
48
|
+
"""Generate code from the template type
|
|
49
|
+
|
|
50
|
+
Parameters
|
|
51
|
+
----------
|
|
52
|
+
data_type : str
|
|
53
|
+
Which template file to use
|
|
54
|
+
isort : bool, optional
|
|
55
|
+
Whether to run isort on the output, by default True
|
|
56
|
+
black : bool, optional
|
|
57
|
+
Whether to run black on the output, by default True
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
ROOT_DIR = Path(root_path)
|
|
61
|
+
template_file = ROOT_DIR / "_generators" / "templates" / f"{data_type}.txt"
|
|
62
|
+
output_file = ROOT_DIR / f"{data_type}.py"
|
|
63
|
+
|
|
64
|
+
data = load_data(data_type, root_path)
|
|
65
|
+
|
|
66
|
+
# Load template
|
|
67
|
+
with open(template_file) as f:
|
|
68
|
+
template = f.read()
|
|
69
|
+
|
|
70
|
+
# Set up Jinja2 environment
|
|
71
|
+
env = Environment()
|
|
72
|
+
env.filters["to_class_name"] = to_class_name
|
|
73
|
+
env.filters["to_class_name_underscored"] = to_class_name_underscored
|
|
74
|
+
env.filters["unique_rows"] = lambda data, key: data.drop_duplicates(subset=key)
|
|
75
|
+
|
|
76
|
+
env.filters["regex_search"] = regex_search
|
|
77
|
+
rendered_template = env.from_string(template)
|
|
78
|
+
|
|
79
|
+
# Render template with data
|
|
80
|
+
rendered_code = rendered_template.render(data=data)
|
|
81
|
+
|
|
82
|
+
# Write generated code to file
|
|
83
|
+
with open(output_file, "w") as f:
|
|
84
|
+
f.write(rendered_code)
|
|
85
|
+
|
|
86
|
+
print(f"Code generated in {output_file}")
|
|
87
|
+
|
|
88
|
+
# Optionally, format with isort and black
|
|
89
|
+
if isort:
|
|
90
|
+
subprocess.run(["isort", str(output_file)])
|
|
91
|
+
|
|
92
|
+
if black:
|
|
93
|
+
subprocess.run(["black", str(output_file)])
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
if __name__ == "__main__":
|
|
97
|
+
check_black_version()
|
|
98
|
+
|
|
99
|
+
parser = argparse.ArgumentParser(description="Generate code from templates.")
|
|
100
|
+
parser.add_argument("--type", required=True, help="The data type to generate code for (e.g., 'platforms').")
|
|
101
|
+
parser.add_argument(
|
|
102
|
+
"--root-path",
|
|
103
|
+
required=False,
|
|
104
|
+
default="./src/biodata_models/",
|
|
105
|
+
help="Path to the source folder of the project",
|
|
106
|
+
)
|
|
107
|
+
args = parser.parse_args()
|
|
108
|
+
|
|
109
|
+
generate_code(args.type, args.root_path)
|