semra 0.1.0.dev0__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {semra-0.1.0.dev0 → semra-0.1.2}/PKG-INFO +4 -3
- {semra-0.1.0.dev0 → semra-0.1.2}/pyproject.toml +5 -4
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/client.py +186 -17
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/database.py +13 -20
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/io.py +67 -47
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/neo4j_io.py +36 -22
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/templates/Dockerfile +22 -19
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/templates/run_on_startup.sh +1 -1
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/templates/startup.sh +1 -1
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/anatomy.py +1 -10
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/cells.py +8 -15
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/cli.py +3 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/complexes.py +1 -11
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/diseases.py +1 -10
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/genes.py +1 -10
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/taxrank.py +1 -10
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/utils.py +3 -2
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/pipeline.py +223 -145
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/clo.py +2 -1
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/struct.py +2 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/concept.html +1 -1
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/home.html +3 -3
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/mapping.html +5 -5
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/utils.html +1 -1
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/utils.py +14 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/version.py +1 -1
- semra-0.1.2/src/semra/web/__init__.py +1 -0
- semra-0.1.2/src/semra/web/fastapi_components.py +101 -0
- semra-0.1.2/src/semra/web/flask_components.py +166 -0
- semra-0.1.2/src/semra/web/shared.py +45 -0
- semra-0.1.2/src/semra/wsgi.py +89 -0
- semra-0.1.0.dev0/src/semra/wsgi.py +0 -268
- {semra-0.1.0.dev0 → semra-0.1.2}/LICENSE +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/README.md +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/.DS_Store +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/__init__.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/__main__.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/api.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/cli.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/gilda_utils.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/inference.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/__init__.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/graph.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/io_utils.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/__init__.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/__main__.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/py.typed +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/rules.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/__init__.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/__main__.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/biopragmatics.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/cbms2019.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/cli.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/compath.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/famplex.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/gilda.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/intact.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/ncit.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/omim.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/pubchem.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/wikidata.py +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/base.html +0 -0
- {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/mapping_set.html +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: semra
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: Semantic mapping reasoner and assembler
|
|
5
5
|
Keywords: snekpack,cookiecutter,sssom,semantic mappings,ontology merging,ontology mappings
|
|
6
6
|
Author: Charles Tapley Hoyt
|
|
@@ -29,7 +29,7 @@ Requires-Dist: tqdm
|
|
|
29
29
|
Requires-Dist: more-itertools
|
|
30
30
|
Requires-Dist: networkx
|
|
31
31
|
Requires-Dist: bioregistry>=0.12.7
|
|
32
|
-
Requires-Dist: pyobo>=0.12.
|
|
32
|
+
Requires-Dist: pyobo>=0.12.1
|
|
33
33
|
Requires-Dist: bioontologies>=0.7.0
|
|
34
34
|
Requires-Dist: typing-extensions
|
|
35
35
|
Requires-Dist: zenodo-client
|
|
@@ -52,7 +52,8 @@ Requires-Dist: matplotlib ; extra == 'landscape-notebooks'
|
|
|
52
52
|
Requires-Dist: pygraphviz ; extra == 'landscape-notebooks'
|
|
53
53
|
Requires-Dist: pytest ; extra == 'tests'
|
|
54
54
|
Requires-Dist: coverage[toml] ; extra == 'tests'
|
|
55
|
-
Requires-Dist: sssom ; extra == 'tests'
|
|
55
|
+
Requires-Dist: sssom>=0.4.16 ; extra == 'tests'
|
|
56
|
+
Requires-Dist: httpx ; extra == 'tests'
|
|
56
57
|
Requires-Dist: fastapi ; extra == 'web'
|
|
57
58
|
Requires-Dist: uvicorn ; extra == 'web'
|
|
58
59
|
Requires-Dist: flask ; extra == 'web'
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "semra"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.2"
|
|
8
8
|
description = "Semantic mapping reasoner and assembler"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [
|
|
@@ -58,7 +58,7 @@ dependencies = [
|
|
|
58
58
|
"more_itertools",
|
|
59
59
|
"networkx",
|
|
60
60
|
"bioregistry>=0.12.7",
|
|
61
|
-
"pyobo>=0.12.
|
|
61
|
+
"pyobo>=0.12.1",
|
|
62
62
|
"bioontologies>=0.7.0",
|
|
63
63
|
"typing_extensions",
|
|
64
64
|
"zenodo_client",
|
|
@@ -73,7 +73,8 @@ dependencies = [
|
|
|
73
73
|
tests = [
|
|
74
74
|
"pytest",
|
|
75
75
|
"coverage[toml]",
|
|
76
|
-
"sssom",
|
|
76
|
+
"sssom>=0.4.16",
|
|
77
|
+
"httpx",
|
|
77
78
|
]
|
|
78
79
|
docs = [
|
|
79
80
|
"sphinx>=8",
|
|
@@ -225,7 +226,7 @@ known-first-party = [
|
|
|
225
226
|
docstring-code-format = true
|
|
226
227
|
|
|
227
228
|
[tool.bumpversion]
|
|
228
|
-
current_version = "0.1.
|
|
229
|
+
current_version = "0.1.2"
|
|
229
230
|
parse = "(?P<major>\\d+)\\.(?P<minor>\\d+)\\.(?P<patch>\\d+)(?:-(?P<release>[0-9A-Za-z-]+(?:\\.[0-9A-Za-z-]+)*))?(?:\\+(?P<build>[0-9A-Za-z-]+(?:\\.[0-9A-Za-z-]+)*))?"
|
|
230
231
|
serialize = [
|
|
231
232
|
"{major}.{minor}.{patch}-{release}+{build}",
|
|
@@ -2,10 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
import dataclasses
|
|
5
6
|
import os
|
|
6
7
|
import typing as t
|
|
7
8
|
from collections import Counter
|
|
8
|
-
from
|
|
9
|
+
from textwrap import dedent
|
|
10
|
+
from typing import Any, NamedTuple, TypeAlias, cast
|
|
9
11
|
|
|
10
12
|
import bioregistry
|
|
11
13
|
import neo4j
|
|
@@ -13,6 +15,7 @@ import neo4j.graph
|
|
|
13
15
|
import networkx as nx
|
|
14
16
|
import pydantic
|
|
15
17
|
from neo4j import ManagedTransaction, unit_of_work
|
|
18
|
+
from typing_extensions import Self
|
|
16
19
|
|
|
17
20
|
import semra
|
|
18
21
|
from semra import Evidence, MappingSet, Reference, SimpleEvidence
|
|
@@ -24,11 +27,12 @@ from semra.rules import (
|
|
|
24
27
|
)
|
|
25
28
|
|
|
26
29
|
__all__ = [
|
|
30
|
+
"BaseClient",
|
|
31
|
+
"FullSummary",
|
|
27
32
|
"Neo4jClient",
|
|
28
33
|
"Node",
|
|
29
34
|
]
|
|
30
35
|
|
|
31
|
-
|
|
32
36
|
Node: TypeAlias = t.Mapping[str, Any]
|
|
33
37
|
|
|
34
38
|
TxResult: TypeAlias = list[list[Any]] | None
|
|
@@ -53,10 +57,125 @@ RELATIONS_CYPHER = "CALL db.relationshipTypes() YIELD relationshipType RETURN re
|
|
|
53
57
|
CONCEPT_NAME_CYPHER = "MATCH (n:concept) WHERE n.curie = $curie RETURN n.name LIMIT 1"
|
|
54
58
|
|
|
55
59
|
|
|
56
|
-
class
|
|
57
|
-
"""
|
|
60
|
+
class BaseClient:
|
|
61
|
+
"""An abstract class defining all of the functionality for a mapping client."""
|
|
62
|
+
|
|
63
|
+
def get_mapping(self, curie: ReferenceHint) -> semra.Mapping | None:
|
|
64
|
+
"""Get a mapping.
|
|
65
|
+
|
|
66
|
+
:param curie: Either a Reference object, a string representing a curie with
|
|
67
|
+
``semra.mapping`` as the prefix, or a local unique identifier representing a
|
|
68
|
+
SeMRA mapping.
|
|
69
|
+
|
|
70
|
+
:returns: A semantic mapping object
|
|
71
|
+
"""
|
|
72
|
+
raise NotImplementedError
|
|
73
|
+
|
|
74
|
+
def get_mapping_sets(self) -> list[MappingSet]:
|
|
75
|
+
"""Get all mappings sets."""
|
|
76
|
+
raise NotImplementedError
|
|
58
77
|
|
|
59
|
-
|
|
78
|
+
def get_mapping_set(self, curie: ReferenceHint) -> MappingSet | None:
|
|
79
|
+
"""Get a mappings set.
|
|
80
|
+
|
|
81
|
+
:param curie: The CURIE for a mapping set, using ``semra.mappingset`` as a
|
|
82
|
+
prefix. For example, use
|
|
83
|
+
``semra.mappingset:7831d5bc95698099fb6471667e5282cd`` for biomappings
|
|
84
|
+
|
|
85
|
+
:returns: A mapping set object
|
|
86
|
+
"""
|
|
87
|
+
raise NotImplementedError
|
|
88
|
+
|
|
89
|
+
def get_evidence(self, curie: ReferenceHint) -> Evidence | None:
|
|
90
|
+
"""Get an evidence.
|
|
91
|
+
|
|
92
|
+
:param curie: The CURIE for a mapping set, using ``semra.evidence`` as a prefix.
|
|
93
|
+
|
|
94
|
+
:returns: An evidence object
|
|
95
|
+
"""
|
|
96
|
+
raise NotImplementedError
|
|
97
|
+
|
|
98
|
+
def summarize_predicates(self) -> t.Counter[str]:
|
|
99
|
+
"""Get a counter of predicates."""
|
|
100
|
+
raise NotImplementedError
|
|
101
|
+
|
|
102
|
+
def summarize_justifications(self) -> t.Counter[str]:
|
|
103
|
+
"""Get a counter of mapping justifications."""
|
|
104
|
+
raise NotImplementedError
|
|
105
|
+
|
|
106
|
+
def summarize_evidence_types(self) -> t.Counter[str]:
|
|
107
|
+
"""Get a counter of evidence types."""
|
|
108
|
+
raise NotImplementedError
|
|
109
|
+
|
|
110
|
+
def summarize_mapping_sets(self) -> t.Counter[str]:
|
|
111
|
+
"""Get the number of evidences in each mapping set."""
|
|
112
|
+
raise NotImplementedError
|
|
113
|
+
|
|
114
|
+
def summarize_nodes(self) -> t.Counter[str]:
|
|
115
|
+
"""Get a counter of node types (concepts, evidences, mappings, mapping sets)."""
|
|
116
|
+
raise NotImplementedError
|
|
117
|
+
|
|
118
|
+
def summarize_concepts(self) -> t.Counter[tuple[str, str]]:
|
|
119
|
+
"""Get a counter of prefixes in concept nodes."""
|
|
120
|
+
raise NotImplementedError
|
|
121
|
+
|
|
122
|
+
def summarize_authors(self) -> t.Counter[tuple[str, str]]:
|
|
123
|
+
"""Get a counter of the number of evidences each author has contributed to."""
|
|
124
|
+
raise NotImplementedError
|
|
125
|
+
|
|
126
|
+
def get_highest_exact_matches(self, limit: int = 10) -> t.Counter[tuple[str, str]]:
|
|
127
|
+
"""Get a counter of concepts with the highest exact matches.
|
|
128
|
+
|
|
129
|
+
:param limit: The number of top concepts to return
|
|
130
|
+
|
|
131
|
+
:returns: A counter with keys that are CURIE/name pairs
|
|
132
|
+
"""
|
|
133
|
+
raise NotImplementedError
|
|
134
|
+
|
|
135
|
+
def get_exact_matches(
|
|
136
|
+
self, curie: ReferenceHint, *, max_distance: int | None = None
|
|
137
|
+
) -> dict[Reference, str] | None:
|
|
138
|
+
"""Get a mapping of references->name for all concepts equivalent to the given concept."""
|
|
139
|
+
raise NotImplementedError
|
|
140
|
+
|
|
141
|
+
def get_connected_component_graph(self, curie: ReferenceHint) -> nx.MultiDiGraph | None:
|
|
142
|
+
"""Get a networkx MultiDiGraph representing the connected component of mappings around the given CURIE.
|
|
143
|
+
|
|
144
|
+
:param curie: A CURIE string or reference
|
|
145
|
+
|
|
146
|
+
:returns: A networkx MultiDiGraph where mappings subject CURIE strings are th
|
|
147
|
+
"""
|
|
148
|
+
raise NotImplementedError
|
|
149
|
+
|
|
150
|
+
def get_concept_name(self, curie: ReferenceHint) -> str | None:
|
|
151
|
+
"""Get the name for a CURIE or reference."""
|
|
152
|
+
raise NotImplementedError
|
|
153
|
+
|
|
154
|
+
def sample_mappings_from_set(self, curie: ReferenceHint, n: int = 10) -> list[ExampleMapping]:
|
|
155
|
+
"""Get n mappings from a given set (by CURIE)."""
|
|
156
|
+
raise NotImplementedError
|
|
157
|
+
|
|
158
|
+
def get_example_mappings(self) -> list[ExampleMapping]:
|
|
159
|
+
"""Get example mappings."""
|
|
160
|
+
raise NotImplementedError
|
|
161
|
+
|
|
162
|
+
def get_full_summary(self) -> FullSummary:
|
|
163
|
+
"""Get a full summary."""
|
|
164
|
+
return FullSummary(
|
|
165
|
+
PREDICATE_COUNTER=self.summarize_predicates(),
|
|
166
|
+
MAPPING_SET_COUNTER=self.summarize_mapping_sets(),
|
|
167
|
+
NODE_COUNTER=self.summarize_nodes(),
|
|
168
|
+
JUSTIFICATION_COUNTER=self.summarize_justifications(),
|
|
169
|
+
EVIDENCE_TYPE_COUNTER=self.summarize_evidence_types(),
|
|
170
|
+
PREFIX_COUNTER=self.summarize_concepts(),
|
|
171
|
+
AUTHOR_COUNTER=self.summarize_authors(),
|
|
172
|
+
HIGH_MATCHES_COUNTER=self.get_highest_exact_matches(),
|
|
173
|
+
example_mappings=self.get_example_mappings(),
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
class Neo4jClient(BaseClient):
|
|
178
|
+
"""A client to Neo4j."""
|
|
60
179
|
|
|
61
180
|
def __init__(
|
|
62
181
|
self,
|
|
@@ -86,6 +205,7 @@ class Neo4jClient:
|
|
|
86
205
|
for reference in RELATIONS
|
|
87
206
|
if reference.curie in self._all_relations
|
|
88
207
|
)
|
|
208
|
+
self._summary = None
|
|
89
209
|
|
|
90
210
|
def __del__(self) -> None:
|
|
91
211
|
"""Ensure driver is shut down when client is destroyed."""
|
|
@@ -142,7 +262,7 @@ class Neo4jClient:
|
|
|
142
262
|
res = self.read_query(query, curie=curie)
|
|
143
263
|
return cast(Node, res[0][0])
|
|
144
264
|
|
|
145
|
-
def get_mapping(self, curie: ReferenceHint) -> semra.Mapping:
|
|
265
|
+
def get_mapping(self, curie: ReferenceHint) -> semra.Mapping | None:
|
|
146
266
|
"""Get a mapping.
|
|
147
267
|
|
|
148
268
|
:param curie: Either a Reference object, a string representing a curie with
|
|
@@ -197,7 +317,7 @@ class Neo4jClient:
|
|
|
197
317
|
records = self.read_query(query)
|
|
198
318
|
return [MappingSet.model_validate(record) for (record,) in records]
|
|
199
319
|
|
|
200
|
-
def get_mapping_set(self, curie: ReferenceHint) -> MappingSet:
|
|
320
|
+
def get_mapping_set(self, curie: ReferenceHint) -> MappingSet | None:
|
|
201
321
|
"""Get a mappings set.
|
|
202
322
|
|
|
203
323
|
:param curie: The CURIE for a mapping set, using ``semra.mappingset`` as a
|
|
@@ -210,7 +330,7 @@ class Neo4jClient:
|
|
|
210
330
|
node = self._get_node_by_curie(curie, "mappingset")
|
|
211
331
|
return MappingSet.model_validate(node)
|
|
212
332
|
|
|
213
|
-
def get_evidence(self, curie: ReferenceHint) -> Evidence:
|
|
333
|
+
def get_evidence(self, curie: ReferenceHint) -> Evidence | None:
|
|
214
334
|
"""Get an evidence.
|
|
215
335
|
|
|
216
336
|
:param curie: The CURIE for a mapping set, using ``semra.evidence`` as a prefix.
|
|
@@ -220,7 +340,7 @@ class Neo4jClient:
|
|
|
220
340
|
curie = _safe_curie(curie, SEMRA_EVIDENCE_PREFIX)
|
|
221
341
|
query = "MATCH (n:evidence {curie: $curie}) RETURN n"
|
|
222
342
|
res = self.read_query(query, curie=curie)
|
|
223
|
-
return SimpleEvidence.model_validate(res[0][0])
|
|
343
|
+
return SimpleEvidence.model_validate(res[0][0])
|
|
224
344
|
|
|
225
345
|
def summarize_predicates(self) -> t.Counter[str]:
|
|
226
346
|
"""Get a counter of predicates."""
|
|
@@ -292,7 +412,7 @@ as label, count UNION ALL
|
|
|
292
412
|
|
|
293
413
|
def get_exact_matches(
|
|
294
414
|
self, curie: ReferenceHint, *, max_distance: int | None = None
|
|
295
|
-
) -> dict[Reference, str]:
|
|
415
|
+
) -> dict[Reference, str] | None:
|
|
296
416
|
"""Get a mapping of references->name for all concepts equivalent to the given concept."""
|
|
297
417
|
if isinstance(curie, Reference):
|
|
298
418
|
curie = curie.curie
|
|
@@ -348,7 +468,7 @@ as label, count UNION ALL
|
|
|
348
468
|
relations = [r[0] for r in self.read_query(edge_query, curies=sorted(component_curies))]
|
|
349
469
|
return nodes, relations
|
|
350
470
|
|
|
351
|
-
def get_connected_component_graph(self, curie: ReferenceHint) -> nx.MultiDiGraph:
|
|
471
|
+
def get_connected_component_graph(self, curie: ReferenceHint) -> nx.MultiDiGraph | None:
|
|
352
472
|
"""Get a networkx MultiDiGraph representing the connected component of mappings around the given CURIE.
|
|
353
473
|
|
|
354
474
|
:param curie: A CURIE string or reference
|
|
@@ -364,8 +484,8 @@ as label, count UNION ALL
|
|
|
364
484
|
g.add_edge(
|
|
365
485
|
path.start_node["curie"],
|
|
366
486
|
path.end_node["curie"],
|
|
367
|
-
key=relationship.id,
|
|
368
|
-
type=relationship.type,
|
|
487
|
+
key=relationship.id, # this is the mapping's CURIE
|
|
488
|
+
type=relationship.type, # this is the predicate CURIE
|
|
369
489
|
)
|
|
370
490
|
return g
|
|
371
491
|
|
|
@@ -380,9 +500,7 @@ as label, count UNION ALL
|
|
|
380
500
|
else:
|
|
381
501
|
return cast(str, name)
|
|
382
502
|
|
|
383
|
-
def sample_mappings_from_set(
|
|
384
|
-
self, curie: ReferenceHint, n: int = 10
|
|
385
|
-
) -> list[tuple[str, str, str, str, str, str]]:
|
|
503
|
+
def sample_mappings_from_set(self, curie: ReferenceHint, n: int = 10) -> list[ExampleMapping]:
|
|
386
504
|
"""Get n mappings from a given set (by CURIE)."""
|
|
387
505
|
if isinstance(curie, Reference):
|
|
388
506
|
curie = curie.curie
|
|
@@ -397,7 +515,58 @@ as label, count UNION ALL
|
|
|
397
515
|
RETURN n.curie, n.predicate, s.curie, s.name, t.curie, t.name
|
|
398
516
|
LIMIT {n}
|
|
399
517
|
"""
|
|
400
|
-
return
|
|
518
|
+
return [ExampleMapping(*row) for row in self.read_query(query, curie=curie)]
|
|
519
|
+
|
|
520
|
+
def get_example_mappings(self) -> list[ExampleMapping]:
|
|
521
|
+
"""Get example mappings."""
|
|
522
|
+
return [ExampleMapping(*row) for row in self.read_query(EXAMPLE_MAPPINGS_QUERY)]
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
EXAMPLE_MAPPINGS_QUERY = dedent("""\
|
|
526
|
+
MATCH
|
|
527
|
+
(t:concept)<-[`owl:annotatedTarget`]-(n:mapping)-[`owl:annotatedSource`]->(s:concept)
|
|
528
|
+
WHERE n.predicate = 'skos:exactMatch'
|
|
529
|
+
RETURN n.curie, n.predicate, s.curie, s.name, t.curie, t.name
|
|
530
|
+
LIMIT 5
|
|
531
|
+
""")
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
class ExampleMapping(NamedTuple):
|
|
535
|
+
"""Example mapping."""
|
|
536
|
+
|
|
537
|
+
mapping_curie: str
|
|
538
|
+
predicate: str
|
|
539
|
+
subject_curie: str
|
|
540
|
+
subject_name: str
|
|
541
|
+
object_curie: str
|
|
542
|
+
object_name: str
|
|
543
|
+
|
|
544
|
+
@classmethod
|
|
545
|
+
def from_mapping(cls, mapping: semra.Mapping) -> Self:
|
|
546
|
+
"""Get from a mapping."""
|
|
547
|
+
return cls(
|
|
548
|
+
mapping.curie,
|
|
549
|
+
mapping.predicate.curie,
|
|
550
|
+
mapping.subject.curie,
|
|
551
|
+
mapping.subject.name or "",
|
|
552
|
+
mapping.object.curie,
|
|
553
|
+
mapping.object.name or "",
|
|
554
|
+
)
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
@dataclasses.dataclass
|
|
558
|
+
class FullSummary:
|
|
559
|
+
"""A full summary object."""
|
|
560
|
+
|
|
561
|
+
PREDICATE_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
|
|
562
|
+
MAPPING_SET_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
|
|
563
|
+
NODE_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
|
|
564
|
+
JUSTIFICATION_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
|
|
565
|
+
EVIDENCE_TYPE_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
|
|
566
|
+
PREFIX_COUNTER: t.Counter[tuple[str, str]] = dataclasses.field(default_factory=t.Counter)
|
|
567
|
+
AUTHOR_COUNTER: t.Counter[tuple[str, str]] = dataclasses.field(default_factory=t.Counter)
|
|
568
|
+
HIGH_MATCHES_COUNTER: t.Counter[tuple[str, str]] = dataclasses.field(default_factory=t.Counter)
|
|
569
|
+
example_mappings: list[ExampleMapping] = dataclasses.field(default_factory=list)
|
|
401
570
|
|
|
402
571
|
|
|
403
572
|
# Follows example here:
|
|
@@ -14,11 +14,10 @@ import pystow
|
|
|
14
14
|
import requests
|
|
15
15
|
from bioontologies.obograph import write_warned
|
|
16
16
|
from bioontologies.robot import write_getter_warnings
|
|
17
|
-
from curies.vocabulary import charlie
|
|
18
17
|
from pyobo.getters import NoBuildError
|
|
19
18
|
from tqdm.auto import tqdm
|
|
20
19
|
from tqdm.contrib.logging import logging_redirect_tqdm
|
|
21
|
-
from zenodo_client import
|
|
20
|
+
from zenodo_client import update_zenodo
|
|
22
21
|
|
|
23
22
|
from semra import Mapping
|
|
24
23
|
from semra.io import from_jsonl, from_pyobo, write_jsonl, write_neo4j, write_sssom
|
|
@@ -26,6 +25,7 @@ from semra.io.io_utils import safe_open_writer
|
|
|
26
25
|
from semra.pipeline import REFRESH_SOURCE_OPTION, UPLOAD_OPTION
|
|
27
26
|
from semra.sources import SOURCE_RESOLVER
|
|
28
27
|
from semra.sources.wikidata import get_wikidata_mappings_by_prefix
|
|
28
|
+
from semra.utils import gzip_path
|
|
29
29
|
|
|
30
30
|
MODULE = pystow.module("semra", "database")
|
|
31
31
|
SOURCES = MODULE.module("sources")
|
|
@@ -133,32 +133,25 @@ def build(
|
|
|
133
133
|
)
|
|
134
134
|
mappings = write_jsonl(mappings, JSONL_PATH, stream=True)
|
|
135
135
|
mappings = write_sssom(mappings, SSSOM_PATH, add_labels=False, prune=False, stream=True)
|
|
136
|
-
# neo4j doesn't need to stream since it's last
|
|
137
|
-
|
|
136
|
+
# neo4j doesn't need to stream since it's last. to avoid SIGKILLs,
|
|
137
|
+
# write the file to disk, then compress after.
|
|
138
|
+
write_neo4j(mappings, NEO4J_DIR, compress="after")
|
|
139
|
+
|
|
140
|
+
# gzip these after the fact to avoid SIGKILLs
|
|
141
|
+
jsonl_gz_path = gzip_path(JSONL_PATH)
|
|
142
|
+
sssom_gz_path = gzip_path(SSSOM_PATH)
|
|
138
143
|
|
|
139
144
|
if upload:
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
title="SeMRA Mapping Database",
|
|
143
|
-
upload_type="dataset",
|
|
144
|
-
description=f"A compendium of mappings extracted from {len(summaries)} database/ontologies. "
|
|
145
|
-
f"Note that primary mappings are marked with the license of their source (when available). "
|
|
146
|
-
f"Inferred mappings are distributed under the CC0 license.",
|
|
147
|
-
creators=[
|
|
148
|
-
Creator(name="Hoyt, Charles Tapley", orcid=charlie.identifier),
|
|
149
|
-
],
|
|
150
|
-
)
|
|
151
|
-
res = ensure_zenodo(
|
|
152
|
-
key="semra-database-test-1",
|
|
153
|
-
data=zenodo_metadata,
|
|
145
|
+
res = update_zenodo(
|
|
146
|
+
deposition_id="11082038",
|
|
154
147
|
paths=[
|
|
155
|
-
|
|
148
|
+
jsonl_gz_path,
|
|
149
|
+
sssom_gz_path,
|
|
156
150
|
WARNINGS_PATH,
|
|
157
151
|
ERRORS_PATH,
|
|
158
152
|
SUMMARY_PATH,
|
|
159
153
|
*NEO4J_DIR.iterdir(),
|
|
160
154
|
],
|
|
161
|
-
sandbox=True,
|
|
162
155
|
)
|
|
163
156
|
click.echo(res.json()["links"]["html"])
|
|
164
157
|
|
|
@@ -11,13 +11,9 @@ from collections.abc import Generator, Iterable
|
|
|
11
11
|
from pathlib import Path
|
|
12
12
|
from typing import Any, Literal, NamedTuple, TextIO, TypeVar, cast, overload
|
|
13
13
|
|
|
14
|
-
import bioontologies
|
|
15
14
|
import bioregistry
|
|
16
|
-
import bioversions
|
|
17
15
|
import pandas as pd
|
|
18
16
|
import pydantic
|
|
19
|
-
import pyobo
|
|
20
|
-
import pyobo.utils
|
|
21
17
|
import requests
|
|
22
18
|
import yaml
|
|
23
19
|
from tqdm.autonotebook import tqdm
|
|
@@ -55,14 +51,6 @@ DEFAULT_ONTOLOGY_CONFIDENCE = 0.9
|
|
|
55
51
|
X = TypeVar("X", bound=pydantic.BaseModel)
|
|
56
52
|
|
|
57
53
|
|
|
58
|
-
def _safe_get_version(prefix: str) -> str | None:
|
|
59
|
-
"""Get a version from Bioversions, or return None if not possible."""
|
|
60
|
-
try:
|
|
61
|
-
return bioversions.get_version(prefix)
|
|
62
|
-
except (KeyError, TypeError):
|
|
63
|
-
return None
|
|
64
|
-
|
|
65
|
-
|
|
66
54
|
# TODO delete this
|
|
67
55
|
def from_cache_df(
|
|
68
56
|
path: str | Path,
|
|
@@ -146,6 +134,8 @@ def from_pyobo(
|
|
|
146
134
|
|
|
147
135
|
:returns: A list of semantic mapping objects
|
|
148
136
|
"""
|
|
137
|
+
import pyobo
|
|
138
|
+
|
|
149
139
|
df: pd.DataFrame = pyobo.get_mappings_df(
|
|
150
140
|
prefix, force_process=force_process, names=False, cache=cache
|
|
151
141
|
)
|
|
@@ -172,6 +162,7 @@ def _from_pyobo_sssom_df(
|
|
|
172
162
|
license: str | None = None,
|
|
173
163
|
justification: Reference | None = None,
|
|
174
164
|
mapping_set_name: str | None = None,
|
|
165
|
+
mapping_set_title: str | None = None,
|
|
175
166
|
) -> list[Mapping]:
|
|
176
167
|
"""Get mappings from a :mod:`pyobo`-flavored cache file.
|
|
177
168
|
|
|
@@ -203,18 +194,22 @@ def _from_pyobo_sssom_df(
|
|
|
203
194
|
confidence = DEFAULT_ONTOLOGY_CONFIDENCE
|
|
204
195
|
if license is None:
|
|
205
196
|
license = bioregistry.get_license(prefix)
|
|
206
|
-
if mapping_set_name is None:
|
|
207
|
-
|
|
197
|
+
if mapping_set_name is not None:
|
|
198
|
+
if mapping_set_title:
|
|
199
|
+
raise ValueError
|
|
200
|
+
mapping_set_title = mapping_set_name
|
|
201
|
+
if mapping_set_title is None:
|
|
202
|
+
mapping_set_title = bioregistry.get_name(prefix)
|
|
208
203
|
if prefixes:
|
|
209
204
|
df = _filter_sssom_by_prefixes(df, prefixes)
|
|
210
205
|
return from_sssom_df(
|
|
211
206
|
df,
|
|
212
207
|
standardize=standardize,
|
|
213
208
|
license=license,
|
|
214
|
-
version=version,
|
|
215
209
|
justification=justification,
|
|
216
210
|
mapping_set_confidence=confidence,
|
|
217
|
-
|
|
211
|
+
mapping_set_title=mapping_set_title,
|
|
212
|
+
mapping_set_version=version,
|
|
218
213
|
)
|
|
219
214
|
|
|
220
215
|
|
|
@@ -234,6 +229,8 @@ def from_bioontologies(
|
|
|
234
229
|
prefix: str, confidence: float | None = None, **kwargs: Any
|
|
235
230
|
) -> list[Mapping]:
|
|
236
231
|
"""Get mappings from a given ontology via :mod:`bioontologies`."""
|
|
232
|
+
import bioontologies
|
|
233
|
+
|
|
237
234
|
if confidence is None:
|
|
238
235
|
confidence = DEFAULT_ONTOLOGY_CONFIDENCE
|
|
239
236
|
o = bioontologies.get_obograph_by_prefix(prefix, **kwargs)
|
|
@@ -734,12 +731,12 @@ def _write_sssom_stream(
|
|
|
734
731
|
mappings: Iterable[Mapping], file: str | Path | TextIO, *, stream: bool = False
|
|
735
732
|
) -> Generator[Mapping] | None:
|
|
736
733
|
fallback_mapping_set_id = _get_fallback_mapping_set_id()
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
734
|
+
it = tqdm(mappings, desc="Writing SSSOM", leave=False, unit="mapping", unit_scale=True)
|
|
735
|
+
if stream:
|
|
736
|
+
return _stream_write_sssom(file, it, fallback_mapping_set_id)
|
|
737
|
+
else:
|
|
738
|
+
with safe_open_writer(file) as writer:
|
|
739
|
+
writer.writerow(SSSOM_DEFAULT_COLUMNS)
|
|
743
740
|
for mapping in it:
|
|
744
741
|
for evidence in mapping.evidence:
|
|
745
742
|
writer.writerow(_get_sssom_row(mapping, evidence, fallback_mapping_set_id))
|
|
@@ -747,12 +744,14 @@ def _write_sssom_stream(
|
|
|
747
744
|
|
|
748
745
|
|
|
749
746
|
def _stream_write_sssom(
|
|
750
|
-
|
|
747
|
+
path: str | Path | TextIO, mappings: Iterable[Mapping], fallback_mapping_set_id: str
|
|
751
748
|
) -> Generator[Mapping]:
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
749
|
+
with safe_open_writer(path) as writer:
|
|
750
|
+
writer.writerow(SSSOM_DEFAULT_COLUMNS)
|
|
751
|
+
for mapping in mappings:
|
|
752
|
+
for evidence in mapping.evidence:
|
|
753
|
+
writer.writerow(_get_sssom_row(mapping, evidence, fallback_mapping_set_id))
|
|
754
|
+
yield mapping
|
|
756
755
|
|
|
757
756
|
|
|
758
757
|
def write_pickle(mappings: list[Mapping], path: str | Path) -> None:
|
|
@@ -811,19 +810,20 @@ def write_jsonl(
|
|
|
811
810
|
unit_scale=True,
|
|
812
811
|
disable=not show_progress,
|
|
813
812
|
)
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
813
|
+
if stream:
|
|
814
|
+
return _stream_write_jsonl(models, path)
|
|
815
|
+
else:
|
|
816
|
+
with safe_open(path, read=False) as file:
|
|
818
817
|
for model in models:
|
|
819
818
|
file.write(f"{model.model_dump_json(exclude_none=True)}\n")
|
|
820
|
-
|
|
819
|
+
return None
|
|
821
820
|
|
|
822
821
|
|
|
823
|
-
def _stream_write_jsonl(models: Iterable[X],
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
822
|
+
def _stream_write_jsonl(models: Iterable[X], path: str | Path) -> Generator[X]:
|
|
823
|
+
with safe_open(path, read=False) as file:
|
|
824
|
+
for model in models:
|
|
825
|
+
file.write(f"{model.model_dump_json(exclude_none=True)}\n")
|
|
826
|
+
yield model
|
|
827
827
|
|
|
828
828
|
|
|
829
829
|
# docstr-coverage:excused `overload`
|
|
@@ -841,25 +841,45 @@ def from_jsonl(
|
|
|
841
841
|
|
|
842
842
|
|
|
843
843
|
def from_jsonl(
|
|
844
|
-
path: str | Path,
|
|
844
|
+
path: str | Path,
|
|
845
|
+
*,
|
|
846
|
+
show_progress: bool = False,
|
|
847
|
+
stream: bool = False,
|
|
848
|
+
failure_action: Literal["raise", "skip"] = "skip",
|
|
845
849
|
) -> list[Mapping] | Generator[Mapping]:
|
|
846
850
|
"""Read a list of Mapping objects from a JSONL file."""
|
|
847
|
-
rv = _iter_read_jsonl(path, show_progress=show_progress)
|
|
851
|
+
rv = _iter_read_jsonl(path, show_progress=show_progress, failure_action=failure_action)
|
|
848
852
|
if stream:
|
|
849
853
|
return rv
|
|
850
854
|
else:
|
|
851
855
|
return list(rv)
|
|
852
856
|
|
|
853
857
|
|
|
854
|
-
def _iter_read_jsonl(
|
|
858
|
+
def _iter_read_jsonl(
|
|
859
|
+
path: str | Path,
|
|
860
|
+
*,
|
|
861
|
+
show_progress: bool = False,
|
|
862
|
+
failure_action: Literal["raise", "skip"] = "skip",
|
|
863
|
+
) -> Generator[Mapping]:
|
|
855
864
|
"""Stream mapping objects from a JSONL file."""
|
|
856
865
|
with safe_open(path, read=True) as file:
|
|
857
|
-
for line in
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
866
|
+
for i, line in enumerate(
|
|
867
|
+
tqdm(
|
|
868
|
+
file,
|
|
869
|
+
desc="Reading mappings",
|
|
870
|
+
leave=False,
|
|
871
|
+
unit="mapping",
|
|
872
|
+
unit_scale=True,
|
|
873
|
+
disable=not show_progress,
|
|
874
|
+
)
|
|
864
875
|
):
|
|
865
|
-
|
|
876
|
+
try:
|
|
877
|
+
yv = Mapping.model_validate_json(line.strip())
|
|
878
|
+
except pydantic.ValidationError:
|
|
879
|
+
if failure_action == "raise":
|
|
880
|
+
raise
|
|
881
|
+
else:
|
|
882
|
+
logger.debug("[line:%d] failed to parse JSON", i)
|
|
883
|
+
continue
|
|
884
|
+
else:
|
|
885
|
+
yield yv
|