semra 0.1.0.dev0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. {semra-0.1.0.dev0 → semra-0.1.2}/PKG-INFO +4 -3
  2. {semra-0.1.0.dev0 → semra-0.1.2}/pyproject.toml +5 -4
  3. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/client.py +186 -17
  4. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/database.py +13 -20
  5. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/io.py +67 -47
  6. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/neo4j_io.py +36 -22
  7. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/templates/Dockerfile +22 -19
  8. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/templates/run_on_startup.sh +1 -1
  9. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/templates/startup.sh +1 -1
  10. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/anatomy.py +1 -10
  11. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/cells.py +8 -15
  12. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/cli.py +3 -0
  13. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/complexes.py +1 -11
  14. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/diseases.py +1 -10
  15. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/genes.py +1 -10
  16. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/taxrank.py +1 -10
  17. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/utils.py +3 -2
  18. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/pipeline.py +223 -145
  19. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/clo.py +2 -1
  20. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/struct.py +2 -0
  21. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/concept.html +1 -1
  22. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/home.html +3 -3
  23. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/mapping.html +5 -5
  24. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/utils.html +1 -1
  25. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/utils.py +14 -0
  26. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/version.py +1 -1
  27. semra-0.1.2/src/semra/web/__init__.py +1 -0
  28. semra-0.1.2/src/semra/web/fastapi_components.py +101 -0
  29. semra-0.1.2/src/semra/web/flask_components.py +166 -0
  30. semra-0.1.2/src/semra/web/shared.py +45 -0
  31. semra-0.1.2/src/semra/wsgi.py +89 -0
  32. semra-0.1.0.dev0/src/semra/wsgi.py +0 -268
  33. {semra-0.1.0.dev0 → semra-0.1.2}/LICENSE +0 -0
  34. {semra-0.1.0.dev0 → semra-0.1.2}/README.md +0 -0
  35. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/.DS_Store +0 -0
  36. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/__init__.py +0 -0
  37. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/__main__.py +0 -0
  38. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/api.py +0 -0
  39. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/cli.py +0 -0
  40. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/gilda_utils.py +0 -0
  41. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/inference.py +0 -0
  42. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/__init__.py +0 -0
  43. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/graph.py +0 -0
  44. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/io/io_utils.py +0 -0
  45. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/__init__.py +0 -0
  46. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/landscape/__main__.py +0 -0
  47. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/py.typed +0 -0
  48. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/rules.py +0 -0
  49. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/__init__.py +0 -0
  50. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/__main__.py +0 -0
  51. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/biopragmatics.py +0 -0
  52. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/cbms2019.py +0 -0
  53. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/cli.py +0 -0
  54. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/compath.py +0 -0
  55. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/famplex.py +0 -0
  56. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/gilda.py +0 -0
  57. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/intact.py +0 -0
  58. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/ncit.py +0 -0
  59. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/omim.py +0 -0
  60. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/pubchem.py +0 -0
  61. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/sources/wikidata.py +0 -0
  62. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/base.html +0 -0
  63. {semra-0.1.0.dev0 → semra-0.1.2}/src/semra/templates/mapping_set.html +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: semra
3
- Version: 0.1.0.dev0
3
+ Version: 0.1.2
4
4
  Summary: Semantic mapping reasoner and assembler
5
5
  Keywords: snekpack,cookiecutter,sssom,semantic mappings,ontology merging,ontology mappings
6
6
  Author: Charles Tapley Hoyt
@@ -29,7 +29,7 @@ Requires-Dist: tqdm
29
29
  Requires-Dist: more-itertools
30
30
  Requires-Dist: networkx
31
31
  Requires-Dist: bioregistry>=0.12.7
32
- Requires-Dist: pyobo>=0.12.0
32
+ Requires-Dist: pyobo>=0.12.1
33
33
  Requires-Dist: bioontologies>=0.7.0
34
34
  Requires-Dist: typing-extensions
35
35
  Requires-Dist: zenodo-client
@@ -52,7 +52,8 @@ Requires-Dist: matplotlib ; extra == 'landscape-notebooks'
52
52
  Requires-Dist: pygraphviz ; extra == 'landscape-notebooks'
53
53
  Requires-Dist: pytest ; extra == 'tests'
54
54
  Requires-Dist: coverage[toml] ; extra == 'tests'
55
- Requires-Dist: sssom ; extra == 'tests'
55
+ Requires-Dist: sssom>=0.4.16 ; extra == 'tests'
56
+ Requires-Dist: httpx ; extra == 'tests'
56
57
  Requires-Dist: fastapi ; extra == 'web'
57
58
  Requires-Dist: uvicorn ; extra == 'web'
58
59
  Requires-Dist: flask ; extra == 'web'
@@ -4,7 +4,7 @@ build-backend = "uv_build"
4
4
 
5
5
  [project]
6
6
  name = "semra"
7
- version = "0.1.0-dev"
7
+ version = "0.1.2"
8
8
  description = "Semantic mapping reasoner and assembler"
9
9
  readme = "README.md"
10
10
  authors = [
@@ -58,7 +58,7 @@ dependencies = [
58
58
  "more_itertools",
59
59
  "networkx",
60
60
  "bioregistry>=0.12.7",
61
- "pyobo>=0.12.0",
61
+ "pyobo>=0.12.1",
62
62
  "bioontologies>=0.7.0",
63
63
  "typing_extensions",
64
64
  "zenodo_client",
@@ -73,7 +73,8 @@ dependencies = [
73
73
  tests = [
74
74
  "pytest",
75
75
  "coverage[toml]",
76
- "sssom",
76
+ "sssom>=0.4.16",
77
+ "httpx",
77
78
  ]
78
79
  docs = [
79
80
  "sphinx>=8",
@@ -225,7 +226,7 @@ known-first-party = [
225
226
  docstring-code-format = true
226
227
 
227
228
  [tool.bumpversion]
228
- current_version = "0.1.0-dev"
229
+ current_version = "0.1.2"
229
230
  parse = "(?P<major>\\d+)\\.(?P<minor>\\d+)\\.(?P<patch>\\d+)(?:-(?P<release>[0-9A-Za-z-]+(?:\\.[0-9A-Za-z-]+)*))?(?:\\+(?P<build>[0-9A-Za-z-]+(?:\\.[0-9A-Za-z-]+)*))?"
230
231
  serialize = [
231
232
  "{major}.{minor}.{patch}-{release}+{build}",
@@ -2,10 +2,12 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ import dataclasses
5
6
  import os
6
7
  import typing as t
7
8
  from collections import Counter
8
- from typing import Any, TypeAlias, cast
9
+ from textwrap import dedent
10
+ from typing import Any, NamedTuple, TypeAlias, cast
9
11
 
10
12
  import bioregistry
11
13
  import neo4j
@@ -13,6 +15,7 @@ import neo4j.graph
13
15
  import networkx as nx
14
16
  import pydantic
15
17
  from neo4j import ManagedTransaction, unit_of_work
18
+ from typing_extensions import Self
16
19
 
17
20
  import semra
18
21
  from semra import Evidence, MappingSet, Reference, SimpleEvidence
@@ -24,11 +27,12 @@ from semra.rules import (
24
27
  )
25
28
 
26
29
  __all__ = [
30
+ "BaseClient",
31
+ "FullSummary",
27
32
  "Neo4jClient",
28
33
  "Node",
29
34
  ]
30
35
 
31
-
32
36
  Node: TypeAlias = t.Mapping[str, Any]
33
37
 
34
38
  TxResult: TypeAlias = list[list[Any]] | None
@@ -53,10 +57,125 @@ RELATIONS_CYPHER = "CALL db.relationshipTypes() YIELD relationshipType RETURN re
53
57
  CONCEPT_NAME_CYPHER = "MATCH (n:concept) WHERE n.curie = $curie RETURN n.name LIMIT 1"
54
58
 
55
59
 
56
- class Neo4jClient:
57
- """A client to Neo4j."""
60
+ class BaseClient:
61
+ """An abstract class defining all of the functionality for a mapping client."""
62
+
63
+ def get_mapping(self, curie: ReferenceHint) -> semra.Mapping | None:
64
+ """Get a mapping.
65
+
66
+ :param curie: Either a Reference object, a string representing a curie with
67
+ ``semra.mapping`` as the prefix, or a local unique identifier representing a
68
+ SeMRA mapping.
69
+
70
+ :returns: A semantic mapping object
71
+ """
72
+ raise NotImplementedError
73
+
74
+ def get_mapping_sets(self) -> list[MappingSet]:
75
+ """Get all mappings sets."""
76
+ raise NotImplementedError
58
77
 
59
- _session: neo4j.Session | None = None
78
+ def get_mapping_set(self, curie: ReferenceHint) -> MappingSet | None:
79
+ """Get a mappings set.
80
+
81
+ :param curie: The CURIE for a mapping set, using ``semra.mappingset`` as a
82
+ prefix. For example, use
83
+ ``semra.mappingset:7831d5bc95698099fb6471667e5282cd`` for biomappings
84
+
85
+ :returns: A mapping set object
86
+ """
87
+ raise NotImplementedError
88
+
89
+ def get_evidence(self, curie: ReferenceHint) -> Evidence | None:
90
+ """Get an evidence.
91
+
92
+ :param curie: The CURIE for a mapping set, using ``semra.evidence`` as a prefix.
93
+
94
+ :returns: An evidence object
95
+ """
96
+ raise NotImplementedError
97
+
98
+ def summarize_predicates(self) -> t.Counter[str]:
99
+ """Get a counter of predicates."""
100
+ raise NotImplementedError
101
+
102
+ def summarize_justifications(self) -> t.Counter[str]:
103
+ """Get a counter of mapping justifications."""
104
+ raise NotImplementedError
105
+
106
+ def summarize_evidence_types(self) -> t.Counter[str]:
107
+ """Get a counter of evidence types."""
108
+ raise NotImplementedError
109
+
110
+ def summarize_mapping_sets(self) -> t.Counter[str]:
111
+ """Get the number of evidences in each mapping set."""
112
+ raise NotImplementedError
113
+
114
+ def summarize_nodes(self) -> t.Counter[str]:
115
+ """Get a counter of node types (concepts, evidences, mappings, mapping sets)."""
116
+ raise NotImplementedError
117
+
118
+ def summarize_concepts(self) -> t.Counter[tuple[str, str]]:
119
+ """Get a counter of prefixes in concept nodes."""
120
+ raise NotImplementedError
121
+
122
+ def summarize_authors(self) -> t.Counter[tuple[str, str]]:
123
+ """Get a counter of the number of evidences each author has contributed to."""
124
+ raise NotImplementedError
125
+
126
+ def get_highest_exact_matches(self, limit: int = 10) -> t.Counter[tuple[str, str]]:
127
+ """Get a counter of concepts with the highest exact matches.
128
+
129
+ :param limit: The number of top concepts to return
130
+
131
+ :returns: A counter with keys that are CURIE/name pairs
132
+ """
133
+ raise NotImplementedError
134
+
135
+ def get_exact_matches(
136
+ self, curie: ReferenceHint, *, max_distance: int | None = None
137
+ ) -> dict[Reference, str] | None:
138
+ """Get a mapping of references->name for all concepts equivalent to the given concept."""
139
+ raise NotImplementedError
140
+
141
+ def get_connected_component_graph(self, curie: ReferenceHint) -> nx.MultiDiGraph | None:
142
+ """Get a networkx MultiDiGraph representing the connected component of mappings around the given CURIE.
143
+
144
+ :param curie: A CURIE string or reference
145
+
146
+ :returns: A networkx MultiDiGraph where mappings subject CURIE strings are th
147
+ """
148
+ raise NotImplementedError
149
+
150
+ def get_concept_name(self, curie: ReferenceHint) -> str | None:
151
+ """Get the name for a CURIE or reference."""
152
+ raise NotImplementedError
153
+
154
+ def sample_mappings_from_set(self, curie: ReferenceHint, n: int = 10) -> list[ExampleMapping]:
155
+ """Get n mappings from a given set (by CURIE)."""
156
+ raise NotImplementedError
157
+
158
+ def get_example_mappings(self) -> list[ExampleMapping]:
159
+ """Get example mappings."""
160
+ raise NotImplementedError
161
+
162
+ def get_full_summary(self) -> FullSummary:
163
+ """Get a full summary."""
164
+ return FullSummary(
165
+ PREDICATE_COUNTER=self.summarize_predicates(),
166
+ MAPPING_SET_COUNTER=self.summarize_mapping_sets(),
167
+ NODE_COUNTER=self.summarize_nodes(),
168
+ JUSTIFICATION_COUNTER=self.summarize_justifications(),
169
+ EVIDENCE_TYPE_COUNTER=self.summarize_evidence_types(),
170
+ PREFIX_COUNTER=self.summarize_concepts(),
171
+ AUTHOR_COUNTER=self.summarize_authors(),
172
+ HIGH_MATCHES_COUNTER=self.get_highest_exact_matches(),
173
+ example_mappings=self.get_example_mappings(),
174
+ )
175
+
176
+
177
+ class Neo4jClient(BaseClient):
178
+ """A client to Neo4j."""
60
179
 
61
180
  def __init__(
62
181
  self,
@@ -86,6 +205,7 @@ class Neo4jClient:
86
205
  for reference in RELATIONS
87
206
  if reference.curie in self._all_relations
88
207
  )
208
+ self._summary = None
89
209
 
90
210
  def __del__(self) -> None:
91
211
  """Ensure driver is shut down when client is destroyed."""
@@ -142,7 +262,7 @@ class Neo4jClient:
142
262
  res = self.read_query(query, curie=curie)
143
263
  return cast(Node, res[0][0])
144
264
 
145
- def get_mapping(self, curie: ReferenceHint) -> semra.Mapping:
265
+ def get_mapping(self, curie: ReferenceHint) -> semra.Mapping | None:
146
266
  """Get a mapping.
147
267
 
148
268
  :param curie: Either a Reference object, a string representing a curie with
@@ -197,7 +317,7 @@ class Neo4jClient:
197
317
  records = self.read_query(query)
198
318
  return [MappingSet.model_validate(record) for (record,) in records]
199
319
 
200
- def get_mapping_set(self, curie: ReferenceHint) -> MappingSet:
320
+ def get_mapping_set(self, curie: ReferenceHint) -> MappingSet | None:
201
321
  """Get a mappings set.
202
322
 
203
323
  :param curie: The CURIE for a mapping set, using ``semra.mappingset`` as a
@@ -210,7 +330,7 @@ class Neo4jClient:
210
330
  node = self._get_node_by_curie(curie, "mappingset")
211
331
  return MappingSet.model_validate(node)
212
332
 
213
- def get_evidence(self, curie: ReferenceHint) -> Evidence:
333
+ def get_evidence(self, curie: ReferenceHint) -> Evidence | None:
214
334
  """Get an evidence.
215
335
 
216
336
  :param curie: The CURIE for a mapping set, using ``semra.evidence`` as a prefix.
@@ -220,7 +340,7 @@ class Neo4jClient:
220
340
  curie = _safe_curie(curie, SEMRA_EVIDENCE_PREFIX)
221
341
  query = "MATCH (n:evidence {curie: $curie}) RETURN n"
222
342
  res = self.read_query(query, curie=curie)
223
- return SimpleEvidence.model_validate(res[0][0]) # FIXME test this?
343
+ return SimpleEvidence.model_validate(res[0][0])
224
344
 
225
345
  def summarize_predicates(self) -> t.Counter[str]:
226
346
  """Get a counter of predicates."""
@@ -292,7 +412,7 @@ as label, count UNION ALL
292
412
 
293
413
  def get_exact_matches(
294
414
  self, curie: ReferenceHint, *, max_distance: int | None = None
295
- ) -> dict[Reference, str]:
415
+ ) -> dict[Reference, str] | None:
296
416
  """Get a mapping of references->name for all concepts equivalent to the given concept."""
297
417
  if isinstance(curie, Reference):
298
418
  curie = curie.curie
@@ -348,7 +468,7 @@ as label, count UNION ALL
348
468
  relations = [r[0] for r in self.read_query(edge_query, curies=sorted(component_curies))]
349
469
  return nodes, relations
350
470
 
351
- def get_connected_component_graph(self, curie: ReferenceHint) -> nx.MultiDiGraph:
471
+ def get_connected_component_graph(self, curie: ReferenceHint) -> nx.MultiDiGraph | None:
352
472
  """Get a networkx MultiDiGraph representing the connected component of mappings around the given CURIE.
353
473
 
354
474
  :param curie: A CURIE string or reference
@@ -364,8 +484,8 @@ as label, count UNION ALL
364
484
  g.add_edge(
365
485
  path.start_node["curie"],
366
486
  path.end_node["curie"],
367
- key=relationship.id,
368
- type=relationship.type,
487
+ key=relationship.id, # this is the mapping's CURIE
488
+ type=relationship.type, # this is the predicate CURIE
369
489
  )
370
490
  return g
371
491
 
@@ -380,9 +500,7 @@ as label, count UNION ALL
380
500
  else:
381
501
  return cast(str, name)
382
502
 
383
- def sample_mappings_from_set(
384
- self, curie: ReferenceHint, n: int = 10
385
- ) -> list[tuple[str, str, str, str, str, str]]:
503
+ def sample_mappings_from_set(self, curie: ReferenceHint, n: int = 10) -> list[ExampleMapping]:
386
504
  """Get n mappings from a given set (by CURIE)."""
387
505
  if isinstance(curie, Reference):
388
506
  curie = curie.curie
@@ -397,7 +515,58 @@ as label, count UNION ALL
397
515
  RETURN n.curie, n.predicate, s.curie, s.name, t.curie, t.name
398
516
  LIMIT {n}
399
517
  """
400
- return list(self.read_query(query, curie=curie)) # type:ignore
518
+ return [ExampleMapping(*row) for row in self.read_query(query, curie=curie)]
519
+
520
+ def get_example_mappings(self) -> list[ExampleMapping]:
521
+ """Get example mappings."""
522
+ return [ExampleMapping(*row) for row in self.read_query(EXAMPLE_MAPPINGS_QUERY)]
523
+
524
+
525
+ EXAMPLE_MAPPINGS_QUERY = dedent("""\
526
+ MATCH
527
+ (t:concept)<-[`owl:annotatedTarget`]-(n:mapping)-[`owl:annotatedSource`]->(s:concept)
528
+ WHERE n.predicate = 'skos:exactMatch'
529
+ RETURN n.curie, n.predicate, s.curie, s.name, t.curie, t.name
530
+ LIMIT 5
531
+ """)
532
+
533
+
534
+ class ExampleMapping(NamedTuple):
535
+ """Example mapping."""
536
+
537
+ mapping_curie: str
538
+ predicate: str
539
+ subject_curie: str
540
+ subject_name: str
541
+ object_curie: str
542
+ object_name: str
543
+
544
+ @classmethod
545
+ def from_mapping(cls, mapping: semra.Mapping) -> Self:
546
+ """Get from a mapping."""
547
+ return cls(
548
+ mapping.curie,
549
+ mapping.predicate.curie,
550
+ mapping.subject.curie,
551
+ mapping.subject.name or "",
552
+ mapping.object.curie,
553
+ mapping.object.name or "",
554
+ )
555
+
556
+
557
+ @dataclasses.dataclass
558
+ class FullSummary:
559
+ """A full summary object."""
560
+
561
+ PREDICATE_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
562
+ MAPPING_SET_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
563
+ NODE_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
564
+ JUSTIFICATION_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
565
+ EVIDENCE_TYPE_COUNTER: t.Counter[str] = dataclasses.field(default_factory=t.Counter)
566
+ PREFIX_COUNTER: t.Counter[tuple[str, str]] = dataclasses.field(default_factory=t.Counter)
567
+ AUTHOR_COUNTER: t.Counter[tuple[str, str]] = dataclasses.field(default_factory=t.Counter)
568
+ HIGH_MATCHES_COUNTER: t.Counter[tuple[str, str]] = dataclasses.field(default_factory=t.Counter)
569
+ example_mappings: list[ExampleMapping] = dataclasses.field(default_factory=list)
401
570
 
402
571
 
403
572
  # Follows example here:
@@ -14,11 +14,10 @@ import pystow
14
14
  import requests
15
15
  from bioontologies.obograph import write_warned
16
16
  from bioontologies.robot import write_getter_warnings
17
- from curies.vocabulary import charlie
18
17
  from pyobo.getters import NoBuildError
19
18
  from tqdm.auto import tqdm
20
19
  from tqdm.contrib.logging import logging_redirect_tqdm
21
- from zenodo_client import Creator, Metadata, ensure_zenodo
20
+ from zenodo_client import update_zenodo
22
21
 
23
22
  from semra import Mapping
24
23
  from semra.io import from_jsonl, from_pyobo, write_jsonl, write_neo4j, write_sssom
@@ -26,6 +25,7 @@ from semra.io.io_utils import safe_open_writer
26
25
  from semra.pipeline import REFRESH_SOURCE_OPTION, UPLOAD_OPTION
27
26
  from semra.sources import SOURCE_RESOLVER
28
27
  from semra.sources.wikidata import get_wikidata_mappings_by_prefix
28
+ from semra.utils import gzip_path
29
29
 
30
30
  MODULE = pystow.module("semra", "database")
31
31
  SOURCES = MODULE.module("sources")
@@ -133,32 +133,25 @@ def build(
133
133
  )
134
134
  mappings = write_jsonl(mappings, JSONL_PATH, stream=True)
135
135
  mappings = write_sssom(mappings, SSSOM_PATH, add_labels=False, prune=False, stream=True)
136
- # neo4j doesn't need to stream since it's last
137
- write_neo4j(mappings, NEO4J_DIR)
136
+ # neo4j doesn't need to stream since it's last. to avoid SIGKILLs,
137
+ # write the file to disk, then compress after.
138
+ write_neo4j(mappings, NEO4J_DIR, compress="after")
139
+
140
+ # gzip these after the fact to avoid SIGKILLs
141
+ jsonl_gz_path = gzip_path(JSONL_PATH)
142
+ sssom_gz_path = gzip_path(SSSOM_PATH)
138
143
 
139
144
  if upload:
140
- # Define the metadata that will be used on initial upload
141
- zenodo_metadata = Metadata(
142
- title="SeMRA Mapping Database",
143
- upload_type="dataset",
144
- description=f"A compendium of mappings extracted from {len(summaries)} database/ontologies. "
145
- f"Note that primary mappings are marked with the license of their source (when available). "
146
- f"Inferred mappings are distributed under the CC0 license.",
147
- creators=[
148
- Creator(name="Hoyt, Charles Tapley", orcid=charlie.identifier),
149
- ],
150
- )
151
- res = ensure_zenodo(
152
- key="semra-database-test-1",
153
- data=zenodo_metadata,
145
+ res = update_zenodo(
146
+ deposition_id="11082038",
154
147
  paths=[
155
- SSSOM_PATH,
148
+ jsonl_gz_path,
149
+ sssom_gz_path,
156
150
  WARNINGS_PATH,
157
151
  ERRORS_PATH,
158
152
  SUMMARY_PATH,
159
153
  *NEO4J_DIR.iterdir(),
160
154
  ],
161
- sandbox=True,
162
155
  )
163
156
  click.echo(res.json()["links"]["html"])
164
157
 
@@ -11,13 +11,9 @@ from collections.abc import Generator, Iterable
11
11
  from pathlib import Path
12
12
  from typing import Any, Literal, NamedTuple, TextIO, TypeVar, cast, overload
13
13
 
14
- import bioontologies
15
14
  import bioregistry
16
- import bioversions
17
15
  import pandas as pd
18
16
  import pydantic
19
- import pyobo
20
- import pyobo.utils
21
17
  import requests
22
18
  import yaml
23
19
  from tqdm.autonotebook import tqdm
@@ -55,14 +51,6 @@ DEFAULT_ONTOLOGY_CONFIDENCE = 0.9
55
51
  X = TypeVar("X", bound=pydantic.BaseModel)
56
52
 
57
53
 
58
- def _safe_get_version(prefix: str) -> str | None:
59
- """Get a version from Bioversions, or return None if not possible."""
60
- try:
61
- return bioversions.get_version(prefix)
62
- except (KeyError, TypeError):
63
- return None
64
-
65
-
66
54
  # TODO delete this
67
55
  def from_cache_df(
68
56
  path: str | Path,
@@ -146,6 +134,8 @@ def from_pyobo(
146
134
 
147
135
  :returns: A list of semantic mapping objects
148
136
  """
137
+ import pyobo
138
+
149
139
  df: pd.DataFrame = pyobo.get_mappings_df(
150
140
  prefix, force_process=force_process, names=False, cache=cache
151
141
  )
@@ -172,6 +162,7 @@ def _from_pyobo_sssom_df(
172
162
  license: str | None = None,
173
163
  justification: Reference | None = None,
174
164
  mapping_set_name: str | None = None,
165
+ mapping_set_title: str | None = None,
175
166
  ) -> list[Mapping]:
176
167
  """Get mappings from a :mod:`pyobo`-flavored cache file.
177
168
 
@@ -203,18 +194,22 @@ def _from_pyobo_sssom_df(
203
194
  confidence = DEFAULT_ONTOLOGY_CONFIDENCE
204
195
  if license is None:
205
196
  license = bioregistry.get_license(prefix)
206
- if mapping_set_name is None:
207
- mapping_set_name = bioregistry.get_name(prefix)
197
+ if mapping_set_name is not None:
198
+ if mapping_set_title:
199
+ raise ValueError
200
+ mapping_set_title = mapping_set_name
201
+ if mapping_set_title is None:
202
+ mapping_set_title = bioregistry.get_name(prefix)
208
203
  if prefixes:
209
204
  df = _filter_sssom_by_prefixes(df, prefixes)
210
205
  return from_sssom_df(
211
206
  df,
212
207
  standardize=standardize,
213
208
  license=license,
214
- version=version,
215
209
  justification=justification,
216
210
  mapping_set_confidence=confidence,
217
- mapping_set_name=mapping_set_name, # TODO rename to mapping_set_title align with SSSOM
211
+ mapping_set_title=mapping_set_title,
212
+ mapping_set_version=version,
218
213
  )
219
214
 
220
215
 
@@ -234,6 +229,8 @@ def from_bioontologies(
234
229
  prefix: str, confidence: float | None = None, **kwargs: Any
235
230
  ) -> list[Mapping]:
236
231
  """Get mappings from a given ontology via :mod:`bioontologies`."""
232
+ import bioontologies
233
+
237
234
  if confidence is None:
238
235
  confidence = DEFAULT_ONTOLOGY_CONFIDENCE
239
236
  o = bioontologies.get_obograph_by_prefix(prefix, **kwargs)
@@ -734,12 +731,12 @@ def _write_sssom_stream(
734
731
  mappings: Iterable[Mapping], file: str | Path | TextIO, *, stream: bool = False
735
732
  ) -> Generator[Mapping] | None:
736
733
  fallback_mapping_set_id = _get_fallback_mapping_set_id()
737
- with safe_open_writer(file) as writer:
738
- writer.writerow(SSSOM_DEFAULT_COLUMNS)
739
- it = tqdm(mappings, desc="Writing SSSOM", leave=False, unit="mapping", unit_scale=True)
740
- if stream:
741
- return _stream_write_sssom(writer, it, fallback_mapping_set_id)
742
- else:
734
+ it = tqdm(mappings, desc="Writing SSSOM", leave=False, unit="mapping", unit_scale=True)
735
+ if stream:
736
+ return _stream_write_sssom(file, it, fallback_mapping_set_id)
737
+ else:
738
+ with safe_open_writer(file) as writer:
739
+ writer.writerow(SSSOM_DEFAULT_COLUMNS)
743
740
  for mapping in it:
744
741
  for evidence in mapping.evidence:
745
742
  writer.writerow(_get_sssom_row(mapping, evidence, fallback_mapping_set_id))
@@ -747,12 +744,14 @@ def _write_sssom_stream(
747
744
 
748
745
 
749
746
  def _stream_write_sssom(
750
- writer: Any, mappings: Iterable[Mapping], fallback_mapping_set_id: str
747
+ path: str | Path | TextIO, mappings: Iterable[Mapping], fallback_mapping_set_id: str
751
748
  ) -> Generator[Mapping]:
752
- for mapping in mappings:
753
- for evidence in mapping.evidence:
754
- writer.writerow(_get_sssom_row(mapping, evidence, fallback_mapping_set_id))
755
- yield mapping
749
+ with safe_open_writer(path) as writer:
750
+ writer.writerow(SSSOM_DEFAULT_COLUMNS)
751
+ for mapping in mappings:
752
+ for evidence in mapping.evidence:
753
+ writer.writerow(_get_sssom_row(mapping, evidence, fallback_mapping_set_id))
754
+ yield mapping
756
755
 
757
756
 
758
757
  def write_pickle(mappings: list[Mapping], path: str | Path) -> None:
@@ -811,19 +810,20 @@ def write_jsonl(
811
810
  unit_scale=True,
812
811
  disable=not show_progress,
813
812
  )
814
- with safe_open(path, read=False) as file:
815
- if stream:
816
- return _stream_write_jsonl(models, file)
817
- else:
813
+ if stream:
814
+ return _stream_write_jsonl(models, path)
815
+ else:
816
+ with safe_open(path, read=False) as file:
818
817
  for model in models:
819
818
  file.write(f"{model.model_dump_json(exclude_none=True)}\n")
820
- return None
819
+ return None
821
820
 
822
821
 
823
- def _stream_write_jsonl(models: Iterable[X], file: TextIO) -> Generator[X]:
824
- for model in models:
825
- file.write(f"{model.model_dump_json(exclude_none=True)}\n")
826
- yield model
822
+ def _stream_write_jsonl(models: Iterable[X], path: str | Path) -> Generator[X]:
823
+ with safe_open(path, read=False) as file:
824
+ for model in models:
825
+ file.write(f"{model.model_dump_json(exclude_none=True)}\n")
826
+ yield model
827
827
 
828
828
 
829
829
  # docstr-coverage:excused `overload`
@@ -841,25 +841,45 @@ def from_jsonl(
841
841
 
842
842
 
843
843
  def from_jsonl(
844
- path: str | Path, *, show_progress: bool = False, stream: bool = False
844
+ path: str | Path,
845
+ *,
846
+ show_progress: bool = False,
847
+ stream: bool = False,
848
+ failure_action: Literal["raise", "skip"] = "skip",
845
849
  ) -> list[Mapping] | Generator[Mapping]:
846
850
  """Read a list of Mapping objects from a JSONL file."""
847
- rv = _iter_read_jsonl(path, show_progress=show_progress)
851
+ rv = _iter_read_jsonl(path, show_progress=show_progress, failure_action=failure_action)
848
852
  if stream:
849
853
  return rv
850
854
  else:
851
855
  return list(rv)
852
856
 
853
857
 
854
- def _iter_read_jsonl(path: str | Path, *, show_progress: bool = False) -> Generator[Mapping]:
858
+ def _iter_read_jsonl(
859
+ path: str | Path,
860
+ *,
861
+ show_progress: bool = False,
862
+ failure_action: Literal["raise", "skip"] = "skip",
863
+ ) -> Generator[Mapping]:
855
864
  """Stream mapping objects from a JSONL file."""
856
865
  with safe_open(path, read=True) as file:
857
- for line in tqdm(
858
- file,
859
- desc="Reading mappings",
860
- leave=False,
861
- unit="mapping",
862
- unit_scale=True,
863
- disable=not show_progress,
866
+ for i, line in enumerate(
867
+ tqdm(
868
+ file,
869
+ desc="Reading mappings",
870
+ leave=False,
871
+ unit="mapping",
872
+ unit_scale=True,
873
+ disable=not show_progress,
874
+ )
864
875
  ):
865
- yield Mapping.model_validate_json(line.strip())
876
+ try:
877
+ yv = Mapping.model_validate_json(line.strip())
878
+ except pydantic.ValidationError:
879
+ if failure_action == "raise":
880
+ raise
881
+ else:
882
+ logger.debug("[line:%d] failed to parse JSON", i)
883
+ continue
884
+ else:
885
+ yield yv