biomapper 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {biomapper-0.3.0 → biomapper-0.3.2}/PKG-INFO +42 -32
- {biomapper-0.3.0 → biomapper-0.3.2}/README.md +33 -27
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/__init__.py +1 -1
- biomapper-0.3.2/biomapper/mapping/llm_mapper.py +130 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/metabolite_name_mapper.py +3 -1
- biomapper-0.3.2/biomapper/mapping/multi_provider_rag.py +341 -0
- biomapper-0.3.2/biomapper/mapping/rag_mapper.py +207 -0
- biomapper-0.3.2/biomapper/mapping/unichem_client.py +438 -0
- biomapper-0.3.2/biomapper/py.typed +1 -0
- biomapper-0.3.2/biomapper/schemas/llm_schema.py +51 -0
- biomapper-0.3.2/biomapper/schemas/provider_schemas.py +74 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/pyproject.toml +12 -7
- biomapper-0.3.0/biomapper/mapping/unichem_client.py +0 -214
- biomapper-0.3.0/biomapper/standardization/tutorial.ipynb +0 -321
- {biomapper-0.3.0 → biomapper-0.3.2}/LICENSE +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/core/__init__.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/core/protein_metadata_comparison.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/core/set_analysis.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/__init__.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/chebi_client.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/refmet_client.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/uniprot_focused_mapper.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/schemas/__init__.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/standardization/__init__.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/standardization/ramp_client.py +0 -0
- {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/utils/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: biomapper
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: A unified Python toolkit for biological data harmonization and ontology mapping
|
|
5
5
|
Home-page: https://github.com/arpanauts/biomapper
|
|
6
6
|
License: MIT
|
|
@@ -20,16 +20,20 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
20
20
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
21
21
|
Provides-Extra: api
|
|
22
22
|
Provides-Extra: full
|
|
23
|
-
|
|
23
|
+
Requires-Dist: cloudpickle (>=3.0.0,<4.0.0)
|
|
24
|
+
Requires-Dist: dspy-ai (>=2.1.8,<3.0.0)
|
|
25
|
+
Requires-Dist: langfuse (>=2.57.1,<3.0.0)
|
|
24
26
|
Requires-Dist: libChEBIpy (==1.0.10)
|
|
25
|
-
Requires-Dist: matplotlib (>=3.8.0,<4.0.0)
|
|
27
|
+
Requires-Dist: matplotlib (>=3.8.0,<4.0.0)
|
|
28
|
+
Requires-Dist: openai (>=1.14.0,<2.0.0)
|
|
26
29
|
Requires-Dist: pandas (>=2.0.0,<3.0.0)
|
|
30
|
+
Requires-Dist: python-dotenv (>=1.0.1,<2.0.0)
|
|
27
31
|
Requires-Dist: pyyaml (>=6.0.1,<7.0.0)
|
|
28
32
|
Requires-Dist: requests (>=2.25.1,<3.0.0)
|
|
29
|
-
Requires-Dist: seaborn (>=0.13.0,<0.14.0)
|
|
33
|
+
Requires-Dist: seaborn (>=0.13.0,<0.14.0)
|
|
30
34
|
Requires-Dist: sqlalchemy (>=1.4.0,<2.0.0)
|
|
31
35
|
Requires-Dist: tqdm (>=4.66.1,<5.0.0)
|
|
32
|
-
Requires-Dist: upsetplot (>=0.8.0,<0.9.0)
|
|
36
|
+
Requires-Dist: upsetplot (>=0.8.0,<0.9.0)
|
|
33
37
|
Requires-Dist: venn (>=0.1.3,<0.2.0)
|
|
34
38
|
Project-URL: Documentation, https://github.com/arpanauts/biomapper/blob/main/README.md
|
|
35
39
|
Project-URL: Repository, https://github.com/arpanauts/biomapper
|
|
@@ -43,21 +47,22 @@ A unified Python toolkit for biological data harmonization and ontology mapping.
|
|
|
43
47
|
|
|
44
48
|
### Core Functionality
|
|
45
49
|
- **ID Standardization**: Unified interface for standardizing biological identifiers
|
|
46
|
-
- **Ontology Mapping**: Comprehensive ontology mapping using major biological databases
|
|
50
|
+
- **Ontology Mapping**: Comprehensive ontology mapping using major biological databases and AI-powered techniques
|
|
47
51
|
- **Data Validation**: Robust validation of input data and mappings
|
|
48
52
|
- **Extensible Architecture**: Easy integration of new data sources and mapping services
|
|
49
53
|
|
|
50
54
|
### Supported Systems
|
|
51
55
|
|
|
52
56
|
#### ID Standardization Tools
|
|
53
|
-
-
|
|
54
|
-
- RefMet
|
|
55
|
-
- RaMP-DB
|
|
57
|
+
- RaMP-DB: Integration with the Rapid Mapping Database for metabolites and pathways
|
|
56
58
|
|
|
57
|
-
####
|
|
58
|
-
-
|
|
59
|
-
-
|
|
60
|
-
-
|
|
59
|
+
#### Mapping Services
|
|
60
|
+
- ChEBI: Chemical Entities of Biological Interest database integration
|
|
61
|
+
- UniChem: Cross-referencing of chemical structure identifiers
|
|
62
|
+
- UniProt: Protein-focused mapping capabilities
|
|
63
|
+
- RefMet: Reference list of metabolite names and identifiers
|
|
64
|
+
- RAG-Based Mapping: AI-powered mapping using Retrieval Augmented Generation
|
|
65
|
+
- Multi-Provider RAG: Combining multiple data sources for improved mapping accuracy
|
|
61
66
|
|
|
62
67
|
## Installation
|
|
63
68
|
|
|
@@ -113,20 +118,18 @@ poetry install
|
|
|
113
118
|
## Quick Start
|
|
114
119
|
|
|
115
120
|
```python
|
|
116
|
-
from biomapper import
|
|
117
|
-
from biomapper.standardization import
|
|
121
|
+
from biomapper.mapping import UniProtFocusedMapper, MetaboliteNameMapper
|
|
122
|
+
from biomapper.standardization import RaMPClient
|
|
118
123
|
|
|
119
|
-
# Example 1: Using
|
|
120
|
-
|
|
121
|
-
|
|
124
|
+
# Example 1: Using UniProt-focused mapping
|
|
125
|
+
uniprot_mapper = UniProtFocusedMapper()
|
|
126
|
+
protein_mapping = uniprot_mapper.map_identifier("P12345")
|
|
122
127
|
|
|
123
|
-
#
|
|
124
|
-
|
|
128
|
+
# Example 2: Using Metabolite Name Mapping
|
|
129
|
+
metabolite_mapper = MetaboliteNameMapper()
|
|
130
|
+
metabolite_mapping = metabolite_mapper.map_name("glucose")
|
|
125
131
|
|
|
126
|
-
#
|
|
127
|
-
results = bridge_handler.standardize(["P12345", "Q67890"])
|
|
128
|
-
|
|
129
|
-
# Example 2: Using RaMP-DB
|
|
132
|
+
# Example 3: Using RaMP-DB
|
|
130
133
|
# Initialize the RaMP client
|
|
131
134
|
ramp_client = RaMPClient()
|
|
132
135
|
|
|
@@ -136,6 +139,12 @@ versions = ramp_client.get_source_versions()
|
|
|
136
139
|
# Get pathways for metabolites
|
|
137
140
|
# Example: Get pathways for Creatine (HMDB0000064)
|
|
138
141
|
pathways = ramp_client.get_pathways_from_analytes(["hmdb:HMDB0000064"])
|
|
142
|
+
|
|
143
|
+
# Example 4: Using RAG-based mapping
|
|
144
|
+
from biomapper.mapping import RagMapper
|
|
145
|
+
|
|
146
|
+
rag_mapper = RagMapper()
|
|
147
|
+
rag_results = rag_mapper.map_name("alpha-D-glucose")
|
|
139
148
|
```
|
|
140
149
|
|
|
141
150
|
## Development
|
|
@@ -215,17 +224,18 @@ For support, please open an issue in the GitHub issue tracker.
|
|
|
215
224
|
|
|
216
225
|
## Roadmap
|
|
217
226
|
|
|
218
|
-
- [
|
|
219
|
-
- [
|
|
220
|
-
- [
|
|
227
|
+
- [x] Initial release with core functionality
|
|
228
|
+
- [x] Implement RAG-based mapping capabilities
|
|
229
|
+
- [x] Add support for major chemical/biological databases (ChEBI, UniChem, UniProt)
|
|
230
|
+
- [ ] Add caching layer for improved performance
|
|
231
|
+
- [ ] Expand RAG capabilities with more specialized models
|
|
221
232
|
- [ ] Add batch processing capabilities
|
|
222
233
|
- [ ] Develop REST API interface
|
|
223
234
|
|
|
224
235
|
## Acknowledgments
|
|
225
236
|
|
|
226
|
-
- [BridgeDb](https://www.bridgedb.org/)
|
|
227
|
-
- [RefMet](https://refmet.metabolomicsworkbench.org/)
|
|
228
237
|
- [RaMP-DB](http://rampdb.org/)
|
|
229
|
-
- [
|
|
230
|
-
- [
|
|
231
|
-
- [
|
|
238
|
+
- [ChEBI](https://www.ebi.ac.uk/chebi/)
|
|
239
|
+
- [UniChem](https://www.ebi.ac.uk/unichem/)
|
|
240
|
+
- [UniProt](https://www.uniprot.org/)
|
|
241
|
+
- [RefMet](https://refmet.metabolomicsworkbench.org/)
|
|
@@ -6,21 +6,22 @@ A unified Python toolkit for biological data harmonization and ontology mapping.
|
|
|
6
6
|
|
|
7
7
|
### Core Functionality
|
|
8
8
|
- **ID Standardization**: Unified interface for standardizing biological identifiers
|
|
9
|
-
- **Ontology Mapping**: Comprehensive ontology mapping using major biological databases
|
|
9
|
+
- **Ontology Mapping**: Comprehensive ontology mapping using major biological databases and AI-powered techniques
|
|
10
10
|
- **Data Validation**: Robust validation of input data and mappings
|
|
11
11
|
- **Extensible Architecture**: Easy integration of new data sources and mapping services
|
|
12
12
|
|
|
13
13
|
### Supported Systems
|
|
14
14
|
|
|
15
15
|
#### ID Standardization Tools
|
|
16
|
-
-
|
|
17
|
-
- RefMet
|
|
18
|
-
- RaMP-DB
|
|
16
|
+
- RaMP-DB: Integration with the Rapid Mapping Database for metabolites and pathways
|
|
19
17
|
|
|
20
|
-
####
|
|
21
|
-
-
|
|
22
|
-
-
|
|
23
|
-
-
|
|
18
|
+
#### Mapping Services
|
|
19
|
+
- ChEBI: Chemical Entities of Biological Interest database integration
|
|
20
|
+
- UniChem: Cross-referencing of chemical structure identifiers
|
|
21
|
+
- UniProt: Protein-focused mapping capabilities
|
|
22
|
+
- RefMet: Reference list of metabolite names and identifiers
|
|
23
|
+
- RAG-Based Mapping: AI-powered mapping using Retrieval Augmented Generation
|
|
24
|
+
- Multi-Provider RAG: Combining multiple data sources for improved mapping accuracy
|
|
24
25
|
|
|
25
26
|
## Installation
|
|
26
27
|
|
|
@@ -76,20 +77,18 @@ poetry install
|
|
|
76
77
|
## Quick Start
|
|
77
78
|
|
|
78
79
|
```python
|
|
79
|
-
from biomapper import
|
|
80
|
-
from biomapper.standardization import
|
|
80
|
+
from biomapper.mapping import UniProtFocusedMapper, MetaboliteNameMapper
|
|
81
|
+
from biomapper.standardization import RaMPClient
|
|
81
82
|
|
|
82
|
-
# Example 1: Using
|
|
83
|
-
|
|
84
|
-
|
|
83
|
+
# Example 1: Using UniProt-focused mapping
|
|
84
|
+
uniprot_mapper = UniProtFocusedMapper()
|
|
85
|
+
protein_mapping = uniprot_mapper.map_identifier("P12345")
|
|
85
86
|
|
|
86
|
-
#
|
|
87
|
-
|
|
87
|
+
# Example 2: Using Metabolite Name Mapping
|
|
88
|
+
metabolite_mapper = MetaboliteNameMapper()
|
|
89
|
+
metabolite_mapping = metabolite_mapper.map_name("glucose")
|
|
88
90
|
|
|
89
|
-
#
|
|
90
|
-
results = bridge_handler.standardize(["P12345", "Q67890"])
|
|
91
|
-
|
|
92
|
-
# Example 2: Using RaMP-DB
|
|
91
|
+
# Example 3: Using RaMP-DB
|
|
93
92
|
# Initialize the RaMP client
|
|
94
93
|
ramp_client = RaMPClient()
|
|
95
94
|
|
|
@@ -99,6 +98,12 @@ versions = ramp_client.get_source_versions()
|
|
|
99
98
|
# Get pathways for metabolites
|
|
100
99
|
# Example: Get pathways for Creatine (HMDB0000064)
|
|
101
100
|
pathways = ramp_client.get_pathways_from_analytes(["hmdb:HMDB0000064"])
|
|
101
|
+
|
|
102
|
+
# Example 4: Using RAG-based mapping
|
|
103
|
+
from biomapper.mapping import RagMapper
|
|
104
|
+
|
|
105
|
+
rag_mapper = RagMapper()
|
|
106
|
+
rag_results = rag_mapper.map_name("alpha-D-glucose")
|
|
102
107
|
```
|
|
103
108
|
|
|
104
109
|
## Development
|
|
@@ -178,17 +183,18 @@ For support, please open an issue in the GitHub issue tracker.
|
|
|
178
183
|
|
|
179
184
|
## Roadmap
|
|
180
185
|
|
|
181
|
-
- [
|
|
182
|
-
- [
|
|
183
|
-
- [
|
|
186
|
+
- [x] Initial release with core functionality
|
|
187
|
+
- [x] Implement RAG-based mapping capabilities
|
|
188
|
+
- [x] Add support for major chemical/biological databases (ChEBI, UniChem, UniProt)
|
|
189
|
+
- [ ] Add caching layer for improved performance
|
|
190
|
+
- [ ] Expand RAG capabilities with more specialized models
|
|
184
191
|
- [ ] Add batch processing capabilities
|
|
185
192
|
- [ ] Develop REST API interface
|
|
186
193
|
|
|
187
194
|
## Acknowledgments
|
|
188
195
|
|
|
189
|
-
- [BridgeDb](https://www.bridgedb.org/)
|
|
190
|
-
- [RefMet](https://refmet.metabolomicsworkbench.org/)
|
|
191
196
|
- [RaMP-DB](http://rampdb.org/)
|
|
192
|
-
- [
|
|
193
|
-
- [
|
|
194
|
-
- [
|
|
197
|
+
- [ChEBI](https://www.ebi.ac.uk/chebi/)
|
|
198
|
+
- [UniChem](https://www.ebi.ac.uk/unichem/)
|
|
199
|
+
- [UniProt](https://www.uniprot.org/)
|
|
200
|
+
- [RefMet](https://refmet.metabolomicsworkbench.org/)
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Base LLM integration for biomapper."""
|
|
2
|
+
|
|
3
|
+
from typing import Dict, List, Optional, TYPE_CHECKING
|
|
4
|
+
|
|
5
|
+
import time
|
|
6
|
+
import os
|
|
7
|
+
|
|
8
|
+
from langfuse import Langfuse # type: ignore
|
|
9
|
+
from openai import OpenAI
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from openai.types.chat import ChatCompletionMessageParam
|
|
13
|
+
|
|
14
|
+
from ..schemas.llm_schema import (
|
|
15
|
+
LLMMatch,
|
|
16
|
+
LLMMapperResult,
|
|
17
|
+
LLMMapperMetrics,
|
|
18
|
+
MatchConfidence,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class LLMMapper:
|
|
23
|
+
"""Base class for LLM-based ontology mapping."""
|
|
24
|
+
|
|
25
|
+
def __init__(
|
|
26
|
+
self,
|
|
27
|
+
model: str = "gpt-4",
|
|
28
|
+
temperature: float = 0.0,
|
|
29
|
+
max_tokens: int = 1000,
|
|
30
|
+
):
|
|
31
|
+
"""Initialize the LLM mapper."""
|
|
32
|
+
self.model = model
|
|
33
|
+
self.temperature = temperature
|
|
34
|
+
self.max_tokens = max_tokens
|
|
35
|
+
|
|
36
|
+
# Initialize OpenAI client
|
|
37
|
+
self.client = OpenAI()
|
|
38
|
+
|
|
39
|
+
# Initialize Langfuse
|
|
40
|
+
self.langfuse = Langfuse(
|
|
41
|
+
public_key=os.getenv("LANGFUSE_PUBLIC_KEY"),
|
|
42
|
+
secret_key=os.getenv("LANGFUSE_SECRET_KEY"),
|
|
43
|
+
host=os.getenv("LANGFUSE_HOST", "https://cloud.langfuse.com"),
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
def _create_system_prompt(self) -> str:
|
|
47
|
+
"""Create the system prompt for ontology mapping."""
|
|
48
|
+
return (
|
|
49
|
+
"You are an expert at mapping chemical and biological terms to ontologies. "
|
|
50
|
+
"Your task is to find the most appropriate ontology term for a given input. "
|
|
51
|
+
"Consider synonyms, related terms, and the hierarchical structure of the ontology."
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
def _estimate_cost(self, tokens_used: int) -> float:
|
|
55
|
+
"""Estimate the cost of the API call in USD."""
|
|
56
|
+
# GPT-4 pricing (as of 2023)
|
|
57
|
+
return tokens_used * 0.00003 # $0.03 per 1000 tokens
|
|
58
|
+
|
|
59
|
+
def map_term(
|
|
60
|
+
self,
|
|
61
|
+
term: str,
|
|
62
|
+
target_ontology: Optional[str] = None,
|
|
63
|
+
metadata: Optional[Dict[str, str]] = None,
|
|
64
|
+
) -> LLMMapperResult:
|
|
65
|
+
"""Map a single term using the LLM.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
term: Input term to map
|
|
69
|
+
target_ontology: Optional target ontology identifier
|
|
70
|
+
metadata: Optional metadata to include
|
|
71
|
+
|
|
72
|
+
Returns:
|
|
73
|
+
LLMMapperResult containing matches and metrics
|
|
74
|
+
"""
|
|
75
|
+
start_time = time.time()
|
|
76
|
+
|
|
77
|
+
# Create messages
|
|
78
|
+
messages: List[ChatCompletionMessageParam] = [
|
|
79
|
+
{"role": "system", "content": self._create_system_prompt()},
|
|
80
|
+
{"role": "user", "content": f"Map the following term: {term}"},
|
|
81
|
+
]
|
|
82
|
+
|
|
83
|
+
if target_ontology:
|
|
84
|
+
messages.append(
|
|
85
|
+
{"role": "user", "content": f"Target ontology: {target_ontology}"}
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
# Make API call
|
|
89
|
+
response = self.client.chat.completions.create(
|
|
90
|
+
model=self.model,
|
|
91
|
+
messages=messages,
|
|
92
|
+
temperature=self.temperature,
|
|
93
|
+
max_tokens=self.max_tokens,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
# Calculate metrics
|
|
97
|
+
end_time = time.time()
|
|
98
|
+
latency = (end_time - start_time) * 1000 # Convert to ms
|
|
99
|
+
tokens_used = response.usage.total_tokens if response.usage else 0
|
|
100
|
+
|
|
101
|
+
# Process response into matches
|
|
102
|
+
content = response.choices[0].message.content if response.choices else None
|
|
103
|
+
if not content:
|
|
104
|
+
raise ValueError("No content in LLM response")
|
|
105
|
+
|
|
106
|
+
matches = [
|
|
107
|
+
LLMMatch(
|
|
108
|
+
target_id="example_id",
|
|
109
|
+
target_name=content,
|
|
110
|
+
confidence=MatchConfidence.MEDIUM,
|
|
111
|
+
score=0.8,
|
|
112
|
+
reasoning="Based on LLM response",
|
|
113
|
+
metadata=metadata or {},
|
|
114
|
+
)
|
|
115
|
+
]
|
|
116
|
+
|
|
117
|
+
# Create result
|
|
118
|
+
return LLMMapperResult(
|
|
119
|
+
query_term=term,
|
|
120
|
+
matches=matches,
|
|
121
|
+
best_match=matches[0] if matches else None,
|
|
122
|
+
metrics=LLMMapperMetrics(
|
|
123
|
+
latency_ms=latency,
|
|
124
|
+
tokens_used=tokens_used,
|
|
125
|
+
provider="openai",
|
|
126
|
+
model=self.model,
|
|
127
|
+
cost=self._estimate_cost(tokens_used),
|
|
128
|
+
),
|
|
129
|
+
trace_id=self.langfuse.trace(name="llm_mapping").id,
|
|
130
|
+
)
|
|
@@ -95,7 +95,9 @@ class MetaboliteNameMapper:
|
|
|
95
95
|
if inchikey:
|
|
96
96
|
try:
|
|
97
97
|
unichem_result = (
|
|
98
|
-
self.unichem_client.
|
|
98
|
+
self.unichem_client.get_compound_info_by_src_id(
|
|
99
|
+
inchikey, "inchikey"
|
|
100
|
+
)
|
|
99
101
|
)
|
|
100
102
|
if unichem_result:
|
|
101
103
|
# Get ChEBI ID if not already found
|