biomapper 0.3.1__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. {biomapper-0.3.1 → biomapper-0.3.2}/PKG-INFO +39 -28
  2. {biomapper-0.3.1 → biomapper-0.3.2}/README.md +33 -27
  3. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/__init__.py +1 -1
  4. biomapper-0.3.2/biomapper/mapping/llm_mapper.py +130 -0
  5. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/mapping/metabolite_name_mapper.py +3 -1
  6. biomapper-0.3.2/biomapper/mapping/multi_provider_rag.py +341 -0
  7. biomapper-0.3.2/biomapper/mapping/rag_mapper.py +207 -0
  8. biomapper-0.3.2/biomapper/mapping/unichem_client.py +438 -0
  9. biomapper-0.3.2/biomapper/py.typed +1 -0
  10. biomapper-0.3.2/biomapper/schemas/llm_schema.py +51 -0
  11. biomapper-0.3.2/biomapper/schemas/provider_schemas.py +74 -0
  12. {biomapper-0.3.1 → biomapper-0.3.2}/pyproject.toml +7 -1
  13. biomapper-0.3.1/biomapper/mapping/unichem_client.py +0 -214
  14. {biomapper-0.3.1 → biomapper-0.3.2}/LICENSE +0 -0
  15. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/core/__init__.py +0 -0
  16. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/core/protein_metadata_comparison.py +0 -0
  17. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/core/set_analysis.py +0 -0
  18. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/mapping/__init__.py +0 -0
  19. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/mapping/chebi_client.py +0 -0
  20. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/mapping/refmet_client.py +0 -0
  21. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/mapping/uniprot_focused_mapper.py +0 -0
  22. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/schemas/__init__.py +0 -0
  23. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/standardization/__init__.py +0 -0
  24. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/standardization/ramp_client.py +0 -0
  25. {biomapper-0.3.1 → biomapper-0.3.2}/biomapper/utils/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: biomapper
3
- Version: 0.3.1
3
+ Version: 0.3.2
4
4
  Summary: A unified Python toolkit for biological data harmonization and ontology mapping
5
5
  Home-page: https://github.com/arpanauts/biomapper
6
6
  License: MIT
@@ -20,9 +20,14 @@ Classifier: Programming Language :: Python :: 3.13
20
20
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
21
21
  Provides-Extra: api
22
22
  Provides-Extra: full
23
+ Requires-Dist: cloudpickle (>=3.0.0,<4.0.0)
24
+ Requires-Dist: dspy-ai (>=2.1.8,<3.0.0)
25
+ Requires-Dist: langfuse (>=2.57.1,<3.0.0)
23
26
  Requires-Dist: libChEBIpy (==1.0.10)
24
27
  Requires-Dist: matplotlib (>=3.8.0,<4.0.0)
28
+ Requires-Dist: openai (>=1.14.0,<2.0.0)
25
29
  Requires-Dist: pandas (>=2.0.0,<3.0.0)
30
+ Requires-Dist: python-dotenv (>=1.0.1,<2.0.0)
26
31
  Requires-Dist: pyyaml (>=6.0.1,<7.0.0)
27
32
  Requires-Dist: requests (>=2.25.1,<3.0.0)
28
33
  Requires-Dist: seaborn (>=0.13.0,<0.14.0)
@@ -42,21 +47,22 @@ A unified Python toolkit for biological data harmonization and ontology mapping.
42
47
 
43
48
  ### Core Functionality
44
49
  - **ID Standardization**: Unified interface for standardizing biological identifiers
45
- - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases
50
+ - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases and AI-powered techniques
46
51
  - **Data Validation**: Robust validation of input data and mappings
47
52
  - **Extensible Architecture**: Easy integration of new data sources and mapping services
48
53
 
49
54
  ### Supported Systems
50
55
 
51
56
  #### ID Standardization Tools
52
- - BridgeDb
53
- - RefMet
54
- - RaMP-DB
57
+ - RaMP-DB: Integration with the Rapid Mapping Database for metabolites and pathways
55
58
 
56
- #### Ontology Mapping Services
57
- - UMLS Metathesaurus
58
- - Ontology Lookup Service (OLS)
59
- - BioPortal
59
+ #### Mapping Services
60
+ - ChEBI: Chemical Entities of Biological Interest database integration
61
+ - UniChem: Cross-referencing of chemical structure identifiers
62
+ - UniProt: Protein-focused mapping capabilities
63
+ - RefMet: Reference list of metabolite names and identifiers
64
+ - RAG-Based Mapping: AI-powered mapping using Retrieval Augmented Generation
65
+ - Multi-Provider RAG: Combining multiple data sources for improved mapping accuracy
60
66
 
61
67
  ## Installation
62
68
 
@@ -112,20 +118,18 @@ poetry install
112
118
  ## Quick Start
113
119
 
114
120
  ```python
115
- from biomapper import AnalyteMetadata
116
- from biomapper.standardization import BridgeDBHandler, RaMPClient
121
+ from biomapper.mapping import UniProtFocusedMapper, MetaboliteNameMapper
122
+ from biomapper.standardization import RaMPClient
117
123
 
118
- # Example 1: Using BridgeDB
119
- # Initialize metadata handler
120
- metadata = AnalyteMetadata()
124
+ # Example 1: Using UniProt-focused mapping
125
+ uniprot_mapper = UniProtFocusedMapper()
126
+ protein_mapping = uniprot_mapper.map_identifier("P12345")
121
127
 
122
- # Create standardization handler
123
- bridge_handler = BridgeDBHandler()
128
+ # Example 2: Using Metabolite Name Mapping
129
+ metabolite_mapper = MetaboliteNameMapper()
130
+ metabolite_mapping = metabolite_mapper.map_name("glucose")
124
131
 
125
- # Process identifiers
126
- results = bridge_handler.standardize(["P12345", "Q67890"])
127
-
128
- # Example 2: Using RaMP-DB
132
+ # Example 3: Using RaMP-DB
129
133
  # Initialize the RaMP client
130
134
  ramp_client = RaMPClient()
131
135
 
@@ -135,6 +139,12 @@ versions = ramp_client.get_source_versions()
135
139
  # Get pathways for metabolites
136
140
  # Example: Get pathways for Creatine (HMDB0000064)
137
141
  pathways = ramp_client.get_pathways_from_analytes(["hmdb:HMDB0000064"])
142
+
143
+ # Example 4: Using RAG-based mapping
144
+ from biomapper.mapping import RagMapper
145
+
146
+ rag_mapper = RagMapper()
147
+ rag_results = rag_mapper.map_name("alpha-D-glucose")
138
148
  ```
139
149
 
140
150
  ## Development
@@ -214,17 +224,18 @@ For support, please open an issue in the GitHub issue tracker.
214
224
 
215
225
  ## Roadmap
216
226
 
217
- - [ ] Initial release with core functionality
218
- - [ ] Add support for additional ontology services
219
- - [ ] Implement caching layer
227
+ - [x] Initial release with core functionality
228
+ - [x] Implement RAG-based mapping capabilities
229
+ - [x] Add support for major chemical/biological databases (ChEBI, UniChem, UniProt)
230
+ - [ ] Add caching layer for improved performance
231
+ - [ ] Expand RAG capabilities with more specialized models
220
232
  - [ ] Add batch processing capabilities
221
233
  - [ ] Develop REST API interface
222
234
 
223
235
  ## Acknowledgments
224
236
 
225
- - [BridgeDb](https://www.bridgedb.org/)
226
- - [RefMet](https://refmet.metabolomicsworkbench.org/)
227
237
  - [RaMP-DB](http://rampdb.org/)
228
- - [UMLS](https://www.nlm.nih.gov/research/umls/index.html)
229
- - [OLS](https://www.ebi.ac.uk/ols/index)
230
- - [BioPortal](https://bioportal.bioontology.org/)
238
+ - [ChEBI](https://www.ebi.ac.uk/chebi/)
239
+ - [UniChem](https://www.ebi.ac.uk/unichem/)
240
+ - [UniProt](https://www.uniprot.org/)
241
+ - [RefMet](https://refmet.metabolomicsworkbench.org/)
@@ -6,21 +6,22 @@ A unified Python toolkit for biological data harmonization and ontology mapping.
6
6
 
7
7
  ### Core Functionality
8
8
  - **ID Standardization**: Unified interface for standardizing biological identifiers
9
- - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases
9
+ - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases and AI-powered techniques
10
10
  - **Data Validation**: Robust validation of input data and mappings
11
11
  - **Extensible Architecture**: Easy integration of new data sources and mapping services
12
12
 
13
13
  ### Supported Systems
14
14
 
15
15
  #### ID Standardization Tools
16
- - BridgeDb
17
- - RefMet
18
- - RaMP-DB
16
+ - RaMP-DB: Integration with the Rapid Mapping Database for metabolites and pathways
19
17
 
20
- #### Ontology Mapping Services
21
- - UMLS Metathesaurus
22
- - Ontology Lookup Service (OLS)
23
- - BioPortal
18
+ #### Mapping Services
19
+ - ChEBI: Chemical Entities of Biological Interest database integration
20
+ - UniChem: Cross-referencing of chemical structure identifiers
21
+ - UniProt: Protein-focused mapping capabilities
22
+ - RefMet: Reference list of metabolite names and identifiers
23
+ - RAG-Based Mapping: AI-powered mapping using Retrieval Augmented Generation
24
+ - Multi-Provider RAG: Combining multiple data sources for improved mapping accuracy
24
25
 
25
26
  ## Installation
26
27
 
@@ -76,20 +77,18 @@ poetry install
76
77
  ## Quick Start
77
78
 
78
79
  ```python
79
- from biomapper import AnalyteMetadata
80
- from biomapper.standardization import BridgeDBHandler, RaMPClient
80
+ from biomapper.mapping import UniProtFocusedMapper, MetaboliteNameMapper
81
+ from biomapper.standardization import RaMPClient
81
82
 
82
- # Example 1: Using BridgeDB
83
- # Initialize metadata handler
84
- metadata = AnalyteMetadata()
83
+ # Example 1: Using UniProt-focused mapping
84
+ uniprot_mapper = UniProtFocusedMapper()
85
+ protein_mapping = uniprot_mapper.map_identifier("P12345")
85
86
 
86
- # Create standardization handler
87
- bridge_handler = BridgeDBHandler()
87
+ # Example 2: Using Metabolite Name Mapping
88
+ metabolite_mapper = MetaboliteNameMapper()
89
+ metabolite_mapping = metabolite_mapper.map_name("glucose")
88
90
 
89
- # Process identifiers
90
- results = bridge_handler.standardize(["P12345", "Q67890"])
91
-
92
- # Example 2: Using RaMP-DB
91
+ # Example 3: Using RaMP-DB
93
92
  # Initialize the RaMP client
94
93
  ramp_client = RaMPClient()
95
94
 
@@ -99,6 +98,12 @@ versions = ramp_client.get_source_versions()
99
98
  # Get pathways for metabolites
100
99
  # Example: Get pathways for Creatine (HMDB0000064)
101
100
  pathways = ramp_client.get_pathways_from_analytes(["hmdb:HMDB0000064"])
101
+
102
+ # Example 4: Using RAG-based mapping
103
+ from biomapper.mapping import RagMapper
104
+
105
+ rag_mapper = RagMapper()
106
+ rag_results = rag_mapper.map_name("alpha-D-glucose")
102
107
  ```
103
108
 
104
109
  ## Development
@@ -178,17 +183,18 @@ For support, please open an issue in the GitHub issue tracker.
178
183
 
179
184
  ## Roadmap
180
185
 
181
- - [ ] Initial release with core functionality
182
- - [ ] Add support for additional ontology services
183
- - [ ] Implement caching layer
186
+ - [x] Initial release with core functionality
187
+ - [x] Implement RAG-based mapping capabilities
188
+ - [x] Add support for major chemical/biological databases (ChEBI, UniChem, UniProt)
189
+ - [ ] Add caching layer for improved performance
190
+ - [ ] Expand RAG capabilities with more specialized models
184
191
  - [ ] Add batch processing capabilities
185
192
  - [ ] Develop REST API interface
186
193
 
187
194
  ## Acknowledgments
188
195
 
189
- - [BridgeDb](https://www.bridgedb.org/)
190
- - [RefMet](https://refmet.metabolomicsworkbench.org/)
191
196
  - [RaMP-DB](http://rampdb.org/)
192
- - [UMLS](https://www.nlm.nih.gov/research/umls/index.html)
193
- - [OLS](https://www.ebi.ac.uk/ols/index)
194
- - [BioPortal](https://bioportal.bioontology.org/)
197
+ - [ChEBI](https://www.ebi.ac.uk/chebi/)
198
+ - [UniChem](https://www.ebi.ac.uk/unichem/)
199
+ - [UniProt](https://www.uniprot.org/)
200
+ - [RefMet](https://refmet.metabolomicsworkbench.org/)
@@ -3,5 +3,5 @@
3
3
  from .standardization import RaMPClient
4
4
  from .core import SetAnalyzer
5
5
 
6
- __version__ = "0.3.1"
6
+ __version__ = "0.3.2"
7
7
  __all__ = ["RaMPClient", "SetAnalyzer"]
@@ -0,0 +1,130 @@
1
+ """Base LLM integration for biomapper."""
2
+
3
+ from typing import Dict, List, Optional, TYPE_CHECKING
4
+
5
+ import time
6
+ import os
7
+
8
+ from langfuse import Langfuse # type: ignore
9
+ from openai import OpenAI
10
+
11
+ if TYPE_CHECKING:
12
+ from openai.types.chat import ChatCompletionMessageParam
13
+
14
+ from ..schemas.llm_schema import (
15
+ LLMMatch,
16
+ LLMMapperResult,
17
+ LLMMapperMetrics,
18
+ MatchConfidence,
19
+ )
20
+
21
+
22
+ class LLMMapper:
23
+ """Base class for LLM-based ontology mapping."""
24
+
25
+ def __init__(
26
+ self,
27
+ model: str = "gpt-4",
28
+ temperature: float = 0.0,
29
+ max_tokens: int = 1000,
30
+ ):
31
+ """Initialize the LLM mapper."""
32
+ self.model = model
33
+ self.temperature = temperature
34
+ self.max_tokens = max_tokens
35
+
36
+ # Initialize OpenAI client
37
+ self.client = OpenAI()
38
+
39
+ # Initialize Langfuse
40
+ self.langfuse = Langfuse(
41
+ public_key=os.getenv("LANGFUSE_PUBLIC_KEY"),
42
+ secret_key=os.getenv("LANGFUSE_SECRET_KEY"),
43
+ host=os.getenv("LANGFUSE_HOST", "https://cloud.langfuse.com"),
44
+ )
45
+
46
+ def _create_system_prompt(self) -> str:
47
+ """Create the system prompt for ontology mapping."""
48
+ return (
49
+ "You are an expert at mapping chemical and biological terms to ontologies. "
50
+ "Your task is to find the most appropriate ontology term for a given input. "
51
+ "Consider synonyms, related terms, and the hierarchical structure of the ontology."
52
+ )
53
+
54
+ def _estimate_cost(self, tokens_used: int) -> float:
55
+ """Estimate the cost of the API call in USD."""
56
+ # GPT-4 pricing (as of 2023)
57
+ return tokens_used * 0.00003 # $0.03 per 1000 tokens
58
+
59
+ def map_term(
60
+ self,
61
+ term: str,
62
+ target_ontology: Optional[str] = None,
63
+ metadata: Optional[Dict[str, str]] = None,
64
+ ) -> LLMMapperResult:
65
+ """Map a single term using the LLM.
66
+
67
+ Args:
68
+ term: Input term to map
69
+ target_ontology: Optional target ontology identifier
70
+ metadata: Optional metadata to include
71
+
72
+ Returns:
73
+ LLMMapperResult containing matches and metrics
74
+ """
75
+ start_time = time.time()
76
+
77
+ # Create messages
78
+ messages: List[ChatCompletionMessageParam] = [
79
+ {"role": "system", "content": self._create_system_prompt()},
80
+ {"role": "user", "content": f"Map the following term: {term}"},
81
+ ]
82
+
83
+ if target_ontology:
84
+ messages.append(
85
+ {"role": "user", "content": f"Target ontology: {target_ontology}"}
86
+ )
87
+
88
+ # Make API call
89
+ response = self.client.chat.completions.create(
90
+ model=self.model,
91
+ messages=messages,
92
+ temperature=self.temperature,
93
+ max_tokens=self.max_tokens,
94
+ )
95
+
96
+ # Calculate metrics
97
+ end_time = time.time()
98
+ latency = (end_time - start_time) * 1000 # Convert to ms
99
+ tokens_used = response.usage.total_tokens if response.usage else 0
100
+
101
+ # Process response into matches
102
+ content = response.choices[0].message.content if response.choices else None
103
+ if not content:
104
+ raise ValueError("No content in LLM response")
105
+
106
+ matches = [
107
+ LLMMatch(
108
+ target_id="example_id",
109
+ target_name=content,
110
+ confidence=MatchConfidence.MEDIUM,
111
+ score=0.8,
112
+ reasoning="Based on LLM response",
113
+ metadata=metadata or {},
114
+ )
115
+ ]
116
+
117
+ # Create result
118
+ return LLMMapperResult(
119
+ query_term=term,
120
+ matches=matches,
121
+ best_match=matches[0] if matches else None,
122
+ metrics=LLMMapperMetrics(
123
+ latency_ms=latency,
124
+ tokens_used=tokens_used,
125
+ provider="openai",
126
+ model=self.model,
127
+ cost=self._estimate_cost(tokens_used),
128
+ ),
129
+ trace_id=self.langfuse.trace(name="llm_mapping").id,
130
+ )
@@ -95,7 +95,9 @@ class MetaboliteNameMapper:
95
95
  if inchikey:
96
96
  try:
97
97
  unichem_result = (
98
- self.unichem_client.get_compound_info_by_inchikey(inchikey)
98
+ self.unichem_client.get_compound_info_by_src_id(
99
+ inchikey, "inchikey"
100
+ )
99
101
  )
100
102
  if unichem_result:
101
103
  # Get ChEBI ID if not already found