biomapper 0.3.0__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {biomapper-0.3.0 → biomapper-0.3.2}/PKG-INFO +42 -32
  2. {biomapper-0.3.0 → biomapper-0.3.2}/README.md +33 -27
  3. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/__init__.py +1 -1
  4. biomapper-0.3.2/biomapper/mapping/llm_mapper.py +130 -0
  5. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/metabolite_name_mapper.py +3 -1
  6. biomapper-0.3.2/biomapper/mapping/multi_provider_rag.py +341 -0
  7. biomapper-0.3.2/biomapper/mapping/rag_mapper.py +207 -0
  8. biomapper-0.3.2/biomapper/mapping/unichem_client.py +438 -0
  9. biomapper-0.3.2/biomapper/py.typed +1 -0
  10. biomapper-0.3.2/biomapper/schemas/llm_schema.py +51 -0
  11. biomapper-0.3.2/biomapper/schemas/provider_schemas.py +74 -0
  12. {biomapper-0.3.0 → biomapper-0.3.2}/pyproject.toml +12 -7
  13. biomapper-0.3.0/biomapper/mapping/unichem_client.py +0 -214
  14. biomapper-0.3.0/biomapper/standardization/tutorial.ipynb +0 -321
  15. {biomapper-0.3.0 → biomapper-0.3.2}/LICENSE +0 -0
  16. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/core/__init__.py +0 -0
  17. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/core/protein_metadata_comparison.py +0 -0
  18. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/core/set_analysis.py +0 -0
  19. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/__init__.py +0 -0
  20. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/chebi_client.py +0 -0
  21. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/refmet_client.py +0 -0
  22. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/mapping/uniprot_focused_mapper.py +0 -0
  23. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/schemas/__init__.py +0 -0
  24. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/standardization/__init__.py +0 -0
  25. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/standardization/ramp_client.py +0 -0
  26. {biomapper-0.3.0 → biomapper-0.3.2}/biomapper/utils/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: biomapper
3
- Version: 0.3.0
3
+ Version: 0.3.2
4
4
  Summary: A unified Python toolkit for biological data harmonization and ontology mapping
5
5
  Home-page: https://github.com/arpanauts/biomapper
6
6
  License: MIT
@@ -20,16 +20,20 @@ Classifier: Programming Language :: Python :: 3.13
20
20
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
21
21
  Provides-Extra: api
22
22
  Provides-Extra: full
23
- Provides-Extra: viz
23
+ Requires-Dist: cloudpickle (>=3.0.0,<4.0.0)
24
+ Requires-Dist: dspy-ai (>=2.1.8,<3.0.0)
25
+ Requires-Dist: langfuse (>=2.57.1,<3.0.0)
24
26
  Requires-Dist: libChEBIpy (==1.0.10)
25
- Requires-Dist: matplotlib (>=3.8.0,<4.0.0) ; extra == "viz" or extra == "full"
27
+ Requires-Dist: matplotlib (>=3.8.0,<4.0.0)
28
+ Requires-Dist: openai (>=1.14.0,<2.0.0)
26
29
  Requires-Dist: pandas (>=2.0.0,<3.0.0)
30
+ Requires-Dist: python-dotenv (>=1.0.1,<2.0.0)
27
31
  Requires-Dist: pyyaml (>=6.0.1,<7.0.0)
28
32
  Requires-Dist: requests (>=2.25.1,<3.0.0)
29
- Requires-Dist: seaborn (>=0.13.0,<0.14.0) ; extra == "viz" or extra == "full"
33
+ Requires-Dist: seaborn (>=0.13.0,<0.14.0)
30
34
  Requires-Dist: sqlalchemy (>=1.4.0,<2.0.0)
31
35
  Requires-Dist: tqdm (>=4.66.1,<5.0.0)
32
- Requires-Dist: upsetplot (>=0.8.0,<0.9.0) ; extra == "viz" or extra == "full"
36
+ Requires-Dist: upsetplot (>=0.8.0,<0.9.0)
33
37
  Requires-Dist: venn (>=0.1.3,<0.2.0)
34
38
  Project-URL: Documentation, https://github.com/arpanauts/biomapper/blob/main/README.md
35
39
  Project-URL: Repository, https://github.com/arpanauts/biomapper
@@ -43,21 +47,22 @@ A unified Python toolkit for biological data harmonization and ontology mapping.
43
47
 
44
48
  ### Core Functionality
45
49
  - **ID Standardization**: Unified interface for standardizing biological identifiers
46
- - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases
50
+ - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases and AI-powered techniques
47
51
  - **Data Validation**: Robust validation of input data and mappings
48
52
  - **Extensible Architecture**: Easy integration of new data sources and mapping services
49
53
 
50
54
  ### Supported Systems
51
55
 
52
56
  #### ID Standardization Tools
53
- - BridgeDb
54
- - RefMet
55
- - RaMP-DB
57
+ - RaMP-DB: Integration with the Rapid Mapping Database for metabolites and pathways
56
58
 
57
- #### Ontology Mapping Services
58
- - UMLS Metathesaurus
59
- - Ontology Lookup Service (OLS)
60
- - BioPortal
59
+ #### Mapping Services
60
+ - ChEBI: Chemical Entities of Biological Interest database integration
61
+ - UniChem: Cross-referencing of chemical structure identifiers
62
+ - UniProt: Protein-focused mapping capabilities
63
+ - RefMet: Reference list of metabolite names and identifiers
64
+ - RAG-Based Mapping: AI-powered mapping using Retrieval Augmented Generation
65
+ - Multi-Provider RAG: Combining multiple data sources for improved mapping accuracy
61
66
 
62
67
  ## Installation
63
68
 
@@ -113,20 +118,18 @@ poetry install
113
118
  ## Quick Start
114
119
 
115
120
  ```python
116
- from biomapper import AnalyteMetadata
117
- from biomapper.standardization import BridgeDBHandler, RaMPClient
121
+ from biomapper.mapping import UniProtFocusedMapper, MetaboliteNameMapper
122
+ from biomapper.standardization import RaMPClient
118
123
 
119
- # Example 1: Using BridgeDB
120
- # Initialize metadata handler
121
- metadata = AnalyteMetadata()
124
+ # Example 1: Using UniProt-focused mapping
125
+ uniprot_mapper = UniProtFocusedMapper()
126
+ protein_mapping = uniprot_mapper.map_identifier("P12345")
122
127
 
123
- # Create standardization handler
124
- bridge_handler = BridgeDBHandler()
128
+ # Example 2: Using Metabolite Name Mapping
129
+ metabolite_mapper = MetaboliteNameMapper()
130
+ metabolite_mapping = metabolite_mapper.map_name("glucose")
125
131
 
126
- # Process identifiers
127
- results = bridge_handler.standardize(["P12345", "Q67890"])
128
-
129
- # Example 2: Using RaMP-DB
132
+ # Example 3: Using RaMP-DB
130
133
  # Initialize the RaMP client
131
134
  ramp_client = RaMPClient()
132
135
 
@@ -136,6 +139,12 @@ versions = ramp_client.get_source_versions()
136
139
  # Get pathways for metabolites
137
140
  # Example: Get pathways for Creatine (HMDB0000064)
138
141
  pathways = ramp_client.get_pathways_from_analytes(["hmdb:HMDB0000064"])
142
+
143
+ # Example 4: Using RAG-based mapping
144
+ from biomapper.mapping import RagMapper
145
+
146
+ rag_mapper = RagMapper()
147
+ rag_results = rag_mapper.map_name("alpha-D-glucose")
139
148
  ```
140
149
 
141
150
  ## Development
@@ -215,17 +224,18 @@ For support, please open an issue in the GitHub issue tracker.
215
224
 
216
225
  ## Roadmap
217
226
 
218
- - [ ] Initial release with core functionality
219
- - [ ] Add support for additional ontology services
220
- - [ ] Implement caching layer
227
+ - [x] Initial release with core functionality
228
+ - [x] Implement RAG-based mapping capabilities
229
+ - [x] Add support for major chemical/biological databases (ChEBI, UniChem, UniProt)
230
+ - [ ] Add caching layer for improved performance
231
+ - [ ] Expand RAG capabilities with more specialized models
221
232
  - [ ] Add batch processing capabilities
222
233
  - [ ] Develop REST API interface
223
234
 
224
235
  ## Acknowledgments
225
236
 
226
- - [BridgeDb](https://www.bridgedb.org/)
227
- - [RefMet](https://refmet.metabolomicsworkbench.org/)
228
237
  - [RaMP-DB](http://rampdb.org/)
229
- - [UMLS](https://www.nlm.nih.gov/research/umls/index.html)
230
- - [OLS](https://www.ebi.ac.uk/ols/index)
231
- - [BioPortal](https://bioportal.bioontology.org/)
238
+ - [ChEBI](https://www.ebi.ac.uk/chebi/)
239
+ - [UniChem](https://www.ebi.ac.uk/unichem/)
240
+ - [UniProt](https://www.uniprot.org/)
241
+ - [RefMet](https://refmet.metabolomicsworkbench.org/)
@@ -6,21 +6,22 @@ A unified Python toolkit for biological data harmonization and ontology mapping.
6
6
 
7
7
  ### Core Functionality
8
8
  - **ID Standardization**: Unified interface for standardizing biological identifiers
9
- - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases
9
+ - **Ontology Mapping**: Comprehensive ontology mapping using major biological databases and AI-powered techniques
10
10
  - **Data Validation**: Robust validation of input data and mappings
11
11
  - **Extensible Architecture**: Easy integration of new data sources and mapping services
12
12
 
13
13
  ### Supported Systems
14
14
 
15
15
  #### ID Standardization Tools
16
- - BridgeDb
17
- - RefMet
18
- - RaMP-DB
16
+ - RaMP-DB: Integration with the Rapid Mapping Database for metabolites and pathways
19
17
 
20
- #### Ontology Mapping Services
21
- - UMLS Metathesaurus
22
- - Ontology Lookup Service (OLS)
23
- - BioPortal
18
+ #### Mapping Services
19
+ - ChEBI: Chemical Entities of Biological Interest database integration
20
+ - UniChem: Cross-referencing of chemical structure identifiers
21
+ - UniProt: Protein-focused mapping capabilities
22
+ - RefMet: Reference list of metabolite names and identifiers
23
+ - RAG-Based Mapping: AI-powered mapping using Retrieval Augmented Generation
24
+ - Multi-Provider RAG: Combining multiple data sources for improved mapping accuracy
24
25
 
25
26
  ## Installation
26
27
 
@@ -76,20 +77,18 @@ poetry install
76
77
  ## Quick Start
77
78
 
78
79
  ```python
79
- from biomapper import AnalyteMetadata
80
- from biomapper.standardization import BridgeDBHandler, RaMPClient
80
+ from biomapper.mapping import UniProtFocusedMapper, MetaboliteNameMapper
81
+ from biomapper.standardization import RaMPClient
81
82
 
82
- # Example 1: Using BridgeDB
83
- # Initialize metadata handler
84
- metadata = AnalyteMetadata()
83
+ # Example 1: Using UniProt-focused mapping
84
+ uniprot_mapper = UniProtFocusedMapper()
85
+ protein_mapping = uniprot_mapper.map_identifier("P12345")
85
86
 
86
- # Create standardization handler
87
- bridge_handler = BridgeDBHandler()
87
+ # Example 2: Using Metabolite Name Mapping
88
+ metabolite_mapper = MetaboliteNameMapper()
89
+ metabolite_mapping = metabolite_mapper.map_name("glucose")
88
90
 
89
- # Process identifiers
90
- results = bridge_handler.standardize(["P12345", "Q67890"])
91
-
92
- # Example 2: Using RaMP-DB
91
+ # Example 3: Using RaMP-DB
93
92
  # Initialize the RaMP client
94
93
  ramp_client = RaMPClient()
95
94
 
@@ -99,6 +98,12 @@ versions = ramp_client.get_source_versions()
99
98
  # Get pathways for metabolites
100
99
  # Example: Get pathways for Creatine (HMDB0000064)
101
100
  pathways = ramp_client.get_pathways_from_analytes(["hmdb:HMDB0000064"])
101
+
102
+ # Example 4: Using RAG-based mapping
103
+ from biomapper.mapping import RagMapper
104
+
105
+ rag_mapper = RagMapper()
106
+ rag_results = rag_mapper.map_name("alpha-D-glucose")
102
107
  ```
103
108
 
104
109
  ## Development
@@ -178,17 +183,18 @@ For support, please open an issue in the GitHub issue tracker.
178
183
 
179
184
  ## Roadmap
180
185
 
181
- - [ ] Initial release with core functionality
182
- - [ ] Add support for additional ontology services
183
- - [ ] Implement caching layer
186
+ - [x] Initial release with core functionality
187
+ - [x] Implement RAG-based mapping capabilities
188
+ - [x] Add support for major chemical/biological databases (ChEBI, UniChem, UniProt)
189
+ - [ ] Add caching layer for improved performance
190
+ - [ ] Expand RAG capabilities with more specialized models
184
191
  - [ ] Add batch processing capabilities
185
192
  - [ ] Develop REST API interface
186
193
 
187
194
  ## Acknowledgments
188
195
 
189
- - [BridgeDb](https://www.bridgedb.org/)
190
- - [RefMet](https://refmet.metabolomicsworkbench.org/)
191
196
  - [RaMP-DB](http://rampdb.org/)
192
- - [UMLS](https://www.nlm.nih.gov/research/umls/index.html)
193
- - [OLS](https://www.ebi.ac.uk/ols/index)
194
- - [BioPortal](https://bioportal.bioontology.org/)
197
+ - [ChEBI](https://www.ebi.ac.uk/chebi/)
198
+ - [UniChem](https://www.ebi.ac.uk/unichem/)
199
+ - [UniProt](https://www.uniprot.org/)
200
+ - [RefMet](https://refmet.metabolomicsworkbench.org/)
@@ -3,5 +3,5 @@
3
3
  from .standardization import RaMPClient
4
4
  from .core import SetAnalyzer
5
5
 
6
- __version__ = "0.3.0"
6
+ __version__ = "0.3.2"
7
7
  __all__ = ["RaMPClient", "SetAnalyzer"]
@@ -0,0 +1,130 @@
1
+ """Base LLM integration for biomapper."""
2
+
3
+ from typing import Dict, List, Optional, TYPE_CHECKING
4
+
5
+ import time
6
+ import os
7
+
8
+ from langfuse import Langfuse # type: ignore
9
+ from openai import OpenAI
10
+
11
+ if TYPE_CHECKING:
12
+ from openai.types.chat import ChatCompletionMessageParam
13
+
14
+ from ..schemas.llm_schema import (
15
+ LLMMatch,
16
+ LLMMapperResult,
17
+ LLMMapperMetrics,
18
+ MatchConfidence,
19
+ )
20
+
21
+
22
+ class LLMMapper:
23
+ """Base class for LLM-based ontology mapping."""
24
+
25
+ def __init__(
26
+ self,
27
+ model: str = "gpt-4",
28
+ temperature: float = 0.0,
29
+ max_tokens: int = 1000,
30
+ ):
31
+ """Initialize the LLM mapper."""
32
+ self.model = model
33
+ self.temperature = temperature
34
+ self.max_tokens = max_tokens
35
+
36
+ # Initialize OpenAI client
37
+ self.client = OpenAI()
38
+
39
+ # Initialize Langfuse
40
+ self.langfuse = Langfuse(
41
+ public_key=os.getenv("LANGFUSE_PUBLIC_KEY"),
42
+ secret_key=os.getenv("LANGFUSE_SECRET_KEY"),
43
+ host=os.getenv("LANGFUSE_HOST", "https://cloud.langfuse.com"),
44
+ )
45
+
46
+ def _create_system_prompt(self) -> str:
47
+ """Create the system prompt for ontology mapping."""
48
+ return (
49
+ "You are an expert at mapping chemical and biological terms to ontologies. "
50
+ "Your task is to find the most appropriate ontology term for a given input. "
51
+ "Consider synonyms, related terms, and the hierarchical structure of the ontology."
52
+ )
53
+
54
+ def _estimate_cost(self, tokens_used: int) -> float:
55
+ """Estimate the cost of the API call in USD."""
56
+ # GPT-4 pricing (as of 2023)
57
+ return tokens_used * 0.00003 # $0.03 per 1000 tokens
58
+
59
+ def map_term(
60
+ self,
61
+ term: str,
62
+ target_ontology: Optional[str] = None,
63
+ metadata: Optional[Dict[str, str]] = None,
64
+ ) -> LLMMapperResult:
65
+ """Map a single term using the LLM.
66
+
67
+ Args:
68
+ term: Input term to map
69
+ target_ontology: Optional target ontology identifier
70
+ metadata: Optional metadata to include
71
+
72
+ Returns:
73
+ LLMMapperResult containing matches and metrics
74
+ """
75
+ start_time = time.time()
76
+
77
+ # Create messages
78
+ messages: List[ChatCompletionMessageParam] = [
79
+ {"role": "system", "content": self._create_system_prompt()},
80
+ {"role": "user", "content": f"Map the following term: {term}"},
81
+ ]
82
+
83
+ if target_ontology:
84
+ messages.append(
85
+ {"role": "user", "content": f"Target ontology: {target_ontology}"}
86
+ )
87
+
88
+ # Make API call
89
+ response = self.client.chat.completions.create(
90
+ model=self.model,
91
+ messages=messages,
92
+ temperature=self.temperature,
93
+ max_tokens=self.max_tokens,
94
+ )
95
+
96
+ # Calculate metrics
97
+ end_time = time.time()
98
+ latency = (end_time - start_time) * 1000 # Convert to ms
99
+ tokens_used = response.usage.total_tokens if response.usage else 0
100
+
101
+ # Process response into matches
102
+ content = response.choices[0].message.content if response.choices else None
103
+ if not content:
104
+ raise ValueError("No content in LLM response")
105
+
106
+ matches = [
107
+ LLMMatch(
108
+ target_id="example_id",
109
+ target_name=content,
110
+ confidence=MatchConfidence.MEDIUM,
111
+ score=0.8,
112
+ reasoning="Based on LLM response",
113
+ metadata=metadata or {},
114
+ )
115
+ ]
116
+
117
+ # Create result
118
+ return LLMMapperResult(
119
+ query_term=term,
120
+ matches=matches,
121
+ best_match=matches[0] if matches else None,
122
+ metrics=LLMMapperMetrics(
123
+ latency_ms=latency,
124
+ tokens_used=tokens_used,
125
+ provider="openai",
126
+ model=self.model,
127
+ cost=self._estimate_cost(tokens_used),
128
+ ),
129
+ trace_id=self.langfuse.trace(name="llm_mapping").id,
130
+ )
@@ -95,7 +95,9 @@ class MetaboliteNameMapper:
95
95
  if inchikey:
96
96
  try:
97
97
  unichem_result = (
98
- self.unichem_client.get_compound_info_by_inchikey(inchikey)
98
+ self.unichem_client.get_compound_info_by_src_id(
99
+ inchikey, "inchikey"
100
+ )
99
101
  )
100
102
  if unichem_result:
101
103
  # Get ChEBI ID if not already found