biomapper 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (21) hide show
  1. {biomapper-0.2.0 → biomapper-0.3.0}/PKG-INFO +5 -1
  2. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/__init__.py +3 -2
  3. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/core/__init__.py +4 -0
  4. biomapper-0.3.0/biomapper/core/set_analysis.py +415 -0
  5. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/chebi_client.py +37 -20
  6. biomapper-0.3.0/biomapper/mapping/metabolite_name_mapper.py +569 -0
  7. biomapper-0.3.0/biomapper/mapping/refmet_client.py +222 -0
  8. {biomapper-0.2.0 → biomapper-0.3.0}/pyproject.toml +7 -3
  9. biomapper-0.2.0/biomapper/mapping/metabolite_name_mapper.py +0 -179
  10. biomapper-0.2.0/biomapper/mapping/refmet_client.py +0 -113
  11. {biomapper-0.2.0 → biomapper-0.3.0}/LICENSE +0 -0
  12. {biomapper-0.2.0 → biomapper-0.3.0}/README.md +0 -0
  13. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/core/protein_metadata_comparison.py +0 -0
  14. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/__init__.py +0 -0
  15. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/unichem_client.py +0 -0
  16. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/uniprot_focused_mapper.py +0 -0
  17. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/schemas/__init__.py +0 -0
  18. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/standardization/__init__.py +0 -0
  19. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/standardization/ramp_client.py +0 -0
  20. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/standardization/tutorial.ipynb +0 -0
  21. {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/utils/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: biomapper
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: A unified Python toolkit for biological data harmonization and ontology mapping
5
5
  Home-page: https://github.com/arpanauts/biomapper
6
6
  License: MIT
@@ -22,11 +22,15 @@ Provides-Extra: api
22
22
  Provides-Extra: full
23
23
  Provides-Extra: viz
24
24
  Requires-Dist: libChEBIpy (==1.0.10)
25
+ Requires-Dist: matplotlib (>=3.8.0,<4.0.0) ; extra == "viz" or extra == "full"
25
26
  Requires-Dist: pandas (>=2.0.0,<3.0.0)
26
27
  Requires-Dist: pyyaml (>=6.0.1,<7.0.0)
27
28
  Requires-Dist: requests (>=2.25.1,<3.0.0)
29
+ Requires-Dist: seaborn (>=0.13.0,<0.14.0) ; extra == "viz" or extra == "full"
28
30
  Requires-Dist: sqlalchemy (>=1.4.0,<2.0.0)
29
31
  Requires-Dist: tqdm (>=4.66.1,<5.0.0)
32
+ Requires-Dist: upsetplot (>=0.8.0,<0.9.0) ; extra == "viz" or extra == "full"
33
+ Requires-Dist: venn (>=0.1.3,<0.2.0)
30
34
  Project-URL: Documentation, https://github.com/arpanauts/biomapper/blob/main/README.md
31
35
  Project-URL: Repository, https://github.com/arpanauts/biomapper
32
36
  Description-Content-Type: text/markdown
@@ -1,6 +1,7 @@
1
1
  """Biomapper package for biological data harmonization and ontology mapping."""
2
2
 
3
3
  from .standardization import RaMPClient
4
+ from .core import SetAnalyzer
4
5
 
5
- __version__ = "0.2.0"
6
- __all__ = ["RaMPClient"]
6
+ __version__ = "0.3.0"
7
+ __all__ = ["RaMPClient", "SetAnalyzer"]
@@ -1 +1,5 @@
1
1
  """Core functionality and base classes for the BioMapper package."""
2
+
3
+ from .set_analysis import SetAnalyzer
4
+
5
+ __all__ = ["SetAnalyzer"]
@@ -0,0 +1,415 @@
1
+ """Module for performing set analysis on omic datasets."""
2
+
3
+ import logging
4
+ from pathlib import Path
5
+ from typing import Any, Dict, List, Optional, TypedDict
6
+ import matplotlib.pyplot as plt
7
+ import pandas as pd
8
+ import venn # type: ignore
9
+ from upsetplot import plot as upsetplot # type: ignore
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+
14
+ class DatasetInfo(TypedDict):
15
+ """Type definition for dataset information."""
16
+
17
+ size: int
18
+ sample_ids: List[str]
19
+
20
+
21
+ class IntersectionInfo(TypedDict):
22
+ """Type definition for intersection information."""
23
+
24
+ sets: List[str]
25
+ size: int
26
+ sample_ids: List[str]
27
+
28
+
29
+ class AnalysisResults(TypedDict):
30
+ """Type definition for analysis results."""
31
+
32
+ datasets: Dict[str, DatasetInfo]
33
+ intersections: List[IntersectionInfo]
34
+ unique: Dict[str, DatasetInfo]
35
+
36
+
37
+ class SetAnalyzer:
38
+ """Analyzer for performing set comparisons across omic datasets."""
39
+
40
+ def __init__(self, id_columns: dict[str, str]) -> None:
41
+ """Initialize the analyzer.
42
+
43
+ Args:
44
+ id_columns: Mapping of dataset names to their ID column names
45
+
46
+ Raises:
47
+ ValueError: If id_columns is empty
48
+ """
49
+ if not id_columns:
50
+ raise ValueError("id_columns mapping cannot be empty")
51
+
52
+ self.id_columns = {} # Initialize empty dict
53
+ self.datasets: dict[str, pd.DataFrame] = {}
54
+ self.id_delimiters: dict[str, list[str]] = {} # Change type to list[str]
55
+ # Use property setter for validation
56
+ for name, col in id_columns.items():
57
+ self.set_id_column(name, col)
58
+
59
+ def set_id_column(self, dataset: str, column: str) -> None:
60
+ """Set ID column for a dataset with validation."""
61
+ if dataset in self.datasets:
62
+ if column not in self.datasets[dataset].columns:
63
+ raise ValueError(
64
+ f"Column '{column}' not found in dataset '{dataset}'. "
65
+ f"Available columns: {', '.join(self.datasets[dataset].columns)}"
66
+ )
67
+ self.id_columns[dataset] = column
68
+
69
+ @property
70
+ def id_columns(self) -> dict[str, str]:
71
+ return self._id_columns
72
+
73
+ @id_columns.setter
74
+ def id_columns(self, value: dict[str, str]) -> None:
75
+ self._id_columns = value
76
+
77
+ def __setitem__(self, key: str, value: str) -> None:
78
+ """Set ID column for a dataset."""
79
+ if key not in self.datasets:
80
+ raise ValueError(f"Dataset '{key}' not found")
81
+
82
+ if value not in self.datasets[key].columns:
83
+ raise ValueError(f"Column '{value}' not found in dataset '{key}'")
84
+
85
+ self._id_columns[key] = value
86
+
87
+ def load_dataset(
88
+ self, name: str, path: Path | str, id_delimiters: list[str] | None = None
89
+ ) -> None:
90
+ """Load a dataset from file.
91
+
92
+ Args:
93
+ name: Name identifier for the dataset
94
+ path: Path to the data file (CSV/TSV)
95
+ id_delimiters: Optional list of delimiters to split IDs
96
+
97
+ Raises:
98
+ ValueError: If dataset name not found in id_columns mapping
99
+ ValueError: If specified ID column not found in dataset
100
+ ValueError: If file format not supported
101
+ """
102
+ name = str(name)
103
+ if name not in self.id_columns:
104
+ raise ValueError(
105
+ f"No ID column mapping provided for dataset '{name}'. "
106
+ f"Available mappings: {', '.join(self.id_columns.keys())}"
107
+ )
108
+
109
+ path_str = str(path)
110
+ if not (path_str.endswith(".csv") or path_str.endswith(".tsv")):
111
+ raise ValueError("File must be .csv or .tsv format")
112
+
113
+ sep = "\t" if path_str.endswith(".tsv") else ","
114
+
115
+ try:
116
+ df = pd.read_csv(path, sep=sep)
117
+ except Exception as e:
118
+ raise ValueError(f"Failed to load dataset: {str(e)}")
119
+
120
+ df.columns = df.columns.astype(str)
121
+
122
+ id_col = self.id_columns[name]
123
+ if id_col not in df.columns:
124
+ raise ValueError(
125
+ f"ID column '{id_col}' not found in dataset '{name}'. "
126
+ f"Available columns: {', '.join(df.columns)}"
127
+ )
128
+
129
+ self.datasets[name] = df
130
+
131
+ if id_delimiters:
132
+ self.id_delimiters[name] = id_delimiters
133
+
134
+ def _split_identifier(self, identifier: str, delimiters: list[str]) -> set[str]:
135
+ """Split an identifier using multiple delimiters.
136
+
137
+ Args:
138
+ identifier: String to split
139
+ delimiters: List of delimiter strings
140
+
141
+ Returns:
142
+ Set of split values
143
+ """
144
+ if not delimiters:
145
+ return {identifier.strip()}
146
+
147
+ # Start with initial split on first delimiter
148
+ result = {v.strip() for v in identifier.split(delimiters[0])}
149
+
150
+ # Apply remaining delimiters
151
+ for delimiter in delimiters[1:]:
152
+ new_values: set[str] = set()
153
+ for value in result:
154
+ new_values.update(v.strip() for v in value.split(delimiter))
155
+ result.update(new_values)
156
+
157
+ # Remove any empty strings that might have been created
158
+ result.discard("")
159
+
160
+ return result
161
+
162
+ def get_sets(self) -> dict[str, set[str]]:
163
+ """Get sets of IDs from each dataset.
164
+
165
+ Returns:
166
+ Dictionary mapping dataset names to their sets of unique IDs
167
+
168
+ Raises:
169
+ ValueError: If no datasets have been loaded
170
+ """
171
+ if not self.datasets:
172
+ raise ValueError("No datasets have been loaded")
173
+
174
+ sets: dict[str, set[str]] = {}
175
+
176
+ for name, df in self.datasets.items():
177
+ id_col = self.id_columns[name]
178
+ # Convert to strings and handle NaN values
179
+ valid_ids = df[id_col].dropna().astype(str).unique()
180
+
181
+ # Apply splitting if delimiters specified
182
+ if name in self.id_delimiters:
183
+ split_values = set()
184
+ for value in valid_ids:
185
+ split_values.update(
186
+ self._split_identifier(value, self.id_delimiters[name])
187
+ )
188
+ sets[name] = split_values
189
+ else:
190
+ sets[name] = set(valid_ids)
191
+
192
+ return sets
193
+
194
+ def analyze(self) -> Dict[str, Any]:
195
+ """Perform set analysis across all datasets."""
196
+ sets = self.get_sets()
197
+
198
+ if not sets:
199
+ return {}
200
+
201
+ set_names = list(sets.keys())
202
+ n_sets = len(set_names)
203
+
204
+ results: Dict[str, Any] = {
205
+ "datasets": {
206
+ name: {"size": len(s), "sample_ids": list(sorted(s))[:5]}
207
+ for name, s in sets.items()
208
+ },
209
+ "intersections": [],
210
+ "unique": {},
211
+ }
212
+
213
+ # Calculate unique elements and intersections
214
+ for i in range(n_sets):
215
+ set_i = sets[set_names[i]]
216
+
217
+ others = set().union(*[s for n, s in sets.items() if n != set_names[i]])
218
+ unique = set_i - others
219
+ results["unique"][set_names[i]] = {
220
+ "size": len(unique),
221
+ "sample_ids": list(sorted(unique))[:5],
222
+ }
223
+
224
+ for j in range(i + 1, n_sets):
225
+ set_j = sets[set_names[j]]
226
+ intersection = set_i & set_j
227
+
228
+ if intersection:
229
+ results["intersections"].append(
230
+ {
231
+ "sets": [set_names[i], set_names[j]],
232
+ "size": len(intersection),
233
+ "sample_ids": list(sorted(intersection))[:5],
234
+ }
235
+ )
236
+
237
+ if n_sets > 2:
238
+ total_intersection = set.intersection(*sets.values())
239
+ if total_intersection:
240
+ results["intersections"].append(
241
+ {
242
+ "sets": list(sets.keys()),
243
+ "size": len(total_intersection),
244
+ "sample_ids": list(sorted(total_intersection))[:5],
245
+ }
246
+ )
247
+
248
+ return results
249
+
250
+ def plot_venn(self, output_path: str | None = None) -> None:
251
+ """Generate a Venn diagram visualization.
252
+
253
+ Args:
254
+ output_path: Optional path to save the plot. If None, displays plot.
255
+
256
+ Raises:
257
+ ValueError: If number of datasets is not 2 or 3.
258
+ """
259
+ if len(self.datasets) not in {2, 3}:
260
+ raise ValueError("Venn diagrams are only supported for 2 or 3 sets")
261
+
262
+ # Get sets and labels
263
+ sets = self.get_sets()
264
+ non_empty_sets = {k: v for k, v in sets.items() if v}
265
+
266
+ # Convert to list of sets for venn
267
+ data = non_empty_sets
268
+
269
+ plt.figure(figsize=(10, 10))
270
+ venn.venn(data)
271
+
272
+ if output_path:
273
+ plt.savefig(output_path)
274
+ plt.close()
275
+
276
+ def plot_upset(self, output_path: Optional[str] = None) -> None:
277
+ """Generate UpSet plot visualization."""
278
+ from upsetplot import from_contents
279
+
280
+ sets = self.get_sets()
281
+ data = from_contents(sets)
282
+
283
+ plt.figure(figsize=(12, 6))
284
+ upsetplot(data)
285
+
286
+ if output_path:
287
+ plt.savefig(output_path)
288
+ else:
289
+ plt.show()
290
+
291
+ def generate_text_report(self, output_path: str | None = None) -> str:
292
+ """Generate detailed text report of set analysis results.
293
+
294
+ Args:
295
+ output_path: Optional path to save report to file
296
+
297
+ Returns:
298
+ Formatted text report
299
+ """
300
+ results = self.analyze()
301
+ sets = self.get_sets()
302
+
303
+ # Build report sections
304
+ lines = ["Set Analysis Report", "=" * 50, ""]
305
+
306
+ # Dataset sizes
307
+ lines.extend(["Dataset Sizes:", "-" * 15])
308
+ for name, items in sets.items():
309
+ lines.append(f"{name}: {len(items):,} unique identifiers")
310
+ lines.append("")
311
+
312
+ # Pairwise overlaps
313
+ lines.extend(["Pairwise Overlaps:", "-" * 20])
314
+ for i, (name1, set1) in enumerate(sets.items()):
315
+ for name2, set2 in list(sets.items())[i + 1 :]:
316
+ overlap = len(set1 & set2)
317
+ pct1 = (overlap / len(set1)) * 100
318
+ pct2 = (overlap / len(set2)) * 100
319
+ lines.append(f"{name1} ∩ {name2}:")
320
+ lines.append(f" {overlap:,} identifiers")
321
+ lines.append(f" {pct1:.1f}% of {name1}")
322
+ lines.append(f" {pct2:.1f}% of {name2}")
323
+ lines.append("")
324
+
325
+ # Unique elements
326
+ lines.extend(["Unique Elements:", "-" * 20])
327
+ for name, result in results["unique"].items():
328
+ lines.append(f"{name}: {result['size']:,} unique identifiers")
329
+ if result["sample_ids"]:
330
+ lines.append("Examples:")
331
+ for id in result["sample_ids"][:5]:
332
+ lines.append(f" - {id}")
333
+ lines.append("")
334
+
335
+ # Total intersection if more than 2 sets
336
+ if len(sets) > 2:
337
+ common = set.intersection(*sets.values())
338
+ lines.extend(["Common to All Sets:", "-" * 20])
339
+ lines.append(f"Total: {len(common):,} identifiers")
340
+ if common:
341
+ lines.append("Examples:")
342
+ for id in sorted(common)[:5]:
343
+ lines.append(f" - {id}")
344
+
345
+ report = "\n".join(lines)
346
+
347
+ if output_path:
348
+ with open(output_path, "w", encoding="utf-8") as f:
349
+ f.write(report)
350
+
351
+ return report
352
+
353
+ def get_merged_dataframe(self, delimiter: str = "|") -> pd.DataFrame:
354
+ """Create DataFrame of merged set data.
355
+
356
+ Args:
357
+ delimiter: Delimiter for source dataset names
358
+
359
+ Returns:
360
+ DataFrame with identifier and sources columns
361
+ """
362
+ sets = self.get_sets()
363
+
364
+ # Create merged data structure
365
+ merged_data: list[dict[str, str]] = []
366
+
367
+ # Process each unique identifier
368
+ all_ids = set().union(*sets.values())
369
+ for identifier in sorted(all_ids):
370
+ # Find which sets contain this identifier
371
+ sources = [name for name, items in sets.items() if identifier in items]
372
+ merged_data.append(
373
+ {"identifier": identifier, "sources": delimiter.join(sorted(sources))}
374
+ )
375
+
376
+ return pd.DataFrame(merged_data)
377
+
378
+ def export_merged_sets(
379
+ self, output_path: str, delimiter: str = "|"
380
+ ) -> pd.DataFrame:
381
+ """Export merged set data to CSV file and return as DataFrame.
382
+
383
+ Args:
384
+ output_path: Path to save merged CSV
385
+ delimiter: Delimiter for source dataset names
386
+
387
+ Returns:
388
+ DataFrame containing the merged data
389
+ """
390
+ df = self.get_merged_dataframe(delimiter)
391
+ df.to_csv(output_path, index=False)
392
+ return df
393
+
394
+ def generate_report(self, output_prefix: str) -> None:
395
+ """Generate a comprehensive analysis report.
396
+
397
+ Args:
398
+ output_prefix: Prefix for output files
399
+ """
400
+ # Generate plots
401
+ self.plot_venn(f"{output_prefix}_venn.png")
402
+ self.plot_upset(f"{output_prefix}_upset.png")
403
+
404
+ # Generate text report
405
+ self.generate_text_report(f"{output_prefix}_report.txt")
406
+
407
+ # Export merged sets
408
+ self.export_merged_sets(f"{output_prefix}_merged.csv")
409
+
410
+ def _clean_identifier(self, identifier: Any) -> set[str]:
411
+ """Clean and validate an identifier."""
412
+ if pd.isna(identifier):
413
+ return set()
414
+ # Convert to string and strip whitespace
415
+ return {str(identifier).strip()}
@@ -2,15 +2,23 @@
2
2
 
3
3
  import logging
4
4
  from dataclasses import dataclass
5
- from typing import Any, Optional, Sequence
5
+ from typing import Any, Optional
6
+ import requests
6
7
 
7
8
  # Add type ignore comment since libchebipy doesn't have type stubs
8
- import libchebipy # type: ignore[import-untyped]
9
- from libchebipy import ChebiEntity # type: ignore[import-untyped, unused-ignore]
9
+ from libchebipy import ChebiEntity, search as chebi_search # type: ignore[import-untyped, unused-ignore]
10
10
 
11
11
  logger = logging.getLogger(__name__)
12
12
 
13
13
 
14
+ @dataclass(frozen=True)
15
+ class ChEBIConfig:
16
+ """Configuration for ChEBI client."""
17
+
18
+ base_url: str = "https://www.ebi.ac.uk/ols/api"
19
+ timeout: int = 30
20
+
21
+
14
22
  @dataclass(frozen=True)
15
23
  class ChEBIResult:
16
24
  """Result from ChEBI entity lookup."""
@@ -31,6 +39,11 @@ class ChEBIError(Exception):
31
39
  class ChEBIClient:
32
40
  """Client for interacting with the ChEBI database."""
33
41
 
42
+ def __init__(self, config: Optional[ChEBIConfig] = None) -> None:
43
+ """Initialize the ChEBI client."""
44
+ self.config = config or ChEBIConfig()
45
+ self.session = requests.Session()
46
+
34
47
  def _get_safe_property(
35
48
  self, getter: Any, error_types: tuple[type[Exception], ...] = (Exception,)
36
49
  ) -> Any:
@@ -126,32 +139,36 @@ class ChEBIClient:
126
139
  logger.error("ChEBI entity lookup failed: %s", str(e))
127
140
  raise ChEBIError(f"Entity lookup failed: {str(e)}") from e
128
141
 
129
- def search_by_name(self, name: str, max_results: int = 5) -> Sequence[ChEBIResult]:
130
- """Search for ChEBI entities by name.
142
+ def search_by_name(
143
+ self, name: str, max_results: int | None = None
144
+ ) -> list[ChEBIResult] | None:
145
+ """Search ChEBI by a compound name, returning a list of ChEBIResult objects.
131
146
 
132
147
  Args:
133
- name: Name to search for
134
- max_results: Maximum number of results to return
148
+ name: The compound name to search for.
149
+ max_results: Limit on the number of results to return.
135
150
 
136
151
  Returns:
137
- List of ChEBIResult objects matching the search
138
-
139
- Raises:
140
- ChEBIError: If search fails
152
+ A list of ChEBIResult objects if successful, an empty list if no results,
153
+ or None if an unexpected error occurs.
141
154
  """
142
155
  try:
143
- entities = libchebipy.search(name)
144
- matches: list[ChEBIResult] = []
145
-
146
- for entity in entities[:max_results]:
156
+ entities = chebi_search(name)
157
+ if not entities:
158
+ return []
159
+
160
+ results: list[ChEBIResult] = []
161
+ for entity in entities:
162
+ if max_results is not None and len(results) >= max_results:
163
+ break
147
164
  try:
148
165
  result = self._get_entity_result(entity)
149
- matches.append(result)
150
- except Exception as e:
151
- logger.warning("Failed to process entity %s: %s", entity, str(e))
166
+ results.append(result)
167
+ except ChEBIError:
168
+ # Skip this entity if it cannot be processed
152
169
  continue
153
170
 
154
- return matches
171
+ return results[:max_results] if max_results is not None else results
155
172
  except Exception as e:
156
173
  logger.error("ChEBI search failed: %s", str(e))
157
- raise ChEBIError(f"Search failed: {str(e)}") from e
174
+ return None