biomapper 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {biomapper-0.2.0 → biomapper-0.3.0}/PKG-INFO +5 -1
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/__init__.py +3 -2
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/core/__init__.py +4 -0
- biomapper-0.3.0/biomapper/core/set_analysis.py +415 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/chebi_client.py +37 -20
- biomapper-0.3.0/biomapper/mapping/metabolite_name_mapper.py +569 -0
- biomapper-0.3.0/biomapper/mapping/refmet_client.py +222 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/pyproject.toml +7 -3
- biomapper-0.2.0/biomapper/mapping/metabolite_name_mapper.py +0 -179
- biomapper-0.2.0/biomapper/mapping/refmet_client.py +0 -113
- {biomapper-0.2.0 → biomapper-0.3.0}/LICENSE +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/README.md +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/core/protein_metadata_comparison.py +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/__init__.py +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/unichem_client.py +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/mapping/uniprot_focused_mapper.py +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/schemas/__init__.py +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/standardization/__init__.py +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/standardization/ramp_client.py +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/standardization/tutorial.ipynb +0 -0
- {biomapper-0.2.0 → biomapper-0.3.0}/biomapper/utils/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: biomapper
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: A unified Python toolkit for biological data harmonization and ontology mapping
|
|
5
5
|
Home-page: https://github.com/arpanauts/biomapper
|
|
6
6
|
License: MIT
|
|
@@ -22,11 +22,15 @@ Provides-Extra: api
|
|
|
22
22
|
Provides-Extra: full
|
|
23
23
|
Provides-Extra: viz
|
|
24
24
|
Requires-Dist: libChEBIpy (==1.0.10)
|
|
25
|
+
Requires-Dist: matplotlib (>=3.8.0,<4.0.0) ; extra == "viz" or extra == "full"
|
|
25
26
|
Requires-Dist: pandas (>=2.0.0,<3.0.0)
|
|
26
27
|
Requires-Dist: pyyaml (>=6.0.1,<7.0.0)
|
|
27
28
|
Requires-Dist: requests (>=2.25.1,<3.0.0)
|
|
29
|
+
Requires-Dist: seaborn (>=0.13.0,<0.14.0) ; extra == "viz" or extra == "full"
|
|
28
30
|
Requires-Dist: sqlalchemy (>=1.4.0,<2.0.0)
|
|
29
31
|
Requires-Dist: tqdm (>=4.66.1,<5.0.0)
|
|
32
|
+
Requires-Dist: upsetplot (>=0.8.0,<0.9.0) ; extra == "viz" or extra == "full"
|
|
33
|
+
Requires-Dist: venn (>=0.1.3,<0.2.0)
|
|
30
34
|
Project-URL: Documentation, https://github.com/arpanauts/biomapper/blob/main/README.md
|
|
31
35
|
Project-URL: Repository, https://github.com/arpanauts/biomapper
|
|
32
36
|
Description-Content-Type: text/markdown
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"""Biomapper package for biological data harmonization and ontology mapping."""
|
|
2
2
|
|
|
3
3
|
from .standardization import RaMPClient
|
|
4
|
+
from .core import SetAnalyzer
|
|
4
5
|
|
|
5
|
-
__version__ = "0.
|
|
6
|
-
__all__ = ["RaMPClient"]
|
|
6
|
+
__version__ = "0.3.0"
|
|
7
|
+
__all__ = ["RaMPClient", "SetAnalyzer"]
|
|
@@ -0,0 +1,415 @@
|
|
|
1
|
+
"""Module for performing set analysis on omic datasets."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any, Dict, List, Optional, TypedDict
|
|
6
|
+
import matplotlib.pyplot as plt
|
|
7
|
+
import pandas as pd
|
|
8
|
+
import venn # type: ignore
|
|
9
|
+
from upsetplot import plot as upsetplot # type: ignore
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class DatasetInfo(TypedDict):
|
|
15
|
+
"""Type definition for dataset information."""
|
|
16
|
+
|
|
17
|
+
size: int
|
|
18
|
+
sample_ids: List[str]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class IntersectionInfo(TypedDict):
|
|
22
|
+
"""Type definition for intersection information."""
|
|
23
|
+
|
|
24
|
+
sets: List[str]
|
|
25
|
+
size: int
|
|
26
|
+
sample_ids: List[str]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class AnalysisResults(TypedDict):
|
|
30
|
+
"""Type definition for analysis results."""
|
|
31
|
+
|
|
32
|
+
datasets: Dict[str, DatasetInfo]
|
|
33
|
+
intersections: List[IntersectionInfo]
|
|
34
|
+
unique: Dict[str, DatasetInfo]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class SetAnalyzer:
|
|
38
|
+
"""Analyzer for performing set comparisons across omic datasets."""
|
|
39
|
+
|
|
40
|
+
def __init__(self, id_columns: dict[str, str]) -> None:
|
|
41
|
+
"""Initialize the analyzer.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
id_columns: Mapping of dataset names to their ID column names
|
|
45
|
+
|
|
46
|
+
Raises:
|
|
47
|
+
ValueError: If id_columns is empty
|
|
48
|
+
"""
|
|
49
|
+
if not id_columns:
|
|
50
|
+
raise ValueError("id_columns mapping cannot be empty")
|
|
51
|
+
|
|
52
|
+
self.id_columns = {} # Initialize empty dict
|
|
53
|
+
self.datasets: dict[str, pd.DataFrame] = {}
|
|
54
|
+
self.id_delimiters: dict[str, list[str]] = {} # Change type to list[str]
|
|
55
|
+
# Use property setter for validation
|
|
56
|
+
for name, col in id_columns.items():
|
|
57
|
+
self.set_id_column(name, col)
|
|
58
|
+
|
|
59
|
+
def set_id_column(self, dataset: str, column: str) -> None:
|
|
60
|
+
"""Set ID column for a dataset with validation."""
|
|
61
|
+
if dataset in self.datasets:
|
|
62
|
+
if column not in self.datasets[dataset].columns:
|
|
63
|
+
raise ValueError(
|
|
64
|
+
f"Column '{column}' not found in dataset '{dataset}'. "
|
|
65
|
+
f"Available columns: {', '.join(self.datasets[dataset].columns)}"
|
|
66
|
+
)
|
|
67
|
+
self.id_columns[dataset] = column
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def id_columns(self) -> dict[str, str]:
|
|
71
|
+
return self._id_columns
|
|
72
|
+
|
|
73
|
+
@id_columns.setter
|
|
74
|
+
def id_columns(self, value: dict[str, str]) -> None:
|
|
75
|
+
self._id_columns = value
|
|
76
|
+
|
|
77
|
+
def __setitem__(self, key: str, value: str) -> None:
|
|
78
|
+
"""Set ID column for a dataset."""
|
|
79
|
+
if key not in self.datasets:
|
|
80
|
+
raise ValueError(f"Dataset '{key}' not found")
|
|
81
|
+
|
|
82
|
+
if value not in self.datasets[key].columns:
|
|
83
|
+
raise ValueError(f"Column '{value}' not found in dataset '{key}'")
|
|
84
|
+
|
|
85
|
+
self._id_columns[key] = value
|
|
86
|
+
|
|
87
|
+
def load_dataset(
|
|
88
|
+
self, name: str, path: Path | str, id_delimiters: list[str] | None = None
|
|
89
|
+
) -> None:
|
|
90
|
+
"""Load a dataset from file.
|
|
91
|
+
|
|
92
|
+
Args:
|
|
93
|
+
name: Name identifier for the dataset
|
|
94
|
+
path: Path to the data file (CSV/TSV)
|
|
95
|
+
id_delimiters: Optional list of delimiters to split IDs
|
|
96
|
+
|
|
97
|
+
Raises:
|
|
98
|
+
ValueError: If dataset name not found in id_columns mapping
|
|
99
|
+
ValueError: If specified ID column not found in dataset
|
|
100
|
+
ValueError: If file format not supported
|
|
101
|
+
"""
|
|
102
|
+
name = str(name)
|
|
103
|
+
if name not in self.id_columns:
|
|
104
|
+
raise ValueError(
|
|
105
|
+
f"No ID column mapping provided for dataset '{name}'. "
|
|
106
|
+
f"Available mappings: {', '.join(self.id_columns.keys())}"
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
path_str = str(path)
|
|
110
|
+
if not (path_str.endswith(".csv") or path_str.endswith(".tsv")):
|
|
111
|
+
raise ValueError("File must be .csv or .tsv format")
|
|
112
|
+
|
|
113
|
+
sep = "\t" if path_str.endswith(".tsv") else ","
|
|
114
|
+
|
|
115
|
+
try:
|
|
116
|
+
df = pd.read_csv(path, sep=sep)
|
|
117
|
+
except Exception as e:
|
|
118
|
+
raise ValueError(f"Failed to load dataset: {str(e)}")
|
|
119
|
+
|
|
120
|
+
df.columns = df.columns.astype(str)
|
|
121
|
+
|
|
122
|
+
id_col = self.id_columns[name]
|
|
123
|
+
if id_col not in df.columns:
|
|
124
|
+
raise ValueError(
|
|
125
|
+
f"ID column '{id_col}' not found in dataset '{name}'. "
|
|
126
|
+
f"Available columns: {', '.join(df.columns)}"
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
self.datasets[name] = df
|
|
130
|
+
|
|
131
|
+
if id_delimiters:
|
|
132
|
+
self.id_delimiters[name] = id_delimiters
|
|
133
|
+
|
|
134
|
+
def _split_identifier(self, identifier: str, delimiters: list[str]) -> set[str]:
|
|
135
|
+
"""Split an identifier using multiple delimiters.
|
|
136
|
+
|
|
137
|
+
Args:
|
|
138
|
+
identifier: String to split
|
|
139
|
+
delimiters: List of delimiter strings
|
|
140
|
+
|
|
141
|
+
Returns:
|
|
142
|
+
Set of split values
|
|
143
|
+
"""
|
|
144
|
+
if not delimiters:
|
|
145
|
+
return {identifier.strip()}
|
|
146
|
+
|
|
147
|
+
# Start with initial split on first delimiter
|
|
148
|
+
result = {v.strip() for v in identifier.split(delimiters[0])}
|
|
149
|
+
|
|
150
|
+
# Apply remaining delimiters
|
|
151
|
+
for delimiter in delimiters[1:]:
|
|
152
|
+
new_values: set[str] = set()
|
|
153
|
+
for value in result:
|
|
154
|
+
new_values.update(v.strip() for v in value.split(delimiter))
|
|
155
|
+
result.update(new_values)
|
|
156
|
+
|
|
157
|
+
# Remove any empty strings that might have been created
|
|
158
|
+
result.discard("")
|
|
159
|
+
|
|
160
|
+
return result
|
|
161
|
+
|
|
162
|
+
def get_sets(self) -> dict[str, set[str]]:
|
|
163
|
+
"""Get sets of IDs from each dataset.
|
|
164
|
+
|
|
165
|
+
Returns:
|
|
166
|
+
Dictionary mapping dataset names to their sets of unique IDs
|
|
167
|
+
|
|
168
|
+
Raises:
|
|
169
|
+
ValueError: If no datasets have been loaded
|
|
170
|
+
"""
|
|
171
|
+
if not self.datasets:
|
|
172
|
+
raise ValueError("No datasets have been loaded")
|
|
173
|
+
|
|
174
|
+
sets: dict[str, set[str]] = {}
|
|
175
|
+
|
|
176
|
+
for name, df in self.datasets.items():
|
|
177
|
+
id_col = self.id_columns[name]
|
|
178
|
+
# Convert to strings and handle NaN values
|
|
179
|
+
valid_ids = df[id_col].dropna().astype(str).unique()
|
|
180
|
+
|
|
181
|
+
# Apply splitting if delimiters specified
|
|
182
|
+
if name in self.id_delimiters:
|
|
183
|
+
split_values = set()
|
|
184
|
+
for value in valid_ids:
|
|
185
|
+
split_values.update(
|
|
186
|
+
self._split_identifier(value, self.id_delimiters[name])
|
|
187
|
+
)
|
|
188
|
+
sets[name] = split_values
|
|
189
|
+
else:
|
|
190
|
+
sets[name] = set(valid_ids)
|
|
191
|
+
|
|
192
|
+
return sets
|
|
193
|
+
|
|
194
|
+
def analyze(self) -> Dict[str, Any]:
|
|
195
|
+
"""Perform set analysis across all datasets."""
|
|
196
|
+
sets = self.get_sets()
|
|
197
|
+
|
|
198
|
+
if not sets:
|
|
199
|
+
return {}
|
|
200
|
+
|
|
201
|
+
set_names = list(sets.keys())
|
|
202
|
+
n_sets = len(set_names)
|
|
203
|
+
|
|
204
|
+
results: Dict[str, Any] = {
|
|
205
|
+
"datasets": {
|
|
206
|
+
name: {"size": len(s), "sample_ids": list(sorted(s))[:5]}
|
|
207
|
+
for name, s in sets.items()
|
|
208
|
+
},
|
|
209
|
+
"intersections": [],
|
|
210
|
+
"unique": {},
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
# Calculate unique elements and intersections
|
|
214
|
+
for i in range(n_sets):
|
|
215
|
+
set_i = sets[set_names[i]]
|
|
216
|
+
|
|
217
|
+
others = set().union(*[s for n, s in sets.items() if n != set_names[i]])
|
|
218
|
+
unique = set_i - others
|
|
219
|
+
results["unique"][set_names[i]] = {
|
|
220
|
+
"size": len(unique),
|
|
221
|
+
"sample_ids": list(sorted(unique))[:5],
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
for j in range(i + 1, n_sets):
|
|
225
|
+
set_j = sets[set_names[j]]
|
|
226
|
+
intersection = set_i & set_j
|
|
227
|
+
|
|
228
|
+
if intersection:
|
|
229
|
+
results["intersections"].append(
|
|
230
|
+
{
|
|
231
|
+
"sets": [set_names[i], set_names[j]],
|
|
232
|
+
"size": len(intersection),
|
|
233
|
+
"sample_ids": list(sorted(intersection))[:5],
|
|
234
|
+
}
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
if n_sets > 2:
|
|
238
|
+
total_intersection = set.intersection(*sets.values())
|
|
239
|
+
if total_intersection:
|
|
240
|
+
results["intersections"].append(
|
|
241
|
+
{
|
|
242
|
+
"sets": list(sets.keys()),
|
|
243
|
+
"size": len(total_intersection),
|
|
244
|
+
"sample_ids": list(sorted(total_intersection))[:5],
|
|
245
|
+
}
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
return results
|
|
249
|
+
|
|
250
|
+
def plot_venn(self, output_path: str | None = None) -> None:
|
|
251
|
+
"""Generate a Venn diagram visualization.
|
|
252
|
+
|
|
253
|
+
Args:
|
|
254
|
+
output_path: Optional path to save the plot. If None, displays plot.
|
|
255
|
+
|
|
256
|
+
Raises:
|
|
257
|
+
ValueError: If number of datasets is not 2 or 3.
|
|
258
|
+
"""
|
|
259
|
+
if len(self.datasets) not in {2, 3}:
|
|
260
|
+
raise ValueError("Venn diagrams are only supported for 2 or 3 sets")
|
|
261
|
+
|
|
262
|
+
# Get sets and labels
|
|
263
|
+
sets = self.get_sets()
|
|
264
|
+
non_empty_sets = {k: v for k, v in sets.items() if v}
|
|
265
|
+
|
|
266
|
+
# Convert to list of sets for venn
|
|
267
|
+
data = non_empty_sets
|
|
268
|
+
|
|
269
|
+
plt.figure(figsize=(10, 10))
|
|
270
|
+
venn.venn(data)
|
|
271
|
+
|
|
272
|
+
if output_path:
|
|
273
|
+
plt.savefig(output_path)
|
|
274
|
+
plt.close()
|
|
275
|
+
|
|
276
|
+
def plot_upset(self, output_path: Optional[str] = None) -> None:
|
|
277
|
+
"""Generate UpSet plot visualization."""
|
|
278
|
+
from upsetplot import from_contents
|
|
279
|
+
|
|
280
|
+
sets = self.get_sets()
|
|
281
|
+
data = from_contents(sets)
|
|
282
|
+
|
|
283
|
+
plt.figure(figsize=(12, 6))
|
|
284
|
+
upsetplot(data)
|
|
285
|
+
|
|
286
|
+
if output_path:
|
|
287
|
+
plt.savefig(output_path)
|
|
288
|
+
else:
|
|
289
|
+
plt.show()
|
|
290
|
+
|
|
291
|
+
def generate_text_report(self, output_path: str | None = None) -> str:
|
|
292
|
+
"""Generate detailed text report of set analysis results.
|
|
293
|
+
|
|
294
|
+
Args:
|
|
295
|
+
output_path: Optional path to save report to file
|
|
296
|
+
|
|
297
|
+
Returns:
|
|
298
|
+
Formatted text report
|
|
299
|
+
"""
|
|
300
|
+
results = self.analyze()
|
|
301
|
+
sets = self.get_sets()
|
|
302
|
+
|
|
303
|
+
# Build report sections
|
|
304
|
+
lines = ["Set Analysis Report", "=" * 50, ""]
|
|
305
|
+
|
|
306
|
+
# Dataset sizes
|
|
307
|
+
lines.extend(["Dataset Sizes:", "-" * 15])
|
|
308
|
+
for name, items in sets.items():
|
|
309
|
+
lines.append(f"{name}: {len(items):,} unique identifiers")
|
|
310
|
+
lines.append("")
|
|
311
|
+
|
|
312
|
+
# Pairwise overlaps
|
|
313
|
+
lines.extend(["Pairwise Overlaps:", "-" * 20])
|
|
314
|
+
for i, (name1, set1) in enumerate(sets.items()):
|
|
315
|
+
for name2, set2 in list(sets.items())[i + 1 :]:
|
|
316
|
+
overlap = len(set1 & set2)
|
|
317
|
+
pct1 = (overlap / len(set1)) * 100
|
|
318
|
+
pct2 = (overlap / len(set2)) * 100
|
|
319
|
+
lines.append(f"{name1} ∩ {name2}:")
|
|
320
|
+
lines.append(f" {overlap:,} identifiers")
|
|
321
|
+
lines.append(f" {pct1:.1f}% of {name1}")
|
|
322
|
+
lines.append(f" {pct2:.1f}% of {name2}")
|
|
323
|
+
lines.append("")
|
|
324
|
+
|
|
325
|
+
# Unique elements
|
|
326
|
+
lines.extend(["Unique Elements:", "-" * 20])
|
|
327
|
+
for name, result in results["unique"].items():
|
|
328
|
+
lines.append(f"{name}: {result['size']:,} unique identifiers")
|
|
329
|
+
if result["sample_ids"]:
|
|
330
|
+
lines.append("Examples:")
|
|
331
|
+
for id in result["sample_ids"][:5]:
|
|
332
|
+
lines.append(f" - {id}")
|
|
333
|
+
lines.append("")
|
|
334
|
+
|
|
335
|
+
# Total intersection if more than 2 sets
|
|
336
|
+
if len(sets) > 2:
|
|
337
|
+
common = set.intersection(*sets.values())
|
|
338
|
+
lines.extend(["Common to All Sets:", "-" * 20])
|
|
339
|
+
lines.append(f"Total: {len(common):,} identifiers")
|
|
340
|
+
if common:
|
|
341
|
+
lines.append("Examples:")
|
|
342
|
+
for id in sorted(common)[:5]:
|
|
343
|
+
lines.append(f" - {id}")
|
|
344
|
+
|
|
345
|
+
report = "\n".join(lines)
|
|
346
|
+
|
|
347
|
+
if output_path:
|
|
348
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
349
|
+
f.write(report)
|
|
350
|
+
|
|
351
|
+
return report
|
|
352
|
+
|
|
353
|
+
def get_merged_dataframe(self, delimiter: str = "|") -> pd.DataFrame:
|
|
354
|
+
"""Create DataFrame of merged set data.
|
|
355
|
+
|
|
356
|
+
Args:
|
|
357
|
+
delimiter: Delimiter for source dataset names
|
|
358
|
+
|
|
359
|
+
Returns:
|
|
360
|
+
DataFrame with identifier and sources columns
|
|
361
|
+
"""
|
|
362
|
+
sets = self.get_sets()
|
|
363
|
+
|
|
364
|
+
# Create merged data structure
|
|
365
|
+
merged_data: list[dict[str, str]] = []
|
|
366
|
+
|
|
367
|
+
# Process each unique identifier
|
|
368
|
+
all_ids = set().union(*sets.values())
|
|
369
|
+
for identifier in sorted(all_ids):
|
|
370
|
+
# Find which sets contain this identifier
|
|
371
|
+
sources = [name for name, items in sets.items() if identifier in items]
|
|
372
|
+
merged_data.append(
|
|
373
|
+
{"identifier": identifier, "sources": delimiter.join(sorted(sources))}
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
return pd.DataFrame(merged_data)
|
|
377
|
+
|
|
378
|
+
def export_merged_sets(
|
|
379
|
+
self, output_path: str, delimiter: str = "|"
|
|
380
|
+
) -> pd.DataFrame:
|
|
381
|
+
"""Export merged set data to CSV file and return as DataFrame.
|
|
382
|
+
|
|
383
|
+
Args:
|
|
384
|
+
output_path: Path to save merged CSV
|
|
385
|
+
delimiter: Delimiter for source dataset names
|
|
386
|
+
|
|
387
|
+
Returns:
|
|
388
|
+
DataFrame containing the merged data
|
|
389
|
+
"""
|
|
390
|
+
df = self.get_merged_dataframe(delimiter)
|
|
391
|
+
df.to_csv(output_path, index=False)
|
|
392
|
+
return df
|
|
393
|
+
|
|
394
|
+
def generate_report(self, output_prefix: str) -> None:
|
|
395
|
+
"""Generate a comprehensive analysis report.
|
|
396
|
+
|
|
397
|
+
Args:
|
|
398
|
+
output_prefix: Prefix for output files
|
|
399
|
+
"""
|
|
400
|
+
# Generate plots
|
|
401
|
+
self.plot_venn(f"{output_prefix}_venn.png")
|
|
402
|
+
self.plot_upset(f"{output_prefix}_upset.png")
|
|
403
|
+
|
|
404
|
+
# Generate text report
|
|
405
|
+
self.generate_text_report(f"{output_prefix}_report.txt")
|
|
406
|
+
|
|
407
|
+
# Export merged sets
|
|
408
|
+
self.export_merged_sets(f"{output_prefix}_merged.csv")
|
|
409
|
+
|
|
410
|
+
def _clean_identifier(self, identifier: Any) -> set[str]:
|
|
411
|
+
"""Clean and validate an identifier."""
|
|
412
|
+
if pd.isna(identifier):
|
|
413
|
+
return set()
|
|
414
|
+
# Convert to string and strip whitespace
|
|
415
|
+
return {str(identifier).strip()}
|
|
@@ -2,15 +2,23 @@
|
|
|
2
2
|
|
|
3
3
|
import logging
|
|
4
4
|
from dataclasses import dataclass
|
|
5
|
-
from typing import Any, Optional
|
|
5
|
+
from typing import Any, Optional
|
|
6
|
+
import requests
|
|
6
7
|
|
|
7
8
|
# Add type ignore comment since libchebipy doesn't have type stubs
|
|
8
|
-
import
|
|
9
|
-
from libchebipy import ChebiEntity # type: ignore[import-untyped, unused-ignore]
|
|
9
|
+
from libchebipy import ChebiEntity, search as chebi_search # type: ignore[import-untyped, unused-ignore]
|
|
10
10
|
|
|
11
11
|
logger = logging.getLogger(__name__)
|
|
12
12
|
|
|
13
13
|
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class ChEBIConfig:
|
|
16
|
+
"""Configuration for ChEBI client."""
|
|
17
|
+
|
|
18
|
+
base_url: str = "https://www.ebi.ac.uk/ols/api"
|
|
19
|
+
timeout: int = 30
|
|
20
|
+
|
|
21
|
+
|
|
14
22
|
@dataclass(frozen=True)
|
|
15
23
|
class ChEBIResult:
|
|
16
24
|
"""Result from ChEBI entity lookup."""
|
|
@@ -31,6 +39,11 @@ class ChEBIError(Exception):
|
|
|
31
39
|
class ChEBIClient:
|
|
32
40
|
"""Client for interacting with the ChEBI database."""
|
|
33
41
|
|
|
42
|
+
def __init__(self, config: Optional[ChEBIConfig] = None) -> None:
|
|
43
|
+
"""Initialize the ChEBI client."""
|
|
44
|
+
self.config = config or ChEBIConfig()
|
|
45
|
+
self.session = requests.Session()
|
|
46
|
+
|
|
34
47
|
def _get_safe_property(
|
|
35
48
|
self, getter: Any, error_types: tuple[type[Exception], ...] = (Exception,)
|
|
36
49
|
) -> Any:
|
|
@@ -126,32 +139,36 @@ class ChEBIClient:
|
|
|
126
139
|
logger.error("ChEBI entity lookup failed: %s", str(e))
|
|
127
140
|
raise ChEBIError(f"Entity lookup failed: {str(e)}") from e
|
|
128
141
|
|
|
129
|
-
def search_by_name(
|
|
130
|
-
|
|
142
|
+
def search_by_name(
|
|
143
|
+
self, name: str, max_results: int | None = None
|
|
144
|
+
) -> list[ChEBIResult] | None:
|
|
145
|
+
"""Search ChEBI by a compound name, returning a list of ChEBIResult objects.
|
|
131
146
|
|
|
132
147
|
Args:
|
|
133
|
-
name:
|
|
134
|
-
max_results:
|
|
148
|
+
name: The compound name to search for.
|
|
149
|
+
max_results: Limit on the number of results to return.
|
|
135
150
|
|
|
136
151
|
Returns:
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
Raises:
|
|
140
|
-
ChEBIError: If search fails
|
|
152
|
+
A list of ChEBIResult objects if successful, an empty list if no results,
|
|
153
|
+
or None if an unexpected error occurs.
|
|
141
154
|
"""
|
|
142
155
|
try:
|
|
143
|
-
entities =
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
156
|
+
entities = chebi_search(name)
|
|
157
|
+
if not entities:
|
|
158
|
+
return []
|
|
159
|
+
|
|
160
|
+
results: list[ChEBIResult] = []
|
|
161
|
+
for entity in entities:
|
|
162
|
+
if max_results is not None and len(results) >= max_results:
|
|
163
|
+
break
|
|
147
164
|
try:
|
|
148
165
|
result = self._get_entity_result(entity)
|
|
149
|
-
|
|
150
|
-
except
|
|
151
|
-
|
|
166
|
+
results.append(result)
|
|
167
|
+
except ChEBIError:
|
|
168
|
+
# Skip this entity if it cannot be processed
|
|
152
169
|
continue
|
|
153
170
|
|
|
154
|
-
return
|
|
171
|
+
return results[:max_results] if max_results is not None else results
|
|
155
172
|
except Exception as e:
|
|
156
173
|
logger.error("ChEBI search failed: %s", str(e))
|
|
157
|
-
|
|
174
|
+
return None
|