PFASGroups 3.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- HalogenGroups/__init__.py +246 -0
- PFASGroups/ComponentsSolverModel.py +977 -0
- PFASGroups/HalogenGroupModel.py +810 -0
- PFASGroups/PFASDefinitionModel.py +393 -0
- PFASGroups/PFASEmbeddings.py +3315 -0
- PFASGroups/__init__.py +21 -0
- PFASGroups/cli.py +618 -0
- PFASGroups/core.py +415 -0
- PFASGroups/data/Halogen_groups_smarts.json +9024 -0
- PFASGroups/data/PFAS_definitions_smarts.json +170 -0
- PFASGroups/data/component_smarts.json +4 -0
- PFASGroups/data/component_smarts_halogens.json +142 -0
- PFASGroups/data/diatomic_bonds_dict.json +12802 -0
- PFASGroups/draw_mols.py +374 -0
- PFASGroups/embeddings.py +150 -0
- PFASGroups/fragmentation.py +548 -0
- PFASGroups/generate_homologues.py +256 -0
- PFASGroups/generate_mol.py +656 -0
- PFASGroups/generate_paper_figures.py +266 -0
- PFASGroups/getter.py +111 -0
- PFASGroups/homologue_series.py +473 -0
- PFASGroups/parser.py +942 -0
- PFASGroups/prioritise.py +439 -0
- pfasgroups-3.2.2.dist-info/METADATA +724 -0
- pfasgroups-3.2.2.dist-info/RECORD +28 -0
- pfasgroups-3.2.2.dist-info/WHEEL +5 -0
- pfasgroups-3.2.2.dist-info/entry_points.txt +3 -0
- pfasgroups-3.2.2.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,810 @@
|
|
|
1
|
+
from rdkit import Chem
|
|
2
|
+
from rdkit.Chem.rdMolDescriptors import CalcMolFormula
|
|
3
|
+
import re
|
|
4
|
+
from .core import mol_to_nx, add_componentSmarts
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class HalogenGroup():
|
|
10
|
+
"""Model class representing a specific halogenated functional group with structural patterns.
|
|
11
|
+
|
|
12
|
+
A HalogenGroup defines a specific halogenated functional group using SMARTS patterns,
|
|
13
|
+
component path types, and molecular formula constraints. Groups are used to classify
|
|
14
|
+
molecules into specific categories (e.g., "Perfluoroalkyl carboxylic acid").
|
|
15
|
+
|
|
16
|
+
Attributes
|
|
17
|
+
----------
|
|
18
|
+
id : int
|
|
19
|
+
Unique identifier for this Halogen group
|
|
20
|
+
name : str
|
|
21
|
+
Human-readable group name (e.g., "Perfluoroalkyl carboxylic acid")
|
|
22
|
+
smarts : Chem.Mol or None
|
|
23
|
+
SMARTS patterns (compiled RDKit molecule) for functional group detection.
|
|
24
|
+
None if group is defined by componentSmarts alone.
|
|
25
|
+
componentSmarts : list, str or None
|
|
26
|
+
componentForm: str or None
|
|
27
|
+
componentHalogens: list, str or None
|
|
28
|
+
componentSaturation: str or None (-> both)
|
|
29
|
+
max_dist_from_comp : int
|
|
30
|
+
Maximum graph distance (number of bonds) from fluorinated component to functional group.
|
|
31
|
+
When > 0, extends component search radius to find nearby functional groups.
|
|
32
|
+
linker_smarts : Chem.Mol or None
|
|
33
|
+
Compiled SMARTS pattern for validating linker atoms between fluorinated component
|
|
34
|
+
and functional group. When None (default), no restriction is applied to linker atoms.
|
|
35
|
+
Only used when max_dist_from_comp > 0.
|
|
36
|
+
constraints : dict
|
|
37
|
+
Molecular formula constraints with keys:
|
|
38
|
+
- 'only': Elements that must be present exclusively (e.g., ['C', 'F', 'O'])
|
|
39
|
+
- 'gte': Minimum element counts (e.g., {'C': 2})
|
|
40
|
+
- 'lte': Maximum element counts (e.g., {'O': 2})
|
|
41
|
+
- 'eq': Exact element counts (e.g., {'N': 1})
|
|
42
|
+
- 'rel': Relational constraints (e.g., {'O': {'atoms': ['C'], 'div': 2, 'add': 0}})
|
|
43
|
+
|
|
44
|
+
Examples
|
|
45
|
+
--------
|
|
46
|
+
>>> # Perfluoroalkyl carboxylic acid: R_F-COOH
|
|
47
|
+
>>> pfaa = HalogenGroup(
|
|
48
|
+
... id=1,
|
|
49
|
+
... name="Perfluoroalkyl carboxylic acid",
|
|
50
|
+
... smarts={"C(=O)O":1}, # Carboxylic acid group
|
|
51
|
+
... componentSmarts="Perfluoroalkyl",
|
|
52
|
+
... constraints={"only": ["C", "F", "O", "H"]},
|
|
53
|
+
... max_dist_from_comp=0,
|
|
54
|
+
... linker_smarts=None
|
|
55
|
+
... )
|
|
56
|
+
|
|
57
|
+
Notes
|
|
58
|
+
-----
|
|
59
|
+
- SMARTS patterns are compiled on initialization for efficient matching
|
|
60
|
+
- Constraints are validated when checking if a molecule belongs to this group
|
|
61
|
+
- max_dist_from_comp allows finding functional groups connected via non-fluorinated linkers
|
|
62
|
+
- linker_smarts restricts which atoms can be in the path between component and functional group
|
|
63
|
+
"""
|
|
64
|
+
@add_componentSmarts()
|
|
65
|
+
def __init__(self, id, name,**kwargs):
|
|
66
|
+
self.id = id
|
|
67
|
+
self.name = name
|
|
68
|
+
smarts = kwargs.get('smarts',{})
|
|
69
|
+
# Save original SMARTS strings for atom counting
|
|
70
|
+
if smarts and len(smarts) > 0:
|
|
71
|
+
self.smarts_str, self.smarts_count = zip(*smarts.items())
|
|
72
|
+
else:
|
|
73
|
+
self.smarts_str = None
|
|
74
|
+
self.smarts_count = None
|
|
75
|
+
self.smarts = [] if self.smarts_str else None
|
|
76
|
+
self.componentSmarts = kwargs.get('componentSmarts',None)
|
|
77
|
+
self.componentSaturation = kwargs.get('componentSaturation',None)
|
|
78
|
+
self.componentHalogens = kwargs.get('componentHalogens', kwargs.get('componentHalogen', None))
|
|
79
|
+
self.componentForm = kwargs.get("componentForm", None)
|
|
80
|
+
self._comp_type_to_halogen = {} # populated by set_component_smarts
|
|
81
|
+
self._comp_type_to_constraints = {} # populated by set_component_smarts
|
|
82
|
+
self.set_component_smarts(kwargs.get('componentSmartss', {}))
|
|
83
|
+
self.excludeHalogens = kwargs.get('excludeHalogens', None)
|
|
84
|
+
self.max_dist_from_comp = kwargs.get('max_dist_from_comp', 0)
|
|
85
|
+
# Compile linker_smarts pattern if provided
|
|
86
|
+
linker_smarts_str = kwargs.get('linker_smarts', None)
|
|
87
|
+
self.linker_smarts = None
|
|
88
|
+
if linker_smarts_str is not None:
|
|
89
|
+
try:
|
|
90
|
+
self.linker_smarts = Chem.MolFromSmarts(linker_smarts_str)
|
|
91
|
+
self.linker_smarts.UpdatePropertyCache()
|
|
92
|
+
Chem.GetSymmSSSR(self.linker_smarts)
|
|
93
|
+
self.linker_smarts.GetRingInfo().NumRings()
|
|
94
|
+
except:
|
|
95
|
+
raise ValueError(f"Invalid linker_smarts pattern '{linker_smarts_str}' for HalogenGroup '{self.name}' (ID: {self.id})")
|
|
96
|
+
if self.smarts_str is not None:
|
|
97
|
+
for smarts_pattern in self.smarts_str:
|
|
98
|
+
if smarts_pattern and smarts_pattern != "":
|
|
99
|
+
try:
|
|
100
|
+
smarts_mol = Chem.MolFromSmarts(smarts_pattern)
|
|
101
|
+
smarts_mol.UpdatePropertyCache()
|
|
102
|
+
Chem.GetSymmSSSR(smarts_mol)
|
|
103
|
+
smarts_mol.GetRingInfo().NumRings()
|
|
104
|
+
self.smarts.append(smarts_mol)
|
|
105
|
+
except:
|
|
106
|
+
raise ValueError(f"Invalid SMARTS pattern(s) for HalogenGroup '{self.name}' (ID: {self.id})")
|
|
107
|
+
self.constraints = kwargs.get('constraints',{})
|
|
108
|
+
# Precompute number of extra atoms in SMARTS patterns (beyond matched atom and H/F/Cl/Br/I)
|
|
109
|
+
self.smarts_extra_atoms = self._count_smarts_extra_atoms(self.smarts_str)
|
|
110
|
+
self.component_specific_extra_atoms = []
|
|
111
|
+
self.all_matches = []
|
|
112
|
+
self.compute = kwargs.get('compute',True)# whether the HalogenGroups needs to be parsed, or is an aggregate group, e.g. telomers
|
|
113
|
+
self.re_search = kwargs.get('re_search',None)# regex for aggregate groups, e.g. telomers
|
|
114
|
+
if self.re_search is not None:
|
|
115
|
+
try:
|
|
116
|
+
self.re_search = re.compile(self.re_search)
|
|
117
|
+
except Exception as e:
|
|
118
|
+
raise Exception(f"Error for agg Group {self.id}: {self.name}\n {e}")
|
|
119
|
+
self.test_dict = kwargs.get('test',None)# test dict for unit tests
|
|
120
|
+
def set_component_smarts(self, componentSmartss):
|
|
121
|
+
"""
|
|
122
|
+
Infers componentSmarts based on componentSmarts, componentSaturation, componentForm and componentHalogen
|
|
123
|
+
"""
|
|
124
|
+
if not componentSmartss:
|
|
125
|
+
return
|
|
126
|
+
# Build comp_type → halogen and comp_type → constraints mappings
|
|
127
|
+
self._comp_type_to_halogen = {
|
|
128
|
+
k: v['halogen']
|
|
129
|
+
for k, v in componentSmartss.items()
|
|
130
|
+
if isinstance(v, dict) and 'halogen' in v
|
|
131
|
+
}
|
|
132
|
+
self._comp_type_to_constraints = {
|
|
133
|
+
k: v['constraints']
|
|
134
|
+
for k, v in componentSmartss.items()
|
|
135
|
+
if isinstance(v, dict) and v.get('constraints')
|
|
136
|
+
}
|
|
137
|
+
# if componentSmarts is None, or is not in (entirely) in the available componentSmartss
|
|
138
|
+
if self.componentSmarts is None or (isinstance(self.componentSmarts,list) and not set(self.componentSmarts).issubset(componentSmartss.keys()) or (isinstance(self.componentSmarts, str) and not self.componentSmarts in componentSmartss.keys())):
|
|
139
|
+
new_CS = []
|
|
140
|
+
try:
|
|
141
|
+
componentSmarts_dict = {}
|
|
142
|
+
for k,v in componentSmartss.items():
|
|
143
|
+
if not isinstance(v, dict):
|
|
144
|
+
continue
|
|
145
|
+
halogen = v.get('halogen')
|
|
146
|
+
form = v.get('form')
|
|
147
|
+
saturation = v.get('saturation')
|
|
148
|
+
if halogen is None or form is None or saturation is None:
|
|
149
|
+
continue
|
|
150
|
+
componentSmarts_dict.setdefault(halogen,{}).setdefault(form,{})[saturation]=k
|
|
151
|
+
except Exception as e:
|
|
152
|
+
raise ValueError(f"Error processing componentSmartss for HalogenGroup '{self.name}' (ID: {self.id}): {e}")
|
|
153
|
+
if not componentSmarts_dict:
|
|
154
|
+
# No metadata available to infer component SMARTS
|
|
155
|
+
return
|
|
156
|
+
# prepare halogens (accepts list, str or None)
|
|
157
|
+
if isinstance(self.componentHalogens,list) and len(set(self.componentHalogens).intersection(['F','Cl','Br','I'])) == 0:
|
|
158
|
+
raise ValueError(f"Invalid componentHalogens for HalogenGroup '{self.name}' (ID: {self.id})")
|
|
159
|
+
if self.componentHalogens is None:
|
|
160
|
+
self.componentHalogens = ['F','Cl','Br','I']
|
|
161
|
+
elif isinstance(self.componentHalogens, str) and self.componentHalogens in ['F','Cl','Br','I']:
|
|
162
|
+
self.componentHalogens = [self.componentHalogens]
|
|
163
|
+
if self.componentSaturation is None:
|
|
164
|
+
self.componentSaturation = ['per','poly']
|
|
165
|
+
elif isinstance(self.componentSaturation, str) and self.componentSaturation in ['per','poly','both']:
|
|
166
|
+
self.componentSaturation = ['per','poly'] if self.componentSaturation == 'both' else [self.componentSaturation]
|
|
167
|
+
elif isinstance(self.componentSaturation, list):
|
|
168
|
+
self.componentSaturation = self.componentSaturation
|
|
169
|
+
else:
|
|
170
|
+
raise ValueError(f"Invalid componentSaturation for HalogenGroup '{self.name}' (ID: {self.id}), expected 'per', 'poly', 'both' or null, got {self.componentSaturation}")
|
|
171
|
+
if self.componentForm is None:
|
|
172
|
+
self.componentForm = 'alkyl'
|
|
173
|
+
for halogen in self.componentHalogens:
|
|
174
|
+
if halogen not in componentSmarts_dict:
|
|
175
|
+
raise ValueError(f"No component SMARTS available for halogen '{halogen}' in HalogenGroup '{self.name}' (ID: {self.id})")
|
|
176
|
+
if self.componentForm not in componentSmarts_dict[halogen]:
|
|
177
|
+
raise ValueError(f"No component SMARTS available for form '{self.componentForm}' and halogen '{halogen}' in HalogenGroup '{self.name}' (ID: {self.id})")
|
|
178
|
+
for saturation in self.componentSaturation:
|
|
179
|
+
if saturation not in componentSmarts_dict[halogen][self.componentForm]:
|
|
180
|
+
raise ValueError(f"No component SMARTS available for saturation '{saturation}', form '{self.componentForm}', halogen '{halogen}' in HalogenGroup '{self.name}' (ID: {self.id})")
|
|
181
|
+
new_CS.append(componentSmarts_dict[halogen][self.componentForm][saturation])
|
|
182
|
+
self.componentSmarts = new_CS
|
|
183
|
+
|
|
184
|
+
def set_componentSmarts(self, componentSmartss):
|
|
185
|
+
"""Backward-compatible alias for set_component_smarts."""
|
|
186
|
+
return self.set_component_smarts(componentSmartss)
|
|
187
|
+
def _count_smarts_extra_atoms(self, smarts_str):
|
|
188
|
+
"""Count number of extra carbon atoms in functional group beyond what's captured by component.
|
|
189
|
+
|
|
190
|
+
Parameters
|
|
191
|
+
----------
|
|
192
|
+
smarts_str : str or None
|
|
193
|
+
Original SMARTS string before compilation
|
|
194
|
+
Returns
|
|
195
|
+
-------
|
|
196
|
+
int
|
|
197
|
+
Number of extra carbon atoms beyond the matched atom
|
|
198
|
+
|
|
199
|
+
Notes
|
|
200
|
+
-----
|
|
201
|
+
The component fraction calculation is now based on carbon atoms only:
|
|
202
|
+
1. Carbon atoms in component
|
|
203
|
+
2. Carbon atoms in SMARTS matches
|
|
204
|
+
3. Additional carbon atoms from SMARTS (this return value)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
For automatic counting (when manual_size is None), this returns 0 since we now focus only
|
|
208
|
+
on carbons and they are already counted in the component and SMARTS matches.
|
|
209
|
+
"""
|
|
210
|
+
PAT_c = re.compile(r'((C(?![adeflmnorsu]))|((?<![TAS])c)|(\#6))') # Match 'C' not followed by a letter, or c not preceded by T,A,S or #6
|
|
211
|
+
return [max(0,len(PAT_c.findall(s))-1) for s in smarts_str] if smarts_str is not None else None
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def __str__(self):
|
|
215
|
+
return self.name
|
|
216
|
+
|
|
217
|
+
@staticmethod
|
|
218
|
+
def _check_component_constraints(comp_formula, constraints):
|
|
219
|
+
"""Check a component's element formula against component-level constraints.
|
|
220
|
+
|
|
221
|
+
Parameters
|
|
222
|
+
----------
|
|
223
|
+
comp_formula : dict
|
|
224
|
+
``{element_symbol: count}`` for all atoms in the full component
|
|
225
|
+
(backbone + attached H/halogens).
|
|
226
|
+
constraints : dict
|
|
227
|
+
Component constraints with optional keys:
|
|
228
|
+
|
|
229
|
+
* ``'gte'`` – ``{element: min_count}``; component must have at least
|
|
230
|
+
*min_count* atoms of *element*.
|
|
231
|
+
* ``'exclude'`` – list of element symbols that must be absent from
|
|
232
|
+
the component.
|
|
233
|
+
|
|
234
|
+
Returns
|
|
235
|
+
-------
|
|
236
|
+
bool
|
|
237
|
+
"""
|
|
238
|
+
for elem, n in constraints.get('gte', {}).items():
|
|
239
|
+
if comp_formula.get(elem, 0) < n:
|
|
240
|
+
return False
|
|
241
|
+
for elem in constraints.get('exclude', []):
|
|
242
|
+
if comp_formula.get(elem, 0) > 0:
|
|
243
|
+
return False
|
|
244
|
+
return True
|
|
245
|
+
|
|
246
|
+
def constraint_gte(self, formula_dict):
|
|
247
|
+
"""Check 'greater than or equal' constraints on element counts.
|
|
248
|
+
|
|
249
|
+
Parameters
|
|
250
|
+
----------
|
|
251
|
+
formula_dict : dict
|
|
252
|
+
Molecular formula as {element: count} dictionary
|
|
253
|
+
|
|
254
|
+
Returns
|
|
255
|
+
-------
|
|
256
|
+
bool
|
|
257
|
+
True if all 'gte' constraints are satisfied, False otherwise
|
|
258
|
+
|
|
259
|
+
Examples
|
|
260
|
+
--------
|
|
261
|
+
>>> # Requires at least 2 carbons and 3 fluorines
|
|
262
|
+
>>> group.constraints = {'gte': {'C': 2, 'F': 3}}
|
|
263
|
+
>>> group.constraint_gte({'C': 3, 'F': 5, 'O': 1}) # True
|
|
264
|
+
>>> group.constraint_gte({'C': 1, 'F': 5, 'O': 1}) # False (C < 2)
|
|
265
|
+
"""
|
|
266
|
+
success = True
|
|
267
|
+
for e,n in self.constraints.get('gte',{}).items():
|
|
268
|
+
success = success and formula_dict.get(e,0)>=n
|
|
269
|
+
return success
|
|
270
|
+
def constraint_lte(self, formula_dict):
|
|
271
|
+
"""Check 'less than or equal' constraints on element counts.
|
|
272
|
+
|
|
273
|
+
Parameters
|
|
274
|
+
----------
|
|
275
|
+
formula_dict : dict
|
|
276
|
+
Molecular formula as {element: count} dictionary
|
|
277
|
+
|
|
278
|
+
Returns
|
|
279
|
+
-------
|
|
280
|
+
bool
|
|
281
|
+
True if all 'lte' constraints are satisfied, False otherwise
|
|
282
|
+
|
|
283
|
+
Examples
|
|
284
|
+
--------
|
|
285
|
+
>>> # Requires at most 2 oxygens
|
|
286
|
+
>>> group.constraints = {'lte': {'O': 2}}
|
|
287
|
+
>>> group.constraint_lte({'C': 8, 'F': 15, 'O': 2}) # True
|
|
288
|
+
>>> group.constraint_lte({'C': 8, 'F': 15, 'O': 3}) # False (O > 2)
|
|
289
|
+
"""
|
|
290
|
+
success = True
|
|
291
|
+
for e,n in self.constraints.get('lte',{}).items():
|
|
292
|
+
success = success and formula_dict.get(e,0)<=n
|
|
293
|
+
return success
|
|
294
|
+
def constraint_eq(self, formula_dict):
|
|
295
|
+
"""Check 'equal to' constraints on element counts.
|
|
296
|
+
|
|
297
|
+
Parameters
|
|
298
|
+
----------
|
|
299
|
+
formula_dict : dict
|
|
300
|
+
Molecular formula as {element: count} dictionary
|
|
301
|
+
|
|
302
|
+
Returns
|
|
303
|
+
-------
|
|
304
|
+
bool
|
|
305
|
+
True if all 'eq' constraints are satisfied, False otherwise
|
|
306
|
+
|
|
307
|
+
Examples
|
|
308
|
+
--------
|
|
309
|
+
>>> # Requires exactly 1 nitrogen
|
|
310
|
+
>>> group.constraints = {'eq': {'N': 1}}
|
|
311
|
+
>>> group.constraint_eq({'C': 8, 'F': 15, 'N': 1}) # True
|
|
312
|
+
>>> group.constraint_eq({'C': 8, 'F': 15, 'N': 2}) # False (N != 1)
|
|
313
|
+
"""
|
|
314
|
+
success = True
|
|
315
|
+
for e,n in self.constraints.get('eq',{}).items():
|
|
316
|
+
success = success and formula_dict.get(e,0)==n
|
|
317
|
+
return success
|
|
318
|
+
def constraint_only(self, formula_dict):
|
|
319
|
+
"""Check 'only' constraint - molecule must contain only specified elements.
|
|
320
|
+
|
|
321
|
+
Parameters
|
|
322
|
+
----------
|
|
323
|
+
formula_dict : dict
|
|
324
|
+
Molecular formula as {element: count} dictionary
|
|
325
|
+
|
|
326
|
+
Returns
|
|
327
|
+
-------
|
|
328
|
+
bool
|
|
329
|
+
True if molecule contains only the allowed elements, False otherwise
|
|
330
|
+
|
|
331
|
+
Examples
|
|
332
|
+
--------
|
|
333
|
+
>>> # Molecule must contain only C, F, O, H
|
|
334
|
+
>>> group.constraints = {'only': ['C', 'F', 'O', 'H']}
|
|
335
|
+
>>> group.constraint_only({'C': 8, 'F': 15, 'O': 2, 'H': 1}) # True
|
|
336
|
+
>>> group.constraint_only({'C': 8, 'F': 15, 'O': 2, 'S': 1}) # False (S not allowed)
|
|
337
|
+
|
|
338
|
+
Notes
|
|
339
|
+
-----
|
|
340
|
+
Checks that sum of allowed elements equals total atoms in molecule.
|
|
341
|
+
"""
|
|
342
|
+
success = True
|
|
343
|
+
if 'only' in self.constraints.keys():
|
|
344
|
+
tot = sum(formula_dict.values())
|
|
345
|
+
nn = 0
|
|
346
|
+
for e in self.constraints['only']:
|
|
347
|
+
nn += formula_dict.get(e,0)
|
|
348
|
+
success = success and tot == nn
|
|
349
|
+
return success
|
|
350
|
+
def constraint_rel(self, formula_dict):
|
|
351
|
+
"""Check relational constraints between element counts.
|
|
352
|
+
|
|
353
|
+
Validates relationships of the form: count(element) = f(other_elements)
|
|
354
|
+
where f can include division, addition, and summing other element counts.
|
|
355
|
+
|
|
356
|
+
Parameters
|
|
357
|
+
----------
|
|
358
|
+
formula_dict : dict
|
|
359
|
+
Molecular formula as {element: count} dictionary
|
|
360
|
+
|
|
361
|
+
Returns
|
|
362
|
+
-------
|
|
363
|
+
bool
|
|
364
|
+
True if all relational constraints are satisfied, False otherwise
|
|
365
|
+
|
|
366
|
+
Notes
|
|
367
|
+
-----
|
|
368
|
+
Constraint Format::
|
|
369
|
+
|
|
370
|
+
'rel': {
|
|
371
|
+
'ElementA': {
|
|
372
|
+
'atoms': ['ElementB', 'ElementC'], # Elements to sum
|
|
373
|
+
'div': int, # Divisor (default 1)
|
|
374
|
+
'add': int, # Additive constant (default 0)
|
|
375
|
+
'add_atoms': ['ElementD'] # Additional elements to add
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
Formula: count(ElementA) = (sum(atoms) / div) + add + sum(add_atoms)
|
|
380
|
+
|
|
381
|
+
Examples
|
|
382
|
+
--------
|
|
383
|
+
>>> # Carbon count must equal half the fluorine count
|
|
384
|
+
>>> group.constraints = {'rel': {'C': {'atoms': ['F'], 'div': 2, 'add': 0}}}
|
|
385
|
+
>>> group.constraint_rel({'C': 4, 'F': 8, 'O': 2}) # True (4 == 8/2)
|
|
386
|
+
>>> group.constraint_rel({'C': 3, 'F': 8, 'O': 2}) # False (3 != 8/2)
|
|
387
|
+
|
|
388
|
+
>>> # Oxygen count must equal carbon count plus 1
|
|
389
|
+
>>> group.constraints = {'rel': {'O': {'atoms': ['C'], 'div': 1, 'add': 1}}}
|
|
390
|
+
>>> group.constraint_rel({'C': 3, 'F': 7, 'O': 4}) # True (4 == 3 + 1)
|
|
391
|
+
"""
|
|
392
|
+
success = True
|
|
393
|
+
for e,v in self.constraints.get('rel',{}).items():
|
|
394
|
+
n = sum([formula_dict.get(x,0) for x in v.get('atoms',[])])
|
|
395
|
+
success = success and formula_dict.get(e,0)==n/v.get('div',1)+v.get('add',0)+sum([formula_dict.get(x,0) for x in v.get('add_atoms',[])])
|
|
396
|
+
return success
|
|
397
|
+
def formula_dict_satisfies_constraints(self,formula_dict):
|
|
398
|
+
"""Check if a molecular formula satisfies all constraints for this PFAS group.
|
|
399
|
+
|
|
400
|
+
Evaluates all constraint types in order: relational → only → equal → lte → gte.
|
|
401
|
+
Stops evaluation at first failure for efficiency.
|
|
402
|
+
|
|
403
|
+
Parameters
|
|
404
|
+
----------
|
|
405
|
+
formula_dict : dict
|
|
406
|
+
Molecular formula as {element: count} dictionary (e.g., {'C': 8, 'F': 17, 'O': 2})
|
|
407
|
+
|
|
408
|
+
Returns
|
|
409
|
+
-------
|
|
410
|
+
bool
|
|
411
|
+
True if all constraints are satisfied, False if any constraint fails
|
|
412
|
+
|
|
413
|
+
Constraint Evaluation Order
|
|
414
|
+
---------------------------
|
|
415
|
+
1. Relational constraints ('rel') - element count relationships
|
|
416
|
+
2. 'Only' constraints - allowed elements
|
|
417
|
+
3. Equality constraints ('eq') - exact element counts
|
|
418
|
+
4. Upper bound constraints ('lte') - maximum element counts
|
|
419
|
+
5. Lower bound constraints ('gte') - minimum element counts
|
|
420
|
+
|
|
421
|
+
Examples
|
|
422
|
+
--------
|
|
423
|
+
>>> # Perfluoroalkyl carboxylic acid constraints
|
|
424
|
+
>>> group.constraints = {
|
|
425
|
+
... 'only': ['C', 'F', 'O', 'H'], # No other elements
|
|
426
|
+
... 'gte': {'C': 2, 'F': 3}, # At least 2 carbons, 3 fluorines
|
|
427
|
+
... 'eq': {'O': 2} # Exactly 2 oxygens
|
|
428
|
+
... }
|
|
429
|
+
>>> group.formula_dict_satisfies_constraints({'C': 8, 'F': 15, 'O': 2, 'H': 1})
|
|
430
|
+
True
|
|
431
|
+
>>> group.formula_dict_satisfies_constraints({'C': 8, 'F': 15, 'O': 3, 'H': 1})
|
|
432
|
+
False # Fails 'eq': {'O': 2}
|
|
433
|
+
|
|
434
|
+
Notes
|
|
435
|
+
-----
|
|
436
|
+
- Returns True immediately if no constraints are defined
|
|
437
|
+
- Short-circuits on first constraint failure for performance
|
|
438
|
+
- Constraint evaluation order is fixed for consistency
|
|
439
|
+
"""
|
|
440
|
+
if len(self.constraints.keys())==0:
|
|
441
|
+
return True
|
|
442
|
+
success = True
|
|
443
|
+
process = [None,self.constraint_rel,self.constraint_only,self.constraint_eq, self.constraint_lte, self.constraint_gte]
|
|
444
|
+
k = process.pop()
|
|
445
|
+
while success and k is not None:
|
|
446
|
+
success = k(formula_dict)
|
|
447
|
+
k = process.pop()
|
|
448
|
+
return success
|
|
449
|
+
def find_matched_atoms(self, mol):
|
|
450
|
+
"""Find all substructure matches of this PFAS group's SMARTS patterns in a molecule.
|
|
451
|
+
|
|
452
|
+
Parameters
|
|
453
|
+
----------
|
|
454
|
+
mol : Chem.Mol
|
|
455
|
+
RDKit molecule object to search for matches
|
|
456
|
+
|
|
457
|
+
Returns
|
|
458
|
+
-------
|
|
459
|
+
List[List[int]]
|
|
460
|
+
List of matches, where each match is a list of atom indices in the molecule
|
|
461
|
+
|
|
462
|
+
Notes
|
|
463
|
+
-----
|
|
464
|
+
- If no SMARTS patterns are defined, returns an empty list.
|
|
465
|
+
- Each SMARTS pattern is searched independently; matches from all patterns are combined.
|
|
466
|
+
"""
|
|
467
|
+
self.all_matches = []
|
|
468
|
+
self.subset = set()
|
|
469
|
+
if self.smarts is not None:
|
|
470
|
+
for smarts_mol,min_count in zip(self.smarts, self.smarts_count):
|
|
471
|
+
matches = mol.GetSubstructMatches(smarts_mol)
|
|
472
|
+
if len(matches) < min_count:
|
|
473
|
+
return False
|
|
474
|
+
if len(matches)>0:
|
|
475
|
+
# Keep matches as list to preserve duplicates when same atoms matched multiple times
|
|
476
|
+
# (e.g., dicarboxylic/disulfonic acids where both groups attach to same carbon)
|
|
477
|
+
self.all_matches.append(list(matches))
|
|
478
|
+
self.subset.update({y for x in matches for y in x if len(x)>0})
|
|
479
|
+
return True
|
|
480
|
+
def component_satisfies_all_smarts(self, component):
|
|
481
|
+
"""Check if a fluorinated component matches all SMARTS patterns of this PFAS group.
|
|
482
|
+
|
|
483
|
+
Parameters
|
|
484
|
+
----------
|
|
485
|
+
component : PFASComponent
|
|
486
|
+
PFASComponent object representing a fluorinated component in the molecule
|
|
487
|
+
|
|
488
|
+
Returns
|
|
489
|
+
-------
|
|
490
|
+
bool
|
|
491
|
+
True if the component matches all SMARTS patterns, False otherwise
|
|
492
|
+
|
|
493
|
+
Notes
|
|
494
|
+
-----
|
|
495
|
+
- If no SMARTS patterns are defined for this group, returns True.
|
|
496
|
+
- Each SMARTS pattern must have at least one match that includes the component's atom.
|
|
497
|
+
"""
|
|
498
|
+
atom_count = 0
|
|
499
|
+
for i, (matches, min_count) in enumerate(zip(self.all_matches,self.smarts_count)):
|
|
500
|
+
# Count how many SMARTS matches have overlap with this component
|
|
501
|
+
# matches is a set of tuples, each tuple represents one SMARTS match
|
|
502
|
+
component_set = set(component)
|
|
503
|
+
found = sum(1 for match_tuple in matches if any(atom_idx in component_set for atom_idx in match_tuple))
|
|
504
|
+
|
|
505
|
+
if found < min_count:
|
|
506
|
+
self.component_specific_extra_atoms.append(0)
|
|
507
|
+
return False
|
|
508
|
+
atom_count += found * self.smarts_extra_atoms[i]
|
|
509
|
+
self.component_specific_extra_atoms.append(atom_count)
|
|
510
|
+
return True
|
|
511
|
+
|
|
512
|
+
def find_alkyl_components(self, mol, component_solver, **kwargs):
|
|
513
|
+
"""Find fluorinated components in a molecule that match this PFAS group's criteria.
|
|
514
|
+
|
|
515
|
+
Parameters
|
|
516
|
+
----------
|
|
517
|
+
mol : Chem.Mol
|
|
518
|
+
RDKit molecule object to search
|
|
519
|
+
components : List[PFASComponent]
|
|
520
|
+
List of PFASComponent objects representing fluorinated components in the molecule
|
|
521
|
+
|
|
522
|
+
Returns
|
|
523
|
+
-------
|
|
524
|
+
List[PFASComponent]
|
|
525
|
+
List of PFASComponent objects that match this PFAS group's criteria
|
|
526
|
+
|
|
527
|
+
Notes
|
|
528
|
+
-----
|
|
529
|
+
- Matches are determined based on componentSmarts and max_dist_from_comp attributes.
|
|
530
|
+
- If componentSmarts is None, all components are considered.
|
|
531
|
+
- max_dist_from_comp allows extending the search radius for functional groups.
|
|
532
|
+
"""
|
|
533
|
+
if not self.find_matched_atoms(mol):
|
|
534
|
+
return 0, [], 0, []
|
|
535
|
+
|
|
536
|
+
# Clear component-specific extra atoms list for this matching attempt
|
|
537
|
+
self.component_specific_extra_atoms = []
|
|
538
|
+
|
|
539
|
+
if self.componentSmarts is None:
|
|
540
|
+
# If no componentSmarts specified, only check alkyl components (not cyclic)
|
|
541
|
+
# This ensures functional groups like carboxylic acid (group 33) are only
|
|
542
|
+
# detected when attached to perfluoroalkyl or polyfluoroalkyl chains,
|
|
543
|
+
# not when attached directly to cyclic structures
|
|
544
|
+
componentSmartss = []
|
|
545
|
+
for path_type, meta in component_solver.componentSmartss.items():
|
|
546
|
+
if isinstance(meta, dict) and meta.get('form') == 'cyclic':
|
|
547
|
+
continue
|
|
548
|
+
componentSmartss.append(path_type)
|
|
549
|
+
elif isinstance(self.componentSmarts, (list, tuple, set)):
|
|
550
|
+
# Only keep paths the solver knows about (respects halogen filtering)
|
|
551
|
+
available = set(component_solver.componentSmartss.keys())
|
|
552
|
+
componentSmartss = [cs for cs in self.componentSmarts if cs in available]
|
|
553
|
+
else:
|
|
554
|
+
componentSmartss = [self.componentSmarts]
|
|
555
|
+
|
|
556
|
+
# Preload components (optional, not strictly required for filtering)
|
|
557
|
+
components = []
|
|
558
|
+
for comp_type in componentSmartss:
|
|
559
|
+
comps = component_solver.get(comp_type, max_dist=self.max_dist_from_comp, default=[])
|
|
560
|
+
components.extend(comps)
|
|
561
|
+
|
|
562
|
+
# Filter components connected to the smarts and get augmented versions
|
|
563
|
+
augmented_matched_components = []
|
|
564
|
+
for _componentSmarts in componentSmartss:
|
|
565
|
+
extended_components = component_solver.get(_componentSmarts, self.max_dist_from_comp, [])
|
|
566
|
+
for i, comp in enumerate(extended_components):
|
|
567
|
+
# Check if this component is connected to SMARTS matches
|
|
568
|
+
if self.component_satisfies_all_smarts(comp):
|
|
569
|
+
augmented = component_solver.get_augmented_component(
|
|
570
|
+
_componentSmarts, self.max_dist_from_comp, i, self.subset, self.linker_smarts
|
|
571
|
+
)
|
|
572
|
+
# Accept augmented component if valid (linker validation already done in get_augmented_component)
|
|
573
|
+
if augmented is not None and len(augmented) > 0:
|
|
574
|
+
augmented_matched_components.append(
|
|
575
|
+
component_solver.get_matched_component_dict(augmented, self.subset, _componentSmarts, self, comp_id = i)
|
|
576
|
+
)
|
|
577
|
+
|
|
578
|
+
if len(augmented_matched_components) == 0:
|
|
579
|
+
return 0, [], 0, []
|
|
580
|
+
|
|
581
|
+
# Get all component sizes from all path types
|
|
582
|
+
all_components = list(set([comp for comps in augmented_matched_components for comp in comps]))
|
|
583
|
+
component_sizes = [len(x) for x in all_components]
|
|
584
|
+
|
|
585
|
+
self.all_matches = [] # Clear matches after use
|
|
586
|
+
self.component_specific_extra_atoms = []
|
|
587
|
+
return max([0] + component_sizes), component_sizes, len(all_components), augmented_matched_components
|
|
588
|
+
|
|
589
|
+
def find_aryl_components(self,mol, component_solver=None, **kwargs):
|
|
590
|
+
"""Find aryl components in a molecule with comprehensive metrics."""
|
|
591
|
+
matches = mol.GetSubstructMatches(self.smarts[0])
|
|
592
|
+
subset = [y for x in matches for y in x]
|
|
593
|
+
if len(subset)==0:
|
|
594
|
+
return 0, [], 0, []
|
|
595
|
+
|
|
596
|
+
components = component_solver._connected_components(subset)
|
|
597
|
+
component_sizes = [len(x) for x in components]
|
|
598
|
+
|
|
599
|
+
# Get molecular graph for metrics calculation
|
|
600
|
+
subset_set = set(subset)
|
|
601
|
+
|
|
602
|
+
# Convert components to the same format as other functions return with comprehensive metrics
|
|
603
|
+
matched_components = []
|
|
604
|
+
for comp in components:
|
|
605
|
+
matched_components.append(
|
|
606
|
+
component_solver.get_matched_component_dict(comp, subset_set, 'cyclic', self)
|
|
607
|
+
)
|
|
608
|
+
|
|
609
|
+
return max([0]+[len(x) for x in components]), component_sizes, len(components), matched_components
|
|
610
|
+
|
|
611
|
+
def find_components(self, mol, fd, component_solver, **kwargs):
|
|
612
|
+
"""Find fluorinated components in a molecule that match this PFAS group's criteria."""
|
|
613
|
+
group_matches = []
|
|
614
|
+
if self.formula_dict_satisfies_constraints(fd) is True:
|
|
615
|
+
component_sizes = []
|
|
616
|
+
# Create molecular graph once for this fragment
|
|
617
|
+
G = mol_to_nx(mol)
|
|
618
|
+
kwargs['G'] = G
|
|
619
|
+
if self.componentSmarts =='cyclic':
|
|
620
|
+
# treat cyclic groups separately
|
|
621
|
+
# Use first SMARTS pattern for cyclic
|
|
622
|
+
match_count, component_sizes, matched1_len, matched_components = self.find_aryl_components(mol, component_solver=component_solver, **kwargs)
|
|
623
|
+
elif self.smarts is not None and len(self.smarts) > 0:
|
|
624
|
+
# Handle groups with SMARTS patterns
|
|
625
|
+
match_count, component_sizes, matched1_len, matched_components = self.find_alkyl_components(mol, component_solver, **kwargs)
|
|
626
|
+
elif self.componentSmarts is not None:
|
|
627
|
+
# treat cases with only componentSmarts defined (no SMARTS patterns), find all components of that path type
|
|
628
|
+
available = set(component_solver.componentSmartss.keys())
|
|
629
|
+
if isinstance(self.componentSmarts, (list, tuple, set)):
|
|
630
|
+
# Only keep paths available in the solver (respects halogen filter)
|
|
631
|
+
component_types = [cs for cs in self.componentSmarts if cs in available]
|
|
632
|
+
elif self.componentSmarts in available:
|
|
633
|
+
component_types = [self.componentSmarts]
|
|
634
|
+
else:
|
|
635
|
+
component_types = []
|
|
636
|
+
# Collect unique components across all types; deduplicate by atom-set so
|
|
637
|
+
# the same carbon substructure is only reported once even when multiple
|
|
638
|
+
# per-halogen SMARTS types are listed (e.g. perhalogenated alkyl groups).
|
|
639
|
+
seen_atom_sets = set()
|
|
640
|
+
matched_components = []
|
|
641
|
+
for comp_type in component_types:
|
|
642
|
+
comps = component_solver.get(comp_type, max_dist=self.max_dist_from_comp, default=[])
|
|
643
|
+
for comp in comps:
|
|
644
|
+
key = frozenset(comp)
|
|
645
|
+
if key in seen_atom_sets:
|
|
646
|
+
continue
|
|
647
|
+
# Per-component formula constraint check defined in
|
|
648
|
+
# component_smarts_halogens.json (e.g. 'gte': {'F': 2}).
|
|
649
|
+
# Done before adding to seen_atom_sets so the same atom-set
|
|
650
|
+
# can still be accepted via a different comp_type.
|
|
651
|
+
comp_constraints = self._comp_type_to_constraints.get(comp_type, {})
|
|
652
|
+
if comp_constraints:
|
|
653
|
+
full_comp = component_solver.get_full_component_atoms(comp)
|
|
654
|
+
comp_formula = {}
|
|
655
|
+
for idx in full_comp:
|
|
656
|
+
sym = mol.GetAtomWithIdx(idx).GetSymbol()
|
|
657
|
+
comp_formula[sym] = comp_formula.get(sym, 0) + 1
|
|
658
|
+
if not self._check_component_constraints(comp_formula, comp_constraints):
|
|
659
|
+
continue
|
|
660
|
+
seen_atom_sets.add(key)
|
|
661
|
+
matched_components.append(
|
|
662
|
+
component_solver.get_matched_component_dict(comp, None, comp_type, self)
|
|
663
|
+
)
|
|
664
|
+
match_count = len(matched_components)
|
|
665
|
+
component_sizes = [comp.get('size', 0) for comp in matched_components]
|
|
666
|
+
matched1_len = match_count # Set matched1_len to enable group matching
|
|
667
|
+
else:
|
|
668
|
+
# treat cases with no SMARTS patterns, just formula constraints
|
|
669
|
+
match_count = 1
|
|
670
|
+
matched_components = []
|
|
671
|
+
matched1_len = 1
|
|
672
|
+
if match_count > 0 and matched1_len > 0:
|
|
673
|
+
# add to matches if functional group was found
|
|
674
|
+
group_matches.append((self, match_count, component_sizes, matched_components))
|
|
675
|
+
else:
|
|
676
|
+
return None
|
|
677
|
+
return group_matches
|
|
678
|
+
return None
|
|
679
|
+
|
|
680
|
+
def test(self, test_data=None):
|
|
681
|
+
"""Test this PFAS group against test molecules from metadata.
|
|
682
|
+
|
|
683
|
+
Validates that the group correctly identifies positive examples and
|
|
684
|
+
rejects negative examples based on test metadata in PFAS_groups_smarts.json.
|
|
685
|
+
|
|
686
|
+
Parameters
|
|
687
|
+
----------
|
|
688
|
+
test_data : dict, optional
|
|
689
|
+
Test metadata dictionary. If None, will be loaded from the group's
|
|
690
|
+
entry in PFAS_groups_smarts.json. Expected keys: ``category``,
|
|
691
|
+
``examples``, ``generate``.
|
|
692
|
+
|
|
693
|
+
Returns
|
|
694
|
+
-------
|
|
695
|
+
dict
|
|
696
|
+
Test results with keys: ``passed`` (bool), ``total_tests`` (int),
|
|
697
|
+
``failures`` (list of dicts), ``category`` (str).
|
|
698
|
+
|
|
699
|
+
Notes
|
|
700
|
+
-----
|
|
701
|
+
- For OECD groups: Tests against curated positive examples
|
|
702
|
+
- For telomer groups: Tests generated molecules based on smiles patterns
|
|
703
|
+
- For generic groups: Tests both positive and negative examples
|
|
704
|
+
- Returns detailed failure information for debugging
|
|
705
|
+
"""
|
|
706
|
+
from .core import n_from_formula
|
|
707
|
+
from .ComponentsSolverModel import ComponentsSolver
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
results = {
|
|
712
|
+
'passed': True,
|
|
713
|
+
'total_tests': 0,
|
|
714
|
+
'failures': [],
|
|
715
|
+
'category': self.test_dict.get('category', 'unknown') if self.test_dict else 'unknown'
|
|
716
|
+
}
|
|
717
|
+
# Load test data if not provided
|
|
718
|
+
if self.test_dict is None:
|
|
719
|
+
return results # No tests defined for this group
|
|
720
|
+
# Test positive examples
|
|
721
|
+
examples = self.test_dict.get('examples', [])
|
|
722
|
+
for smiles in examples:
|
|
723
|
+
results['total_tests'] += 1
|
|
724
|
+
try:
|
|
725
|
+
mol = Chem.MolFromSmiles(smiles)
|
|
726
|
+
if mol is None:
|
|
727
|
+
results['passed'] = False
|
|
728
|
+
results['failures'].append({
|
|
729
|
+
'smiles': smiles,
|
|
730
|
+
'expected': True,
|
|
731
|
+
'got': None,
|
|
732
|
+
'error': 'Invalid SMILES'
|
|
733
|
+
})
|
|
734
|
+
continue
|
|
735
|
+
|
|
736
|
+
# Add hydrogens as done in parser
|
|
737
|
+
mol = Chem.AddHs(mol)
|
|
738
|
+
|
|
739
|
+
# Create ComponentsSolver for this molecule
|
|
740
|
+
with ComponentsSolver(mol) as component_solver:
|
|
741
|
+
# Get formula dict
|
|
742
|
+
formula = CalcMolFormula(mol)
|
|
743
|
+
fd = n_from_formula(formula)
|
|
744
|
+
|
|
745
|
+
# Use find_components to check if group matches
|
|
746
|
+
matches = self.find_components(mol, fd, component_solver)
|
|
747
|
+
is_match = matches is not None and len(matches) > 0
|
|
748
|
+
|
|
749
|
+
if not is_match:
|
|
750
|
+
results['passed'] = False
|
|
751
|
+
results['failures'].append({
|
|
752
|
+
'smiles': smiles,
|
|
753
|
+
'expected': True,
|
|
754
|
+
'got': False,
|
|
755
|
+
'error': 'Group should match but did not'
|
|
756
|
+
})
|
|
757
|
+
except Exception as e:
|
|
758
|
+
results['passed'] = False
|
|
759
|
+
results['failures'].append({
|
|
760
|
+
'smiles': smiles,
|
|
761
|
+
'expected': True,
|
|
762
|
+
'got': None,
|
|
763
|
+
'error': f'Exception: {str(e)}'
|
|
764
|
+
})
|
|
765
|
+
# Test positive examples
|
|
766
|
+
examples = self.test_dict.get('counter-examples', [])
|
|
767
|
+
for smiles in examples:
|
|
768
|
+
results['total_tests'] += 1
|
|
769
|
+
try:
|
|
770
|
+
mol = Chem.MolFromSmiles(smiles)
|
|
771
|
+
if mol is None:
|
|
772
|
+
results['passed'] = False
|
|
773
|
+
results['failures'].append({
|
|
774
|
+
'smiles': smiles,
|
|
775
|
+
'expected': False,
|
|
776
|
+
'got': None,
|
|
777
|
+
'error': 'Invalid SMILES'
|
|
778
|
+
})
|
|
779
|
+
continue
|
|
780
|
+
|
|
781
|
+
# Add hydrogens as done in parser
|
|
782
|
+
mol = Chem.AddHs(mol)
|
|
783
|
+
|
|
784
|
+
# Create ComponentsSolver for this molecule
|
|
785
|
+
with ComponentsSolver(mol) as component_solver:
|
|
786
|
+
# Get formula dict
|
|
787
|
+
formula = CalcMolFormula(mol)
|
|
788
|
+
fd = n_from_formula(formula)
|
|
789
|
+
|
|
790
|
+
# Use find_components to check if group matches
|
|
791
|
+
matches = self.find_components(mol, fd, component_solver)
|
|
792
|
+
is_match = matches is not None and len(matches) > 0
|
|
793
|
+
|
|
794
|
+
if is_match:
|
|
795
|
+
results['passed'] = False
|
|
796
|
+
results['failures'].append({
|
|
797
|
+
'smiles': smiles,
|
|
798
|
+
'expected': False,
|
|
799
|
+
'got': True,
|
|
800
|
+
'error': 'Group should match but did not'
|
|
801
|
+
})
|
|
802
|
+
except Exception as e:
|
|
803
|
+
results['passed'] = False
|
|
804
|
+
results['failures'].append({
|
|
805
|
+
'smiles': smiles,
|
|
806
|
+
'expected': True,
|
|
807
|
+
'got': None,
|
|
808
|
+
'error': f'Exception: {str(e)}'
|
|
809
|
+
})
|
|
810
|
+
return results
|