PFASGroups 3.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- HalogenGroups/__init__.py +246 -0
- PFASGroups/ComponentsSolverModel.py +977 -0
- PFASGroups/HalogenGroupModel.py +810 -0
- PFASGroups/PFASDefinitionModel.py +393 -0
- PFASGroups/PFASEmbeddings.py +3315 -0
- PFASGroups/__init__.py +21 -0
- PFASGroups/cli.py +618 -0
- PFASGroups/core.py +415 -0
- PFASGroups/data/Halogen_groups_smarts.json +9024 -0
- PFASGroups/data/PFAS_definitions_smarts.json +170 -0
- PFASGroups/data/component_smarts.json +4 -0
- PFASGroups/data/component_smarts_halogens.json +142 -0
- PFASGroups/data/diatomic_bonds_dict.json +12802 -0
- PFASGroups/draw_mols.py +374 -0
- PFASGroups/embeddings.py +150 -0
- PFASGroups/fragmentation.py +548 -0
- PFASGroups/generate_homologues.py +256 -0
- PFASGroups/generate_mol.py +656 -0
- PFASGroups/generate_paper_figures.py +266 -0
- PFASGroups/getter.py +111 -0
- PFASGroups/homologue_series.py +473 -0
- PFASGroups/parser.py +942 -0
- PFASGroups/prioritise.py +439 -0
- pfasgroups-3.2.2.dist-info/METADATA +724 -0
- pfasgroups-3.2.2.dist-info/RECORD +28 -0
- pfasgroups-3.2.2.dist-info/WHEEL +5 -0
- pfasgroups-3.2.2.dist-info/entry_points.txt +3 -0
- pfasgroups-3.2.2.dist-info/top_level.txt +2 -0
PFASGroups/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# PFASGroups package
|
|
2
|
+
from .HalogenGroupModel import HalogenGroup
|
|
3
|
+
from .PFASDefinitionModel import PFASDefinition
|
|
4
|
+
from .ComponentsSolverModel import ComponentsSolver
|
|
5
|
+
from .core import rdkit_disable_log, HALOGEN_GROUPS_FILE
|
|
6
|
+
from .parser import parse_smiles, parse_mols, parse_mol, parse_groups_in_mol, parse_from_database, setup_halogen_groups_database, load_HalogenGroups
|
|
7
|
+
# PFASFingerprint: convenience alias for parse_smiles — returns a PFASEmbeddingSet
|
|
8
|
+
PFASFingerprint = parse_smiles
|
|
9
|
+
from .draw_mols import plot_mol, plot_mols, plot_HalogenGroups
|
|
10
|
+
from .getter import get_componentSMARTSs, get_HalogenGroups, get_compiled_HalogenGroups, get_compiled_PFASGroups, get_PFASDefinitions, get_compiled_componentSMARTSs
|
|
11
|
+
from .embeddings import FINGERPRINT_PRESETS, EMBEDDING_PRESETS
|
|
12
|
+
from .generate_homologues import generate_homologues
|
|
13
|
+
from .homologue_series import HomologueSeries, HomologueEntry
|
|
14
|
+
from .fragmentation import generate_degradation_products
|
|
15
|
+
# PFASEmbedding (dict subclass, primary) must be imported after embeddings to take precedence
|
|
16
|
+
from .PFASEmbeddings import PFASEmbedding, PFASEmbeddingSet, EmbeddingArray, ResultsModel, MoleculeResult, generate_fingerprint
|
|
17
|
+
from .prioritise import prioritise_molecules, prioritize_molecules, get_priority_statistics
|
|
18
|
+
__version__ = "3.2.0"
|
|
19
|
+
__all__ = ['HalogenGroup', 'PFASDefinition', 'parse_smiles', 'parse_mols','parse_mol', 'parse_groups_in_mol', 'parse_from_database', 'setup_halogen_groups_database', 'plot_HalogenGroups', 'plot_mol','plot_mols', 'FINGERPRINT_PRESETS', 'PFASFingerprint', 'generate_fingerprint', 'get_compiled_componentSMARTSs', 'get_componentSMARTSs', 'get_HalogenGroups', 'get_compiled_HalogenGroups', 'get_compiled_PFASGroups', 'get_PFASDefinitions' ,'ComponentsSolver', 'generate_homologues', 'generate_degradation_products',"rdkit_disable_log","load_HalogenGroups", "HALOGEN_GROUPS_FILE"]
|
|
20
|
+
__all__.extend(['PFASEmbedding', 'PFASEmbeddingSet', 'EmbeddingArray', 'ResultsModel', 'MoleculeResult', 'prioritise_molecules', 'prioritize_molecules', 'get_priority_statistics'])
|
|
21
|
+
__all__.extend(['HomologueSeries', 'HomologueEntry'])
|
PFASGroups/cli.py
ADDED
|
@@ -0,0 +1,618 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command-line interface for PFASGroups.
|
|
3
|
+
|
|
4
|
+
Provides command-line tools for parsing PFAS structures and generating fingerprints.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import argparse
|
|
8
|
+
import csv
|
|
9
|
+
import sys
|
|
10
|
+
import json
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
from .parser import parse_smiles
|
|
14
|
+
from .getter import get_componentSMARTSs, get_HalogenGroups
|
|
15
|
+
from .PFASEmbeddings import PFASEmbedding
|
|
16
|
+
from rdkit import Chem
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def parse_args():
|
|
20
|
+
"""Parse command-line arguments."""
|
|
21
|
+
parser = argparse.ArgumentParser(
|
|
22
|
+
description='PFASGroups - Parse and analyze PFAS structures',
|
|
23
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
24
|
+
epilog="""
|
|
25
|
+
Examples:
|
|
26
|
+
# Parse SMILES from command line
|
|
27
|
+
PFASGroups parse "C(C(F)(F)F)F" "FC(F)(F)C(F)(F)C(=O)O"
|
|
28
|
+
|
|
29
|
+
# Parse with component metrics
|
|
30
|
+
PFASGroups parse --bycomponent "FC(F)(F)C(F)(F)C(=O)O" --pretty
|
|
31
|
+
|
|
32
|
+
# Parse SMILES from file
|
|
33
|
+
PFASGroups parse --input smiles.txt --output results.json
|
|
34
|
+
|
|
35
|
+
# Use custom configuration files
|
|
36
|
+
PFASGroups parse --groups-file custom_groups.json "CCF"
|
|
37
|
+
|
|
38
|
+
# Generate fingerprints
|
|
39
|
+
PFASGroups fingerprint "C(C(F)(F)F)F" --output fp.json
|
|
40
|
+
|
|
41
|
+
# Generate fingerprints with custom groups
|
|
42
|
+
PFASGroups fingerprint --input smiles.txt --groups 28-52 --format dict
|
|
43
|
+
|
|
44
|
+
# List available groups
|
|
45
|
+
PFASGroups list-groups
|
|
46
|
+
|
|
47
|
+
# List available path types
|
|
48
|
+
PFASGroups list-paths
|
|
49
|
+
|
|
50
|
+
Note: Use get_componentSMARTSs() and get_PFASGroups() in Python to extend defaults.
|
|
51
|
+
"""
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
# Global options
|
|
55
|
+
parser.add_argument(
|
|
56
|
+
'--component_smarts-file',
|
|
57
|
+
type=str,
|
|
58
|
+
help='Path to custom component_smarts.json file (default: use package default)'
|
|
59
|
+
)
|
|
60
|
+
parser.add_argument(
|
|
61
|
+
'--groups-file',
|
|
62
|
+
type=str,
|
|
63
|
+
help='Path to custom PFAS_groups_smarts.json file (default: use package default)'
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
subparsers = parser.add_subparsers(dest='command', help='Available commands')
|
|
67
|
+
|
|
68
|
+
# Parse command
|
|
69
|
+
parse_parser = subparsers.add_parser(
|
|
70
|
+
'parse',
|
|
71
|
+
help='Parse SMILES strings and identify PFAS groups'
|
|
72
|
+
)
|
|
73
|
+
parse_parser.add_argument(
|
|
74
|
+
'smiles',
|
|
75
|
+
nargs='*',
|
|
76
|
+
help='SMILES strings to parse (use --input for file input)'
|
|
77
|
+
)
|
|
78
|
+
parse_parser.add_argument(
|
|
79
|
+
'-i', '--input',
|
|
80
|
+
type=str,
|
|
81
|
+
help='Input file containing SMILES strings (one per line)'
|
|
82
|
+
)
|
|
83
|
+
parse_parser.add_argument(
|
|
84
|
+
'-o', '--output',
|
|
85
|
+
type=str,
|
|
86
|
+
help='Output file for results (JSON format, default: stdout)'
|
|
87
|
+
)
|
|
88
|
+
parse_parser.add_argument(
|
|
89
|
+
'--bycomponent',
|
|
90
|
+
action='store_true',
|
|
91
|
+
help='Use component-based analysis (provides comprehensive metrics including component_fraction, branching, eccentricity)'
|
|
92
|
+
)
|
|
93
|
+
parse_parser.add_argument(
|
|
94
|
+
'--no-component-metrics',
|
|
95
|
+
action='store_true',
|
|
96
|
+
help='Skip all component graph metrics (fastest)'
|
|
97
|
+
)
|
|
98
|
+
parse_parser.add_argument(
|
|
99
|
+
'--limit-effective-graph-resistance',
|
|
100
|
+
type=int,
|
|
101
|
+
help='Only compute effective graph resistance for components smaller than this size (0 disables it)'
|
|
102
|
+
)
|
|
103
|
+
parse_parser.add_argument(
|
|
104
|
+
'--halogens',
|
|
105
|
+
nargs='+',
|
|
106
|
+
help='Filter components by halogen element symbol(s), e.g. F or F Cl'
|
|
107
|
+
)
|
|
108
|
+
parse_parser.add_argument(
|
|
109
|
+
'--form',
|
|
110
|
+
nargs='+',
|
|
111
|
+
choices=['alkyl', 'cyclic'],
|
|
112
|
+
help='Filter components by form (alkyl, cyclic)'
|
|
113
|
+
)
|
|
114
|
+
parse_parser.add_argument(
|
|
115
|
+
'--saturation',
|
|
116
|
+
nargs='+',
|
|
117
|
+
choices=['per', 'poly'],
|
|
118
|
+
help='Filter components by saturation (per, poly)'
|
|
119
|
+
)
|
|
120
|
+
parse_parser.add_argument(
|
|
121
|
+
'--format',
|
|
122
|
+
choices=['json', 'csv'],
|
|
123
|
+
default='json',
|
|
124
|
+
help='Output format (default: json)'
|
|
125
|
+
)
|
|
126
|
+
parse_parser.add_argument(
|
|
127
|
+
'--pretty',
|
|
128
|
+
action='store_true',
|
|
129
|
+
help='Pretty-print JSON output (only for JSON format)'
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
# Fingerprint command
|
|
133
|
+
fp_parser = subparsers.add_parser(
|
|
134
|
+
'fingerprint',
|
|
135
|
+
help='Generate PFAS group fingerprints'
|
|
136
|
+
)
|
|
137
|
+
fp_parser.add_argument(
|
|
138
|
+
'smiles',
|
|
139
|
+
nargs='*',
|
|
140
|
+
help='SMILES strings to fingerprint (use --input for file input)'
|
|
141
|
+
)
|
|
142
|
+
fp_parser.add_argument(
|
|
143
|
+
'-i', '--input',
|
|
144
|
+
type=str,
|
|
145
|
+
help='Input file containing SMILES strings (one per line)'
|
|
146
|
+
)
|
|
147
|
+
fp_parser.add_argument(
|
|
148
|
+
'-o', '--output',
|
|
149
|
+
type=str,
|
|
150
|
+
help='Output file for fingerprints (JSON format, default: stdout)'
|
|
151
|
+
)
|
|
152
|
+
fp_parser.add_argument(
|
|
153
|
+
'-g', '--groups',
|
|
154
|
+
type=str,
|
|
155
|
+
help='Selected groups as range (e.g., "28-52") or comma-separated indices (e.g., "28,29,30")'
|
|
156
|
+
)
|
|
157
|
+
fp_parser.add_argument(
|
|
158
|
+
'--halogens',
|
|
159
|
+
nargs='+',
|
|
160
|
+
help='Filter components by halogen element symbol(s), e.g. F or F Cl Br I'
|
|
161
|
+
)
|
|
162
|
+
fp_parser.add_argument(
|
|
163
|
+
'-f', '--format',
|
|
164
|
+
choices=['vector', 'dict', 'sparse', 'detailed', 'int'],
|
|
165
|
+
default='vector',
|
|
166
|
+
help='Fingerprint representation format (default: vector)'
|
|
167
|
+
)
|
|
168
|
+
fp_parser.add_argument(
|
|
169
|
+
'--count-mode',
|
|
170
|
+
choices=['binary', 'count', 'max_chain'],
|
|
171
|
+
default='binary',
|
|
172
|
+
help='How to count matches (default: binary)'
|
|
173
|
+
)
|
|
174
|
+
fp_parser.add_argument(
|
|
175
|
+
'--output-format',
|
|
176
|
+
choices=['json', 'csv'],
|
|
177
|
+
default='json',
|
|
178
|
+
help='Output file format (default: json)'
|
|
179
|
+
)
|
|
180
|
+
fp_parser.add_argument(
|
|
181
|
+
'--pretty',
|
|
182
|
+
action='store_true',
|
|
183
|
+
help='Pretty-print JSON output (only for JSON format)'
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
# List groups command
|
|
187
|
+
list_parser = subparsers.add_parser(
|
|
188
|
+
'list-groups',
|
|
189
|
+
help='List available PFAS groups (use in Python to extend with get_PFASGroups)'
|
|
190
|
+
)
|
|
191
|
+
list_parser.add_argument(
|
|
192
|
+
'-o', '--output',
|
|
193
|
+
type=str,
|
|
194
|
+
help='Output file (default: stdout)'
|
|
195
|
+
)
|
|
196
|
+
list_parser.add_argument(
|
|
197
|
+
'--pretty',
|
|
198
|
+
action='store_true',
|
|
199
|
+
default=True,
|
|
200
|
+
help='Pretty-print JSON output'
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
# List paths command
|
|
204
|
+
list_paths_parser = subparsers.add_parser(
|
|
205
|
+
'list-paths',
|
|
206
|
+
help='List available path types (use in Python to extend with get_componentSMARTSs)'
|
|
207
|
+
)
|
|
208
|
+
list_paths_parser.add_argument(
|
|
209
|
+
'-o', '--output',
|
|
210
|
+
type=str,
|
|
211
|
+
help='Output file (default: stdout)'
|
|
212
|
+
)
|
|
213
|
+
list_paths_parser.add_argument(
|
|
214
|
+
'--pretty',
|
|
215
|
+
action='store_true',
|
|
216
|
+
default=True,
|
|
217
|
+
help='Pretty-print JSON output'
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
# Validate config command
|
|
221
|
+
validate_parser = subparsers.add_parser( # pylint: disable=unused-variable
|
|
222
|
+
'validate-config',
|
|
223
|
+
help='Validate custom configuration files'
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
return parser.parse_args()
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def read_smiles_file(filepath: str) -> list:
|
|
230
|
+
"""Read SMILES strings from file."""
|
|
231
|
+
with open(filepath, 'r') as f:
|
|
232
|
+
return [line.strip() for line in f if line.strip()]
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def parse_group_selection(groups_str: str) -> list:
|
|
236
|
+
"""
|
|
237
|
+
Parse group selection string.
|
|
238
|
+
|
|
239
|
+
Examples:
|
|
240
|
+
"28-52" -> range(28, 53)
|
|
241
|
+
"28,29,30" -> [28, 29, 30]
|
|
242
|
+
"""
|
|
243
|
+
if '-' in groups_str:
|
|
244
|
+
start, end = groups_str.split('-')
|
|
245
|
+
return list(range(int(start), int(end) + 1))
|
|
246
|
+
if ',' in groups_str:
|
|
247
|
+
return [int(x.strip()) for x in groups_str.split(',')]
|
|
248
|
+
else:
|
|
249
|
+
return [int(groups_str)]
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def cmd_parse(args):
|
|
253
|
+
"""Execute parse command."""
|
|
254
|
+
# Load custom configuration if provided
|
|
255
|
+
kwargs = {}
|
|
256
|
+
if args.component_smarts_file:
|
|
257
|
+
kwargs['componentSmartss'] = get_componentSMARTSs(filename=args.component_smarts_file)
|
|
258
|
+
if args.groups_file:
|
|
259
|
+
kwargs['pfas_groups'] = get_HalogenGroups(filename=args.groups_file)
|
|
260
|
+
kwargs['compute_component_metrics'] = not args.no_component_metrics
|
|
261
|
+
kwargs['limit_effective_graph_resistance'] = args.limit_effective_graph_resistance
|
|
262
|
+
if args.halogens:
|
|
263
|
+
kwargs['halogens'] = args.halogens
|
|
264
|
+
if args.form:
|
|
265
|
+
kwargs['form'] = args.form
|
|
266
|
+
if args.saturation:
|
|
267
|
+
kwargs['saturation'] = args.saturation
|
|
268
|
+
|
|
269
|
+
# Get SMILES from command line or file
|
|
270
|
+
if args.input:
|
|
271
|
+
smiles_list = read_smiles_file(args.input)
|
|
272
|
+
elif args.smiles:
|
|
273
|
+
smiles_list = args.smiles
|
|
274
|
+
else:
|
|
275
|
+
print("Error: Provide SMILES as arguments or use --input", file=sys.stderr)
|
|
276
|
+
sys.exit(1)
|
|
277
|
+
|
|
278
|
+
# Determine output format
|
|
279
|
+
if args.format == 'csv':
|
|
280
|
+
output_format = 'csv'
|
|
281
|
+
else:
|
|
282
|
+
output_format = 'list' # pylint: disable=unused-variable
|
|
283
|
+
|
|
284
|
+
# Parse PFAS — always returns a PFASEmbeddingSet
|
|
285
|
+
results = parse_smiles(smiles_list, bycomponent=args.bycomponent,
|
|
286
|
+
**kwargs)
|
|
287
|
+
|
|
288
|
+
if args.format == 'csv':
|
|
289
|
+
# Build CSV output from PFASEmbeddingSet
|
|
290
|
+
import io
|
|
291
|
+
writer_buf = io.StringIO()
|
|
292
|
+
csv_writer = csv.writer(writer_buf)
|
|
293
|
+
csv_writer.writerow(['smiles', 'group_id', 'group_name', 'match_count',
|
|
294
|
+
'component_idx', 'component_smarts', 'size',
|
|
295
|
+
'branching', 'mean_eccentricity', 'component_fraction',
|
|
296
|
+
'diameter', 'radius', 'effective_graph_resistance',
|
|
297
|
+
'n_spacer', 'ring_size'])
|
|
298
|
+
for embedding in results:
|
|
299
|
+
smiles_val = embedding['smiles']
|
|
300
|
+
for match in embedding.get('matches', []):
|
|
301
|
+
if match.get('type') != 'HalogenGroup':
|
|
302
|
+
continue
|
|
303
|
+
components = match.get('components', [])
|
|
304
|
+
if not components:
|
|
305
|
+
csv_writer.writerow([
|
|
306
|
+
smiles_val, match['id'], match['group_name'],
|
|
307
|
+
match['match_count'], '', '', '', '', '', '', '', '', '', '', ''
|
|
308
|
+
])
|
|
309
|
+
for idx, comp in enumerate(components):
|
|
310
|
+
csv_writer.writerow([
|
|
311
|
+
smiles_val,
|
|
312
|
+
match['id'],
|
|
313
|
+
match['group_name'],
|
|
314
|
+
match['match_count'],
|
|
315
|
+
idx,
|
|
316
|
+
comp.get('SMARTS', ''),
|
|
317
|
+
comp.get('size', ''),
|
|
318
|
+
comp.get('branching', ''),
|
|
319
|
+
comp.get('mean_eccentricity', ''),
|
|
320
|
+
comp.get('component_fraction', ''),
|
|
321
|
+
comp.get('diameter', ''),
|
|
322
|
+
comp.get('radius', ''),
|
|
323
|
+
comp.get('effective_graph_resistance', ''),
|
|
324
|
+
comp.get('n_spacer', ''),
|
|
325
|
+
comp.get('ring_size', ''),
|
|
326
|
+
])
|
|
327
|
+
result = writer_buf.getvalue()
|
|
328
|
+
|
|
329
|
+
if args.output:
|
|
330
|
+
with open(args.output, 'w', encoding='utf-8') as f:
|
|
331
|
+
f.write(result)
|
|
332
|
+
print(f"Results written to {args.output}")
|
|
333
|
+
else:
|
|
334
|
+
print(result, end='')
|
|
335
|
+
else:
|
|
336
|
+
# Convert to JSON format
|
|
337
|
+
output_data = []
|
|
338
|
+
for embedding in results:
|
|
339
|
+
result_entry = {
|
|
340
|
+
'smiles': embedding['smiles'],
|
|
341
|
+
'groups': []
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
for match in embedding.get('matches', []):
|
|
345
|
+
if match.get('type') != 'HalogenGroup':
|
|
346
|
+
continue
|
|
347
|
+
components_out = []
|
|
348
|
+
for comp in match.get('components', []):
|
|
349
|
+
components_out.append({
|
|
350
|
+
'smarts': comp.get('SMARTS'),
|
|
351
|
+
'size': comp.get('size'),
|
|
352
|
+
'branching': comp.get('branching'),
|
|
353
|
+
'mean_eccentricity': comp.get('mean_eccentricity'),
|
|
354
|
+
'component_fraction': comp.get('component_fraction'),
|
|
355
|
+
'diameter': comp.get('diameter'),
|
|
356
|
+
'radius': comp.get('radius'),
|
|
357
|
+
'effective_graph_resistance': comp.get('effective_graph_resistance'),
|
|
358
|
+
'n_spacer': comp.get('n_spacer'),
|
|
359
|
+
'ring_size': comp.get('ring_size'),
|
|
360
|
+
})
|
|
361
|
+
result_entry['groups'].append({
|
|
362
|
+
'name': match['group_name'],
|
|
363
|
+
'id': match['id'],
|
|
364
|
+
'match_count': match['match_count'],
|
|
365
|
+
'num_components': match['num_components'],
|
|
366
|
+
'components_types': match.get('components_types', []),
|
|
367
|
+
'components': components_out,
|
|
368
|
+
})
|
|
369
|
+
|
|
370
|
+
output_data.append(result_entry)
|
|
371
|
+
|
|
372
|
+
# Output JSON
|
|
373
|
+
indent = 2 if args.pretty else None
|
|
374
|
+
output_json = json.dumps(output_data, indent=indent)
|
|
375
|
+
|
|
376
|
+
if args.output:
|
|
377
|
+
with open(args.output, 'w') as f:
|
|
378
|
+
f.write(output_json)
|
|
379
|
+
print(f"Results written to {args.output}")
|
|
380
|
+
else:
|
|
381
|
+
print(output_json)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def cmd_fingerprint(args):
|
|
385
|
+
"""Execute fingerprint command."""
|
|
386
|
+
# Load custom configuration if provided
|
|
387
|
+
kwargs = {}
|
|
388
|
+
if args.component_smarts_file:
|
|
389
|
+
kwargs['componentSmartss'] = get_componentSMARTSs(filename=args.component_smarts_file)
|
|
390
|
+
if args.groups_file:
|
|
391
|
+
kwargs['pfas_groups'] = get_HalogenGroups(filename=args.groups_file)
|
|
392
|
+
|
|
393
|
+
# Get SMILES from command line or file
|
|
394
|
+
if args.input:
|
|
395
|
+
smiles_list = read_smiles_file(args.input)
|
|
396
|
+
elif args.smiles:
|
|
397
|
+
smiles_list = args.smiles
|
|
398
|
+
else:
|
|
399
|
+
print("Error: Provide SMILES as arguments or use --input", file=sys.stderr)
|
|
400
|
+
sys.exit(1)
|
|
401
|
+
|
|
402
|
+
# Determine halogens
|
|
403
|
+
halogens = getattr(args, 'halogens', None) or 'F'
|
|
404
|
+
|
|
405
|
+
# Parse SMILES → PFASEmbeddingSet
|
|
406
|
+
embs = parse_smiles(smiles_list, halogens=halogens, **kwargs)
|
|
407
|
+
|
|
408
|
+
# Build fingerprint array and column names
|
|
409
|
+
array_kwargs = {}
|
|
410
|
+
if args.groups:
|
|
411
|
+
array_kwargs['selected_group_ids'] = parse_group_selection(args.groups)
|
|
412
|
+
arr = embs.to_array(**array_kwargs)
|
|
413
|
+
col_names = embs.column_names(**{k: v for k, v in array_kwargs.items()
|
|
414
|
+
if k in ('selected_group_ids',)})
|
|
415
|
+
|
|
416
|
+
# Format fingerprints per molecule
|
|
417
|
+
fingerprints = []
|
|
418
|
+
for i, smiles_val in enumerate(smiles_list):
|
|
419
|
+
row = arr[i].tolist()
|
|
420
|
+
if args.format == 'dict':
|
|
421
|
+
fingerprints.append({
|
|
422
|
+
'smiles': smiles_val,
|
|
423
|
+
'fingerprint': {col_names[j]: row[j] for j in range(len(col_names))}
|
|
424
|
+
})
|
|
425
|
+
else: # vector
|
|
426
|
+
fingerprints.append({
|
|
427
|
+
'smiles': smiles_val,
|
|
428
|
+
'fingerprint': row
|
|
429
|
+
})
|
|
430
|
+
|
|
431
|
+
# Build output
|
|
432
|
+
output_data = {
|
|
433
|
+
'smiles': smiles_list,
|
|
434
|
+
'column_names': col_names,
|
|
435
|
+
'fingerprints': fingerprints
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
# Output based on requested format
|
|
439
|
+
if args.output_format == 'csv':
|
|
440
|
+
csv_rows = []
|
|
441
|
+
for entry in fingerprints:
|
|
442
|
+
row = {'smiles': entry['smiles']}
|
|
443
|
+
fp = entry['fingerprint']
|
|
444
|
+
if isinstance(fp, dict):
|
|
445
|
+
row.update(fp)
|
|
446
|
+
else:
|
|
447
|
+
for j, val in enumerate(fp):
|
|
448
|
+
row[col_names[j] if j < len(col_names) else f'col_{j}'] = val
|
|
449
|
+
csv_rows.append(row)
|
|
450
|
+
|
|
451
|
+
fieldnames = list(csv_rows[0].keys()) if csv_rows else ['smiles']
|
|
452
|
+
if args.output:
|
|
453
|
+
with open(args.output, 'w', newline='', encoding='utf-8') as f:
|
|
454
|
+
writer = csv.DictWriter(f, fieldnames=fieldnames)
|
|
455
|
+
writer.writeheader()
|
|
456
|
+
writer.writerows(csv_rows)
|
|
457
|
+
print(f"Results written to {args.output}")
|
|
458
|
+
else:
|
|
459
|
+
writer = csv.DictWriter(sys.stdout, fieldnames=fieldnames)
|
|
460
|
+
writer.writeheader()
|
|
461
|
+
writer.writerows(csv_rows)
|
|
462
|
+
return
|
|
463
|
+
|
|
464
|
+
# JSON output (default)
|
|
465
|
+
indent = 2 if args.pretty else None
|
|
466
|
+
output_json = json.dumps(output_data, indent=indent)
|
|
467
|
+
|
|
468
|
+
if args.output:
|
|
469
|
+
with open(args.output, 'w') as f:
|
|
470
|
+
f.write(output_json)
|
|
471
|
+
print(f"Fingerprints written to {args.output}")
|
|
472
|
+
else:
|
|
473
|
+
print(output_json)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def cmd_list_groups(args):
|
|
477
|
+
"""Execute list-groups command."""
|
|
478
|
+
# Load groups (custom or default)
|
|
479
|
+
if args.groups_file:
|
|
480
|
+
groups = get_HalogenGroups(filename=args.groups_file)
|
|
481
|
+
else:
|
|
482
|
+
groups = get_HalogenGroups()
|
|
483
|
+
|
|
484
|
+
# Create output
|
|
485
|
+
groups_list = []
|
|
486
|
+
for i, group in enumerate(groups):
|
|
487
|
+
groups_list.append({
|
|
488
|
+
'index': i,
|
|
489
|
+
'id': group['id'],
|
|
490
|
+
'name': group['name'],
|
|
491
|
+
'smarts': list(group.get('smarts', {}).keys()),
|
|
492
|
+
'componentSmarts': group.get('componentSmarts')
|
|
493
|
+
})
|
|
494
|
+
|
|
495
|
+
output_data = {
|
|
496
|
+
'total_groups': len(groups),
|
|
497
|
+
'groups': groups_list,
|
|
498
|
+
'note': 'Use get_HalogenGroups() in Python to load and extend these groups'
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
# Output results
|
|
502
|
+
indent = 2 if args.pretty else None
|
|
503
|
+
output_json = json.dumps(output_data, indent=indent)
|
|
504
|
+
|
|
505
|
+
if args.output:
|
|
506
|
+
with open(args.output, 'w') as f:
|
|
507
|
+
f.write(output_json)
|
|
508
|
+
print(f"Groups list written to {args.output}")
|
|
509
|
+
else:
|
|
510
|
+
print(output_json)
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
def cmd_list_paths(args):
|
|
514
|
+
"""Execute list-paths command."""
|
|
515
|
+
# Load paths (custom or default)
|
|
516
|
+
if args.component_smarts_file:
|
|
517
|
+
paths = get_componentSMARTSs(filename=args.component_smarts_file)
|
|
518
|
+
else:
|
|
519
|
+
paths = get_componentSMARTSs()
|
|
520
|
+
|
|
521
|
+
# Create output - convert RDKit mols to SMARTS strings for display
|
|
522
|
+
paths_list = []
|
|
523
|
+
for name, path_info in paths.items():
|
|
524
|
+
component_mol = path_info.get('component')
|
|
525
|
+
paths_list.append({
|
|
526
|
+
'name': name,
|
|
527
|
+
'smarts': Chem.MolToSmarts(component_mol) if component_mol else None,
|
|
528
|
+
'halogen': path_info.get('halogen'),
|
|
529
|
+
'form': path_info.get('form'),
|
|
530
|
+
'saturation': path_info.get('saturation')
|
|
531
|
+
})
|
|
532
|
+
|
|
533
|
+
output_data = {
|
|
534
|
+
'total_paths': len(paths),
|
|
535
|
+
'paths': paths_list,
|
|
536
|
+
'note': 'Use get_componentSMARTSs() in Python to load and extend these paths'
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
# Output results
|
|
540
|
+
indent = 2 if args.pretty else None
|
|
541
|
+
output_json = json.dumps(output_data, indent=indent)
|
|
542
|
+
|
|
543
|
+
if args.output:
|
|
544
|
+
with open(args.output, 'w') as f:
|
|
545
|
+
f.write(output_json)
|
|
546
|
+
print(f"Paths list written to {args.output}")
|
|
547
|
+
else:
|
|
548
|
+
print(output_json)
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def cmd_validate_config(args):
|
|
552
|
+
"""Execute validate-config command."""
|
|
553
|
+
try:
|
|
554
|
+
print("Validating configuration files...")
|
|
555
|
+
|
|
556
|
+
# Validate component_smarts if provided
|
|
557
|
+
if args.component_smarts_file:
|
|
558
|
+
paths = get_componentSMARTSs(filename=args.component_smarts_file)
|
|
559
|
+
print(f"✓ component_smarts.json loaded successfully from: {args.component_smarts_file}")
|
|
560
|
+
print(f" Found {len(paths)} path types")
|
|
561
|
+
|
|
562
|
+
# Validate groups if provided
|
|
563
|
+
if args.groups_file:
|
|
564
|
+
groups = get_HalogenGroups(filename=args.groups_file)
|
|
565
|
+
print(f"✓ PFAS_groups_smarts.json loaded successfully from: {args.groups_file}")
|
|
566
|
+
print(f" Found {len(groups)} PFAS groups")
|
|
567
|
+
|
|
568
|
+
if not args.component_smarts_file and not args.groups_file:
|
|
569
|
+
# Validate defaults
|
|
570
|
+
paths = get_componentSMARTSs()
|
|
571
|
+
groups = get_HalogenGroups()
|
|
572
|
+
print("✓ Default configuration loaded successfully")
|
|
573
|
+
print(f" Path types: {len(paths)}")
|
|
574
|
+
print(f" PFAS groups: {len(groups)}")
|
|
575
|
+
|
|
576
|
+
print("\nConfiguration is valid!")
|
|
577
|
+
|
|
578
|
+
except FileNotFoundError as e:
|
|
579
|
+
print("✗ Error: " + str(e), file=sys.stderr)
|
|
580
|
+
sys.exit(1)
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def main(default_halogens=None):
|
|
584
|
+
"""Main CLI entry point."""
|
|
585
|
+
args = parse_args()
|
|
586
|
+
|
|
587
|
+
# Apply default halogens when the entry point provides one and the user
|
|
588
|
+
# did not explicitly pass --halogens on the command line.
|
|
589
|
+
if default_halogens and args.command in ('parse', 'fingerprint'):
|
|
590
|
+
if not getattr(args, 'halogens', None):
|
|
591
|
+
args.halogens = default_halogens
|
|
592
|
+
|
|
593
|
+
if args.command == 'parse':
|
|
594
|
+
cmd_parse(args)
|
|
595
|
+
elif args.command == 'fingerprint':
|
|
596
|
+
cmd_fingerprint(args)
|
|
597
|
+
elif args.command == 'list-groups':
|
|
598
|
+
cmd_list_groups(args)
|
|
599
|
+
elif args.command == 'list-paths':
|
|
600
|
+
cmd_list_paths(args)
|
|
601
|
+
elif args.command == 'validate-config':
|
|
602
|
+
cmd_validate_config(args)
|
|
603
|
+
else:
|
|
604
|
+
print("Error: No command specified. Use --help for usage information.", file=sys.stderr)
|
|
605
|
+
sys.exit(1)
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def main_halogen():
|
|
609
|
+
"""Entry point for the ``halogengroups`` CLI.
|
|
610
|
+
|
|
611
|
+
Identical to ``pfasgroups`` but defaults to all four halogens
|
|
612
|
+
(F, Cl, Br, I) when ``--halogens`` is not specified on the command line.
|
|
613
|
+
"""
|
|
614
|
+
main(default_halogens=['F', 'Cl', 'Br', 'I'])
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
if __name__ == '__main__':
|
|
618
|
+
main()
|