quantui 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantui/__init__.py +311 -0
- quantui/analytics.py +609 -0
- quantui/app.py +5650 -0
- quantui/app_analysis.py +662 -0
- quantui/app_builders.py +2465 -0
- quantui/app_exports.py +194 -0
- quantui/app_formatters.py +493 -0
- quantui/app_history.py +624 -0
- quantui/app_runflow.py +1544 -0
- quantui/app_visualization.py +2620 -0
- quantui/ase_bridge.py +236 -0
- quantui/benchmarks.py +1543 -0
- quantui/c_stderr.py +124 -0
- quantui/cactus.py +88 -0
- quantui/calc_log.py +1116 -0
- quantui/calculator.py +204 -0
- quantui/cancellation.py +88 -0
- quantui/cli.py +288 -0
- quantui/comparison.py +306 -0
- quantui/config.py +725 -0
- quantui/data/js/3Dmol-min.js +2 -0
- quantui/data/js/3Dmol-min.js.LICENSE.txt +5 -0
- quantui/data/library/library.sqlite +0 -0
- quantui/data/manifests/bulk_qm9.json +1 -0
- quantui/data/manifests/curated.json +15482 -0
- quantui/data/manifests/presets.json +816 -0
- quantui/descriptor_cards.py +186 -0
- quantui/freq_calc.py +712 -0
- quantui/freq_ir_workers.py +229 -0
- quantui/gpu_offload.py +278 -0
- quantui/help_content.py +474 -0
- quantui/ir_plot.py +130 -0
- quantui/issue_tracker.py +170 -0
- quantui/live_log.py +387 -0
- quantui/log_utils.py +492 -0
- quantui/molecule.py +577 -0
- quantui/molecule_library.py +433 -0
- quantui/nmr_calc.py +437 -0
- quantui/optimizer.py +670 -0
- quantui/orbital_visualization.py +1102 -0
- quantui/pes_scan.py +420 -0
- quantui/preopt.py +355 -0
- quantui/progress.py +111 -0
- quantui/pubchem.py +1157 -0
- quantui/reorganization_energy.py +435 -0
- quantui/results_storage.py +902 -0
- quantui/security.py +14 -0
- quantui/session_calc.py +622 -0
- quantui/structure_providers.py +277 -0
- quantui/tddft_calc.py +307 -0
- quantui/user_settings.py +238 -0
- quantui/utils.py +287 -0
- quantui/vib_cache.py +247 -0
- quantui/visualization_py3dmol.py +593 -0
- quantui/viz_assets.py +101 -0
- quantui/viz_backend_router.py +243 -0
- quantui-0.5.1.dist-info/METADATA +533 -0
- quantui-0.5.1.dist-info/RECORD +62 -0
- quantui-0.5.1.dist-info/WHEEL +5 -0
- quantui-0.5.1.dist-info/entry_points.txt +2 -0
- quantui-0.5.1.dist-info/licenses/LICENSE +21 -0
- quantui-0.5.1.dist-info/top_level.txt +1 -0
quantui/pubchem.py
ADDED
|
@@ -0,0 +1,1157 @@
|
|
|
1
|
+
"""
|
|
2
|
+
PubChem Integration Module
|
|
3
|
+
|
|
4
|
+
Provides functions to search and retrieve molecular structures from PubChem
|
|
5
|
+
for educational use in quantum chemistry calculations.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
import re
|
|
10
|
+
import threading
|
|
11
|
+
import time
|
|
12
|
+
from functools import lru_cache
|
|
13
|
+
from typing import Any, Dict, Optional, Tuple
|
|
14
|
+
from urllib.parse import quote
|
|
15
|
+
|
|
16
|
+
import requests
|
|
17
|
+
|
|
18
|
+
from . import config
|
|
19
|
+
|
|
20
|
+
try:
|
|
21
|
+
from rdkit import Chem
|
|
22
|
+
from rdkit.Chem import AllChem, Descriptors
|
|
23
|
+
|
|
24
|
+
RDKIT_AVAILABLE = True
|
|
25
|
+
except ImportError:
|
|
26
|
+
RDKIT_AVAILABLE = False
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
logger = logging.getLogger(__name__)
|
|
30
|
+
|
|
31
|
+
# PubChem API endpoints
|
|
32
|
+
PUBCHEM_BASE_URL = "https://pubchem.ncbi.nlm.nih.gov/rest/pug"
|
|
33
|
+
# Back-compat alias; canonical value lives in config (constraint #5).
|
|
34
|
+
PUBCHEM_TIMEOUT = config.PUBCHEM_TIMEOUT_S
|
|
35
|
+
|
|
36
|
+
# ── HTTP client: client-side throttle + bounded 503 back-off ─────────────────
|
|
37
|
+
# A single process-wide limiter keeps us under PUG-REST's ~5 req/s ceiling even
|
|
38
|
+
# when several search threads fire at once.
|
|
39
|
+
_request_lock = threading.Lock()
|
|
40
|
+
_last_request_time = 0.0
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _throttle() -> None:
|
|
44
|
+
"""Block just long enough to honor the client-side minimum request gap."""
|
|
45
|
+
global _last_request_time
|
|
46
|
+
with _request_lock:
|
|
47
|
+
wait = config.PUBCHEM_MIN_REQUEST_INTERVAL_S - (
|
|
48
|
+
time.monotonic() - _last_request_time
|
|
49
|
+
)
|
|
50
|
+
if wait > 0:
|
|
51
|
+
time.sleep(wait)
|
|
52
|
+
_last_request_time = time.monotonic()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _http_get(
|
|
56
|
+
url: str,
|
|
57
|
+
*,
|
|
58
|
+
params: Optional[Dict[str, Any]] = None,
|
|
59
|
+
timeout: Optional[float] = None,
|
|
60
|
+
) -> requests.Response:
|
|
61
|
+
"""GET with client-side throttle + exponential back-off on 503 throttling.
|
|
62
|
+
|
|
63
|
+
Retries only on HTTP 503 (PUG-REST's throttle signal). All other status
|
|
64
|
+
codes are returned to the caller unchanged; network exceptions
|
|
65
|
+
(``Timeout`` / ``ConnectionError`` / ...) propagate so callers can map them
|
|
66
|
+
to :class:`PubChemAPIError` exactly as before.
|
|
67
|
+
"""
|
|
68
|
+
timeout = timeout if timeout is not None else config.PUBCHEM_TIMEOUT_S
|
|
69
|
+
# M10 audit fix (2026-07-14): if config.PUBCHEM_MAX_RETRIES were ever 0
|
|
70
|
+
# (or negative), `range(config.PUBCHEM_MAX_RETRIES)` would iterate zero
|
|
71
|
+
# times, leaving `response` at its None initializer and returning None
|
|
72
|
+
# from a function typed to return requests.Response — every caller
|
|
73
|
+
# then hits AttributeError on response.status_code. Always attempt at
|
|
74
|
+
# least once regardless of the configured retry count.
|
|
75
|
+
max_attempts = max(1, config.PUBCHEM_MAX_RETRIES)
|
|
76
|
+
response = None
|
|
77
|
+
for attempt in range(max_attempts):
|
|
78
|
+
_throttle()
|
|
79
|
+
response = requests.get(url, params=params, timeout=timeout)
|
|
80
|
+
if response.status_code != 503:
|
|
81
|
+
return response
|
|
82
|
+
# Throttled — back off (capped) and retry, unless this was the last try.
|
|
83
|
+
if attempt < max_attempts - 1:
|
|
84
|
+
backoff = min(
|
|
85
|
+
config.PUBCHEM_BACKOFF_BASE_S * (2**attempt),
|
|
86
|
+
config.PUBCHEM_BACKOFF_MAX_S,
|
|
87
|
+
)
|
|
88
|
+
logger.warning(
|
|
89
|
+
"PubChem throttled (503); retrying in %.1fs (attempt %d/%d)",
|
|
90
|
+
backoff,
|
|
91
|
+
attempt + 1,
|
|
92
|
+
max_attempts,
|
|
93
|
+
)
|
|
94
|
+
time.sleep(backoff)
|
|
95
|
+
# Exhausted retries — hand the last 503 back; caller raises via raise_for_status.
|
|
96
|
+
return response # type: ignore[return-value]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class PubChemError(Exception):
|
|
100
|
+
"""Base exception for PubChem-related errors."""
|
|
101
|
+
|
|
102
|
+
pass
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class MoleculeNotFoundError(PubChemError):
|
|
106
|
+
"""Raised when a molecule cannot be found in PubChem."""
|
|
107
|
+
|
|
108
|
+
pass
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class PubChemAPIError(PubChemError):
|
|
112
|
+
"""Raised when PubChem API request fails."""
|
|
113
|
+
|
|
114
|
+
pass
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def search_molecule_by_name(name: str) -> int:
|
|
118
|
+
"""
|
|
119
|
+
Search for a molecule in PubChem by name and return its CID.
|
|
120
|
+
|
|
121
|
+
Args:
|
|
122
|
+
name: Common name or IUPAC name of the molecule
|
|
123
|
+
|
|
124
|
+
Returns:
|
|
125
|
+
int: PubChem Compound ID (CID)
|
|
126
|
+
|
|
127
|
+
Raises:
|
|
128
|
+
PubChemAPIError: If API request fails
|
|
129
|
+
MoleculeNotFoundError: If molecule not found
|
|
130
|
+
"""
|
|
131
|
+
url = f"{PUBCHEM_BASE_URL}/compound/name/{quote(name, safe='')}/cids/JSON"
|
|
132
|
+
|
|
133
|
+
try:
|
|
134
|
+
logger.debug(f"Searching PubChem for: {name}")
|
|
135
|
+
response = _http_get(url)
|
|
136
|
+
|
|
137
|
+
if response.status_code == 404:
|
|
138
|
+
raise MoleculeNotFoundError(f"Molecule '{name}' not found in PubChem")
|
|
139
|
+
|
|
140
|
+
response.raise_for_status()
|
|
141
|
+
data = response.json()
|
|
142
|
+
|
|
143
|
+
cids = data.get("IdentifierList", {}).get("CID", [])
|
|
144
|
+
if not cids:
|
|
145
|
+
raise MoleculeNotFoundError(f"No CID found for '{name}'")
|
|
146
|
+
|
|
147
|
+
cid: int = int(cids[0]) # Take first match
|
|
148
|
+
logger.info(f"Found CID {cid} for '{name}'")
|
|
149
|
+
return cid
|
|
150
|
+
|
|
151
|
+
except requests.RequestException as e:
|
|
152
|
+
logger.error(f"PubChem API request failed: {e}")
|
|
153
|
+
raise PubChemAPIError(f"Failed to connect to PubChem: {e}") from e
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def search_cid_by_inchikey(inchikey: str) -> int:
|
|
157
|
+
"""Resolve a standard InChIKey to a PubChem CID.
|
|
158
|
+
|
|
159
|
+
InChIKeys are hashes and cannot be inverted to a structure locally, so this
|
|
160
|
+
is the one identifier type that always requires the network.
|
|
161
|
+
"""
|
|
162
|
+
url = f"{PUBCHEM_BASE_URL}/compound/inchikey/{quote(inchikey, safe='')}/cids/JSON"
|
|
163
|
+
try:
|
|
164
|
+
response = _http_get(url)
|
|
165
|
+
if response.status_code == 404:
|
|
166
|
+
raise MoleculeNotFoundError(f"InChIKey '{inchikey}' not found in PubChem")
|
|
167
|
+
response.raise_for_status()
|
|
168
|
+
cids = response.json().get("IdentifierList", {}).get("CID", [])
|
|
169
|
+
if not cids:
|
|
170
|
+
raise MoleculeNotFoundError(f"No CID found for InChIKey '{inchikey}'")
|
|
171
|
+
return int(cids[0])
|
|
172
|
+
except requests.RequestException as e:
|
|
173
|
+
logger.error(f"PubChem InChIKey request failed: {e}")
|
|
174
|
+
raise PubChemAPIError(f"Failed to connect to PubChem: {e}") from e
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def search_cids_by_name(name: str) -> list:
|
|
178
|
+
"""Return ALL PubChem CIDs matching a name (best-match order); [] if none.
|
|
179
|
+
|
|
180
|
+
Unlike :func:`search_molecule_by_name` (which returns just the first hit),
|
|
181
|
+
this exposes every match so the UI can disambiguate.
|
|
182
|
+
"""
|
|
183
|
+
url = f"{PUBCHEM_BASE_URL}/compound/name/{quote(name, safe='')}/cids/JSON"
|
|
184
|
+
try:
|
|
185
|
+
response = _http_get(url)
|
|
186
|
+
if response.status_code == 404:
|
|
187
|
+
return []
|
|
188
|
+
response.raise_for_status()
|
|
189
|
+
return [
|
|
190
|
+
int(c) for c in response.json().get("IdentifierList", {}).get("CID", [])
|
|
191
|
+
]
|
|
192
|
+
except requests.RequestException as e:
|
|
193
|
+
logger.error(f"PubChem CID-list request failed: {e}")
|
|
194
|
+
raise PubChemAPIError(f"Failed to connect to PubChem: {e}") from e
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def search_pubchem_candidates(query: str, max_results: int = 10) -> list:
|
|
198
|
+
"""Lightweight candidate descriptors for an ambiguous name query.
|
|
199
|
+
|
|
200
|
+
Returns a list of ``{cid, title, formula, mw}`` dicts (best-match order,
|
|
201
|
+
capped at ``max_results``), or ``[]`` if nothing matches. Uses one batch
|
|
202
|
+
property request rather than fetching each full structure.
|
|
203
|
+
"""
|
|
204
|
+
cids = search_cids_by_name(query)[:max_results]
|
|
205
|
+
if not cids:
|
|
206
|
+
return []
|
|
207
|
+
cid_str = ",".join(str(c) for c in cids)
|
|
208
|
+
url = (
|
|
209
|
+
f"{PUBCHEM_BASE_URL}/compound/cid/{cid_str}"
|
|
210
|
+
f"/property/MolecularFormula,MolecularWeight,Title/JSON"
|
|
211
|
+
)
|
|
212
|
+
try:
|
|
213
|
+
response = _http_get(url)
|
|
214
|
+
response.raise_for_status()
|
|
215
|
+
props = response.json().get("PropertyTable", {}).get("Properties", [])
|
|
216
|
+
except requests.RequestException as e:
|
|
217
|
+
logger.error(f"PubChem property request failed: {e}")
|
|
218
|
+
raise PubChemAPIError(f"Failed to connect to PubChem: {e}") from e
|
|
219
|
+
|
|
220
|
+
# Preserve the CID search order (the property endpoint may reorder).
|
|
221
|
+
by_cid = {int(p.get("CID")): p for p in props if p.get("CID") is not None}
|
|
222
|
+
out = []
|
|
223
|
+
for cid in cids:
|
|
224
|
+
p = by_cid.get(cid)
|
|
225
|
+
if p is None:
|
|
226
|
+
continue
|
|
227
|
+
try:
|
|
228
|
+
mw = float(p.get("MolecularWeight", 0) or 0)
|
|
229
|
+
except (TypeError, ValueError):
|
|
230
|
+
mw = 0.0
|
|
231
|
+
out.append(
|
|
232
|
+
{
|
|
233
|
+
"cid": cid,
|
|
234
|
+
"title": p.get("Title") or f"CID {cid}",
|
|
235
|
+
"formula": p.get("MolecularFormula", "?"),
|
|
236
|
+
"mw": mw,
|
|
237
|
+
}
|
|
238
|
+
)
|
|
239
|
+
return out
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@lru_cache(maxsize=50)
|
|
243
|
+
def get_molecule_sdf(cid: int, conformer_3d: bool = True) -> str:
|
|
244
|
+
"""
|
|
245
|
+
Retrieve molecule SDF from PubChem by CID.
|
|
246
|
+
|
|
247
|
+
Args:
|
|
248
|
+
cid: PubChem Compound ID
|
|
249
|
+
conformer_3d: If True, fetch 3D conformer; if False, fetch 2D structure
|
|
250
|
+
|
|
251
|
+
Returns:
|
|
252
|
+
str: SDF file content
|
|
253
|
+
|
|
254
|
+
Raises:
|
|
255
|
+
PubChemAPIError: If API request fails
|
|
256
|
+
MoleculeNotFoundError: If CID not found
|
|
257
|
+
"""
|
|
258
|
+
record_type = "3d" if conformer_3d else "2d"
|
|
259
|
+
url = f"{PUBCHEM_BASE_URL}/compound/cid/{cid}/record/SDF"
|
|
260
|
+
|
|
261
|
+
params = {}
|
|
262
|
+
if conformer_3d:
|
|
263
|
+
params["record_type"] = "3d"
|
|
264
|
+
|
|
265
|
+
try:
|
|
266
|
+
logger.debug(f"Fetching {record_type.upper()} SDF for CID {cid}")
|
|
267
|
+
response = _http_get(url, params=params)
|
|
268
|
+
|
|
269
|
+
if response.status_code == 404:
|
|
270
|
+
if conformer_3d:
|
|
271
|
+
# Try falling back to 2D if 3D not available
|
|
272
|
+
logger.warning(f"No 3D structure for CID {cid}, trying 2D")
|
|
273
|
+
return get_molecule_sdf(cid, conformer_3d=False)
|
|
274
|
+
raise MoleculeNotFoundError(f"CID {cid} not found in PubChem")
|
|
275
|
+
|
|
276
|
+
response.raise_for_status()
|
|
277
|
+
sdf_content: str = str(response.text)
|
|
278
|
+
|
|
279
|
+
logger.info(f"Retrieved {record_type.upper()} SDF for CID {cid}")
|
|
280
|
+
return sdf_content
|
|
281
|
+
|
|
282
|
+
except requests.RequestException as e:
|
|
283
|
+
logger.error(f"PubChem SDF request failed: {e}")
|
|
284
|
+
raise PubChemAPIError(f"Failed to retrieve molecule: {e}") from e
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _separate_fragments(mol: Any, min_gap: float = 3.0) -> None:
|
|
288
|
+
"""Push disconnected fragments (e.g. a salt's counterion) apart, in place.
|
|
289
|
+
|
|
290
|
+
RDKit's ``EmbedMolecule`` places multiple fragments in one coordinate frame
|
|
291
|
+
and frequently overlaps them — a counterion can land ~1.4 Å from the cation,
|
|
292
|
+
which distance-based bond perception then reads as a (hyper)valent bond and
|
|
293
|
+
the renderer rejects ("Valence of atom N is …, larger than allowed"). This
|
|
294
|
+
is why salts like methylene blue (cation + Cl⁻) failed.
|
|
295
|
+
|
|
296
|
+
After embedding, translate every non-largest fragment radially outward from
|
|
297
|
+
the main fragment so the closest inter-fragment gap is at least ``min_gap``
|
|
298
|
+
Å. Operates on the existing conformer; atom order is preserved. No-op for
|
|
299
|
+
single-fragment molecules.
|
|
300
|
+
"""
|
|
301
|
+
if not RDKIT_AVAILABLE or mol.GetNumConformers() == 0:
|
|
302
|
+
return
|
|
303
|
+
frags = Chem.GetMolFrags(mol) # tuple of atom-index tuples, order preserved
|
|
304
|
+
if len(frags) <= 1:
|
|
305
|
+
return
|
|
306
|
+
|
|
307
|
+
import numpy as np
|
|
308
|
+
from rdkit.Geometry import Point3D
|
|
309
|
+
|
|
310
|
+
conf = mol.GetConformer()
|
|
311
|
+
pos = {i: np.array(conf.GetAtomPosition(i)) for i in range(mol.GetNumAtoms())}
|
|
312
|
+
main = max(frags, key=len)
|
|
313
|
+
main_c = np.mean([pos[i] for i in main], axis=0)
|
|
314
|
+
main_r = max((float(np.linalg.norm(pos[i] - main_c)) for i in main), default=0.0)
|
|
315
|
+
for frag in frags:
|
|
316
|
+
if frag is main:
|
|
317
|
+
continue
|
|
318
|
+
fc = np.mean([pos[i] for i in frag], axis=0)
|
|
319
|
+
fr = max((float(np.linalg.norm(pos[i] - fc)) for i in frag), default=0.0)
|
|
320
|
+
direction = fc - main_c
|
|
321
|
+
norm = float(np.linalg.norm(direction))
|
|
322
|
+
direction = direction / norm if norm > 1e-6 else np.array([1.0, 0.0, 0.0])
|
|
323
|
+
shift = (main_c + direction * (main_r + fr + min_gap)) - fc
|
|
324
|
+
for i in frag:
|
|
325
|
+
new = pos[i] + shift
|
|
326
|
+
conf.SetAtomPosition(
|
|
327
|
+
i, Point3D(float(new[0]), float(new[1]), float(new[2]))
|
|
328
|
+
)
|
|
329
|
+
pos[i] = new
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def sdf_to_xyz(sdf_content: str) -> Tuple[str, Dict[str, Any]]:
|
|
333
|
+
"""
|
|
334
|
+
Convert SDF content to XYZ format string.
|
|
335
|
+
|
|
336
|
+
Args:
|
|
337
|
+
sdf_content: SDF file content as string
|
|
338
|
+
|
|
339
|
+
Returns:
|
|
340
|
+
Tuple of (xyz_string, metadata_dict)
|
|
341
|
+
xyz_string format: "n_atoms\\ncomment\\natom x y z\\n..."
|
|
342
|
+
metadata includes: formula, molecular_weight, charge
|
|
343
|
+
|
|
344
|
+
Raises:
|
|
345
|
+
ValueError: If SDF parsing fails
|
|
346
|
+
"""
|
|
347
|
+
if not RDKIT_AVAILABLE:
|
|
348
|
+
raise ImportError("RDKit is required for SDF to XYZ conversion")
|
|
349
|
+
|
|
350
|
+
try:
|
|
351
|
+
# Parse SDF with RDKit, keeping any explicit hydrogens (3D PubChem SDFs
|
|
352
|
+
# already carry them with real coordinates).
|
|
353
|
+
mol = Chem.MolFromMolBlock(sdf_content, removeHs=False)
|
|
354
|
+
|
|
355
|
+
if mol is None:
|
|
356
|
+
raise ValueError("Failed to parse SDF content")
|
|
357
|
+
|
|
358
|
+
# Add any missing hydrogens *with* coordinates. Without addCoords the
|
|
359
|
+
# new H default to the origin, which — combined with a 2D SDF — yields a
|
|
360
|
+
# degenerate geometry (atoms piled at 0,0,0) that bond perception then
|
|
361
|
+
# reads as absurd valences.
|
|
362
|
+
mol = Chem.AddHs(mol, addCoords=True)
|
|
363
|
+
|
|
364
|
+
# Re-embed in 3D whenever there is no conformer, or the conformer is 2D
|
|
365
|
+
# (PubChem's 3D→2D fallback). A flat conformer must not be returned as a
|
|
366
|
+
# "3D" structure.
|
|
367
|
+
conf = mol.GetConformer() if mol.GetNumConformers() else None
|
|
368
|
+
coords_embedded = conf is None or not conf.Is3D()
|
|
369
|
+
if coords_embedded:
|
|
370
|
+
if AllChem.EmbedMolecule(mol, randomSeed=42) != 0:
|
|
371
|
+
AllChem.EmbedMolecule(mol, randomSeed=42, useRandomCoords=True)
|
|
372
|
+
try:
|
|
373
|
+
AllChem.MMFFOptimizeMolecule(mol)
|
|
374
|
+
except Exception:
|
|
375
|
+
try:
|
|
376
|
+
AllChem.UFFOptimizeMolecule(mol)
|
|
377
|
+
except Exception:
|
|
378
|
+
pass
|
|
379
|
+
# Salts/counterions embed jammed together — separate them so bond
|
|
380
|
+
# perception doesn't see a bonded counterion.
|
|
381
|
+
_separate_fragments(mol)
|
|
382
|
+
|
|
383
|
+
# Extract coordinates and build XYZ string
|
|
384
|
+
conf = mol.GetConformer()
|
|
385
|
+
xyz_lines = [str(mol.GetNumAtoms())]
|
|
386
|
+
|
|
387
|
+
# Get molecular formula
|
|
388
|
+
formula = Chem.rdMolDescriptors.CalcMolFormula(mol)
|
|
389
|
+
xyz_lines.append(f"PubChem molecule: {formula}")
|
|
390
|
+
|
|
391
|
+
for atom in mol.GetAtoms():
|
|
392
|
+
pos = conf.GetAtomPosition(atom.GetIdx())
|
|
393
|
+
symbol = atom.GetSymbol()
|
|
394
|
+
xyz_lines.append(f"{symbol:3s} {pos.x:12.6f} {pos.y:12.6f} {pos.z:12.6f}")
|
|
395
|
+
|
|
396
|
+
xyz_string = "\n".join(xyz_lines)
|
|
397
|
+
|
|
398
|
+
# Gather metadata
|
|
399
|
+
metadata = {
|
|
400
|
+
"formula": formula,
|
|
401
|
+
"molecular_weight": Descriptors.MolWt(mol),
|
|
402
|
+
"charge": Chem.GetFormalCharge(mol),
|
|
403
|
+
"num_atoms": mol.GetNumAtoms(),
|
|
404
|
+
"num_heavy_atoms": mol.GetNumHeavyAtoms(),
|
|
405
|
+
"coords_embedded": coords_embedded,
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
logger.debug(f"Converted SDF to XYZ: {metadata['formula']}")
|
|
409
|
+
return xyz_string, metadata
|
|
410
|
+
|
|
411
|
+
except Exception as e:
|
|
412
|
+
logger.error(f"SDF to XYZ conversion failed: {e}")
|
|
413
|
+
raise ValueError(f"Failed to convert SDF to XYZ: {e}") from e
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def fetch_molecule(
|
|
417
|
+
name: str, conformer_3d: bool = True
|
|
418
|
+
) -> Tuple[str, Dict[str, Any], int]:
|
|
419
|
+
"""
|
|
420
|
+
High-level function to fetch molecule from PubChem by name.
|
|
421
|
+
|
|
422
|
+
Performs search, retrieves SDF, and converts to XYZ in one call.
|
|
423
|
+
|
|
424
|
+
Args:
|
|
425
|
+
name: Molecule name (common or IUPAC)
|
|
426
|
+
conformer_3d: If True, fetch 3D structure; if False, 2D
|
|
427
|
+
|
|
428
|
+
Returns:
|
|
429
|
+
Tuple of (xyz_string, metadata_dict, cid)
|
|
430
|
+
|
|
431
|
+
Raises:
|
|
432
|
+
PubChemError: If any step fails
|
|
433
|
+
"""
|
|
434
|
+
logger.info(f"Fetching molecule '{name}' from PubChem")
|
|
435
|
+
|
|
436
|
+
# Search for CID
|
|
437
|
+
cid = search_molecule_by_name(name)
|
|
438
|
+
|
|
439
|
+
# Get SDF
|
|
440
|
+
sdf_content = get_molecule_sdf(cid, conformer_3d=conformer_3d)
|
|
441
|
+
|
|
442
|
+
# Convert to XYZ
|
|
443
|
+
xyz_string, metadata = sdf_to_xyz(sdf_content)
|
|
444
|
+
|
|
445
|
+
# Add CID to metadata
|
|
446
|
+
metadata["pubchem_cid"] = cid
|
|
447
|
+
metadata["pubchem_name"] = name
|
|
448
|
+
|
|
449
|
+
logger.info(f"Successfully fetched '{name}' (CID: {cid})")
|
|
450
|
+
return xyz_string, metadata, cid
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def get_common_molecules() -> Dict[str, str]:
|
|
454
|
+
"""
|
|
455
|
+
Get a curated list of common molecules for educational use.
|
|
456
|
+
|
|
457
|
+
Returns:
|
|
458
|
+
Dict mapping display names to PubChem search names
|
|
459
|
+
"""
|
|
460
|
+
return {
|
|
461
|
+
# Simple molecules
|
|
462
|
+
"Water (H₂O)": "water",
|
|
463
|
+
"Hydrogen (H₂)": "hydrogen",
|
|
464
|
+
"Oxygen (O₂)": "oxygen",
|
|
465
|
+
"Nitrogen (N₂)": "nitrogen",
|
|
466
|
+
"Carbon Dioxide (CO₂)": "carbon dioxide",
|
|
467
|
+
"Ammonia (NH₃)": "ammonia",
|
|
468
|
+
"Methane (CH₄)": "methane",
|
|
469
|
+
# Organic molecules
|
|
470
|
+
"Ethanol (CH₃CH₂OH)": "ethanol",
|
|
471
|
+
"Acetic Acid (CH₃COOH)": "acetic acid",
|
|
472
|
+
"Acetone (CH₃COCH₃)": "acetone",
|
|
473
|
+
"Benzene (C₆H₆)": "benzene",
|
|
474
|
+
"Toluene (C₆H₅CH₃)": "toluene",
|
|
475
|
+
"Phenol (C₆H₅OH)": "phenol",
|
|
476
|
+
# Biochemical molecules
|
|
477
|
+
"Glucose (C₆H₁₂O₆)": "glucose",
|
|
478
|
+
"Glycine (NH₂CH₂COOH)": "glycine",
|
|
479
|
+
"Alanine (CH₃CH(NH₂)COOH)": "alanine",
|
|
480
|
+
"Caffeine": "caffeine",
|
|
481
|
+
"Aspirin": "aspirin",
|
|
482
|
+
"Vitamin C": "ascorbic acid",
|
|
483
|
+
# Ions (may need special handling)
|
|
484
|
+
"Hydronium (H₃O⁺)": "hydronium",
|
|
485
|
+
"Hydroxide (OH⁻)": "hydroxide",
|
|
486
|
+
"Ammonium (NH₄⁺)": "ammonium",
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def student_friendly_fetch(name: str) -> Tuple[Optional[str], str]:
|
|
491
|
+
"""
|
|
492
|
+
Fetch molecule with student-friendly error messages.
|
|
493
|
+
|
|
494
|
+
Args:
|
|
495
|
+
name: Molecule name to search
|
|
496
|
+
|
|
497
|
+
Returns:
|
|
498
|
+
Tuple of (xyz_string, message)
|
|
499
|
+
xyz_string is None if fetch failed
|
|
500
|
+
message describes success or failure
|
|
501
|
+
"""
|
|
502
|
+
try:
|
|
503
|
+
xyz_string, metadata, cid = fetch_molecule(name, conformer_3d=True)
|
|
504
|
+
|
|
505
|
+
message = (
|
|
506
|
+
f"✓ Found '{name}' in PubChem!\n"
|
|
507
|
+
f" CID: {cid}\n"
|
|
508
|
+
f" Formula: {metadata['formula']}\n"
|
|
509
|
+
f" Atoms: {metadata['num_atoms']} "
|
|
510
|
+
f"({metadata['num_heavy_atoms']} heavy atoms)\n"
|
|
511
|
+
f" Molecular Weight: {metadata['molecular_weight']:.2f} g/mol"
|
|
512
|
+
)
|
|
513
|
+
|
|
514
|
+
return xyz_string, message
|
|
515
|
+
|
|
516
|
+
except MoleculeNotFoundError:
|
|
517
|
+
message = (
|
|
518
|
+
f"❌ Could not find '{name}' in PubChem.\n"
|
|
519
|
+
f" Try:\n"
|
|
520
|
+
f" • Check spelling (e.g., 'ethanol' not 'ethonal')\n"
|
|
521
|
+
f" • Use IUPAC name (e.g., 'ethanol' not 'alcohol')\n"
|
|
522
|
+
f" • Use common name (e.g., 'water' not 'dihydrogen monoxide')\n"
|
|
523
|
+
f" • Search manually at: https://pubchem.ncbi.nlm.nih.gov/"
|
|
524
|
+
)
|
|
525
|
+
return None, message
|
|
526
|
+
|
|
527
|
+
except PubChemAPIError:
|
|
528
|
+
message = (
|
|
529
|
+
"❌ Connection to PubChem failed.\n"
|
|
530
|
+
" • Check your internet connection\n"
|
|
531
|
+
" • Try again in a moment\n"
|
|
532
|
+
" • Use preset molecules if problem persists"
|
|
533
|
+
)
|
|
534
|
+
return None, message
|
|
535
|
+
|
|
536
|
+
except Exception as e:
|
|
537
|
+
message = (
|
|
538
|
+
f"❌ Error fetching molecule: {str(e)}\n"
|
|
539
|
+
f" Please try a different molecule or contact your instructor."
|
|
540
|
+
)
|
|
541
|
+
logger.error(f"Unexpected error in student_friendly_fetch: {e}", exc_info=True)
|
|
542
|
+
return None, message
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def check_pubchem_availability() -> bool:
|
|
546
|
+
"""
|
|
547
|
+
Check if PubChem API is accessible.
|
|
548
|
+
|
|
549
|
+
Returns:
|
|
550
|
+
bool: True if PubChem is accessible, False otherwise
|
|
551
|
+
"""
|
|
552
|
+
# M10 audit fix (2026-07-14): this used to call requests.get() directly,
|
|
553
|
+
# bypassing the shared client-side rate limiter (_throttle()) that every
|
|
554
|
+
# other function in this module goes through — a burst of concurrent
|
|
555
|
+
# availability checks (e.g. several students in a classroom clicking
|
|
556
|
+
# "check connection" around the same time) could exceed PubChem's
|
|
557
|
+
# server-side throttle. It also hardcoded timeout=5 instead of the
|
|
558
|
+
# config.PUBCHEM_AVAILABILITY_TIMEOUT_S constant defined specifically
|
|
559
|
+
# for this probe, so changing that constant silently had no effect
|
|
560
|
+
# here. Calls _throttle() directly (not the full _http_get retry loop —
|
|
561
|
+
# this is meant to be a quick, no-retry reachability probe) and uses
|
|
562
|
+
# the configured timeout.
|
|
563
|
+
try:
|
|
564
|
+
url = f"{PUBCHEM_BASE_URL}/compound/cid/962/property/MolecularFormula/JSON"
|
|
565
|
+
_throttle()
|
|
566
|
+
response = requests.get(url, timeout=config.PUBCHEM_AVAILABILITY_TIMEOUT_S)
|
|
567
|
+
return bool(response.status_code == 200)
|
|
568
|
+
except Exception:
|
|
569
|
+
return False
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
# ============================================================================
|
|
573
|
+
# SMILES Input and 2D Structure Rendering
|
|
574
|
+
# ============================================================================
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def smiles_to_xyz(smiles: str, optimize_3d: bool = True) -> Tuple[str, Dict[str, Any]]:
|
|
578
|
+
"""
|
|
579
|
+
Convert SMILES string to XYZ coordinates with 3D structure generation.
|
|
580
|
+
|
|
581
|
+
Args:
|
|
582
|
+
smiles: SMILES string (e.g., "CCO" for ethanol)
|
|
583
|
+
optimize_3d: If True, generate and optimize 3D coordinates with UFF
|
|
584
|
+
|
|
585
|
+
Returns:
|
|
586
|
+
Tuple of (xyz_string, metadata_dict)
|
|
587
|
+
xyz_string format: "n_atoms\\ncomment\\natom x y z\\n..."
|
|
588
|
+
metadata includes: formula, molecular_weight, charge, smiles
|
|
589
|
+
|
|
590
|
+
Raises:
|
|
591
|
+
ValueError: If SMILES parsing fails or 3D generation fails
|
|
592
|
+
ImportError: If RDKit is not available
|
|
593
|
+
"""
|
|
594
|
+
if not RDKIT_AVAILABLE:
|
|
595
|
+
raise ImportError("RDKit is required for SMILES conversion")
|
|
596
|
+
|
|
597
|
+
try:
|
|
598
|
+
# Parse SMILES
|
|
599
|
+
mol = Chem.MolFromSmiles(smiles)
|
|
600
|
+
if mol is None:
|
|
601
|
+
raise ValueError(f"Invalid SMILES string: {smiles}")
|
|
602
|
+
|
|
603
|
+
# Add hydrogens
|
|
604
|
+
mol = Chem.AddHs(mol)
|
|
605
|
+
|
|
606
|
+
# Generate 3D coordinates
|
|
607
|
+
if optimize_3d:
|
|
608
|
+
# Generate conformer
|
|
609
|
+
result = AllChem.EmbedMolecule(mol, randomSeed=42)
|
|
610
|
+
if result != 0:
|
|
611
|
+
# Try with random coords if embedding fails
|
|
612
|
+
AllChem.EmbedMolecule(mol, randomSeed=42, useRandomCoords=True)
|
|
613
|
+
|
|
614
|
+
# Optimize with UFF force field
|
|
615
|
+
try:
|
|
616
|
+
AllChem.UFFOptimizeMolecule(mol)
|
|
617
|
+
except Exception:
|
|
618
|
+
logger.warning("UFF optimization failed, using unoptimized coordinates")
|
|
619
|
+
|
|
620
|
+
_separate_fragments(mol) # keep salt counterions apart
|
|
621
|
+
|
|
622
|
+
# Extract coordinates
|
|
623
|
+
if mol.GetNumConformers() == 0:
|
|
624
|
+
raise ValueError("Failed to generate 3D coordinates")
|
|
625
|
+
|
|
626
|
+
conf = mol.GetConformer()
|
|
627
|
+
xyz_lines = [str(mol.GetNumAtoms())]
|
|
628
|
+
|
|
629
|
+
# Get molecular formula
|
|
630
|
+
formula = Chem.rdMolDescriptors.CalcMolFormula(mol)
|
|
631
|
+
xyz_lines.append(f"Generated from SMILES: {smiles} ({formula})")
|
|
632
|
+
|
|
633
|
+
# Build XYZ string
|
|
634
|
+
for atom in mol.GetAtoms():
|
|
635
|
+
pos = conf.GetAtomPosition(atom.GetIdx())
|
|
636
|
+
symbol = atom.GetSymbol()
|
|
637
|
+
xyz_lines.append(f"{symbol:3s} {pos.x:12.6f} {pos.y:12.6f} {pos.z:12.6f}")
|
|
638
|
+
|
|
639
|
+
xyz_string = "\n".join(xyz_lines)
|
|
640
|
+
|
|
641
|
+
# Gather metadata
|
|
642
|
+
metadata = {
|
|
643
|
+
"formula": formula,
|
|
644
|
+
"molecular_weight": Descriptors.MolWt(mol),
|
|
645
|
+
"charge": Chem.GetFormalCharge(mol),
|
|
646
|
+
"num_atoms": mol.GetNumAtoms(),
|
|
647
|
+
"num_heavy_atoms": mol.GetNumHeavyAtoms(),
|
|
648
|
+
"smiles": smiles,
|
|
649
|
+
"canonical_smiles": Chem.MolToSmiles(mol),
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
logger.info(f"Converted SMILES '{smiles}' to XYZ: {metadata['formula']}")
|
|
653
|
+
return xyz_string, metadata
|
|
654
|
+
|
|
655
|
+
except Exception as e:
|
|
656
|
+
logger.error(f"SMILES to XYZ conversion failed: {e}")
|
|
657
|
+
raise ValueError(f"Failed to convert SMILES to XYZ: {e}") from e
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
def inchi_to_xyz(inchi: str, optimize_3d: bool = True) -> Tuple[str, Dict[str, Any]]:
|
|
661
|
+
"""Convert an InChI string to XYZ coordinates via RDKit (no network).
|
|
662
|
+
|
|
663
|
+
Mirrors :func:`smiles_to_xyz`: parse → add H → embed (ETKDG, seed 42) →
|
|
664
|
+
UFF-optimize. Returns ``(xyz_string, metadata)``; metadata carries the
|
|
665
|
+
canonical SMILES so the caller can label provenance consistently.
|
|
666
|
+
"""
|
|
667
|
+
if not RDKIT_AVAILABLE:
|
|
668
|
+
raise ImportError("RDKit is required for InChI conversion")
|
|
669
|
+
|
|
670
|
+
try:
|
|
671
|
+
mol = Chem.MolFromInchi(inchi)
|
|
672
|
+
if mol is None:
|
|
673
|
+
raise ValueError(f"Invalid InChI string: {inchi}")
|
|
674
|
+
|
|
675
|
+
mol = Chem.AddHs(mol)
|
|
676
|
+
if optimize_3d:
|
|
677
|
+
result = AllChem.EmbedMolecule(mol, randomSeed=42)
|
|
678
|
+
if result != 0:
|
|
679
|
+
AllChem.EmbedMolecule(mol, randomSeed=42, useRandomCoords=True)
|
|
680
|
+
try:
|
|
681
|
+
AllChem.UFFOptimizeMolecule(mol)
|
|
682
|
+
except Exception:
|
|
683
|
+
logger.warning("UFF optimization failed, using unoptimized coordinates")
|
|
684
|
+
|
|
685
|
+
_separate_fragments(mol) # keep salt counterions apart
|
|
686
|
+
|
|
687
|
+
if mol.GetNumConformers() == 0:
|
|
688
|
+
raise ValueError("Failed to generate 3D coordinates")
|
|
689
|
+
|
|
690
|
+
conf = mol.GetConformer()
|
|
691
|
+
formula = Chem.rdMolDescriptors.CalcMolFormula(mol)
|
|
692
|
+
xyz_lines = [str(mol.GetNumAtoms()), f"Generated from InChI ({formula})"]
|
|
693
|
+
for atom in mol.GetAtoms():
|
|
694
|
+
pos = conf.GetAtomPosition(atom.GetIdx())
|
|
695
|
+
xyz_lines.append(
|
|
696
|
+
f"{atom.GetSymbol():3s} {pos.x:12.6f} {pos.y:12.6f} {pos.z:12.6f}"
|
|
697
|
+
)
|
|
698
|
+
|
|
699
|
+
metadata = {
|
|
700
|
+
"formula": formula,
|
|
701
|
+
"molecular_weight": Descriptors.MolWt(mol),
|
|
702
|
+
"charge": Chem.GetFormalCharge(mol),
|
|
703
|
+
"num_atoms": mol.GetNumAtoms(),
|
|
704
|
+
"num_heavy_atoms": mol.GetNumHeavyAtoms(),
|
|
705
|
+
"inchi": inchi,
|
|
706
|
+
"canonical_smiles": Chem.MolToSmiles(mol),
|
|
707
|
+
}
|
|
708
|
+
logger.info(f"Converted InChI to XYZ: {formula}")
|
|
709
|
+
return "\n".join(xyz_lines), metadata
|
|
710
|
+
|
|
711
|
+
except Exception as e:
|
|
712
|
+
logger.error(f"InChI to XYZ conversion failed: {e}")
|
|
713
|
+
raise ValueError(f"Failed to convert InChI to XYZ: {e}") from e
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
def student_friendly_smiles_to_xyz(smiles: str) -> Tuple[Optional[str], str]:
|
|
717
|
+
"""
|
|
718
|
+
Convert SMILES to XYZ with student-friendly error messages.
|
|
719
|
+
|
|
720
|
+
Args:
|
|
721
|
+
smiles: SMILES string
|
|
722
|
+
|
|
723
|
+
Returns:
|
|
724
|
+
Tuple of (xyz_string, message)
|
|
725
|
+
xyz_string is None if conversion failed
|
|
726
|
+
message describes success or failure
|
|
727
|
+
"""
|
|
728
|
+
try:
|
|
729
|
+
xyz_string, metadata = smiles_to_xyz(smiles, optimize_3d=True)
|
|
730
|
+
|
|
731
|
+
message = (
|
|
732
|
+
f"✓ Converted SMILES to 3D structure!\n"
|
|
733
|
+
f" SMILES: {smiles}\n"
|
|
734
|
+
f" Formula: {metadata['formula']}\n"
|
|
735
|
+
f" Atoms: {metadata['num_atoms']} "
|
|
736
|
+
f"({metadata['num_heavy_atoms']} heavy atoms)\n"
|
|
737
|
+
f" Molecular Weight: {metadata['molecular_weight']:.2f} g/mol\n"
|
|
738
|
+
f" Canonical SMILES: {metadata['canonical_smiles']}"
|
|
739
|
+
)
|
|
740
|
+
|
|
741
|
+
return xyz_string, message
|
|
742
|
+
|
|
743
|
+
except ValueError as e:
|
|
744
|
+
message = (
|
|
745
|
+
f"❌ Invalid SMILES string: {smiles}\n"
|
|
746
|
+
f" Error: {str(e)}\n\n"
|
|
747
|
+
f" SMILES Tips:\n"
|
|
748
|
+
f" • Check syntax (use RDKit/OpenBabel style)\n"
|
|
749
|
+
f" • Ethanol: CCO or C(C)O\n"
|
|
750
|
+
f" • Benzene: c1ccccc1 or C1=CC=CC=C1\n"
|
|
751
|
+
f" • Water: O (just the atom symbol)\n"
|
|
752
|
+
f" • Methane: C\n\n"
|
|
753
|
+
f" Resources:\n"
|
|
754
|
+
f" • SMILES Tutorial: https://www.daylight.com/dayhtml/doc/theory/theory.smiles.html\n"
|
|
755
|
+
f" • Draw structure: https://pubchem.ncbi.nlm.nih.gov/edit3/index.html"
|
|
756
|
+
)
|
|
757
|
+
return None, message
|
|
758
|
+
|
|
759
|
+
except ImportError:
|
|
760
|
+
message = (
|
|
761
|
+
"❌ RDKit is required for SMILES conversion.\n"
|
|
762
|
+
" Install with: conda install -c conda-forge rdkit"
|
|
763
|
+
)
|
|
764
|
+
return None, message
|
|
765
|
+
|
|
766
|
+
except Exception as e:
|
|
767
|
+
message = (
|
|
768
|
+
f"❌ Error converting SMILES: {str(e)}\n"
|
|
769
|
+
f" Please try a different molecule or contact your instructor."
|
|
770
|
+
)
|
|
771
|
+
logger.error(
|
|
772
|
+
f"Unexpected error in student_friendly_smiles_to_xyz: {e}", exc_info=True
|
|
773
|
+
)
|
|
774
|
+
return None, message
|
|
775
|
+
|
|
776
|
+
|
|
777
|
+
def generate_2d_structure_svg(
|
|
778
|
+
smiles: Optional[str] = None,
|
|
779
|
+
mol: Optional[object] = None,
|
|
780
|
+
xyz_string: Optional[str] = None,
|
|
781
|
+
width: int = 300,
|
|
782
|
+
height: int = 300,
|
|
783
|
+
) -> Optional[str]:
|
|
784
|
+
"""
|
|
785
|
+
Generate 2D structure diagram as SVG string.
|
|
786
|
+
|
|
787
|
+
Can accept input as SMILES, RDKit Mol object, or XYZ string.
|
|
788
|
+
|
|
789
|
+
Args:
|
|
790
|
+
smiles: SMILES string (if provided)
|
|
791
|
+
mol: RDKit Mol object (if provided)
|
|
792
|
+
xyz_string: XYZ coordinate string (if provided)
|
|
793
|
+
width: Image width in pixels
|
|
794
|
+
height: Image height in pixels
|
|
795
|
+
|
|
796
|
+
Returns:
|
|
797
|
+
SVG string of 2D structure, or None if generation fails
|
|
798
|
+
|
|
799
|
+
Raises:
|
|
800
|
+
ImportError: If RDKit is not available
|
|
801
|
+
ValueError: If no valid input provided
|
|
802
|
+
"""
|
|
803
|
+
if not RDKIT_AVAILABLE:
|
|
804
|
+
raise ImportError("RDKit is required for 2D structure rendering")
|
|
805
|
+
|
|
806
|
+
try:
|
|
807
|
+
|
|
808
|
+
from rdkit.Chem import Draw
|
|
809
|
+
|
|
810
|
+
# Get RDKit molecule from input
|
|
811
|
+
if mol is not None:
|
|
812
|
+
rdkit_mol = mol
|
|
813
|
+
elif smiles is not None:
|
|
814
|
+
rdkit_mol = Chem.MolFromSmiles(smiles)
|
|
815
|
+
if rdkit_mol is None:
|
|
816
|
+
raise ValueError(f"Invalid SMILES: {smiles}")
|
|
817
|
+
elif xyz_string is not None:
|
|
818
|
+
# Convert XYZ to SMILES (requires RDKit bond perception)
|
|
819
|
+
# Parse XYZ
|
|
820
|
+
lines = xyz_string.strip().split("\n")
|
|
821
|
+
if len(lines) < 3:
|
|
822
|
+
raise ValueError("Invalid XYZ format")
|
|
823
|
+
|
|
824
|
+
# Build mol from XYZ
|
|
825
|
+
from rdkit.Chem import rdDetermineBonds
|
|
826
|
+
|
|
827
|
+
# Collect valid atom lines first so the conformer can be sized
|
|
828
|
+
# up front and the atom index used for SetAtomPosition always
|
|
829
|
+
# matches the just-added atom (skipped malformed lines used to
|
|
830
|
+
# desync the two).
|
|
831
|
+
atom_lines = []
|
|
832
|
+
for line in lines[2:]: # Skip first 2 lines (count + comment)
|
|
833
|
+
parts = line.split()
|
|
834
|
+
if len(parts) < 4:
|
|
835
|
+
continue
|
|
836
|
+
atom_lines.append(parts)
|
|
837
|
+
|
|
838
|
+
if not atom_lines:
|
|
839
|
+
raise ValueError("No atom lines found in XYZ string")
|
|
840
|
+
|
|
841
|
+
# Chem.Mol() is immutable — AddAtom only exists on RWMol.
|
|
842
|
+
# Build on RWMol, then convert to an immutable Mol via
|
|
843
|
+
# GetMol() before DetermineBonds (mirrors the working
|
|
844
|
+
# Chem.MolFromXYZBlock() + DetermineBonds() pattern used
|
|
845
|
+
# elsewhere in QuantUI, e.g. preopt.py).
|
|
846
|
+
rw_mol = Chem.RWMol()
|
|
847
|
+
conf = Chem.Conformer(len(atom_lines))
|
|
848
|
+
|
|
849
|
+
for i, parts in enumerate(atom_lines):
|
|
850
|
+
symbol = parts[0]
|
|
851
|
+
x, y, z = float(parts[1]), float(parts[2]), float(parts[3])
|
|
852
|
+
|
|
853
|
+
atom = Chem.Atom(symbol)
|
|
854
|
+
rw_mol.AddAtom(atom)
|
|
855
|
+
conf.SetAtomPosition(i, (x, y, z))
|
|
856
|
+
|
|
857
|
+
rw_mol.AddConformer(conf)
|
|
858
|
+
rdkit_mol = rw_mol.GetMol()
|
|
859
|
+
|
|
860
|
+
# Determine bonds
|
|
861
|
+
rdDetermineBonds.DetermineBonds(rdkit_mol)
|
|
862
|
+
else:
|
|
863
|
+
raise ValueError("Must provide smiles, mol, or xyz_string")
|
|
864
|
+
|
|
865
|
+
# Generate 2D coordinates for nice layout
|
|
866
|
+
AllChem.Compute2DCoords(rdkit_mol)
|
|
867
|
+
|
|
868
|
+
# Draw molecule to SVG
|
|
869
|
+
drawer = Draw.MolDraw2DSVG(width, height)
|
|
870
|
+
drawer.DrawMolecule(rdkit_mol)
|
|
871
|
+
drawer.FinishDrawing()
|
|
872
|
+
svg: str = str(drawer.GetDrawingText())
|
|
873
|
+
|
|
874
|
+
logger.debug("Generated 2D structure SVG")
|
|
875
|
+
return svg
|
|
876
|
+
|
|
877
|
+
except Exception as e:
|
|
878
|
+
logger.error(f"2D structure generation failed: {e}")
|
|
879
|
+
return None
|
|
880
|
+
|
|
881
|
+
|
|
882
|
+
def display_2d_structure(
|
|
883
|
+
smiles: Optional[str] = None,
|
|
884
|
+
mol: Optional[object] = None,
|
|
885
|
+
xyz_string: Optional[str] = None,
|
|
886
|
+
width: int = 400,
|
|
887
|
+
height: int = 300,
|
|
888
|
+
):
|
|
889
|
+
"""
|
|
890
|
+
Display 2D structure diagram in Jupyter notebook.
|
|
891
|
+
|
|
892
|
+
Args:
|
|
893
|
+
smiles: SMILES string (if provided)
|
|
894
|
+
mol: RDKit Mol object (if provided)
|
|
895
|
+
xyz_string: XYZ coordinate string (if provided)
|
|
896
|
+
width: Image width in pixels
|
|
897
|
+
height: Image height in pixels
|
|
898
|
+
|
|
899
|
+
Returns:
|
|
900
|
+
IPython display object or None if fails
|
|
901
|
+
"""
|
|
902
|
+
try:
|
|
903
|
+
from IPython.display import SVG
|
|
904
|
+
from IPython.display import display as ipython_display
|
|
905
|
+
|
|
906
|
+
svg = generate_2d_structure_svg(
|
|
907
|
+
smiles=smiles, mol=mol, xyz_string=xyz_string, width=width, height=height
|
|
908
|
+
)
|
|
909
|
+
|
|
910
|
+
if svg:
|
|
911
|
+
ipython_display(SVG(svg))
|
|
912
|
+
return True
|
|
913
|
+
else:
|
|
914
|
+
print("⚠️ Could not generate 2D structure")
|
|
915
|
+
return False
|
|
916
|
+
|
|
917
|
+
except ImportError as e:
|
|
918
|
+
logger.warning(f"Could not display 2D structure: {e}")
|
|
919
|
+
print("⚠️ IPython display not available")
|
|
920
|
+
return False
|
|
921
|
+
except Exception as e:
|
|
922
|
+
logger.error(f"2D structure display failed: {e}")
|
|
923
|
+
print(f"⚠️ 2D structure display failed: {e}")
|
|
924
|
+
return False
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def get_smiles_examples() -> Dict[str, str]:
|
|
928
|
+
"""
|
|
929
|
+
Get example SMILES strings for educational use.
|
|
930
|
+
|
|
931
|
+
Returns:
|
|
932
|
+
Dict mapping molecule names to SMILES strings
|
|
933
|
+
"""
|
|
934
|
+
return {
|
|
935
|
+
# Simple molecules
|
|
936
|
+
"Water": "O",
|
|
937
|
+
"Ammonia": "N",
|
|
938
|
+
"Methane": "C",
|
|
939
|
+
"Ethane": "CC",
|
|
940
|
+
"Propane": "CCC",
|
|
941
|
+
# Functional groups
|
|
942
|
+
"Methanol": "CO",
|
|
943
|
+
"Ethanol": "CCO",
|
|
944
|
+
"Acetic Acid": "CC(=O)O",
|
|
945
|
+
"Acetone": "CC(=O)C",
|
|
946
|
+
"Formaldehyde": "C=O",
|
|
947
|
+
# Aromatics
|
|
948
|
+
"Benzene": "c1ccccc1",
|
|
949
|
+
"Toluene": "Cc1ccccc1",
|
|
950
|
+
"Phenol": "Oc1ccccc1",
|
|
951
|
+
"Aniline": "Nc1ccccc1",
|
|
952
|
+
# Biochemical
|
|
953
|
+
"Glycine": "NCC(=O)O",
|
|
954
|
+
"Alanine": "CC(N)C(=O)O",
|
|
955
|
+
"Glucose": "C(C1C(C(C(C(O1)O)O)O)O)O",
|
|
956
|
+
# Common molecules
|
|
957
|
+
"Carbon Dioxide": "O=C=O",
|
|
958
|
+
"Hydrogen Peroxide": "OO",
|
|
959
|
+
"Ethylene": "C=C",
|
|
960
|
+
"Acetylene": "C#C",
|
|
961
|
+
}
|
|
962
|
+
|
|
963
|
+
|
|
964
|
+
def validate_smiles(smiles: str) -> Tuple[bool, str]:
|
|
965
|
+
"""
|
|
966
|
+
Validate a SMILES string.
|
|
967
|
+
|
|
968
|
+
Args:
|
|
969
|
+
smiles: SMILES string to validate
|
|
970
|
+
|
|
971
|
+
Returns:
|
|
972
|
+
Tuple of (is_valid, message)
|
|
973
|
+
"""
|
|
974
|
+
if not RDKIT_AVAILABLE:
|
|
975
|
+
return False, "RDKit not available"
|
|
976
|
+
|
|
977
|
+
try:
|
|
978
|
+
mol = Chem.MolFromSmiles(smiles)
|
|
979
|
+
if mol is None:
|
|
980
|
+
return False, "Invalid SMILES syntax"
|
|
981
|
+
|
|
982
|
+
# Check if molecule is reasonable
|
|
983
|
+
num_atoms = mol.GetNumAtoms()
|
|
984
|
+
if num_atoms == 0:
|
|
985
|
+
return False, "Molecule has no atoms"
|
|
986
|
+
|
|
987
|
+
if num_atoms > 200:
|
|
988
|
+
return (
|
|
989
|
+
False,
|
|
990
|
+
f"Molecule too large ({num_atoms} atoms). Consider smaller molecules for calculations.",
|
|
991
|
+
)
|
|
992
|
+
|
|
993
|
+
return True, f"Valid SMILES ({num_atoms} atoms)"
|
|
994
|
+
|
|
995
|
+
except Exception as e:
|
|
996
|
+
return False, f"Validation error: {str(e)}"
|
|
997
|
+
|
|
998
|
+
|
|
999
|
+
# ============================================================================
|
|
1000
|
+
# Smart input routing
|
|
1001
|
+
# ============================================================================
|
|
1002
|
+
|
|
1003
|
+
# Standard InChIKey: 14 block chars - 10 block chars - 1 flag char.
|
|
1004
|
+
_INCHIKEY_RE = re.compile(r"^[A-Z]{14}-[A-Z]{10}-[A-Z]$")
|
|
1005
|
+
# A bare molecular formula, e.g. "H2O", "C6H6", "CO2", "Fe".
|
|
1006
|
+
_FORMULA_RE = re.compile(r"^(?:[A-Z][a-z]?\d*)+$")
|
|
1007
|
+
# Characters that only appear in SMILES, never in a common/IUPAC molecule name.
|
|
1008
|
+
_SMILES_STRUCTURAL = set("=#()[]/\\@.%+")
|
|
1009
|
+
|
|
1010
|
+
|
|
1011
|
+
def _looks_like_smiles(query: str) -> bool:
|
|
1012
|
+
"""High-precision SMILES check: only ``True`` when RDKit parses it *and* it
|
|
1013
|
+
carries SMILES-only signals (structural punctuation or ring digits).
|
|
1014
|
+
|
|
1015
|
+
Deliberately conservative — bare-letter tokens like ``CCO`` (which are valid
|
|
1016
|
+
SMILES *and* plausible names/formulas) are left for the provider chain
|
|
1017
|
+
to disambiguate, so we never misroute a plain name to a local parse.
|
|
1018
|
+
"""
|
|
1019
|
+
if not RDKIT_AVAILABLE or " " in query:
|
|
1020
|
+
return False
|
|
1021
|
+
has_structural = any(c in _SMILES_STRUCTURAL for c in query)
|
|
1022
|
+
has_digit = any(c.isdigit() for c in query)
|
|
1023
|
+
if not (has_structural or has_digit):
|
|
1024
|
+
return False
|
|
1025
|
+
return Chem.MolFromSmiles(query) is not None
|
|
1026
|
+
|
|
1027
|
+
|
|
1028
|
+
def classify_query(query: str) -> str:
|
|
1029
|
+
"""Classify a user structure query so it can be routed to the right resolver.
|
|
1030
|
+
|
|
1031
|
+
Returns one of ``"cid"``, ``"inchikey"``, ``"inchi"``, ``"smiles"``,
|
|
1032
|
+
``"formula"``, or ``"name"``. ``smiles`` / ``inchi`` resolve locally via
|
|
1033
|
+
RDKit (no network); the rest go to PubChem. ``formula`` is currently routed
|
|
1034
|
+
like ``name`` (PubChem's async fastformula search is not used here).
|
|
1035
|
+
"""
|
|
1036
|
+
q = query.strip()
|
|
1037
|
+
if not q:
|
|
1038
|
+
raise ValueError("Empty query")
|
|
1039
|
+
if q.startswith("InChI="):
|
|
1040
|
+
return "inchi"
|
|
1041
|
+
if _INCHIKEY_RE.match(q):
|
|
1042
|
+
return "inchikey"
|
|
1043
|
+
if re.fullmatch(r"(?:CID:?\s*)?\d+", q, flags=re.IGNORECASE):
|
|
1044
|
+
return "cid"
|
|
1045
|
+
if _looks_like_smiles(q):
|
|
1046
|
+
return "smiles"
|
|
1047
|
+
if _FORMULA_RE.match(q):
|
|
1048
|
+
return "formula"
|
|
1049
|
+
return "name"
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
def _coerce_cid(query: str) -> int:
|
|
1053
|
+
"""Extract the integer CID from ``123`` / ``CID123`` / ``cid: 123`` forms."""
|
|
1054
|
+
return int(re.sub(r"[^\d]", "", query))
|
|
1055
|
+
|
|
1056
|
+
|
|
1057
|
+
def fetch_structure(
|
|
1058
|
+
query: str, conformer_3d: bool = True
|
|
1059
|
+
) -> Tuple[str, Dict[str, Any]]:
|
|
1060
|
+
"""Resolve any supported query type to ``(xyz_string, metadata)``.
|
|
1061
|
+
|
|
1062
|
+
Routes by :func:`classify_query`: SMILES/InChI resolve locally via RDKit
|
|
1063
|
+
(no network); CID/InChIKey/name/formula go to PubChem. ``metadata`` always
|
|
1064
|
+
carries a ``source`` key (``"rdkit-smiles"`` / ``"rdkit-inchi"`` /
|
|
1065
|
+
``"pubchem"``) and a ``conformer_origin`` key describing where the 3D
|
|
1066
|
+
coordinates came from, so the UI can be honest about provenance.
|
|
1067
|
+
|
|
1068
|
+
Raises :class:`PubChemError` / :class:`ValueError` on failure (the
|
|
1069
|
+
student-friendly wrapper :func:`student_friendly_resolve` maps these to
|
|
1070
|
+
messages).
|
|
1071
|
+
"""
|
|
1072
|
+
qtype = classify_query(query)
|
|
1073
|
+
q = query.strip()
|
|
1074
|
+
logger.info(f"Resolving structure query '{q}' classified as '{qtype}'")
|
|
1075
|
+
|
|
1076
|
+
if qtype == "smiles":
|
|
1077
|
+
xyz, metadata = smiles_to_xyz(q, optimize_3d=conformer_3d)
|
|
1078
|
+
metadata["source"] = "rdkit-smiles"
|
|
1079
|
+
metadata["conformer_origin"] = "rdkit-embedded"
|
|
1080
|
+
return xyz, metadata
|
|
1081
|
+
|
|
1082
|
+
if qtype == "inchi":
|
|
1083
|
+
xyz, metadata = inchi_to_xyz(q, optimize_3d=conformer_3d)
|
|
1084
|
+
metadata["source"] = "rdkit-inchi"
|
|
1085
|
+
metadata["conformer_origin"] = "rdkit-embedded"
|
|
1086
|
+
return xyz, metadata
|
|
1087
|
+
|
|
1088
|
+
# Network branch: resolve to a CID, then fetch + convert the SDF.
|
|
1089
|
+
if qtype == "cid":
|
|
1090
|
+
cid = _coerce_cid(q)
|
|
1091
|
+
elif qtype == "inchikey":
|
|
1092
|
+
cid = search_cid_by_inchikey(q)
|
|
1093
|
+
else: # "name" or "formula"
|
|
1094
|
+
cid = search_molecule_by_name(q)
|
|
1095
|
+
|
|
1096
|
+
sdf_content = get_molecule_sdf(cid, conformer_3d=conformer_3d)
|
|
1097
|
+
xyz, metadata = sdf_to_xyz(sdf_content)
|
|
1098
|
+
metadata["source"] = "pubchem"
|
|
1099
|
+
metadata["pubchem_cid"] = cid
|
|
1100
|
+
metadata["query_type"] = qtype
|
|
1101
|
+
metadata["conformer_origin"] = (
|
|
1102
|
+
"rdkit-embedded" if metadata.get("coords_embedded") else "pubchem"
|
|
1103
|
+
)
|
|
1104
|
+
return xyz, metadata
|
|
1105
|
+
|
|
1106
|
+
|
|
1107
|
+
def student_friendly_resolve(query: str) -> Tuple[Optional[str], str]:
|
|
1108
|
+
"""Smart, type-aware structure fetch with student-friendly messages.
|
|
1109
|
+
|
|
1110
|
+
Drop-in replacement for :func:`student_friendly_fetch` that also handles
|
|
1111
|
+
SMILES / InChI / CID / InChIKey input (the plain-name path is unchanged).
|
|
1112
|
+
Returns ``(xyz_string_or_None, message)``.
|
|
1113
|
+
"""
|
|
1114
|
+
try:
|
|
1115
|
+
xyz_string, metadata = fetch_structure(query, conformer_3d=True)
|
|
1116
|
+
except (MoleculeNotFoundError, ValueError) as exc:
|
|
1117
|
+
return None, (
|
|
1118
|
+
f"❌ Could not resolve '{query}'.\n"
|
|
1119
|
+
f" {exc}\n"
|
|
1120
|
+
f" Try a different name, a SMILES (e.g. CCO), or check spelling.\n"
|
|
1121
|
+
f" Search manually at: https://pubchem.ncbi.nlm.nih.gov/"
|
|
1122
|
+
)
|
|
1123
|
+
except PubChemAPIError:
|
|
1124
|
+
return None, (
|
|
1125
|
+
"❌ Connection to PubChem failed.\n"
|
|
1126
|
+
" • Check your internet connection\n"
|
|
1127
|
+
" • Try again in a moment\n"
|
|
1128
|
+
" • Use a preset molecule if the problem persists"
|
|
1129
|
+
)
|
|
1130
|
+
except ImportError:
|
|
1131
|
+
return None, (
|
|
1132
|
+
"❌ RDKit is required for SMILES / InChI input.\n"
|
|
1133
|
+
" Install with: conda install -c conda-forge rdkit"
|
|
1134
|
+
)
|
|
1135
|
+
except Exception as exc: # pragma: no cover - unexpected
|
|
1136
|
+
logger.error(
|
|
1137
|
+
f"Unexpected error in student_friendly_resolve: {exc}", exc_info=True
|
|
1138
|
+
)
|
|
1139
|
+
return None, f"❌ Error resolving '{query}': {exc}"
|
|
1140
|
+
|
|
1141
|
+
source = {
|
|
1142
|
+
"rdkit-smiles": "generated locally from SMILES",
|
|
1143
|
+
"rdkit-inchi": "generated locally from InChI",
|
|
1144
|
+
"pubchem": "PubChem",
|
|
1145
|
+
}.get(metadata.get("source", ""), metadata.get("source", "?"))
|
|
1146
|
+
origin = metadata.get("conformer_origin", "")
|
|
1147
|
+
origin_note = (
|
|
1148
|
+
" (2D structure embedded by RDKit)" if origin == "rdkit-embedded" else ""
|
|
1149
|
+
)
|
|
1150
|
+
message = (
|
|
1151
|
+
f"✓ Resolved '{query}' via {source}.\n"
|
|
1152
|
+
f" Formula: {metadata.get('formula', '?')}\n"
|
|
1153
|
+
f" Atoms: {metadata.get('num_atoms', '?')} "
|
|
1154
|
+
f"({metadata.get('num_heavy_atoms', '?')} heavy)\n"
|
|
1155
|
+
f" Molecular weight: {metadata.get('molecular_weight', 0):.2f} g/mol{origin_note}"
|
|
1156
|
+
)
|
|
1157
|
+
return xyz_string, message
|