quantui 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantui/__init__.py +311 -0
- quantui/analytics.py +609 -0
- quantui/app.py +5650 -0
- quantui/app_analysis.py +662 -0
- quantui/app_builders.py +2465 -0
- quantui/app_exports.py +194 -0
- quantui/app_formatters.py +493 -0
- quantui/app_history.py +624 -0
- quantui/app_runflow.py +1544 -0
- quantui/app_visualization.py +2620 -0
- quantui/ase_bridge.py +236 -0
- quantui/benchmarks.py +1543 -0
- quantui/c_stderr.py +124 -0
- quantui/cactus.py +88 -0
- quantui/calc_log.py +1116 -0
- quantui/calculator.py +204 -0
- quantui/cancellation.py +88 -0
- quantui/cli.py +288 -0
- quantui/comparison.py +306 -0
- quantui/config.py +725 -0
- quantui/data/js/3Dmol-min.js +2 -0
- quantui/data/js/3Dmol-min.js.LICENSE.txt +5 -0
- quantui/data/library/library.sqlite +0 -0
- quantui/data/manifests/bulk_qm9.json +1 -0
- quantui/data/manifests/curated.json +15482 -0
- quantui/data/manifests/presets.json +816 -0
- quantui/descriptor_cards.py +186 -0
- quantui/freq_calc.py +712 -0
- quantui/freq_ir_workers.py +229 -0
- quantui/gpu_offload.py +278 -0
- quantui/help_content.py +474 -0
- quantui/ir_plot.py +130 -0
- quantui/issue_tracker.py +170 -0
- quantui/live_log.py +387 -0
- quantui/log_utils.py +492 -0
- quantui/molecule.py +577 -0
- quantui/molecule_library.py +433 -0
- quantui/nmr_calc.py +437 -0
- quantui/optimizer.py +670 -0
- quantui/orbital_visualization.py +1102 -0
- quantui/pes_scan.py +420 -0
- quantui/preopt.py +355 -0
- quantui/progress.py +111 -0
- quantui/pubchem.py +1157 -0
- quantui/reorganization_energy.py +435 -0
- quantui/results_storage.py +902 -0
- quantui/security.py +14 -0
- quantui/session_calc.py +622 -0
- quantui/structure_providers.py +277 -0
- quantui/tddft_calc.py +307 -0
- quantui/user_settings.py +238 -0
- quantui/utils.py +287 -0
- quantui/vib_cache.py +247 -0
- quantui/visualization_py3dmol.py +593 -0
- quantui/viz_assets.py +101 -0
- quantui/viz_backend_router.py +243 -0
- quantui-0.5.1.dist-info/METADATA +533 -0
- quantui-0.5.1.dist-info/RECORD +62 -0
- quantui-0.5.1.dist-info/WHEEL +5 -0
- quantui-0.5.1.dist-info/entry_points.txt +2 -0
- quantui-0.5.1.dist-info/licenses/LICENSE +21 -0
- quantui-0.5.1.dist-info/top_level.txt +1 -0
quantui/molecule.py
ADDED
|
@@ -0,0 +1,577 @@
|
|
|
1
|
+
"""
|
|
2
|
+
QuantUI Molecule Module
|
|
3
|
+
|
|
4
|
+
Handles molecule input, validation, and coordinate processing.
|
|
5
|
+
Provides classes and functions for representing molecular systems.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import logging
|
|
9
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
10
|
+
|
|
11
|
+
from . import config, utils
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
# Single source of truth lives in config.ATOMIC_NUMBERS (covers the full
|
|
16
|
+
# periodic table, Z=1..118, and derives config.VALID_ATOMS so the two can't
|
|
17
|
+
# drift apart). Re-exported here so existing `from .molecule import
|
|
18
|
+
# ATOMIC_NUMBERS` call sites (e.g. orbital_visualization's charge/spin
|
|
19
|
+
# inference) keep working unchanged.
|
|
20
|
+
ATOMIC_NUMBERS: Dict[str, int] = config.ATOMIC_NUMBERS
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Molecule:
|
|
24
|
+
"""
|
|
25
|
+
Represents a molecular system with atoms, coordinates, charge, and multiplicity.
|
|
26
|
+
|
|
27
|
+
Provides validation and formatting methods for quantum chemistry calculations.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(
|
|
31
|
+
self,
|
|
32
|
+
atoms: List[str],
|
|
33
|
+
coordinates: List[List[float]],
|
|
34
|
+
charge: int = 0,
|
|
35
|
+
multiplicity: int = 1,
|
|
36
|
+
):
|
|
37
|
+
"""
|
|
38
|
+
Initialize a molecule.
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
atoms: List of atomic symbols (e.g., ['H', 'H', 'O'])
|
|
42
|
+
coordinates: List of [x, y, z] coordinates in Angstroms
|
|
43
|
+
charge: Total molecular charge
|
|
44
|
+
multiplicity: Spin multiplicity (2S+1)
|
|
45
|
+
|
|
46
|
+
Raises:
|
|
47
|
+
ValueError: If molecule data is invalid
|
|
48
|
+
"""
|
|
49
|
+
self.atoms = atoms
|
|
50
|
+
self.coordinates = coordinates
|
|
51
|
+
self.charge = charge
|
|
52
|
+
self.multiplicity = multiplicity
|
|
53
|
+
|
|
54
|
+
# Validate molecule
|
|
55
|
+
self._validate()
|
|
56
|
+
|
|
57
|
+
logger.info(
|
|
58
|
+
f"Created molecule: {self.get_formula()} (charge={charge}, mult={multiplicity})"
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
def _validate(self):
|
|
62
|
+
"""
|
|
63
|
+
Validate molecular data.
|
|
64
|
+
|
|
65
|
+
Raises:
|
|
66
|
+
ValueError: If any validation fails
|
|
67
|
+
"""
|
|
68
|
+
# Check atoms and coordinates have same length
|
|
69
|
+
if len(self.atoms) != len(self.coordinates):
|
|
70
|
+
raise ValueError(
|
|
71
|
+
f"Number of atoms ({len(self.atoms)}) does not match "
|
|
72
|
+
f"number of coordinates ({len(self.coordinates)})"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
# Check at least one atom
|
|
76
|
+
if len(self.atoms) == 0:
|
|
77
|
+
raise ValueError("Molecule must have at least one atom")
|
|
78
|
+
|
|
79
|
+
# Validate each atom symbol
|
|
80
|
+
for i, atom in enumerate(self.atoms):
|
|
81
|
+
if not utils.validate_atom_symbol(atom):
|
|
82
|
+
raise ValueError(
|
|
83
|
+
f"Invalid atom symbol '{atom}' at position {i+1}. "
|
|
84
|
+
f"Must be a valid element symbol."
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
# Validate each coordinate
|
|
88
|
+
for i, coord in enumerate(self.coordinates):
|
|
89
|
+
if not utils.validate_coordinates(coord):
|
|
90
|
+
raise ValueError(
|
|
91
|
+
f"Invalid coordinates at position {i+1}. "
|
|
92
|
+
f"Must be a list of 3 numbers: [x, y, z]"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
# Validate charge
|
|
96
|
+
if not utils.validate_charge(self.charge):
|
|
97
|
+
raise ValueError(
|
|
98
|
+
f"Invalid charge {self.charge}. "
|
|
99
|
+
f"Must be an integer between -10 and 10."
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# Validate multiplicity
|
|
103
|
+
if not utils.validate_multiplicity(self.multiplicity):
|
|
104
|
+
raise ValueError(
|
|
105
|
+
f"Invalid multiplicity {self.multiplicity}. "
|
|
106
|
+
f"Must be a positive integer."
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# Check multiplicity compatibility with electron count
|
|
110
|
+
num_electrons = self.get_electron_count()
|
|
111
|
+
if (num_electrons + self.multiplicity) % 2 != 1:
|
|
112
|
+
# Generate list of valid multiplicities (up to 5 options)
|
|
113
|
+
valid_mults = []
|
|
114
|
+
if num_electrons % 2 == 0:
|
|
115
|
+
# Even electrons -> odd multiplicities (1, 3, 5, 7, 9)
|
|
116
|
+
valid_mults = [1, 3, 5, 7, 9]
|
|
117
|
+
explanation = "even number of electrons requires odd multiplicity"
|
|
118
|
+
else:
|
|
119
|
+
# Odd electrons -> even multiplicities (2, 4, 6, 8, 10)
|
|
120
|
+
valid_mults = [2, 4, 6, 8, 10]
|
|
121
|
+
explanation = "odd number of electrons requires even multiplicity"
|
|
122
|
+
|
|
123
|
+
# Format valid options nicely
|
|
124
|
+
valid_str = ", ".join(str(m) for m in valid_mults)
|
|
125
|
+
|
|
126
|
+
raise ValueError(
|
|
127
|
+
f"ā Multiplicity Error: Multiplicity {self.multiplicity} is incompatible with "
|
|
128
|
+
f"{num_electrons} electrons.\n\n"
|
|
129
|
+
f"š” Explanation: This molecule has {num_electrons} electrons (an "
|
|
130
|
+
f"{'even' if num_electrons % 2 == 0 else 'odd'} number), so the {explanation}.\n\n"
|
|
131
|
+
f"ā
Valid multiplicities for this molecule: {valid_str}\n\n"
|
|
132
|
+
f"Common values:\n"
|
|
133
|
+
f" ⢠Multiplicity 1 (singlet) = all electrons paired\n"
|
|
134
|
+
f" ⢠Multiplicity 2 (doublet) = 1 unpaired electron (radical)\n"
|
|
135
|
+
f" ⢠Multiplicity 3 (triplet) = 2 unpaired electrons\n"
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
def get_electron_count(self) -> int:
|
|
139
|
+
"""
|
|
140
|
+
Calculate total number of electrons.
|
|
141
|
+
|
|
142
|
+
Returns:
|
|
143
|
+
int: Number of electrons (nuclear charges - charge)
|
|
144
|
+
"""
|
|
145
|
+
nuclear_charge = sum(ATOMIC_NUMBERS.get(atom, 0) for atom in self.atoms)
|
|
146
|
+
return nuclear_charge - self.charge
|
|
147
|
+
|
|
148
|
+
def get_formula(self) -> str:
|
|
149
|
+
"""
|
|
150
|
+
Get molecular formula (e.g., 'H2O', 'CH4').
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
str: Molecular formula
|
|
154
|
+
"""
|
|
155
|
+
# Count atoms
|
|
156
|
+
atom_counts: dict[str, int] = {}
|
|
157
|
+
for atom in self.atoms:
|
|
158
|
+
atom_counts[atom] = atom_counts.get(atom, 0) + 1
|
|
159
|
+
|
|
160
|
+
# Build formula (C, H, then alphabetical)
|
|
161
|
+
formula_parts = []
|
|
162
|
+
|
|
163
|
+
# Carbon first (if present)
|
|
164
|
+
if "C" in atom_counts:
|
|
165
|
+
count = atom_counts["C"]
|
|
166
|
+
formula_parts.append(f"C{count if count > 1 else ''}")
|
|
167
|
+
del atom_counts["C"]
|
|
168
|
+
|
|
169
|
+
# Hydrogen second (if present)
|
|
170
|
+
if "H" in atom_counts:
|
|
171
|
+
count = atom_counts["H"]
|
|
172
|
+
formula_parts.append(f"H{count if count > 1 else ''}")
|
|
173
|
+
del atom_counts["H"]
|
|
174
|
+
|
|
175
|
+
# Rest alphabetically
|
|
176
|
+
for atom in sorted(atom_counts.keys()):
|
|
177
|
+
count = atom_counts[atom]
|
|
178
|
+
formula_parts.append(f"{atom}{count if count > 1 else ''}")
|
|
179
|
+
|
|
180
|
+
return "".join(formula_parts)
|
|
181
|
+
|
|
182
|
+
def to_pyscf_format(self) -> str:
|
|
183
|
+
"""
|
|
184
|
+
Format molecule for PySCF input.
|
|
185
|
+
|
|
186
|
+
Returns:
|
|
187
|
+
str: Molecule string in PySCF format (atom symbol, x, y, z)
|
|
188
|
+
"""
|
|
189
|
+
lines = []
|
|
190
|
+
for atom, coord in zip(self.atoms, self.coordinates):
|
|
191
|
+
x, y, z = coord
|
|
192
|
+
lines.append(f"{atom:2s} {x:12.8f} {y:12.8f} {z:12.8f}")
|
|
193
|
+
|
|
194
|
+
return "\n".join(lines)
|
|
195
|
+
|
|
196
|
+
def to_xyz_string(self) -> str:
|
|
197
|
+
"""
|
|
198
|
+
Format molecule as XYZ string (without atom count header).
|
|
199
|
+
|
|
200
|
+
This is the simple format expected by PlotlyMol and other
|
|
201
|
+
visualization tools.
|
|
202
|
+
|
|
203
|
+
Returns:
|
|
204
|
+
str: XYZ format string (atom symbol, x, y, z per line)
|
|
205
|
+
|
|
206
|
+
Example:
|
|
207
|
+
>>> mol = Molecule(['H', 'H'], [[0, 0, 0], [0, 0, 0.74]])
|
|
208
|
+
>>> print(mol.to_xyz_string())
|
|
209
|
+
H 0.0 0.0 0.0
|
|
210
|
+
H 0.0 0.0 0.74
|
|
211
|
+
"""
|
|
212
|
+
lines = []
|
|
213
|
+
for atom, coord in zip(self.atoms, self.coordinates):
|
|
214
|
+
x, y, z = coord
|
|
215
|
+
lines.append(f"{atom} {x:.10f} {y:.10f} {z:.10f}")
|
|
216
|
+
|
|
217
|
+
return "\n".join(lines)
|
|
218
|
+
|
|
219
|
+
def count_electrons(self) -> int:
|
|
220
|
+
"""
|
|
221
|
+
Calculate total number of electrons (alias for get_electron_count).
|
|
222
|
+
|
|
223
|
+
Returns:
|
|
224
|
+
int: Number of electrons (nuclear charges - charge)
|
|
225
|
+
|
|
226
|
+
Note:
|
|
227
|
+
This is an alias for get_electron_count() to maintain
|
|
228
|
+
compatibility with visualization module.
|
|
229
|
+
"""
|
|
230
|
+
return self.get_electron_count()
|
|
231
|
+
|
|
232
|
+
def get_spin(self) -> int:
|
|
233
|
+
"""
|
|
234
|
+
Get spin quantum number S from multiplicity (2S+1).
|
|
235
|
+
|
|
236
|
+
Returns:
|
|
237
|
+
int: Number of unpaired electrons / 2
|
|
238
|
+
"""
|
|
239
|
+
return (self.multiplicity - 1) // 2
|
|
240
|
+
|
|
241
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
242
|
+
"""
|
|
243
|
+
Convert molecule to dictionary for storage.
|
|
244
|
+
|
|
245
|
+
Returns:
|
|
246
|
+
dict: Molecule data
|
|
247
|
+
"""
|
|
248
|
+
return {
|
|
249
|
+
"atoms": self.atoms,
|
|
250
|
+
"coordinates": self.coordinates,
|
|
251
|
+
"charge": self.charge,
|
|
252
|
+
"multiplicity": self.multiplicity,
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
@classmethod
|
|
256
|
+
def from_dict(cls, data: Dict[str, Any]) -> "Molecule":
|
|
257
|
+
"""
|
|
258
|
+
Create Molecule from dictionary.
|
|
259
|
+
|
|
260
|
+
Args:
|
|
261
|
+
data: Dictionary with molecule data
|
|
262
|
+
|
|
263
|
+
Returns:
|
|
264
|
+
Molecule: Reconstructed molecule object
|
|
265
|
+
"""
|
|
266
|
+
return cls(
|
|
267
|
+
atoms=data["atoms"],
|
|
268
|
+
coordinates=data["coordinates"],
|
|
269
|
+
charge=data.get("charge", 0),
|
|
270
|
+
multiplicity=data.get("multiplicity", 1),
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
def __str__(self) -> str:
|
|
274
|
+
"""String representation."""
|
|
275
|
+
return (
|
|
276
|
+
f"Molecule({self.get_formula()}, "
|
|
277
|
+
f"{len(self.atoms)} atoms, "
|
|
278
|
+
f"charge={self.charge}, "
|
|
279
|
+
f"mult={self.multiplicity})"
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
def __repr__(self) -> str:
|
|
283
|
+
"""Developer representation."""
|
|
284
|
+
return self.__str__()
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def parse_xyz_input(xyz_text: str) -> Tuple[List[str], List[List[float]]]:
|
|
288
|
+
"""
|
|
289
|
+
Parse XYZ coordinate input from text.
|
|
290
|
+
|
|
291
|
+
Supports multiple formats:
|
|
292
|
+
|
|
293
|
+
1. Simple format (one atom per line):
|
|
294
|
+
H 0.0 0.0 0.0
|
|
295
|
+
H 0.0 0.0 0.74
|
|
296
|
+
|
|
297
|
+
2. XYZ file format (with header):
|
|
298
|
+
2
|
|
299
|
+
Hydrogen molecule
|
|
300
|
+
H 0.0 0.0 0.0
|
|
301
|
+
H 0.0 0.0 0.74
|
|
302
|
+
|
|
303
|
+
3. With comments (lines starting with # or !):
|
|
304
|
+
# This is a water molecule
|
|
305
|
+
O 0.0 0.0 0.0
|
|
306
|
+
H 0.757 0.587 0.0 # First hydrogen
|
|
307
|
+
H -0.757 0.587 0.0 ! Second hydrogen
|
|
308
|
+
|
|
309
|
+
Args:
|
|
310
|
+
xyz_text: Multi-line text with coordinates
|
|
311
|
+
|
|
312
|
+
Returns:
|
|
313
|
+
tuple: (atoms, coordinates) where atoms is list of symbols
|
|
314
|
+
and coordinates is list of [x, y, z] lists
|
|
315
|
+
|
|
316
|
+
Raises:
|
|
317
|
+
ValueError: If input format is invalid
|
|
318
|
+
"""
|
|
319
|
+
if not xyz_text or not xyz_text.strip():
|
|
320
|
+
raise ValueError(
|
|
321
|
+
"ā Empty Input: Please provide XYZ coordinates.\n\n"
|
|
322
|
+
"Expected format:\n"
|
|
323
|
+
" ATOM X Y Z\n\n"
|
|
324
|
+
"Example:\n"
|
|
325
|
+
" H 0.0 0.0 0.0\n"
|
|
326
|
+
" H 0.0 0.0 0.74"
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
lines = xyz_text.strip().split("\n")
|
|
330
|
+
|
|
331
|
+
# āā Step 1: XYZ-file header detection, on RAW lines āāāāāāāāāāāāāāāāāā
|
|
332
|
+
# The header (count line + title line) is POSITIONAL: whichever raw
|
|
333
|
+
# line is the first non-blank, non-full-line-comment line, if it
|
|
334
|
+
# parses as a bare integer, is the atom count ā and the very next raw
|
|
335
|
+
# line is the title, regardless of what that title line itself
|
|
336
|
+
# contains (blank, "#"/"!"-prefixed, or free text).
|
|
337
|
+
#
|
|
338
|
+
# This must run BEFORE the blank/comment filtering below: filtering
|
|
339
|
+
# first would remove a blank or comment title line from the list,
|
|
340
|
+
# shifting the next real atom line into the "title" slot, where it
|
|
341
|
+
# was then silently discarded ā dropping the first atom of every
|
|
342
|
+
# standard XYZ file whose title line happened to be blank or a
|
|
343
|
+
# comment (both very common in practice).
|
|
344
|
+
expected_atoms = None
|
|
345
|
+
body_start = 0
|
|
346
|
+
for idx, raw in enumerate(lines):
|
|
347
|
+
stripped = raw.strip()
|
|
348
|
+
if not stripped:
|
|
349
|
+
continue
|
|
350
|
+
if stripped.startswith("#") or stripped.startswith("!"):
|
|
351
|
+
continue
|
|
352
|
+
try:
|
|
353
|
+
expected_atoms = int(stripped)
|
|
354
|
+
body_start = idx + 2 # count line + title line, whatever it is
|
|
355
|
+
logger.debug(f"Detected XYZ file format expecting {expected_atoms} atoms")
|
|
356
|
+
except ValueError:
|
|
357
|
+
pass # first content line isn't a bare count -> no header
|
|
358
|
+
break # only the first non-blank/non-comment line is eligible
|
|
359
|
+
|
|
360
|
+
body_lines = lines[body_start:]
|
|
361
|
+
|
|
362
|
+
# āā Step 2: filter blank lines + comments from the body āāāāāāāāāāāāā
|
|
363
|
+
processed_lines = []
|
|
364
|
+
original_line_numbers = [] # Track original line numbers for error reporting
|
|
365
|
+
|
|
366
|
+
for offset, line in enumerate(body_lines):
|
|
367
|
+
line_num = body_start + offset + 1 # 1-based original line number
|
|
368
|
+
line = line.strip()
|
|
369
|
+
|
|
370
|
+
# Skip empty lines
|
|
371
|
+
if not line:
|
|
372
|
+
continue
|
|
373
|
+
|
|
374
|
+
# Skip comment lines (starting with # or !)
|
|
375
|
+
if line.startswith("#") or line.startswith("!"):
|
|
376
|
+
logger.debug(f"Skipping comment line {line_num}: {line}")
|
|
377
|
+
continue
|
|
378
|
+
|
|
379
|
+
# Remove inline comments (everything after # or !)
|
|
380
|
+
for comment_char in ["#", "!"]:
|
|
381
|
+
if comment_char in line:
|
|
382
|
+
line = line.split(comment_char)[0].strip()
|
|
383
|
+
|
|
384
|
+
if line: # Only add non-empty lines after comment removal
|
|
385
|
+
processed_lines.append(line)
|
|
386
|
+
original_line_numbers.append(line_num)
|
|
387
|
+
|
|
388
|
+
if not processed_lines:
|
|
389
|
+
raise ValueError(
|
|
390
|
+
"ā No Data: All lines are empty or comments.\n\n"
|
|
391
|
+
"Please provide at least 1 atom with coordinates."
|
|
392
|
+
)
|
|
393
|
+
|
|
394
|
+
atoms = []
|
|
395
|
+
coordinates = []
|
|
396
|
+
|
|
397
|
+
# Parse coordinate lines (the header, if any, was already consumed
|
|
398
|
+
# positionally above, so every processed line here is an atom row).
|
|
399
|
+
for i, line in enumerate(processed_lines):
|
|
400
|
+
orig_line_num = original_line_numbers[i]
|
|
401
|
+
parts = line.split()
|
|
402
|
+
|
|
403
|
+
# Check minimum parts (atom symbol + 3 coordinates)
|
|
404
|
+
if len(parts) < 4:
|
|
405
|
+
raise ValueError(
|
|
406
|
+
f"ā Line {orig_line_num}: Invalid format - not enough values.\n\n"
|
|
407
|
+
f"Got: {line}\n"
|
|
408
|
+
f"Expected format: ATOM X Y Z\n\n"
|
|
409
|
+
f"Example: H 0.0 0.0 0.74\n\n"
|
|
410
|
+
f"š” Make sure each line has:\n"
|
|
411
|
+
f" 1. Atom symbol (H, C, N, O, etc.)\n"
|
|
412
|
+
f" 2. Three coordinate values (X, Y, Z)"
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
atom_symbol = parts[0]
|
|
416
|
+
|
|
417
|
+
# Validate atom symbol with helpful suggestions
|
|
418
|
+
if not utils.validate_atom_symbol(atom_symbol):
|
|
419
|
+
# Try to suggest corrections for common mistakes
|
|
420
|
+
suggestions = []
|
|
421
|
+
|
|
422
|
+
# Case sensitivity: check if lowercase/uppercase version exists
|
|
423
|
+
if atom_symbol.capitalize() in config.VALID_ATOMS:
|
|
424
|
+
suggestions.append(
|
|
425
|
+
f"Did you mean '{atom_symbol.capitalize()}'? (check capitalization)"
|
|
426
|
+
)
|
|
427
|
+
elif atom_symbol.upper() in config.VALID_ATOMS:
|
|
428
|
+
suggestions.append(f"Did you mean '{atom_symbol.upper()}'?")
|
|
429
|
+
elif atom_symbol.lower().capitalize() in config.VALID_ATOMS:
|
|
430
|
+
suggestions.append(
|
|
431
|
+
f"Did you mean '{atom_symbol.lower().capitalize()}'?"
|
|
432
|
+
)
|
|
433
|
+
|
|
434
|
+
# Common typos
|
|
435
|
+
common_typos = {
|
|
436
|
+
"he": "He",
|
|
437
|
+
"li": "Li",
|
|
438
|
+
"be": "Be",
|
|
439
|
+
"ne": "Ne",
|
|
440
|
+
"na": "Na",
|
|
441
|
+
"mg": "Mg",
|
|
442
|
+
"al": "Al",
|
|
443
|
+
"si": "Si",
|
|
444
|
+
"cl": "Cl",
|
|
445
|
+
"ar": "Ar",
|
|
446
|
+
"ca": "Ca",
|
|
447
|
+
"fe": "Fe",
|
|
448
|
+
"cu": "Cu",
|
|
449
|
+
"zn": "Zn",
|
|
450
|
+
"br": "Br",
|
|
451
|
+
"kr": "Kr",
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
if atom_symbol.lower() in common_typos:
|
|
455
|
+
correct = common_typos[atom_symbol.lower()]
|
|
456
|
+
if correct not in suggestions:
|
|
457
|
+
suggestions.append(f"Did you mean '{correct}'?")
|
|
458
|
+
|
|
459
|
+
# Build error message
|
|
460
|
+
error_msg = (
|
|
461
|
+
f"ā Line {orig_line_num}: Invalid atom symbol '{atom_symbol}'.\n\n"
|
|
462
|
+
f"Got: {line}\n"
|
|
463
|
+
)
|
|
464
|
+
|
|
465
|
+
if suggestions:
|
|
466
|
+
error_msg += "\nš” Suggestions:\n"
|
|
467
|
+
for suggestion in suggestions:
|
|
468
|
+
error_msg += f" ⢠{suggestion}\n"
|
|
469
|
+
else:
|
|
470
|
+
error_msg += (
|
|
471
|
+
"\nš” Valid atom symbols include:\n"
|
|
472
|
+
" ⢠H, C, N, O, F, P, S, Cl, Br, I\n"
|
|
473
|
+
" ⢠Li, Be, B, Na, Mg, Al, Si\n"
|
|
474
|
+
" ⢠K, Ca, Fe, Cu, Zn, etc.\n"
|
|
475
|
+
"\nNote: Symbols are case-sensitive (e.g., 'C' not 'c')"
|
|
476
|
+
)
|
|
477
|
+
|
|
478
|
+
raise ValueError(error_msg)
|
|
479
|
+
|
|
480
|
+
# Parse coordinates
|
|
481
|
+
try:
|
|
482
|
+
x, y, z = float(parts[1]), float(parts[2]), float(parts[3])
|
|
483
|
+
except ValueError as e:
|
|
484
|
+
raise ValueError(
|
|
485
|
+
f"ā Line {orig_line_num}: Could not parse coordinates as numbers.\n\n"
|
|
486
|
+
f"Got: {line}\n"
|
|
487
|
+
f"Values: X={parts[1]}, Y={parts[2]}, Z={parts[3]}\n\n"
|
|
488
|
+
f"š” Coordinates must be numbers (integers or decimals).\n"
|
|
489
|
+
f"Examples: 0.0, 1.5, -2.3, 0.757\n\n"
|
|
490
|
+
f"Error details: {e}"
|
|
491
|
+
) from e
|
|
492
|
+
|
|
493
|
+
atoms.append(atom_symbol)
|
|
494
|
+
coordinates.append([x, y, z])
|
|
495
|
+
|
|
496
|
+
# Validate we got atoms
|
|
497
|
+
if not atoms:
|
|
498
|
+
raise ValueError(
|
|
499
|
+
"ā No atoms found in input after parsing.\n\n"
|
|
500
|
+
"Please check your coordinate format."
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
# Verify expected count if XYZ file format
|
|
504
|
+
if expected_atoms is not None and len(atoms) != expected_atoms:
|
|
505
|
+
logger.warning(
|
|
506
|
+
f"XYZ file header specified {expected_atoms} atoms, "
|
|
507
|
+
f"but found {len(atoms)} atoms"
|
|
508
|
+
)
|
|
509
|
+
|
|
510
|
+
logger.info(f"Successfully parsed {len(atoms)} atoms from XYZ input")
|
|
511
|
+
return atoms, coordinates
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def suggest_multiplicity(atoms: List[str], charge: int) -> int:
|
|
515
|
+
"""
|
|
516
|
+
Suggest default spin multiplicity based on molecule composition.
|
|
517
|
+
|
|
518
|
+
For simple molecules, suggests singlet (1) or doublet (2) based on
|
|
519
|
+
whether total electron count is even or odd.
|
|
520
|
+
|
|
521
|
+
Args:
|
|
522
|
+
atoms: List of atomic symbols
|
|
523
|
+
charge: Molecular charge
|
|
524
|
+
|
|
525
|
+
Returns:
|
|
526
|
+
int: Suggested multiplicity
|
|
527
|
+
"""
|
|
528
|
+
# Calculate electron count directly without creating Molecule
|
|
529
|
+
# (to avoid validation errors with incompatible multiplicity)
|
|
530
|
+
try:
|
|
531
|
+
nuclear_charge = sum(ATOMIC_NUMBERS.get(atom, 0) for atom in atoms)
|
|
532
|
+
num_electrons = nuclear_charge - charge
|
|
533
|
+
|
|
534
|
+
# Even electrons -> singlet, odd electrons -> doublet
|
|
535
|
+
suggested = 1 if num_electrons % 2 == 0 else 2
|
|
536
|
+
logger.debug(
|
|
537
|
+
f"Suggested multiplicity {suggested} for {num_electrons} electrons"
|
|
538
|
+
)
|
|
539
|
+
return suggested
|
|
540
|
+
|
|
541
|
+
except Exception as e:
|
|
542
|
+
logger.warning(f"Could not suggest multiplicity: {e}")
|
|
543
|
+
return 1 # Default to singlet
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def get_preset_molecule(name: str) -> Optional[Molecule]:
|
|
547
|
+
"""
|
|
548
|
+
Get a preset molecule from the library.
|
|
549
|
+
|
|
550
|
+
Args:
|
|
551
|
+
name: Molecule name (e.g., 'H2', 'H2O')
|
|
552
|
+
|
|
553
|
+
Returns:
|
|
554
|
+
Molecule: Preset molecule, or None if not found
|
|
555
|
+
"""
|
|
556
|
+
preset = config.MOLECULE_LIBRARY.get(name)
|
|
557
|
+
|
|
558
|
+
if preset is None:
|
|
559
|
+
logger.warning(f"Preset molecule '{name}' not found")
|
|
560
|
+
return None
|
|
561
|
+
|
|
562
|
+
return Molecule(
|
|
563
|
+
atoms=preset["atoms"],
|
|
564
|
+
coordinates=preset["coordinates"],
|
|
565
|
+
charge=preset["charge"],
|
|
566
|
+
multiplicity=preset["multiplicity"],
|
|
567
|
+
)
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def list_preset_molecules() -> List[str]:
|
|
571
|
+
"""
|
|
572
|
+
Get list of available preset molecule names.
|
|
573
|
+
|
|
574
|
+
Returns:
|
|
575
|
+
list: Molecule names
|
|
576
|
+
"""
|
|
577
|
+
return list(config.MOLECULE_LIBRARY.keys())
|