quantui 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantui/__init__.py +311 -0
- quantui/analytics.py +609 -0
- quantui/app.py +5650 -0
- quantui/app_analysis.py +662 -0
- quantui/app_builders.py +2465 -0
- quantui/app_exports.py +194 -0
- quantui/app_formatters.py +493 -0
- quantui/app_history.py +624 -0
- quantui/app_runflow.py +1544 -0
- quantui/app_visualization.py +2620 -0
- quantui/ase_bridge.py +236 -0
- quantui/benchmarks.py +1543 -0
- quantui/c_stderr.py +124 -0
- quantui/cactus.py +88 -0
- quantui/calc_log.py +1116 -0
- quantui/calculator.py +204 -0
- quantui/cancellation.py +88 -0
- quantui/cli.py +288 -0
- quantui/comparison.py +306 -0
- quantui/config.py +725 -0
- quantui/data/js/3Dmol-min.js +2 -0
- quantui/data/js/3Dmol-min.js.LICENSE.txt +5 -0
- quantui/data/library/library.sqlite +0 -0
- quantui/data/manifests/bulk_qm9.json +1 -0
- quantui/data/manifests/curated.json +15482 -0
- quantui/data/manifests/presets.json +816 -0
- quantui/descriptor_cards.py +186 -0
- quantui/freq_calc.py +712 -0
- quantui/freq_ir_workers.py +229 -0
- quantui/gpu_offload.py +278 -0
- quantui/help_content.py +474 -0
- quantui/ir_plot.py +130 -0
- quantui/issue_tracker.py +170 -0
- quantui/live_log.py +387 -0
- quantui/log_utils.py +492 -0
- quantui/molecule.py +577 -0
- quantui/molecule_library.py +433 -0
- quantui/nmr_calc.py +437 -0
- quantui/optimizer.py +670 -0
- quantui/orbital_visualization.py +1102 -0
- quantui/pes_scan.py +420 -0
- quantui/preopt.py +355 -0
- quantui/progress.py +111 -0
- quantui/pubchem.py +1157 -0
- quantui/reorganization_energy.py +435 -0
- quantui/results_storage.py +902 -0
- quantui/security.py +14 -0
- quantui/session_calc.py +622 -0
- quantui/structure_providers.py +277 -0
- quantui/tddft_calc.py +307 -0
- quantui/user_settings.py +238 -0
- quantui/utils.py +287 -0
- quantui/vib_cache.py +247 -0
- quantui/visualization_py3dmol.py +593 -0
- quantui/viz_assets.py +101 -0
- quantui/viz_backend_router.py +243 -0
- quantui-0.5.1.dist-info/METADATA +533 -0
- quantui-0.5.1.dist-info/RECORD +62 -0
- quantui-0.5.1.dist-info/WHEEL +5 -0
- quantui-0.5.1.dist-info/entry_points.txt +2 -0
- quantui-0.5.1.dist-info/licenses/LICENSE +21 -0
- quantui-0.5.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""ProcessPoolExecutor workers for the IR-intensity displacement loop.
|
|
2
|
+
|
|
3
|
+
The Frequency calculation's IR-intensity step requires ``6N`` SCFs over
|
|
4
|
+
finite-difference geometries (one per Cartesian displacement of each atom,
|
|
5
|
+
+Δ and −Δ). The default path in :mod:`quantui.freq_calc` runs them
|
|
6
|
+
serially with each SCF internally parallelized via BLAS + libcint OpenMP.
|
|
7
|
+
|
|
8
|
+
When the user opts in via ``QUANTUI_FREQ_PARALLEL=1`` AND no GPU is
|
|
9
|
+
available AND the host has ``>= 4`` cores AND the molecule has
|
|
10
|
+
``>= 2`` atoms (i.e. ``>= 6`` displacements), the freq_calc driver hands
|
|
11
|
+
this loop off to a ``ProcessPoolExecutor`` whose workers each call
|
|
12
|
+
:func:`run_displaced_scf` on one displaced geometry. Each worker process
|
|
13
|
+
re-imports PySCF, rebuilds the ``gto.Mole`` from the same atom string /
|
|
14
|
+
basis / charge / spin as the parent, applies the displacement, and runs
|
|
15
|
+
the SCF. The initial guess ``dm0`` is shared once per worker via a temp
|
|
16
|
+
pickle file (the path is passed through ``initargs``) so we don't pay
|
|
17
|
+
per-task IPC for a 100×100 matrix.
|
|
18
|
+
|
|
19
|
+
The functions in this module are intentionally top-level (not nested in
|
|
20
|
+
``freq_calc.py``) because ``ProcessPoolExecutor`` requires picklable
|
|
21
|
+
references for both ``initializer`` and the task callable. Nested
|
|
22
|
+
functions cannot be pickled.
|
|
23
|
+
|
|
24
|
+
POSIX-first design note: on Linux/macOS the parent process has already
|
|
25
|
+
imported NumPy + PySCF by the time we spawn workers. We use
|
|
26
|
+
``multiprocessing.get_context("spawn")`` so each worker starts with a
|
|
27
|
+
fresh Python interpreter, reads the BLAS-thread env vars BEFORE NumPy is
|
|
28
|
+
imported, and therefore actually honors the configured thread budget.
|
|
29
|
+
Without ``spawn``, on Linux the default ``fork`` would inherit the
|
|
30
|
+
parent's NumPy thread pool and ignore any env-var changes the worker
|
|
31
|
+
makes.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import os
|
|
37
|
+
from typing import Any, Dict
|
|
38
|
+
|
|
39
|
+
# Process-global state, populated by :func:`init_worker` once per worker.
|
|
40
|
+
# Kept as a module-level dict (not class state) so workers don't need to
|
|
41
|
+
# import any container class to access it.
|
|
42
|
+
_WORKER_STATE: Dict[str, Any] = {}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def init_worker(
|
|
46
|
+
atom_str: str,
|
|
47
|
+
basis: str,
|
|
48
|
+
charge: int,
|
|
49
|
+
spin: int,
|
|
50
|
+
xc: str | None,
|
|
51
|
+
dm0_pickle_path: str,
|
|
52
|
+
omp_threads: int,
|
|
53
|
+
) -> None:
|
|
54
|
+
"""ProcessPoolExecutor worker initializer.
|
|
55
|
+
|
|
56
|
+
Runs once per worker process. **Sets BLAS-thread env vars BEFORE
|
|
57
|
+
importing NumPy** — this is the whole point of the ``spawn`` start
|
|
58
|
+
method: each worker reads the env vars on its fresh interpreter
|
|
59
|
+
startup, NOT on the parent's already-imported NumPy state. Then loads
|
|
60
|
+
the shared initial-guess density matrix from the parent's tempfile
|
|
61
|
+
into ``_WORKER_STATE`` so per-task IPC stays tiny.
|
|
62
|
+
|
|
63
|
+
Parameters
|
|
64
|
+
----------
|
|
65
|
+
atom_str:
|
|
66
|
+
Pyscf-format atom string ("O 0 0 0; H 0.96 0 0; ..."). Used to
|
|
67
|
+
rebuild the Mole in the worker.
|
|
68
|
+
basis:
|
|
69
|
+
Basis set name (e.g. ``"STO-3G"``).
|
|
70
|
+
charge, spin:
|
|
71
|
+
Molecular charge and 2S (spin) for the Mole.
|
|
72
|
+
xc:
|
|
73
|
+
DFT functional name when running a KS calculation; ``None`` for
|
|
74
|
+
plain HF.
|
|
75
|
+
dm0_pickle_path:
|
|
76
|
+
Path to a tempfile containing the parent's converged density
|
|
77
|
+
matrix as a NumPy array, used as the SCF initial guess in every
|
|
78
|
+
displaced calculation. Read once here, then kept in
|
|
79
|
+
``_WORKER_STATE`` for all subsequent task calls.
|
|
80
|
+
omp_threads:
|
|
81
|
+
BLAS thread budget for this worker. Set as ``OMP_NUM_THREADS`` /
|
|
82
|
+
``MKL_NUM_THREADS`` / ``OPENBLAS_NUM_THREADS`` / ``PYSCF_NUM_THREADS``.
|
|
83
|
+
"""
|
|
84
|
+
# Order matters: set env vars before any NumPy / PySCF import.
|
|
85
|
+
threads = str(int(omp_threads))
|
|
86
|
+
os.environ["OMP_NUM_THREADS"] = threads
|
|
87
|
+
os.environ["OPENBLAS_NUM_THREADS"] = threads
|
|
88
|
+
os.environ["MKL_NUM_THREADS"] = threads
|
|
89
|
+
os.environ["PYSCF_NUM_THREADS"] = threads
|
|
90
|
+
|
|
91
|
+
import pickle
|
|
92
|
+
|
|
93
|
+
with open(dm0_pickle_path, "rb") as fh:
|
|
94
|
+
dm0 = pickle.load(fh)
|
|
95
|
+
|
|
96
|
+
_WORKER_STATE.update(
|
|
97
|
+
atom_str=atom_str,
|
|
98
|
+
basis=basis,
|
|
99
|
+
charge=int(charge),
|
|
100
|
+
spin=int(spin),
|
|
101
|
+
xc=xc,
|
|
102
|
+
dm0=dm0,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def run_displaced_scf(coords_bohr_flat) -> Any:
|
|
107
|
+
"""Run one SCF at the displaced geometry; return the dipole as ndarray.
|
|
108
|
+
|
|
109
|
+
Called by :class:`concurrent.futures.ProcessPoolExecutor` once per
|
|
110
|
+
submitted displacement task. ``coords_bohr_flat`` is the displaced
|
|
111
|
+
geometry packed as a flat Python list (``[x0, y0, z0, x1, y1, z1, ...]``)
|
|
112
|
+
for cheap pickling — reshaped to ``(N_atoms, 3)`` inside the worker.
|
|
113
|
+
|
|
114
|
+
Uses ``_WORKER_STATE`` populated by :func:`init_worker` for the
|
|
115
|
+
invariant inputs (atom string, basis, etc.) + the shared initial-guess
|
|
116
|
+
density matrix.
|
|
117
|
+
|
|
118
|
+
Returns
|
|
119
|
+
-------
|
|
120
|
+
np.ndarray
|
|
121
|
+
Three-component dipole moment in Debye.
|
|
122
|
+
|
|
123
|
+
Notes
|
|
124
|
+
-----
|
|
125
|
+
Any exception raised here propagates to the parent via the
|
|
126
|
+
``Future.result()`` call. The freq_calc driver catches such failures
|
|
127
|
+
and falls back to the serial loop so the user's calc still completes.
|
|
128
|
+
"""
|
|
129
|
+
import numpy as np
|
|
130
|
+
from pyscf import dft, gto, scf
|
|
131
|
+
|
|
132
|
+
state = _WORKER_STATE
|
|
133
|
+
coords = np.asarray(coords_bohr_flat, dtype=float).reshape(-1, 3)
|
|
134
|
+
|
|
135
|
+
mol = gto.Mole()
|
|
136
|
+
mol.atom = state["atom_str"]
|
|
137
|
+
mol.basis = state["basis"]
|
|
138
|
+
mol.charge = state["charge"]
|
|
139
|
+
mol.spin = state["spin"]
|
|
140
|
+
mol.verbose = 0
|
|
141
|
+
mol.build()
|
|
142
|
+
mol.set_geom_(coords, unit="Bohr")
|
|
143
|
+
|
|
144
|
+
# M5 audit fix (2026-07-14): whether this displaced SCF needs an
|
|
145
|
+
# unrestricted (UHF/UKS) object is determined by the shared dm0's
|
|
146
|
+
# actual shape -- (2, nao, nao) for UHF/UKS/ROHF, (nao, nao) for
|
|
147
|
+
# RHF/RKS -- NOT by mol.spin == 0. Those two signals only agree when
|
|
148
|
+
# the user's method choice matches the molecule's natural spin state.
|
|
149
|
+
# They diverge when a user explicitly selects UHF for a closed-shell
|
|
150
|
+
# molecule (mol.spin == 0 but the parent mf, and therefore dm0, is
|
|
151
|
+
# still UHF-shaped): building RHF from mol.spin == 0 and then feeding
|
|
152
|
+
# it the UHF-shaped dm0 raises a shape-mismatch ValueError inside
|
|
153
|
+
# PySCF. Mirrors the serial-path fix in freq_calc.py.
|
|
154
|
+
dm0 = state.get("dm0")
|
|
155
|
+
dm0_is_unrestricted = dm0 is not None and np.asarray(dm0).ndim == 3
|
|
156
|
+
|
|
157
|
+
xc = state.get("xc")
|
|
158
|
+
if xc is not None:
|
|
159
|
+
mf = dft.UKS(mol) if dm0_is_unrestricted else dft.RKS(mol)
|
|
160
|
+
mf.xc = xc
|
|
161
|
+
else:
|
|
162
|
+
mf = scf.UHF(mol) if dm0_is_unrestricted else scf.RHF(mol)
|
|
163
|
+
mf.verbose = 0
|
|
164
|
+
mf.kernel(dm0=dm0)
|
|
165
|
+
return np.array(mf.dip_moment(verbose=0))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def parallel_enabled_for_run(
|
|
169
|
+
cpu_count: int,
|
|
170
|
+
displacement_count: int,
|
|
171
|
+
gpu_available: bool,
|
|
172
|
+
) -> bool:
|
|
173
|
+
"""Decide whether the freq_calc IR loop should use the parallel path.
|
|
174
|
+
|
|
175
|
+
Centralised in this module so both the driver and the tests can
|
|
176
|
+
consult the same predicate. The current rules:
|
|
177
|
+
|
|
178
|
+
- **Opt-in**: ``QUANTUI_FREQ_PARALLEL`` env var must be truthy
|
|
179
|
+
(``"1"`` / ``"true"`` / ``"True"``). Shipping this off-by-default
|
|
180
|
+
while the parallel path matures.
|
|
181
|
+
- **No GPU**: if gpu4pyscf is doing the offload, each SCF is already
|
|
182
|
+
~10× faster; running multiple in parallel would compete for one
|
|
183
|
+
GPU's VRAM and is not worth the complexity. Stay serial.
|
|
184
|
+
- **Cores threshold**: at least 4 cores. Below that, the BLAS
|
|
185
|
+
oversubscription tradeoff doesn't pay off.
|
|
186
|
+
- **Displacement threshold**: at least 6 (i.e. ``>= 2`` atoms). For a
|
|
187
|
+
diatomic the serial loop is 12 SCFs at most and parallel overhead
|
|
188
|
+
dominates.
|
|
189
|
+
"""
|
|
190
|
+
if not _truthy(os.environ.get("QUANTUI_FREQ_PARALLEL", "")):
|
|
191
|
+
return False
|
|
192
|
+
if gpu_available:
|
|
193
|
+
return False
|
|
194
|
+
if cpu_count < 4:
|
|
195
|
+
return False
|
|
196
|
+
if displacement_count < 6:
|
|
197
|
+
return False
|
|
198
|
+
return True
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def pick_worker_count(cpu_count: int, displacement_count: int) -> int:
|
|
202
|
+
"""Pick a worker count that balances parallelism vs BLAS oversubscription.
|
|
203
|
+
|
|
204
|
+
Heuristic: use half the available cores, capped by the number of
|
|
205
|
+
displacement tasks. This leaves room for each worker to have ``>= 2``
|
|
206
|
+
BLAS threads on common 4/8/16-core configurations:
|
|
207
|
+
|
|
208
|
+
- 4 cores, 18 displacements → 2 workers × 2 threads each.
|
|
209
|
+
- 8 cores, 60 displacements → 4 workers × 2 threads each.
|
|
210
|
+
- 16 cores, 60 displacements → 8 workers × 2 threads each.
|
|
211
|
+
"""
|
|
212
|
+
half = max(1, cpu_count // 2)
|
|
213
|
+
return min(half, displacement_count)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def threads_per_worker(cpu_count: int, n_workers: int) -> int:
|
|
217
|
+
"""How many BLAS threads each worker process should get.
|
|
218
|
+
|
|
219
|
+
Floors to 1 to avoid setting ``OMP_NUM_THREADS=0`` (which BLAS
|
|
220
|
+
interprets as "use the runtime default" — defeating the budgeting).
|
|
221
|
+
"""
|
|
222
|
+
if n_workers <= 0:
|
|
223
|
+
return 1
|
|
224
|
+
return max(1, cpu_count // n_workers)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _truthy(value: str) -> bool:
|
|
228
|
+
"""Match the truthy convention used by ``QUANTUI_DISABLE_GPU`` etc."""
|
|
229
|
+
return str(value).strip().lower() in ("1", "true", "yes", "on")
|
quantui/gpu_offload.py
ADDED
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
"""GPU offload helpers.
|
|
2
|
+
|
|
3
|
+
Wraps the runtime decision "should this SCF object be migrated to GPU?".
|
|
4
|
+
Detection probes ``gpu4pyscf`` + ``cupy`` for a CUDA-capable device; if
|
|
5
|
+
anything is missing or broken the helpers silently report "no GPU" so the
|
|
6
|
+
caller falls back to CPU. This means GPU integration is safe to leave
|
|
7
|
+
enabled by default on every platform — Windows users without CUDA, WSL
|
|
8
|
+
users without gpu4pyscf installed, and remote machines with broken NVIDIA
|
|
9
|
+
drivers all converge to the same "CPU" outcome with no exception leakage.
|
|
10
|
+
|
|
11
|
+
The companion ``log_utils._detect_gpu`` reports system-level GPU info for
|
|
12
|
+
the run banner (nvidia-smi name + memory). This module's job is narrower:
|
|
13
|
+
"can QuantUI's PySCF dispatcher offload to that GPU right now?".
|
|
14
|
+
|
|
15
|
+
Method coverage (verified against the gpu4pyscf README 2026-05):
|
|
16
|
+
|
|
17
|
+
- RHF / UHF / RKS / UKS — fully supported, ``mf.to_gpu()`` is canonical.
|
|
18
|
+
- MP2, CCSD — listed as experimental; ``.to_gpu()`` may succeed but the
|
|
19
|
+
post-HF kernel may still fall back to CPU. ``try_to_gpu`` honours the
|
|
20
|
+
user's intent (offload the SCF; let gpu4pyscf decide the rest).
|
|
21
|
+
- CCSD(T), double hybrids — explicitly not supported. ``try_to_gpu`` skips
|
|
22
|
+
GPU for these methods so the SCF + (T) step stays on CPU.
|
|
23
|
+
|
|
24
|
+
User opt-out, two ways:
|
|
25
|
+
|
|
26
|
+
- ``QUANTUI_DISABLE_GPU=1`` in the environment — process-wide, wins over
|
|
27
|
+
everything. Useful for benchmarks, regression debugging, and "first run
|
|
28
|
+
as student" comparisons, and it propagates to subprocess workers.
|
|
29
|
+
- ``compute.gpu_enabled = false`` in user settings — the persistent UI
|
|
30
|
+
toggle (Status tab → Settings). Survives restarts.
|
|
31
|
+
|
|
32
|
+
Not every CUDA device is worth using: double-precision throughput on
|
|
33
|
+
consumer cards is a small fraction of their single-precision, and PySCF is
|
|
34
|
+
FP64 throughout. See ``is_low_fp64_device`` — QuantUI reports an advisory
|
|
35
|
+
but never overrides the user's choice.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
from __future__ import annotations
|
|
39
|
+
|
|
40
|
+
import logging
|
|
41
|
+
import os
|
|
42
|
+
from functools import lru_cache
|
|
43
|
+
from typing import Any, Optional, Tuple
|
|
44
|
+
|
|
45
|
+
logger = logging.getLogger(__name__)
|
|
46
|
+
|
|
47
|
+
# Methods for which gpu4pyscf has zero or known-broken support.
|
|
48
|
+
#
|
|
49
|
+
# - ``CCSD(T)`` is documented as unsupported in the gpu4pyscf README.
|
|
50
|
+
# - ``MP2`` and ``CCSD`` are labelled "experimental" by gpu4pyscf and
|
|
51
|
+
# were observed (2026-05-25) to fail
|
|
52
|
+
# immediately after a successful RHF reference on GPU — the failure
|
|
53
|
+
# fingerprint was "step completed in RHF wall time + small delta,
|
|
54
|
+
# then errored", which fits the post-HF code choking on a
|
|
55
|
+
# GPU-migrated mf object. Until the upstream support matures, route
|
|
56
|
+
# these through CPU so calibration data accrues reliably. The RHF
|
|
57
|
+
# reference still benefits from GPU because ``try_to_gpu`` only
|
|
58
|
+
# short-circuits BEFORE the migration.
|
|
59
|
+
# - Double-hybrids would belong here too, but QuantUI doesn't expose
|
|
60
|
+
# any double-hybrid methods today.
|
|
61
|
+
_GPU_UNSUPPORTED_METHODS: frozenset = frozenset({"MP2", "CCSD", "CCSD(T)"})
|
|
62
|
+
|
|
63
|
+
# Datacenter GPU families with usable double-precision throughput (FP64 at
|
|
64
|
+
# roughly 1/2 of FP32). Everything else — GeForce, RTX/Quadro workstation, the
|
|
65
|
+
# inference-oriented T4/L4/L40/A10/A40 line — gates FP64 to about 1/32–1/64.
|
|
66
|
+
#
|
|
67
|
+
# This matters because PySCF/gpu4pyscf SCF is FP64 throughout, so a consumer
|
|
68
|
+
# card can be genuinely *slower* than a many-core CPU. Measured 2026-07-29 on an
|
|
69
|
+
# RTX 5060 Ti: 346 GFLOP/s FP64 versus a 20-core CPU's 792 (0.44×), with real
|
|
70
|
+
# B3LYP single points landing at 0.44–0.91× of CPU wall time.
|
|
71
|
+
#
|
|
72
|
+
# Matching is by device-name substring. Anything unrecognised is reported as
|
|
73
|
+
# low-FP64 **deliberately**: consumer hardware is the common case for students,
|
|
74
|
+
# and a spurious advisory on a brand-new datacenter card costs far less than
|
|
75
|
+
# silently halving someone's throughput. This only ever drives an advisory
|
|
76
|
+
# string — it never changes whether GPU offload runs.
|
|
77
|
+
_STRONG_FP64_MARKERS: tuple = (
|
|
78
|
+
"A100",
|
|
79
|
+
"A30",
|
|
80
|
+
"H100",
|
|
81
|
+
"H200",
|
|
82
|
+
"B100",
|
|
83
|
+
"B200",
|
|
84
|
+
"GB200",
|
|
85
|
+
"GH200",
|
|
86
|
+
"V100",
|
|
87
|
+
"P100",
|
|
88
|
+
"GV100",
|
|
89
|
+
"GP100",
|
|
90
|
+
"TITAN V",
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# Reason strings for the "GPU not in use" cases, surfaced by `quantui gpu check`
|
|
94
|
+
# and the Status tab. Kept here so the CLI and the UI can't drift from the
|
|
95
|
+
# probe's actual logic (they previously re-derived it, which is how a broken
|
|
96
|
+
# CUDA install came to be reported as "gpu4pyscf not installed").
|
|
97
|
+
_REASON_OK = ""
|
|
98
|
+
_REASON_ENV_DISABLED = "QUANTUI_DISABLE_GPU is set in the environment"
|
|
99
|
+
_REASON_SETTINGS_DISABLED = (
|
|
100
|
+
"GPU offload is switched off in QuantUI settings "
|
|
101
|
+
"(Status tab → Settings → GPU offload)"
|
|
102
|
+
)
|
|
103
|
+
_REASON_NOT_INSTALLED = (
|
|
104
|
+
"gpu4pyscf is not installed — install the extra matching your driver's "
|
|
105
|
+
"CUDA version, e.g. pip install 'quantui[gpu-cuda13x]' "
|
|
106
|
+
"(see README → 'Optional: GPU acceleration')"
|
|
107
|
+
)
|
|
108
|
+
_REASON_NO_DEVICE = "cupy reports 0 CUDA devices"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def is_low_fp64_device(name: Optional[str]) -> bool:
|
|
112
|
+
"""Return True when *name* is a GPU with crippled double-precision.
|
|
113
|
+
|
|
114
|
+
See ``_STRONG_FP64_MARKERS`` for the rationale, the measured numbers, and
|
|
115
|
+
why an unknown device is treated as low-FP64 rather than assumed fast.
|
|
116
|
+
"""
|
|
117
|
+
if not name:
|
|
118
|
+
return False
|
|
119
|
+
upper = name.upper()
|
|
120
|
+
return not any(marker in upper for marker in _STRONG_FP64_MARKERS)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _gpu_enabled_in_settings() -> bool:
|
|
124
|
+
"""Read the persistent ``compute.gpu_enabled`` preference.
|
|
125
|
+
|
|
126
|
+
Any failure reading settings returns True — a broken or unreadable
|
|
127
|
+
settings file must never be what silently disables the GPU.
|
|
128
|
+
"""
|
|
129
|
+
try:
|
|
130
|
+
from quantui.user_settings import UserSettings
|
|
131
|
+
|
|
132
|
+
return bool(UserSettings.load().compute.gpu_enabled)
|
|
133
|
+
except Exception as exc: # noqa: BLE001 — settings must never gate compute
|
|
134
|
+
logger.debug("could not read compute.gpu_enabled, assuming True: %s", exc)
|
|
135
|
+
return True
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@lru_cache(maxsize=1)
|
|
139
|
+
def _probe_gpu() -> Tuple[bool, Optional[str], str]:
|
|
140
|
+
"""Probe for a usable GPU. Returns ``(available, name, reason)``.
|
|
141
|
+
|
|
142
|
+
``reason`` is empty when available and otherwise a human-readable
|
|
143
|
+
explanation suitable for showing a user directly.
|
|
144
|
+
|
|
145
|
+
The check sequence:
|
|
146
|
+
|
|
147
|
+
1. ``QUANTUI_DISABLE_GPU=1`` → disabled.
|
|
148
|
+
2. ``compute.gpu_enabled = false`` in user settings → disabled.
|
|
149
|
+
3. ``import gpu4pyscf`` — distinguishing "package absent"
|
|
150
|
+
(``ModuleNotFoundError``) from "present but its import chain is broken"
|
|
151
|
+
(any other ``ImportError``, typically missing CUDA math libraries).
|
|
152
|
+
These are very different problems and must not share a message.
|
|
153
|
+
4. ``import cupy`` + ``cupy.cuda.runtime.getDeviceCount()``.
|
|
154
|
+
5. Read device 0's properties for a friendly name.
|
|
155
|
+
|
|
156
|
+
Failures at any step are swallowed and logged; this never raises.
|
|
157
|
+
"""
|
|
158
|
+
if os.environ.get("QUANTUI_DISABLE_GPU", "").strip() in ("1", "true", "True"):
|
|
159
|
+
return (False, None, _REASON_ENV_DISABLED)
|
|
160
|
+
if not _gpu_enabled_in_settings():
|
|
161
|
+
return (False, None, _REASON_SETTINGS_DISABLED)
|
|
162
|
+
try:
|
|
163
|
+
import gpu4pyscf # noqa: F401
|
|
164
|
+
except ModuleNotFoundError:
|
|
165
|
+
# The package genuinely isn't installed — the common "user didn't opt
|
|
166
|
+
# into the extra" path. Not an error worth logging at warning level.
|
|
167
|
+
logger.debug("gpu4pyscf is not installed")
|
|
168
|
+
return (False, None, _REASON_NOT_INSTALLED)
|
|
169
|
+
except ImportError as exc:
|
|
170
|
+
# gpu4pyscf IS installed but its import chain is broken — nearly always
|
|
171
|
+
# missing NVIDIA CUDA math libraries (e.g. libnvJitLink.so,
|
|
172
|
+
# libcublasLt.so), which the gpu4pyscf wheels do not declare as
|
|
173
|
+
# dependencies. Surfacing the real exception is the whole point: the old
|
|
174
|
+
# code reported this as "not installed" and sent users back to an
|
|
175
|
+
# install step they had already completed.
|
|
176
|
+
logger.warning("gpu4pyscf is installed but failed to import: %s", exc)
|
|
177
|
+
return (
|
|
178
|
+
False,
|
|
179
|
+
None,
|
|
180
|
+
f"gpu4pyscf is installed but failed to import: {exc}. This usually "
|
|
181
|
+
"means its CUDA libraries are missing — see README → 'Optional: "
|
|
182
|
+
"GPU acceleration'",
|
|
183
|
+
)
|
|
184
|
+
except Exception as exc: # noqa: BLE001 — any other import-chain breakage
|
|
185
|
+
logger.warning("gpu4pyscf import raised %s: %s", type(exc).__name__, exc)
|
|
186
|
+
return (
|
|
187
|
+
False,
|
|
188
|
+
None,
|
|
189
|
+
f"gpu4pyscf import raised {type(exc).__name__}: {exc}",
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
try:
|
|
193
|
+
import cupy as _cupy
|
|
194
|
+
|
|
195
|
+
n = int(_cupy.cuda.runtime.getDeviceCount())
|
|
196
|
+
if n < 1:
|
|
197
|
+
logger.debug("cupy reports 0 CUDA devices")
|
|
198
|
+
return (False, None, _REASON_NO_DEVICE)
|
|
199
|
+
props = _cupy.cuda.runtime.getDeviceProperties(0)
|
|
200
|
+
name_raw = props.get("name", b"GPU")
|
|
201
|
+
if isinstance(name_raw, bytes):
|
|
202
|
+
name = name_raw.decode("utf-8", errors="replace")
|
|
203
|
+
else:
|
|
204
|
+
name = str(name_raw)
|
|
205
|
+
return (True, name, _REASON_OK)
|
|
206
|
+
except Exception as exc: # noqa: BLE001 — fall-back to CPU on probe failure
|
|
207
|
+
logger.warning("cupy device probe failed: %s", exc)
|
|
208
|
+
return (False, None, f"cupy device probe failed: {exc}")
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def probe_gpu() -> Tuple[bool, Optional[str], str]:
|
|
212
|
+
"""Return ``(available, gpu_name, reason)`` — the full probe result.
|
|
213
|
+
|
|
214
|
+
Prefer this over :func:`is_gpu_available` when you need to tell the user
|
|
215
|
+
*why* the GPU isn't being used.
|
|
216
|
+
"""
|
|
217
|
+
return _probe_gpu()
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def is_gpu_available() -> Tuple[bool, Optional[str]]:
|
|
221
|
+
"""Return ``(available, gpu_name)`` for the current process.
|
|
222
|
+
|
|
223
|
+
Thin view over :func:`probe_gpu` that drops the reason, kept because it is
|
|
224
|
+
the long-standing call used by the run dispatcher and the Status tab.
|
|
225
|
+
Cached for the process lifetime — the answer doesn't change once the kernel
|
|
226
|
+
is up. Callers that need a re-check (a settings toggle, or a test
|
|
227
|
+
simulating driver loss) call ``is_gpu_available.cache_clear()``.
|
|
228
|
+
"""
|
|
229
|
+
available, name, _reason = _probe_gpu()
|
|
230
|
+
return (available, name)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# The cache lives on ``_probe_gpu`` now that two accessors share it. Forward the
|
|
234
|
+
# lru_cache surface so the historical ``is_gpu_available.cache_clear()`` /
|
|
235
|
+
# ``.cache_info()`` keep working — without this, existing callers would silently
|
|
236
|
+
# clear nothing and read a stale result.
|
|
237
|
+
is_gpu_available.cache_clear = _probe_gpu.cache_clear # type: ignore[attr-defined]
|
|
238
|
+
is_gpu_available.cache_info = _probe_gpu.cache_info # type: ignore[attr-defined]
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def try_to_gpu(mf: Any, method_upper: str) -> Tuple[Any, bool, Optional[str]]:
|
|
242
|
+
"""Attempt to migrate a PySCF SCF object to GPU. Safe CPU fallback.
|
|
243
|
+
|
|
244
|
+
Parameters
|
|
245
|
+
----------
|
|
246
|
+
mf:
|
|
247
|
+
A constructed PySCF mean-field object (``scf.RHF(mol)``,
|
|
248
|
+
``dft.RKS(mol)``, …) BEFORE ``mf.kernel()`` is called. ``to_gpu``
|
|
249
|
+
on a converged object is undefined behaviour in current gpu4pyscf.
|
|
250
|
+
method_upper:
|
|
251
|
+
Upper-cased method name (e.g. ``"RHF"``, ``"B3LYP"``, ``"CCSD(T)"``).
|
|
252
|
+
Used only to skip GPU for methods that gpu4pyscf doesn't support.
|
|
253
|
+
|
|
254
|
+
Returns
|
|
255
|
+
-------
|
|
256
|
+
tuple ``(maybe_gpu_mf, used_gpu, gpu_name)``:
|
|
257
|
+
- ``maybe_gpu_mf`` is the (possibly converted) SCF object the
|
|
258
|
+
caller should use for ``.kernel()``. Always usable — the
|
|
259
|
+
original ``mf`` is returned unchanged on any failure.
|
|
260
|
+
- ``used_gpu`` is ``True`` only when conversion succeeded.
|
|
261
|
+
- ``gpu_name`` is the device name when ``used_gpu`` is True,
|
|
262
|
+
``None`` otherwise.
|
|
263
|
+
"""
|
|
264
|
+
if method_upper in _GPU_UNSUPPORTED_METHODS:
|
|
265
|
+
return (mf, False, None)
|
|
266
|
+
available, gpu_name = is_gpu_available()
|
|
267
|
+
if not available:
|
|
268
|
+
return (mf, False, None)
|
|
269
|
+
try:
|
|
270
|
+
mf_gpu = mf.to_gpu()
|
|
271
|
+
return (mf_gpu, True, gpu_name)
|
|
272
|
+
except Exception as exc:
|
|
273
|
+
# gpu4pyscf migration can fail for many reasons (unsupported method
|
|
274
|
+
# variant, density-fitting requirement, basis-set quirk). On any
|
|
275
|
+
# failure we fall back to CPU — the calc still runs. Log so the
|
|
276
|
+
# user can `quantui log tail` and see why offload didn't happen.
|
|
277
|
+
logger.warning("mf.to_gpu() migration failed, falling back to CPU: %s", exc)
|
|
278
|
+
return (mf, False, None)
|